preflight

What a fine-tune of size N actually costs on this machine, before you start it.

Answers the three questions that decide whether a run finishes: does it fit in VRAM, does the fp16 merge fit in RAM, and does the export fit on disk. Printed up front so a 27B run fails in a second with numbers rather than four hours in with an OOM.

Measure, never assume. os.statvfs is taken on the directory the artifacts are actually written to — on 2026-09-07 df /home inside a sandboxed shell reported 95 GB while the repo's own filesystem had 7 TB, which is the difference between "27B is impossible here" and "27B is fine".

python finetune/preflight.py --preset qwen-27b
python finetune/preflight.py --preset qwen-7b --device cpu
  1#!/usr/bin/env python3
  2"""What a fine-tune of size N actually costs on this machine, before you start it.
  3
  4Answers the three questions that decide whether a run finishes: does it fit in VRAM, does the
  5fp16 merge fit in RAM, and does the export fit on disk. Printed up front so a 27B run fails in a
  6second with numbers rather than four hours in with an OOM.
  7
  8Measure, never assume. `os.statvfs` is taken on the directory the artifacts are actually written
  9to — on 2026-09-07 `df /home` inside a sandboxed shell reported 95 GB while the repo's own
 10filesystem had 7 TB, which is the difference between "27B is impossible here" and "27B is fine".
 11
 12    python finetune/preflight.py --preset qwen-27b
 13    python finetune/preflight.py --preset qwen-7b --device cpu
 14"""
 15
 16from __future__ import annotations
 17
 18import argparse
 19import os
 20import shutil
 21import sys
 22from pathlib import Path
 23
 24sys.path.insert(0, str(Path(__file__).parent))
 25from train_lora import PRESETS, preset_for, run_dir  # noqa: E402
 26
 27HERE = Path(__file__).parent
 28
 29# Bytes per parameter, by role. QLoRA holds the base in 4-bit (~0.55 B/param including the
 30# quantization constants) plus LoRA weights, gradients and optimizer state for the adapter only —
 31# the adapter is <1% of params, so activations and the KV cache dominate the remainder.
 32BYTES_4BIT = 0.55
 33BYTES_FP16 = 2.0
 34QUANT_RATIO = {"Q4_K_M": 0.60, "Q8_0": 1.10}
 35
 36
 37def free_disk_gb(path: Path) -> float:
 38    s = os.statvfs(path)
 39    return s.f_frsize * s.f_bavail / 1e9
 40
 41
 42def total_ram_gb() -> float:
 43    try:
 44        return os.sysconf("SC_PAGE_SIZE") * os.sysconf("SC_PHYS_PAGES") / 1e9
 45    except (ValueError, OSError):  # pragma: no cover - platform-dependent
 46        return 0.0
 47
 48
 49def vram_gb() -> float:
 50    """Total VRAM of device 0, or 0.0 when there is no usable CUDA device. Deliberately does not
 51    import torch — preflight must run on a machine that has not installed it yet."""
 52    exe = shutil.which("nvidia-smi")
 53    if not exe:
 54        return 0.0
 55    import subprocess
 56
 57    try:
 58        out = subprocess.run(
 59            [exe, "--query-gpu=memory.total", "--format=csv,noheader,nounits"],
 60            capture_output=True,
 61            text=True,
 62            timeout=15,
 63        )
 64        if out.returncode != 0:
 65            return 0.0
 66        return float(out.stdout.strip().splitlines()[0]) / 1024
 67    except (ValueError, IndexError, OSError, subprocess.SubprocessError):
 68        return 0.0
 69
 70
 71def estimate(params_b: float, device: str, low_disk: bool) -> dict:
 72    """Peak VRAM / RAM / disk for one full train -> merge -> export cycle."""
 73    train_vram = params_b * BYTES_4BIT + 2.5 if device != "cpu" else 0.0
 74    train_ram = params_b * BYTES_FP16 + 4 if device == "cpu" else 4.0
 75    merge_ram = params_b * BYTES_FP16 + 4
 76    merged = params_b * BYTES_FP16
 77    quants = sum(params_b * r for r in QUANT_RATIO.values())
 78    # Default keeps the f16 GGUF alongside the merged model; --low-disk converts straight to Q8_0
 79    # and drops the merged copy first, so the peak is one large artifact rather than two.
 80    disk_peak = merged + (quants if low_disk else merged + quants)
 81    return {
 82        "train_vram": train_vram,
 83        "ram": max(train_ram, merge_ram),
 84        "disk_peak": disk_peak,
 85        "disk_final": quants,
 86    }
 87
 88
 89# Rough wall-clock for 3 SFT epochs + 1 DPO epoch over ~2.8k short examples. Anchored on the one
 90# run this repo has actually measured (SmolLM2-360M, CPU) and scaled by parameter count; treat as
 91# an order of magnitude, not a promise.
 92def eta_hours(params_b: float, device: str) -> float:
 93    if device == "cpu":
 94        return 0.25 * (params_b / 0.36) ** 1.1
 95    return 0.02 + 0.09 * params_b
 96
 97
 98def main(argv: list[str] | None = None) -> int:
 99    ap = argparse.ArgumentParser(
100        description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
101    )
102    ap.add_argument("--preset", choices=sorted(PRESETS), default="tiny")
103    ap.add_argument(
104        "--base",
105        default=None,
106        help="explicit HF model id; overrides --preset, exactly as in train_lora.py — so the "
107        "estimate is for the run you are actually about to start",
108    )
109    ap.add_argument("--device", choices=("auto", "cuda", "cpu"), default="auto")
110    ap.add_argument("--low-disk", action="store_true")
111    ap.add_argument(
112        "--out",
113        default=None,
114        help="the run directory to measure free disk against (default: finetune/out/<preset>)",
115    )
116    ap.add_argument("--strict", action="store_true", help="exit 1 when the run does not fit")
117    ap.add_argument("--self-test", action="store_true")
118    args = ap.parse_args(argv)
119
120    if args.self_test:
121        return self_test()
122
123    base = args.base or PRESETS[args.preset]["hf"]
124    preset_name = preset_for(base) or args.preset
125    preset = PRESETS.get(preset_name, {})
126    if args.base and not preset:
127        raise SystemExit(
128            f"preflight: {base!r} is not in the preset table, so its size is unknown — add it to "
129            "train_lora.PRESETS, or use --preset for an estimate of a comparable model"
130        )
131    params_b = preset["params_b"]
132    have_vram = vram_gb()
133    device = args.device
134    if device == "auto":
135        device = "cuda" if have_vram > 0 else "cpu"
136
137    need = estimate(params_b, device, args.low_disk)
138    out = Path(args.out) if args.out else run_dir(base)
139    out.mkdir(parents=True, exist_ok=True)
140    have = {"vram": have_vram, "ram": total_ram_gb(), "disk": free_disk_gb(out)}
141
142    print(f"preset      : {preset_name}  ({preset['hf']}, {params_b:g}B, {preset['licence']})")
143    print(f"run dir     : {out}")
144    if preset.get("thinking"):
145        print(
146            "              NOTE: this base's template emits reasoning blocks, so the fine-tune "
147            "teaches that shape too"
148        )
149    print(
150        f"device      : {device}{'  (no usable CUDA device found)' if device == 'cpu' and args.device == 'auto' else ''}"
151    )
152    print()
153    rows = [
154        ("VRAM (train)", need["train_vram"], have["vram"], device == "cuda"),
155        ("RAM  (merge)", need["ram"], have["ram"], True),
156        ("disk (peak) ", need["disk_peak"], have["disk"], True),
157    ]
158    ok = True
159    for label, want, got, applies in rows:
160        if not applies:
161            print(f"  {label}  n/a on cpu")
162            continue
163        verdict = "OK" if got >= want else "DOES NOT FIT"
164        if got < want:
165            ok = False
166        print(f"  {label}  need {want:7.1f} GB   have {got:7.1f} GB   {verdict}")
167    print(f"  disk (kept) {need['disk_final']:7.1f} GB of GGUFs after cleanup")
168    print()
169    print(
170        f"estimated wall clock: ~{eta_hours(params_b, device):.1f} h for 3 SFT + 1 DPO epoch (rough)"
171    )
172    if device == "cpu" and params_b > 3:
173        print(
174            "  WARNING: CPU training above ~3B is measured in days. Use a GPU or a smaller preset."
175        )
176    if not ok:
177        print(
178            "\ndoes not fit — try --low-disk, a smaller preset, or free the resource above",
179            file=sys.stderr,
180        )
181        if args.strict:
182            return 1
183    return 0
184
185
186def self_test() -> int:
187    fails = 0
188
189    def ok(got, want, label):
190        nonlocal fails
191        if got == want:
192            print(f"  ok   {label}")
193        else:
194            fails += 1
195            print(f"  FAIL {label} — got {got!r}, want {want!r}")
196
197    e = estimate(27.0, "cuda", low_disk=False)
198    ok(round(e["train_vram"]) == 17, True, "27B QLoRA needs ~17 GB VRAM — fits a 24 GB card")
199    ok(round(e["disk_peak"]) == 154, True, "27B full export peaks at ~154 GB of disk")
200    low = estimate(27.0, "cuda", low_disk=True)
201    ok(low["disk_peak"] < e["disk_peak"], True, "--low-disk lowers the disk peak")
202    ok(round(low["disk_peak"]) == 100, True, "--low-disk peaks at ~100 GB instead of ~154 GB")
203    ok(estimate(0.36, "cpu", False)["train_vram"] == 0.0, True, "cpu training asks for no VRAM")
204    ok(
205        eta_hours(27.0, "cpu") > eta_hours(27.0, "cuda"),
206        True,
207        "cpu is slower than cuda at the same size",
208    )
209    ok(eta_hours(7.6, "cuda") < 1.0, True, "a 7B QLoRA run is under an hour on a GPU")
210    ok(free_disk_gb(HERE) > 0, True, "free_disk_gb reads the real filesystem, not a parent mount")
211    ok(vram_gb() >= 0.0, True, "vram_gb degrades to 0.0 rather than raising without a driver")
212    if fails:
213        print(f"preflight self-test: {fails} failure(s)", file=sys.stderr)
214        return 1
215    print("preflight self-test: ok")
216    return 0
217
218
219if __name__ == "__main__":
220    sys.exit(main())
HERE = PosixPath('/home/runner/work/marola/marola/finetune')
BYTES_4BIT = 0.55
BYTES_FP16 = 2.0
QUANT_RATIO = {'Q4_K_M': 0.6, 'Q8_0': 1.1}
def free_disk_gb(path: pathlib.Path) -> float:
38def free_disk_gb(path: Path) -> float:
39    s = os.statvfs(path)
40    return s.f_frsize * s.f_bavail / 1e9
def total_ram_gb() -> float:
43def total_ram_gb() -> float:
44    try:
45        return os.sysconf("SC_PAGE_SIZE") * os.sysconf("SC_PHYS_PAGES") / 1e9
46    except (ValueError, OSError):  # pragma: no cover - platform-dependent
47        return 0.0
def vram_gb() -> float:
50def vram_gb() -> float:
51    """Total VRAM of device 0, or 0.0 when there is no usable CUDA device. Deliberately does not
52    import torch — preflight must run on a machine that has not installed it yet."""
53    exe = shutil.which("nvidia-smi")
54    if not exe:
55        return 0.0
56    import subprocess
57
58    try:
59        out = subprocess.run(
60            [exe, "--query-gpu=memory.total", "--format=csv,noheader,nounits"],
61            capture_output=True,
62            text=True,
63            timeout=15,
64        )
65        if out.returncode != 0:
66            return 0.0
67        return float(out.stdout.strip().splitlines()[0]) / 1024
68    except (ValueError, IndexError, OSError, subprocess.SubprocessError):
69        return 0.0

Total VRAM of device 0, or 0.0 when there is no usable CUDA device. Deliberately does not import torch — preflight must run on a machine that has not installed it yet.

def estimate(params_b: float, device: str, low_disk: bool) -> dict:
72def estimate(params_b: float, device: str, low_disk: bool) -> dict:
73    """Peak VRAM / RAM / disk for one full train -> merge -> export cycle."""
74    train_vram = params_b * BYTES_4BIT + 2.5 if device != "cpu" else 0.0
75    train_ram = params_b * BYTES_FP16 + 4 if device == "cpu" else 4.0
76    merge_ram = params_b * BYTES_FP16 + 4
77    merged = params_b * BYTES_FP16
78    quants = sum(params_b * r for r in QUANT_RATIO.values())
79    # Default keeps the f16 GGUF alongside the merged model; --low-disk converts straight to Q8_0
80    # and drops the merged copy first, so the peak is one large artifact rather than two.
81    disk_peak = merged + (quants if low_disk else merged + quants)
82    return {
83        "train_vram": train_vram,
84        "ram": max(train_ram, merge_ram),
85        "disk_peak": disk_peak,
86        "disk_final": quants,
87    }

Peak VRAM / RAM / disk for one full train -> merge -> export cycle.

def eta_hours(params_b: float, device: str) -> float:
93def eta_hours(params_b: float, device: str) -> float:
94    if device == "cpu":
95        return 0.25 * (params_b / 0.36) ** 1.1
96    return 0.02 + 0.09 * params_b
def main(argv: list[str] | None = None) -> int:
 99def main(argv: list[str] | None = None) -> int:
100    ap = argparse.ArgumentParser(
101        description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
102    )
103    ap.add_argument("--preset", choices=sorted(PRESETS), default="tiny")
104    ap.add_argument(
105        "--base",
106        default=None,
107        help="explicit HF model id; overrides --preset, exactly as in train_lora.py — so the "
108        "estimate is for the run you are actually about to start",
109    )
110    ap.add_argument("--device", choices=("auto", "cuda", "cpu"), default="auto")
111    ap.add_argument("--low-disk", action="store_true")
112    ap.add_argument(
113        "--out",
114        default=None,
115        help="the run directory to measure free disk against (default: finetune/out/<preset>)",
116    )
117    ap.add_argument("--strict", action="store_true", help="exit 1 when the run does not fit")
118    ap.add_argument("--self-test", action="store_true")
119    args = ap.parse_args(argv)
120
121    if args.self_test:
122        return self_test()
123
124    base = args.base or PRESETS[args.preset]["hf"]
125    preset_name = preset_for(base) or args.preset
126    preset = PRESETS.get(preset_name, {})
127    if args.base and not preset:
128        raise SystemExit(
129            f"preflight: {base!r} is not in the preset table, so its size is unknown — add it to "
130            "train_lora.PRESETS, or use --preset for an estimate of a comparable model"
131        )
132    params_b = preset["params_b"]
133    have_vram = vram_gb()
134    device = args.device
135    if device == "auto":
136        device = "cuda" if have_vram > 0 else "cpu"
137
138    need = estimate(params_b, device, args.low_disk)
139    out = Path(args.out) if args.out else run_dir(base)
140    out.mkdir(parents=True, exist_ok=True)
141    have = {"vram": have_vram, "ram": total_ram_gb(), "disk": free_disk_gb(out)}
142
143    print(f"preset      : {preset_name}  ({preset['hf']}, {params_b:g}B, {preset['licence']})")
144    print(f"run dir     : {out}")
145    if preset.get("thinking"):
146        print(
147            "              NOTE: this base's template emits reasoning blocks, so the fine-tune "
148            "teaches that shape too"
149        )
150    print(
151        f"device      : {device}{'  (no usable CUDA device found)' if device == 'cpu' and args.device == 'auto' else ''}"
152    )
153    print()
154    rows = [
155        ("VRAM (train)", need["train_vram"], have["vram"], device == "cuda"),
156        ("RAM  (merge)", need["ram"], have["ram"], True),
157        ("disk (peak) ", need["disk_peak"], have["disk"], True),
158    ]
159    ok = True
160    for label, want, got, applies in rows:
161        if not applies:
162            print(f"  {label}  n/a on cpu")
163            continue
164        verdict = "OK" if got >= want else "DOES NOT FIT"
165        if got < want:
166            ok = False
167        print(f"  {label}  need {want:7.1f} GB   have {got:7.1f} GB   {verdict}")
168    print(f"  disk (kept) {need['disk_final']:7.1f} GB of GGUFs after cleanup")
169    print()
170    print(
171        f"estimated wall clock: ~{eta_hours(params_b, device):.1f} h for 3 SFT + 1 DPO epoch (rough)"
172    )
173    if device == "cpu" and params_b > 3:
174        print(
175            "  WARNING: CPU training above ~3B is measured in days. Use a GPU or a smaller preset."
176        )
177    if not ok:
178        print(
179            "\ndoes not fit — try --low-disk, a smaller preset, or free the resource above",
180            file=sys.stderr,
181        )
182        if args.strict:
183            return 1
184    return 0
def self_test() -> int:
187def self_test() -> int:
188    fails = 0
189
190    def ok(got, want, label):
191        nonlocal fails
192        if got == want:
193            print(f"  ok   {label}")
194        else:
195            fails += 1
196            print(f"  FAIL {label} — got {got!r}, want {want!r}")
197
198    e = estimate(27.0, "cuda", low_disk=False)
199    ok(round(e["train_vram"]) == 17, True, "27B QLoRA needs ~17 GB VRAM — fits a 24 GB card")
200    ok(round(e["disk_peak"]) == 154, True, "27B full export peaks at ~154 GB of disk")
201    low = estimate(27.0, "cuda", low_disk=True)
202    ok(low["disk_peak"] < e["disk_peak"], True, "--low-disk lowers the disk peak")
203    ok(round(low["disk_peak"]) == 100, True, "--low-disk peaks at ~100 GB instead of ~154 GB")
204    ok(estimate(0.36, "cpu", False)["train_vram"] == 0.0, True, "cpu training asks for no VRAM")
205    ok(
206        eta_hours(27.0, "cpu") > eta_hours(27.0, "cuda"),
207        True,
208        "cpu is slower than cuda at the same size",
209    )
210    ok(eta_hours(7.6, "cuda") < 1.0, True, "a 7B QLoRA run is under an hour on a GPU")
211    ok(free_disk_gb(HERE) > 0, True, "free_disk_gb reads the real filesystem, not a parent mount")
212    ok(vram_gb() >= 0.0, True, "vram_gb degrades to 0.0 rather than raising without a driver")
213    if fails:
214        print(f"preflight self-test: {fails} failure(s)", file=sys.stderr)
215        return 1
216    print("preflight self-test: ok")
217    return 0