preflight
What a fine-tune of size N actually costs on this machine, before you start it.
Answers the three questions that decide whether a run finishes: does it fit in VRAM, does the fp16 merge fit in RAM, and does the export fit on disk. Printed up front so a 27B run fails in a second with numbers rather than four hours in with an OOM.
Measure, never assume. os.statvfs is taken on the directory the artifacts are actually written
to — on 2026-09-07 df /home inside a sandboxed shell reported 95 GB while the repo's own
filesystem had 7 TB, which is the difference between "27B is impossible here" and "27B is fine".
python finetune/preflight.py --preset qwen-27b
python finetune/preflight.py --preset qwen-7b --device cpu
1#!/usr/bin/env python3 2"""What a fine-tune of size N actually costs on this machine, before you start it. 3 4Answers the three questions that decide whether a run finishes: does it fit in VRAM, does the 5fp16 merge fit in RAM, and does the export fit on disk. Printed up front so a 27B run fails in a 6second with numbers rather than four hours in with an OOM. 7 8Measure, never assume. `os.statvfs` is taken on the directory the artifacts are actually written 9to — on 2026-09-07 `df /home` inside a sandboxed shell reported 95 GB while the repo's own 10filesystem had 7 TB, which is the difference between "27B is impossible here" and "27B is fine". 11 12 python finetune/preflight.py --preset qwen-27b 13 python finetune/preflight.py --preset qwen-7b --device cpu 14""" 15 16from __future__ import annotations 17 18import argparse 19import os 20import shutil 21import sys 22from pathlib import Path 23 24sys.path.insert(0, str(Path(__file__).parent)) 25from train_lora import PRESETS, preset_for, run_dir # noqa: E402 26 27HERE = Path(__file__).parent 28 29# Bytes per parameter, by role. QLoRA holds the base in 4-bit (~0.55 B/param including the 30# quantization constants) plus LoRA weights, gradients and optimizer state for the adapter only — 31# the adapter is <1% of params, so activations and the KV cache dominate the remainder. 32BYTES_4BIT = 0.55 33BYTES_FP16 = 2.0 34QUANT_RATIO = {"Q4_K_M": 0.60, "Q8_0": 1.10} 35 36 37def free_disk_gb(path: Path) -> float: 38 s = os.statvfs(path) 39 return s.f_frsize * s.f_bavail / 1e9 40 41 42def total_ram_gb() -> float: 43 try: 44 return os.sysconf("SC_PAGE_SIZE") * os.sysconf("SC_PHYS_PAGES") / 1e9 45 except (ValueError, OSError): # pragma: no cover - platform-dependent 46 return 0.0 47 48 49def vram_gb() -> float: 50 """Total VRAM of device 0, or 0.0 when there is no usable CUDA device. Deliberately does not 51 import torch — preflight must run on a machine that has not installed it yet.""" 52 exe = shutil.which("nvidia-smi") 53 if not exe: 54 return 0.0 55 import subprocess 56 57 try: 58 out = subprocess.run( 59 [exe, "--query-gpu=memory.total", "--format=csv,noheader,nounits"], 60 capture_output=True, 61 text=True, 62 timeout=15, 63 ) 64 if out.returncode != 0: 65 return 0.0 66 return float(out.stdout.strip().splitlines()[0]) / 1024 67 except (ValueError, IndexError, OSError, subprocess.SubprocessError): 68 return 0.0 69 70 71def estimate(params_b: float, device: str, low_disk: bool) -> dict: 72 """Peak VRAM / RAM / disk for one full train -> merge -> export cycle.""" 73 train_vram = params_b * BYTES_4BIT + 2.5 if device != "cpu" else 0.0 74 train_ram = params_b * BYTES_FP16 + 4 if device == "cpu" else 4.0 75 merge_ram = params_b * BYTES_FP16 + 4 76 merged = params_b * BYTES_FP16 77 quants = sum(params_b * r for r in QUANT_RATIO.values()) 78 # Default keeps the f16 GGUF alongside the merged model; --low-disk converts straight to Q8_0 79 # and drops the merged copy first, so the peak is one large artifact rather than two. 80 disk_peak = merged + (quants if low_disk else merged + quants) 81 return { 82 "train_vram": train_vram, 83 "ram": max(train_ram, merge_ram), 84 "disk_peak": disk_peak, 85 "disk_final": quants, 86 } 87 88 89# Rough wall-clock for 3 SFT epochs + 1 DPO epoch over ~2.8k short examples. Anchored on the one 90# run this repo has actually measured (SmolLM2-360M, CPU) and scaled by parameter count; treat as 91# an order of magnitude, not a promise. 92def eta_hours(params_b: float, device: str) -> float: 93 if device == "cpu": 94 return 0.25 * (params_b / 0.36) ** 1.1 95 return 0.02 + 0.09 * params_b 96 97 98def main(argv: list[str] | None = None) -> int: 99 ap = argparse.ArgumentParser( 100 description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter 101 ) 102 ap.add_argument("--preset", choices=sorted(PRESETS), default="tiny") 103 ap.add_argument( 104 "--base", 105 default=None, 106 help="explicit HF model id; overrides --preset, exactly as in train_lora.py — so the " 107 "estimate is for the run you are actually about to start", 108 ) 109 ap.add_argument("--device", choices=("auto", "cuda", "cpu"), default="auto") 110 ap.add_argument("--low-disk", action="store_true") 111 ap.add_argument( 112 "--out", 113 default=None, 114 help="the run directory to measure free disk against (default: finetune/out/<preset>)", 115 ) 116 ap.add_argument("--strict", action="store_true", help="exit 1 when the run does not fit") 117 ap.add_argument("--self-test", action="store_true") 118 args = ap.parse_args(argv) 119 120 if args.self_test: 121 return self_test() 122 123 base = args.base or PRESETS[args.preset]["hf"] 124 preset_name = preset_for(base) or args.preset 125 preset = PRESETS.get(preset_name, {}) 126 if args.base and not preset: 127 raise SystemExit( 128 f"preflight: {base!r} is not in the preset table, so its size is unknown — add it to " 129 "train_lora.PRESETS, or use --preset for an estimate of a comparable model" 130 ) 131 params_b = preset["params_b"] 132 have_vram = vram_gb() 133 device = args.device 134 if device == "auto": 135 device = "cuda" if have_vram > 0 else "cpu" 136 137 need = estimate(params_b, device, args.low_disk) 138 out = Path(args.out) if args.out else run_dir(base) 139 out.mkdir(parents=True, exist_ok=True) 140 have = {"vram": have_vram, "ram": total_ram_gb(), "disk": free_disk_gb(out)} 141 142 print(f"preset : {preset_name} ({preset['hf']}, {params_b:g}B, {preset['licence']})") 143 print(f"run dir : {out}") 144 if preset.get("thinking"): 145 print( 146 " NOTE: this base's template emits reasoning blocks, so the fine-tune " 147 "teaches that shape too" 148 ) 149 print( 150 f"device : {device}{' (no usable CUDA device found)' if device == 'cpu' and args.device == 'auto' else ''}" 151 ) 152 print() 153 rows = [ 154 ("VRAM (train)", need["train_vram"], have["vram"], device == "cuda"), 155 ("RAM (merge)", need["ram"], have["ram"], True), 156 ("disk (peak) ", need["disk_peak"], have["disk"], True), 157 ] 158 ok = True 159 for label, want, got, applies in rows: 160 if not applies: 161 print(f" {label} n/a on cpu") 162 continue 163 verdict = "OK" if got >= want else "DOES NOT FIT" 164 if got < want: 165 ok = False 166 print(f" {label} need {want:7.1f} GB have {got:7.1f} GB {verdict}") 167 print(f" disk (kept) {need['disk_final']:7.1f} GB of GGUFs after cleanup") 168 print() 169 print( 170 f"estimated wall clock: ~{eta_hours(params_b, device):.1f} h for 3 SFT + 1 DPO epoch (rough)" 171 ) 172 if device == "cpu" and params_b > 3: 173 print( 174 " WARNING: CPU training above ~3B is measured in days. Use a GPU or a smaller preset." 175 ) 176 if not ok: 177 print( 178 "\ndoes not fit — try --low-disk, a smaller preset, or free the resource above", 179 file=sys.stderr, 180 ) 181 if args.strict: 182 return 1 183 return 0 184 185 186def self_test() -> int: 187 fails = 0 188 189 def ok(got, want, label): 190 nonlocal fails 191 if got == want: 192 print(f" ok {label}") 193 else: 194 fails += 1 195 print(f" FAIL {label} — got {got!r}, want {want!r}") 196 197 e = estimate(27.0, "cuda", low_disk=False) 198 ok(round(e["train_vram"]) == 17, True, "27B QLoRA needs ~17 GB VRAM — fits a 24 GB card") 199 ok(round(e["disk_peak"]) == 154, True, "27B full export peaks at ~154 GB of disk") 200 low = estimate(27.0, "cuda", low_disk=True) 201 ok(low["disk_peak"] < e["disk_peak"], True, "--low-disk lowers the disk peak") 202 ok(round(low["disk_peak"]) == 100, True, "--low-disk peaks at ~100 GB instead of ~154 GB") 203 ok(estimate(0.36, "cpu", False)["train_vram"] == 0.0, True, "cpu training asks for no VRAM") 204 ok( 205 eta_hours(27.0, "cpu") > eta_hours(27.0, "cuda"), 206 True, 207 "cpu is slower than cuda at the same size", 208 ) 209 ok(eta_hours(7.6, "cuda") < 1.0, True, "a 7B QLoRA run is under an hour on a GPU") 210 ok(free_disk_gb(HERE) > 0, True, "free_disk_gb reads the real filesystem, not a parent mount") 211 ok(vram_gb() >= 0.0, True, "vram_gb degrades to 0.0 rather than raising without a driver") 212 if fails: 213 print(f"preflight self-test: {fails} failure(s)", file=sys.stderr) 214 return 1 215 print("preflight self-test: ok") 216 return 0 217 218 219if __name__ == "__main__": 220 sys.exit(main())
50def vram_gb() -> float: 51 """Total VRAM of device 0, or 0.0 when there is no usable CUDA device. Deliberately does not 52 import torch — preflight must run on a machine that has not installed it yet.""" 53 exe = shutil.which("nvidia-smi") 54 if not exe: 55 return 0.0 56 import subprocess 57 58 try: 59 out = subprocess.run( 60 [exe, "--query-gpu=memory.total", "--format=csv,noheader,nounits"], 61 capture_output=True, 62 text=True, 63 timeout=15, 64 ) 65 if out.returncode != 0: 66 return 0.0 67 return float(out.stdout.strip().splitlines()[0]) / 1024 68 except (ValueError, IndexError, OSError, subprocess.SubprocessError): 69 return 0.0
Total VRAM of device 0, or 0.0 when there is no usable CUDA device. Deliberately does not import torch — preflight must run on a machine that has not installed it yet.
72def estimate(params_b: float, device: str, low_disk: bool) -> dict: 73 """Peak VRAM / RAM / disk for one full train -> merge -> export cycle.""" 74 train_vram = params_b * BYTES_4BIT + 2.5 if device != "cpu" else 0.0 75 train_ram = params_b * BYTES_FP16 + 4 if device == "cpu" else 4.0 76 merge_ram = params_b * BYTES_FP16 + 4 77 merged = params_b * BYTES_FP16 78 quants = sum(params_b * r for r in QUANT_RATIO.values()) 79 # Default keeps the f16 GGUF alongside the merged model; --low-disk converts straight to Q8_0 80 # and drops the merged copy first, so the peak is one large artifact rather than two. 81 disk_peak = merged + (quants if low_disk else merged + quants) 82 return { 83 "train_vram": train_vram, 84 "ram": max(train_ram, merge_ram), 85 "disk_peak": disk_peak, 86 "disk_final": quants, 87 }
Peak VRAM / RAM / disk for one full train -> merge -> export cycle.
99def main(argv: list[str] | None = None) -> int: 100 ap = argparse.ArgumentParser( 101 description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter 102 ) 103 ap.add_argument("--preset", choices=sorted(PRESETS), default="tiny") 104 ap.add_argument( 105 "--base", 106 default=None, 107 help="explicit HF model id; overrides --preset, exactly as in train_lora.py — so the " 108 "estimate is for the run you are actually about to start", 109 ) 110 ap.add_argument("--device", choices=("auto", "cuda", "cpu"), default="auto") 111 ap.add_argument("--low-disk", action="store_true") 112 ap.add_argument( 113 "--out", 114 default=None, 115 help="the run directory to measure free disk against (default: finetune/out/<preset>)", 116 ) 117 ap.add_argument("--strict", action="store_true", help="exit 1 when the run does not fit") 118 ap.add_argument("--self-test", action="store_true") 119 args = ap.parse_args(argv) 120 121 if args.self_test: 122 return self_test() 123 124 base = args.base or PRESETS[args.preset]["hf"] 125 preset_name = preset_for(base) or args.preset 126 preset = PRESETS.get(preset_name, {}) 127 if args.base and not preset: 128 raise SystemExit( 129 f"preflight: {base!r} is not in the preset table, so its size is unknown — add it to " 130 "train_lora.PRESETS, or use --preset for an estimate of a comparable model" 131 ) 132 params_b = preset["params_b"] 133 have_vram = vram_gb() 134 device = args.device 135 if device == "auto": 136 device = "cuda" if have_vram > 0 else "cpu" 137 138 need = estimate(params_b, device, args.low_disk) 139 out = Path(args.out) if args.out else run_dir(base) 140 out.mkdir(parents=True, exist_ok=True) 141 have = {"vram": have_vram, "ram": total_ram_gb(), "disk": free_disk_gb(out)} 142 143 print(f"preset : {preset_name} ({preset['hf']}, {params_b:g}B, {preset['licence']})") 144 print(f"run dir : {out}") 145 if preset.get("thinking"): 146 print( 147 " NOTE: this base's template emits reasoning blocks, so the fine-tune " 148 "teaches that shape too" 149 ) 150 print( 151 f"device : {device}{' (no usable CUDA device found)' if device == 'cpu' and args.device == 'auto' else ''}" 152 ) 153 print() 154 rows = [ 155 ("VRAM (train)", need["train_vram"], have["vram"], device == "cuda"), 156 ("RAM (merge)", need["ram"], have["ram"], True), 157 ("disk (peak) ", need["disk_peak"], have["disk"], True), 158 ] 159 ok = True 160 for label, want, got, applies in rows: 161 if not applies: 162 print(f" {label} n/a on cpu") 163 continue 164 verdict = "OK" if got >= want else "DOES NOT FIT" 165 if got < want: 166 ok = False 167 print(f" {label} need {want:7.1f} GB have {got:7.1f} GB {verdict}") 168 print(f" disk (kept) {need['disk_final']:7.1f} GB of GGUFs after cleanup") 169 print() 170 print( 171 f"estimated wall clock: ~{eta_hours(params_b, device):.1f} h for 3 SFT + 1 DPO epoch (rough)" 172 ) 173 if device == "cpu" and params_b > 3: 174 print( 175 " WARNING: CPU training above ~3B is measured in days. Use a GPU or a smaller preset." 176 ) 177 if not ok: 178 print( 179 "\ndoes not fit — try --low-disk, a smaller preset, or free the resource above", 180 file=sys.stderr, 181 ) 182 if args.strict: 183 return 1 184 return 0
187def self_test() -> int: 188 fails = 0 189 190 def ok(got, want, label): 191 nonlocal fails 192 if got == want: 193 print(f" ok {label}") 194 else: 195 fails += 1 196 print(f" FAIL {label} — got {got!r}, want {want!r}") 197 198 e = estimate(27.0, "cuda", low_disk=False) 199 ok(round(e["train_vram"]) == 17, True, "27B QLoRA needs ~17 GB VRAM — fits a 24 GB card") 200 ok(round(e["disk_peak"]) == 154, True, "27B full export peaks at ~154 GB of disk") 201 low = estimate(27.0, "cuda", low_disk=True) 202 ok(low["disk_peak"] < e["disk_peak"], True, "--low-disk lowers the disk peak") 203 ok(round(low["disk_peak"]) == 100, True, "--low-disk peaks at ~100 GB instead of ~154 GB") 204 ok(estimate(0.36, "cpu", False)["train_vram"] == 0.0, True, "cpu training asks for no VRAM") 205 ok( 206 eta_hours(27.0, "cpu") > eta_hours(27.0, "cuda"), 207 True, 208 "cpu is slower than cuda at the same size", 209 ) 210 ok(eta_hours(7.6, "cuda") < 1.0, True, "a 7B QLoRA run is under an hour on a GPU") 211 ok(free_disk_gb(HERE) > 0, True, "free_disk_gb reads the real filesystem, not a parent mount") 212 ok(vram_gb() >= 0.0, True, "vram_gb degrades to 0.0 rather than raising without a driver") 213 if fails: 214 print(f"preflight self-test: {fails} failure(s)", file=sys.stderr) 215 return 1 216 print("preflight self-test: ok") 217 return 0