merge_export
Merge a marola LoRA adapter into its base model and export runnable GGUFs (MIP-0025 §5.1).
This is the step between training and publishing, and the one that was missing: train_lora.py
and train_dpo.py produce a LoRA adapter, and convert_lora_to_gguf.py turns that adapter into
an adapter-GGUF. An adapter is not a model. It runs locally only because Ollama already holds the
base weights and Modelfile.adapter names them (FROM llama3.2:1b + ADAPTER ...).
That distinction is what breaks publishing. ollama run hf.co/<user>/<repo> pulls a repo and
expects standalone model GGUFs; handed a 17 MB adapter it has no base to attach it to. MIP-0025
§5.1's chain therefore starts with a peft merge — fold the adapter's weights into the base, then
convert, then quantize — which is what this script does:
adapter + base -> merged HF model -> f16 GGUF -> Q4_K_M + Q8_0 GGUF
Then publish_hf.py uploads the quantized files and ollama run hf.co/... actually works.
python finetune/merge_export.py --preset tiny --dry-run # print the plan, touch nothing
python finetune/merge_export.py --preset tiny --llama-cpp ~/src/llama.cpp
--dry-run and --self-test deliberately need neither torch nor peft nor llama.cpp: the planning
half is pure and testable on any machine, and only the merge itself needs the heavy dependencies
(pip install -r finetune/requirements.txt). Same lazy-import discipline as publish_hf.py.
1#!/usr/bin/env python3 2"""Merge a marola LoRA adapter into its base model and export runnable GGUFs (MIP-0025 §5.1). 3 4This is the step between training and publishing, and the one that was missing: `train_lora.py` 5and `train_dpo.py` produce a LoRA *adapter*, and `convert_lora_to_gguf.py` turns that adapter into 6an adapter-GGUF. An adapter is not a model. It runs locally only because Ollama already holds the 7base weights and `Modelfile.adapter` names them (`FROM llama3.2:1b` + `ADAPTER ...`). 8 9That distinction is what breaks publishing. `ollama run hf.co/<user>/<repo>` pulls a repo and 10expects standalone model GGUFs; handed a 17 MB adapter it has no base to attach it to. MIP-0025 11§5.1's chain therefore starts with a `peft` merge — fold the adapter's weights into the base, then 12convert, then quantize — which is what this script does: 13 14 adapter + base -> merged HF model -> f16 GGUF -> Q4_K_M + Q8_0 GGUF 15 16Then `publish_hf.py` uploads the quantized files and `ollama run hf.co/...` actually works. 17 18 python finetune/merge_export.py --preset tiny --dry-run # print the plan, touch nothing 19 python finetune/merge_export.py --preset tiny --llama-cpp ~/src/llama.cpp 20 21`--dry-run` and `--self-test` deliberately need neither torch nor peft nor llama.cpp: the planning 22half is pure and testable on any machine, and only the merge itself needs the heavy dependencies 23(`pip install -r finetune/requirements.txt`). Same lazy-import discipline as publish_hf.py. 24""" 25 26from __future__ import annotations 27 28import argparse 29import json 30import subprocess 31import sys 32from pathlib import Path 33 34sys.path.insert(0, str(Path(__file__).parent)) 35from train_lora import ( # noqa: E402 — same directory, shares the preset table 36 PRESETS, 37 check_adapter_base, 38 preset_for, 39 run_dir, 40 run_slug, 41) 42 43QUANTIZATIONS = ("Q4_K_M", "Q8_0") 44 45 46def default_adapter(out_dir: Path) -> Path: 47 """The best adapter present: DPO continues training *from* the SFT adapter, so when both 48 exist `out/<preset>/dpo-adapter` already contains the SFT weights and is the one to publish.""" 49 dpo, sft = out_dir / "dpo-adapter", out_dir / "adapter" 50 return dpo if dpo.exists() else sft 51 52 53def model_name(preset: str) -> str: 54 """`marola-sea-tiny` — the published model's stem, per MIP-0025 §5.1(3).""" 55 return f"marola-sea-{preset}" 56 57 58def prefixed_name(base_hf_id: str, name: str) -> str: 59 """Meta's Community Licence requires a Llama-derived model's name to START with `Llama-` 60 (MIP-0025 §5.1(3)). The obligation is a property of the base, so it is declared once in 61 train_lora.PRESETS (`name_prefix`) rather than re-guessed from the id here.""" 62 preset = PRESETS.get(preset_for(base_hf_id) or "", {}) 63 return preset.get("name_prefix", "") + name 64 65 66def gguf_paths(out_dir: Path, name: str) -> dict[str, Path]: 67 paths = {"f16": out_dir / f"{name}-f16.gguf"} 68 for q in QUANTIZATIONS: 69 paths[q] = out_dir / f"{name}-{q}.gguf" 70 return paths 71 72 73def convert_argv(llama_cpp: Path, merged: Path, out_f16: Path) -> list[str]: 74 return [ 75 sys.executable, 76 str(llama_cpp / "convert_hf_to_gguf.py"), 77 str(merged), 78 "--outfile", 79 str(out_f16), 80 "--outtype", 81 "f16", 82 ] 83 84 85def quantize_argv(llama_cpp: Path, out_f16: Path, out_q: Path, quant: str) -> list[str]: 86 """llama.cpp's quantizer moved from `./quantize` to `llama-quantize` and into build/bin; take 87 whichever exists so a user's checkout layout doesn't matter.""" 88 for candidate in ( 89 llama_cpp / "llama-quantize", 90 llama_cpp / "build" / "bin" / "llama-quantize", 91 llama_cpp / "quantize", 92 ): 93 if candidate.exists(): 94 return [str(candidate), str(out_f16), str(out_q), quant] 95 return ["llama-quantize", str(out_f16), str(out_q), quant] 96 97 98def plan(args) -> dict: 99 base = args.base or PRESETS[args.preset]["hf"] 100 # One directory per base — the same `out/<preset>/` layout train_lora.py writes into, so a 101 # Qwen merge can never pick up a SmolLM2 adapter or overwrite its GGUFs. 102 out_dir = Path(args.out) if args.out else run_dir(base) 103 name = prefixed_name(base, model_name(run_slug(base))) 104 return { 105 "base": base, 106 "adapter": Path(args.adapter) if args.adapter else default_adapter(out_dir), 107 "merged": out_dir / "merged", 108 "name": name, 109 "licence": PRESETS.get(preset_for(base) or "", {}).get("licence", "apache-2.0"), 110 "gguf": gguf_paths(out_dir, name), 111 } 112 113 114def plan_json(p: dict) -> dict: 115 """The publish step reads this instead of rebuilding names, paths and licence itself.""" 116 return { 117 "base": p["base"], 118 "name": p["name"], 119 "licence": p["licence"], 120 "adapter": str(p["adapter"]), 121 "merged": str(p["merged"]), 122 "gguf": {k: str(v) for k, v in p["gguf"].items()}, 123 } 124 125 126def merge(base_id: str, adapter: Path, merged_out: Path) -> None: 127 """The only part that needs the heavy dependencies — imported here so --dry-run/--self-test 128 stay runnable on a machine with neither.""" 129 try: 130 import torch 131 from peft import PeftModel 132 from transformers import AutoModelForCausalLM, AutoTokenizer 133 except ImportError as exc: # pragma: no cover - environment-dependent 134 raise SystemExit( 135 f"merge_export: {exc} — run `pip install -r finetune/requirements.txt` first " 136 "(torch/transformers/peft). --dry-run needs none of them." 137 ) from exc 138 139 print(f"loading base {base_id} ...") 140 model = AutoModelForCausalLM.from_pretrained(base_id, torch_dtype=torch.float16) 141 print(f"applying adapter {adapter} ...") 142 model = PeftModel.from_pretrained(model, str(adapter)) 143 print("merging adapter weights into the base ...") 144 model = model.merge_and_unload() 145 merged_out.mkdir(parents=True, exist_ok=True) 146 model.save_pretrained(str(merged_out)) 147 AutoTokenizer.from_pretrained(base_id).save_pretrained(str(merged_out)) 148 print(f"merged model written to {merged_out}") 149 150 151def self_test() -> int: 152 fails = 0 153 154 def ok(got, want, label): 155 nonlocal fails 156 if got == want: 157 print(f" ok {label}") 158 else: 159 fails += 1 160 print(f" FAIL {label} — got {got!r}, want {want!r}") 161 162 ok(model_name("tiny"), "marola-sea-tiny", "model_name follows MIP-0025 §5.1(3)'s stem") 163 ok( 164 prefixed_name("HuggingFaceTB/SmolLM2-360M-Instruct", "marola-sea-tiny"), 165 "marola-sea-tiny", 166 "a SmolLM2-derived model needs no Llama- prefix", 167 ) 168 ok( 169 prefixed_name("unsloth/Llama-3.2-1B-Instruct", "marola-sea-small"), 170 "Llama-marola-sea-small", 171 "a Llama-derived model MUST start with Llama- (Meta Community Licence)", 172 ) 173 ok( 174 prefixed_name("meta-llama/Llama-3.2-3B-Instruct", "marola-sea-base"), 175 "Llama-marola-sea-base", 176 "the gated base preset is Llama-derived too", 177 ) 178 ok( 179 prefixed_name("Qwen/Qwen2.5-7B-Instruct", "marola-sea-qwen-7b"), 180 "marola-sea-qwen-7b", 181 "an Apache-2.0 Qwen base carries no naming obligation", 182 ) 183 184 class _Args: 185 preset, base, adapter, out = "qwen-7b", None, None, None 186 187 j = plan_json(plan(_Args())) 188 ok(j["name"], "marola-sea-qwen-7b", "the plan JSON carries the published name") 189 ok(j["licence"], "apache-2.0", "the plan JSON carries the base's licence id") 190 ok( 191 j["gguf"]["Q4_K_M"].endswith("out/qwen-7b/marola-sea-qwen-7b-Q4_K_M.gguf"), 192 True, 193 "the plan JSON's GGUF path is the one merge_export actually writes", 194 ) 195 ok(all(isinstance(v, str) for v in j["gguf"].values()), True, "the plan JSON is serialisable") 196 197 names = sorted(p.name for p in gguf_paths(Path("/o"), "marola-sea-tiny").values()) 198 ok( 199 names, 200 ["marola-sea-tiny-Q4_K_M.gguf", "marola-sea-tiny-Q8_0.gguf", "marola-sea-tiny-f16.gguf"], 201 "exports f16 plus both quantizations MIP-0025 §5.1 asks for", 202 ) 203 argv = convert_argv(Path("/llama"), Path("/o/merged"), Path("/o/m-f16.gguf")) 204 ok( 205 argv[1], 206 "/llama/convert_hf_to_gguf.py", 207 "converts with convert_hf_to_gguf.py, not convert_lora_to_gguf.py", 208 ) 209 ok(argv[-1], "f16", "converts at f16 before quantizing") 210 ok( 211 quantize_argv(Path("/nope"), Path("/o/a.gguf"), Path("/o/b.gguf"), "Q4_K_M")[-1], 212 "Q4_K_M", 213 "quantize_argv passes the quantization type through", 214 ) 215 ok( 216 default_adapter(Path("/definitely/missing")).name, 217 "adapter", 218 "with no DPO adapter present, the SFT adapter is the one merged", 219 ) 220 # merge_export shells out to llama.cpp's convert_hf_to_gguf.py, whose vocab probe catches only 221 # FileNotFoundError: a missing sentencepiece surfaces as ModuleNotFoundError and kills the run. 222 # setup-ml-venv (labs/cuda) installs the requirements file as given and knows nothing of this. 223 req = (Path(__file__).parent / "requirements.txt").read_text() 224 for dep in ("gguf", "sentencepiece", "protobuf"): 225 ok( 226 dep in req, 227 True, 228 f"{dep} is in finetune/requirements.txt — convert_hf_to_gguf.py needs it", 229 ) 230 231 if fails: 232 print(f"merge_export self-test: {fails} failure(s)", file=sys.stderr) 233 return 1 234 print("merge_export self-test: ok") 235 return 0 236 237 238def main(argv: list[str] | None = None) -> int: 239 ap = argparse.ArgumentParser( 240 description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter 241 ) 242 ap.add_argument("--preset", choices=sorted(PRESETS), default="tiny") 243 ap.add_argument("--base", default=None, help="explicit HF model id; overrides --preset") 244 ap.add_argument( 245 "--adapter", 246 default=None, 247 help="LoRA adapter dir (default: out/<preset>/dpo-adapter, else out/<preset>/adapter)", 248 ) 249 ap.add_argument( 250 "--out", 251 default=None, 252 help="the run directory holding the adapter and the GGUFs " 253 "(default: finetune/out/<preset>, matching train_lora.py)", 254 ) 255 ap.add_argument( 256 "--llama-cpp", default=None, help="path to a llama.cpp checkout (for convert + quantize)" 257 ) 258 ap.add_argument( 259 "--skip-convert", action="store_true", help="merge only; leave GGUF conversion to you" 260 ) 261 ap.add_argument( 262 "--dry-run", action="store_true", help="print the plan and the exact commands, run nothing" 263 ) 264 ap.add_argument( 265 "--plan-json", 266 default=None, 267 help="also write the plan (name, licence, base, GGUF paths) here, for the publish step to " 268 "consume instead of rebuilding those strings itself", 269 ) 270 ap.add_argument("--self-test", action="store_true") 271 args = ap.parse_args(argv) 272 273 if args.self_test: 274 return self_test() 275 276 p = plan(args) 277 if args.plan_json: 278 dest = Path(args.plan_json) 279 dest.parent.mkdir(parents=True, exist_ok=True) 280 dest.write_text(json.dumps(plan_json(p), indent=2) + "\n") 281 print(f"plan : {dest}") 282 llama_cpp = Path(args.llama_cpp) if args.llama_cpp else None 283 print(f"base : {p['base']} ({p['licence']})") 284 print(f"adapter : {p['adapter']}") 285 print(f"merged : {p['merged']}") 286 for k, v in p["gguf"].items(): 287 print(f"gguf : {k:7} {v}") 288 289 if args.dry_run: 290 print("\n--- commands this would run ---") 291 print(f"# merge: peft merge_and_unload({p['base']} + {p['adapter']}) -> {p['merged']}") 292 if not args.skip_convert: 293 cc = llama_cpp or Path("<--llama-cpp>") 294 print(" ".join(convert_argv(cc, p["merged"], p["gguf"]["f16"]))) 295 for q in QUANTIZATIONS: 296 print(" ".join(quantize_argv(cc, p["gguf"]["f16"], p["gguf"][q], q))) 297 print("\n# then publish the QUANTIZED files (never the adapter):") 298 print( 299 f"just finetune-publish repo=<you>/{p['name']}-GGUF " 300 f"gguf={p['gguf']['Q4_K_M']} base={p['base']} --base-license {p['licence']}" 301 ) 302 print("dry run: nothing was written") 303 return 0 304 305 if not p["adapter"].exists(): 306 raise SystemExit( 307 f"merge_export: no adapter at {p['adapter']} — run `just finetune-train` (and " 308 "optionally `just finetune-train-dpo`) first" 309 ) 310 check_adapter_base(p["adapter"], p["base"]) 311 merge(p["base"], p["adapter"], p["merged"]) 312 313 if args.skip_convert: 314 print("--skip-convert: merged model only, no GGUF written") 315 return 0 316 if llama_cpp is None or not (llama_cpp / "convert_hf_to_gguf.py").exists(): 317 print( 318 f"\nmerged model is at {p['merged']}. Pass --llama-cpp <path to a llama.cpp checkout> " 319 "to convert and quantize it here, or run these yourself:", 320 file=sys.stderr, 321 ) 322 cc = llama_cpp or Path("<llama.cpp>") 323 print(" ".join(convert_argv(cc, p["merged"], p["gguf"]["f16"])), file=sys.stderr) 324 return 1 325 subprocess.run(convert_argv(llama_cpp, p["merged"], p["gguf"]["f16"]), check=True) 326 for q in QUANTIZATIONS: 327 subprocess.run(quantize_argv(llama_cpp, p["gguf"]["f16"], p["gguf"][q], q), check=True) 328 print(f"quantized {p['gguf'][q].name}") 329 print( 330 f"\nnext: just finetune-publish repo=<you>/{p['name']}-GGUF gguf={p['gguf']['Q4_K_M']} " 331 f"base={p['base']} --base-license {p['licence']}" 332 ) 333 return 0 334 335 336if __name__ == "__main__": 337 sys.exit(main())
47def default_adapter(out_dir: Path) -> Path: 48 """The best adapter present: DPO continues training *from* the SFT adapter, so when both 49 exist `out/<preset>/dpo-adapter` already contains the SFT weights and is the one to publish.""" 50 dpo, sft = out_dir / "dpo-adapter", out_dir / "adapter" 51 return dpo if dpo.exists() else sft
The best adapter present: DPO continues training from the SFT adapter, so when both
exist out/<preset>/dpo-adapter already contains the SFT weights and is the one to publish.
54def model_name(preset: str) -> str: 55 """`marola-sea-tiny` — the published model's stem, per MIP-0025 §5.1(3).""" 56 return f"marola-sea-{preset}"
marola-sea-tiny — the published model's stem, per MIP-0025 §5.1(3).
59def prefixed_name(base_hf_id: str, name: str) -> str: 60 """Meta's Community Licence requires a Llama-derived model's name to START with `Llama-` 61 (MIP-0025 §5.1(3)). The obligation is a property of the base, so it is declared once in 62 train_lora.PRESETS (`name_prefix`) rather than re-guessed from the id here.""" 63 preset = PRESETS.get(preset_for(base_hf_id) or "", {}) 64 return preset.get("name_prefix", "") + name
Meta's Community Licence requires a Llama-derived model's name to START with Llama-
(MIP-0025 §5.1(3)). The obligation is a property of the base, so it is declared once in
train_lora.PRESETS (name_prefix) rather than re-guessed from the id here.
86def quantize_argv(llama_cpp: Path, out_f16: Path, out_q: Path, quant: str) -> list[str]: 87 """llama.cpp's quantizer moved from `./quantize` to `llama-quantize` and into build/bin; take 88 whichever exists so a user's checkout layout doesn't matter.""" 89 for candidate in ( 90 llama_cpp / "llama-quantize", 91 llama_cpp / "build" / "bin" / "llama-quantize", 92 llama_cpp / "quantize", 93 ): 94 if candidate.exists(): 95 return [str(candidate), str(out_f16), str(out_q), quant] 96 return ["llama-quantize", str(out_f16), str(out_q), quant]
llama.cpp's quantizer moved from ./quantize to llama-quantize and into build/bin; take
whichever exists so a user's checkout layout doesn't matter.
99def plan(args) -> dict: 100 base = args.base or PRESETS[args.preset]["hf"] 101 # One directory per base — the same `out/<preset>/` layout train_lora.py writes into, so a 102 # Qwen merge can never pick up a SmolLM2 adapter or overwrite its GGUFs. 103 out_dir = Path(args.out) if args.out else run_dir(base) 104 name = prefixed_name(base, model_name(run_slug(base))) 105 return { 106 "base": base, 107 "adapter": Path(args.adapter) if args.adapter else default_adapter(out_dir), 108 "merged": out_dir / "merged", 109 "name": name, 110 "licence": PRESETS.get(preset_for(base) or "", {}).get("licence", "apache-2.0"), 111 "gguf": gguf_paths(out_dir, name), 112 }
115def plan_json(p: dict) -> dict: 116 """The publish step reads this instead of rebuilding names, paths and licence itself.""" 117 return { 118 "base": p["base"], 119 "name": p["name"], 120 "licence": p["licence"], 121 "adapter": str(p["adapter"]), 122 "merged": str(p["merged"]), 123 "gguf": {k: str(v) for k, v in p["gguf"].items()}, 124 }
The publish step reads this instead of rebuilding names, paths and licence itself.
127def merge(base_id: str, adapter: Path, merged_out: Path) -> None: 128 """The only part that needs the heavy dependencies — imported here so --dry-run/--self-test 129 stay runnable on a machine with neither.""" 130 try: 131 import torch 132 from peft import PeftModel 133 from transformers import AutoModelForCausalLM, AutoTokenizer 134 except ImportError as exc: # pragma: no cover - environment-dependent 135 raise SystemExit( 136 f"merge_export: {exc} — run `pip install -r finetune/requirements.txt` first " 137 "(torch/transformers/peft). --dry-run needs none of them." 138 ) from exc 139 140 print(f"loading base {base_id} ...") 141 model = AutoModelForCausalLM.from_pretrained(base_id, torch_dtype=torch.float16) 142 print(f"applying adapter {adapter} ...") 143 model = PeftModel.from_pretrained(model, str(adapter)) 144 print("merging adapter weights into the base ...") 145 model = model.merge_and_unload() 146 merged_out.mkdir(parents=True, exist_ok=True) 147 model.save_pretrained(str(merged_out)) 148 AutoTokenizer.from_pretrained(base_id).save_pretrained(str(merged_out)) 149 print(f"merged model written to {merged_out}")
The only part that needs the heavy dependencies — imported here so --dry-run/--self-test stay runnable on a machine with neither.
152def self_test() -> int: 153 fails = 0 154 155 def ok(got, want, label): 156 nonlocal fails 157 if got == want: 158 print(f" ok {label}") 159 else: 160 fails += 1 161 print(f" FAIL {label} — got {got!r}, want {want!r}") 162 163 ok(model_name("tiny"), "marola-sea-tiny", "model_name follows MIP-0025 §5.1(3)'s stem") 164 ok( 165 prefixed_name("HuggingFaceTB/SmolLM2-360M-Instruct", "marola-sea-tiny"), 166 "marola-sea-tiny", 167 "a SmolLM2-derived model needs no Llama- prefix", 168 ) 169 ok( 170 prefixed_name("unsloth/Llama-3.2-1B-Instruct", "marola-sea-small"), 171 "Llama-marola-sea-small", 172 "a Llama-derived model MUST start with Llama- (Meta Community Licence)", 173 ) 174 ok( 175 prefixed_name("meta-llama/Llama-3.2-3B-Instruct", "marola-sea-base"), 176 "Llama-marola-sea-base", 177 "the gated base preset is Llama-derived too", 178 ) 179 ok( 180 prefixed_name("Qwen/Qwen2.5-7B-Instruct", "marola-sea-qwen-7b"), 181 "marola-sea-qwen-7b", 182 "an Apache-2.0 Qwen base carries no naming obligation", 183 ) 184 185 class _Args: 186 preset, base, adapter, out = "qwen-7b", None, None, None 187 188 j = plan_json(plan(_Args())) 189 ok(j["name"], "marola-sea-qwen-7b", "the plan JSON carries the published name") 190 ok(j["licence"], "apache-2.0", "the plan JSON carries the base's licence id") 191 ok( 192 j["gguf"]["Q4_K_M"].endswith("out/qwen-7b/marola-sea-qwen-7b-Q4_K_M.gguf"), 193 True, 194 "the plan JSON's GGUF path is the one merge_export actually writes", 195 ) 196 ok(all(isinstance(v, str) for v in j["gguf"].values()), True, "the plan JSON is serialisable") 197 198 names = sorted(p.name for p in gguf_paths(Path("/o"), "marola-sea-tiny").values()) 199 ok( 200 names, 201 ["marola-sea-tiny-Q4_K_M.gguf", "marola-sea-tiny-Q8_0.gguf", "marola-sea-tiny-f16.gguf"], 202 "exports f16 plus both quantizations MIP-0025 §5.1 asks for", 203 ) 204 argv = convert_argv(Path("/llama"), Path("/o/merged"), Path("/o/m-f16.gguf")) 205 ok( 206 argv[1], 207 "/llama/convert_hf_to_gguf.py", 208 "converts with convert_hf_to_gguf.py, not convert_lora_to_gguf.py", 209 ) 210 ok(argv[-1], "f16", "converts at f16 before quantizing") 211 ok( 212 quantize_argv(Path("/nope"), Path("/o/a.gguf"), Path("/o/b.gguf"), "Q4_K_M")[-1], 213 "Q4_K_M", 214 "quantize_argv passes the quantization type through", 215 ) 216 ok( 217 default_adapter(Path("/definitely/missing")).name, 218 "adapter", 219 "with no DPO adapter present, the SFT adapter is the one merged", 220 ) 221 # merge_export shells out to llama.cpp's convert_hf_to_gguf.py, whose vocab probe catches only 222 # FileNotFoundError: a missing sentencepiece surfaces as ModuleNotFoundError and kills the run. 223 # setup-ml-venv (labs/cuda) installs the requirements file as given and knows nothing of this. 224 req = (Path(__file__).parent / "requirements.txt").read_text() 225 for dep in ("gguf", "sentencepiece", "protobuf"): 226 ok( 227 dep in req, 228 True, 229 f"{dep} is in finetune/requirements.txt — convert_hf_to_gguf.py needs it", 230 ) 231 232 if fails: 233 print(f"merge_export self-test: {fails} failure(s)", file=sys.stderr) 234 return 1 235 print("merge_export self-test: ok") 236 return 0
239def main(argv: list[str] | None = None) -> int: 240 ap = argparse.ArgumentParser( 241 description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter 242 ) 243 ap.add_argument("--preset", choices=sorted(PRESETS), default="tiny") 244 ap.add_argument("--base", default=None, help="explicit HF model id; overrides --preset") 245 ap.add_argument( 246 "--adapter", 247 default=None, 248 help="LoRA adapter dir (default: out/<preset>/dpo-adapter, else out/<preset>/adapter)", 249 ) 250 ap.add_argument( 251 "--out", 252 default=None, 253 help="the run directory holding the adapter and the GGUFs " 254 "(default: finetune/out/<preset>, matching train_lora.py)", 255 ) 256 ap.add_argument( 257 "--llama-cpp", default=None, help="path to a llama.cpp checkout (for convert + quantize)" 258 ) 259 ap.add_argument( 260 "--skip-convert", action="store_true", help="merge only; leave GGUF conversion to you" 261 ) 262 ap.add_argument( 263 "--dry-run", action="store_true", help="print the plan and the exact commands, run nothing" 264 ) 265 ap.add_argument( 266 "--plan-json", 267 default=None, 268 help="also write the plan (name, licence, base, GGUF paths) here, for the publish step to " 269 "consume instead of rebuilding those strings itself", 270 ) 271 ap.add_argument("--self-test", action="store_true") 272 args = ap.parse_args(argv) 273 274 if args.self_test: 275 return self_test() 276 277 p = plan(args) 278 if args.plan_json: 279 dest = Path(args.plan_json) 280 dest.parent.mkdir(parents=True, exist_ok=True) 281 dest.write_text(json.dumps(plan_json(p), indent=2) + "\n") 282 print(f"plan : {dest}") 283 llama_cpp = Path(args.llama_cpp) if args.llama_cpp else None 284 print(f"base : {p['base']} ({p['licence']})") 285 print(f"adapter : {p['adapter']}") 286 print(f"merged : {p['merged']}") 287 for k, v in p["gguf"].items(): 288 print(f"gguf : {k:7} {v}") 289 290 if args.dry_run: 291 print("\n--- commands this would run ---") 292 print(f"# merge: peft merge_and_unload({p['base']} + {p['adapter']}) -> {p['merged']}") 293 if not args.skip_convert: 294 cc = llama_cpp or Path("<--llama-cpp>") 295 print(" ".join(convert_argv(cc, p["merged"], p["gguf"]["f16"]))) 296 for q in QUANTIZATIONS: 297 print(" ".join(quantize_argv(cc, p["gguf"]["f16"], p["gguf"][q], q))) 298 print("\n# then publish the QUANTIZED files (never the adapter):") 299 print( 300 f"just finetune-publish repo=<you>/{p['name']}-GGUF " 301 f"gguf={p['gguf']['Q4_K_M']} base={p['base']} --base-license {p['licence']}" 302 ) 303 print("dry run: nothing was written") 304 return 0 305 306 if not p["adapter"].exists(): 307 raise SystemExit( 308 f"merge_export: no adapter at {p['adapter']} — run `just finetune-train` (and " 309 "optionally `just finetune-train-dpo`) first" 310 ) 311 check_adapter_base(p["adapter"], p["base"]) 312 merge(p["base"], p["adapter"], p["merged"]) 313 314 if args.skip_convert: 315 print("--skip-convert: merged model only, no GGUF written") 316 return 0 317 if llama_cpp is None or not (llama_cpp / "convert_hf_to_gguf.py").exists(): 318 print( 319 f"\nmerged model is at {p['merged']}. Pass --llama-cpp <path to a llama.cpp checkout> " 320 "to convert and quantize it here, or run these yourself:", 321 file=sys.stderr, 322 ) 323 cc = llama_cpp or Path("<llama.cpp>") 324 print(" ".join(convert_argv(cc, p["merged"], p["gguf"]["f16"])), file=sys.stderr) 325 return 1 326 subprocess.run(convert_argv(llama_cpp, p["merged"], p["gguf"]["f16"]), check=True) 327 for q in QUANTIZATIONS: 328 subprocess.run(quantize_argv(llama_cpp, p["gguf"]["f16"], p["gguf"][q], q), check=True) 329 print(f"quantized {p['gguf'][q].name}") 330 print( 331 f"\nnext: just finetune-publish repo=<you>/{p['name']}-GGUF gguf={p['gguf']['Q4_K_M']} " 332 f"base={p['base']} --base-license {p['licence']}" 333 ) 334 return 0