merge_export

Merge a marola LoRA adapter into its base model and export runnable GGUFs (MIP-0025 §5.1).

This is the step between training and publishing, and the one that was missing: train_lora.py and train_dpo.py produce a LoRA adapter, and convert_lora_to_gguf.py turns that adapter into an adapter-GGUF. An adapter is not a model. It runs locally only because Ollama already holds the base weights and Modelfile.adapter names them (FROM llama3.2:1b + ADAPTER ...).

That distinction is what breaks publishing. ollama run hf.co/<user>/<repo> pulls a repo and expects standalone model GGUFs; handed a 17 MB adapter it has no base to attach it to. MIP-0025 §5.1's chain therefore starts with a peft merge — fold the adapter's weights into the base, then convert, then quantize — which is what this script does:

adapter + base  ->  merged HF model  ->  f16 GGUF  ->  Q4_K_M + Q8_0 GGUF

Then publish_hf.py uploads the quantized files and ollama run hf.co/... actually works.

python finetune/merge_export.py --preset tiny --dry-run     # print the plan, touch nothing
python finetune/merge_export.py --preset tiny --llama-cpp ~/src/llama.cpp

--dry-run and --self-test deliberately need neither torch nor peft nor llama.cpp: the planning half is pure and testable on any machine, and only the merge itself needs the heavy dependencies (pip install -r finetune/requirements.txt). Same lazy-import discipline as publish_hf.py.

  1#!/usr/bin/env python3
  2"""Merge a marola LoRA adapter into its base model and export runnable GGUFs (MIP-0025 §5.1).
  3
  4This is the step between training and publishing, and the one that was missing: `train_lora.py`
  5and `train_dpo.py` produce a LoRA *adapter*, and `convert_lora_to_gguf.py` turns that adapter into
  6an adapter-GGUF. An adapter is not a model. It runs locally only because Ollama already holds the
  7base weights and `Modelfile.adapter` names them (`FROM llama3.2:1b` + `ADAPTER ...`).
  8
  9That distinction is what breaks publishing. `ollama run hf.co/<user>/<repo>` pulls a repo and
 10expects standalone model GGUFs; handed a 17 MB adapter it has no base to attach it to. MIP-0025
 11§5.1's chain therefore starts with a `peft` merge — fold the adapter's weights into the base, then
 12convert, then quantize — which is what this script does:
 13
 14    adapter + base  ->  merged HF model  ->  f16 GGUF  ->  Q4_K_M + Q8_0 GGUF
 15
 16Then `publish_hf.py` uploads the quantized files and `ollama run hf.co/...` actually works.
 17
 18    python finetune/merge_export.py --preset tiny --dry-run     # print the plan, touch nothing
 19    python finetune/merge_export.py --preset tiny --llama-cpp ~/src/llama.cpp
 20
 21`--dry-run` and `--self-test` deliberately need neither torch nor peft nor llama.cpp: the planning
 22half is pure and testable on any machine, and only the merge itself needs the heavy dependencies
 23(`pip install -r finetune/requirements.txt`). Same lazy-import discipline as publish_hf.py.
 24"""
 25
 26from __future__ import annotations
 27
 28import argparse
 29import json
 30import subprocess
 31import sys
 32from pathlib import Path
 33
 34sys.path.insert(0, str(Path(__file__).parent))
 35from train_lora import (  # noqa: E402  — same directory, shares the preset table
 36    PRESETS,
 37    check_adapter_base,
 38    preset_for,
 39    run_dir,
 40    run_slug,
 41)
 42
 43QUANTIZATIONS = ("Q4_K_M", "Q8_0")
 44
 45
 46def default_adapter(out_dir: Path) -> Path:
 47    """The best adapter present: DPO continues training *from* the SFT adapter, so when both
 48    exist `out/<preset>/dpo-adapter` already contains the SFT weights and is the one to publish."""
 49    dpo, sft = out_dir / "dpo-adapter", out_dir / "adapter"
 50    return dpo if dpo.exists() else sft
 51
 52
 53def model_name(preset: str) -> str:
 54    """`marola-sea-tiny` — the published model's stem, per MIP-0025 §5.1(3)."""
 55    return f"marola-sea-{preset}"
 56
 57
 58def prefixed_name(base_hf_id: str, name: str) -> str:
 59    """Meta's Community Licence requires a Llama-derived model's name to START with `Llama-`
 60    (MIP-0025 §5.1(3)). The obligation is a property of the base, so it is declared once in
 61    train_lora.PRESETS (`name_prefix`) rather than re-guessed from the id here."""
 62    preset = PRESETS.get(preset_for(base_hf_id) or "", {})
 63    return preset.get("name_prefix", "") + name
 64
 65
 66def gguf_paths(out_dir: Path, name: str) -> dict[str, Path]:
 67    paths = {"f16": out_dir / f"{name}-f16.gguf"}
 68    for q in QUANTIZATIONS:
 69        paths[q] = out_dir / f"{name}-{q}.gguf"
 70    return paths
 71
 72
 73def convert_argv(llama_cpp: Path, merged: Path, out_f16: Path) -> list[str]:
 74    return [
 75        sys.executable,
 76        str(llama_cpp / "convert_hf_to_gguf.py"),
 77        str(merged),
 78        "--outfile",
 79        str(out_f16),
 80        "--outtype",
 81        "f16",
 82    ]
 83
 84
 85def quantize_argv(llama_cpp: Path, out_f16: Path, out_q: Path, quant: str) -> list[str]:
 86    """llama.cpp's quantizer moved from `./quantize` to `llama-quantize` and into build/bin; take
 87    whichever exists so a user's checkout layout doesn't matter."""
 88    for candidate in (
 89        llama_cpp / "llama-quantize",
 90        llama_cpp / "build" / "bin" / "llama-quantize",
 91        llama_cpp / "quantize",
 92    ):
 93        if candidate.exists():
 94            return [str(candidate), str(out_f16), str(out_q), quant]
 95    return ["llama-quantize", str(out_f16), str(out_q), quant]
 96
 97
 98def plan(args) -> dict:
 99    base = args.base or PRESETS[args.preset]["hf"]
100    # One directory per base — the same `out/<preset>/` layout train_lora.py writes into, so a
101    # Qwen merge can never pick up a SmolLM2 adapter or overwrite its GGUFs.
102    out_dir = Path(args.out) if args.out else run_dir(base)
103    name = prefixed_name(base, model_name(run_slug(base)))
104    return {
105        "base": base,
106        "adapter": Path(args.adapter) if args.adapter else default_adapter(out_dir),
107        "merged": out_dir / "merged",
108        "name": name,
109        "licence": PRESETS.get(preset_for(base) or "", {}).get("licence", "apache-2.0"),
110        "gguf": gguf_paths(out_dir, name),
111    }
112
113
114def plan_json(p: dict) -> dict:
115    """The publish step reads this instead of rebuilding names, paths and licence itself."""
116    return {
117        "base": p["base"],
118        "name": p["name"],
119        "licence": p["licence"],
120        "adapter": str(p["adapter"]),
121        "merged": str(p["merged"]),
122        "gguf": {k: str(v) for k, v in p["gguf"].items()},
123    }
124
125
126def merge(base_id: str, adapter: Path, merged_out: Path) -> None:
127    """The only part that needs the heavy dependencies — imported here so --dry-run/--self-test
128    stay runnable on a machine with neither."""
129    try:
130        import torch
131        from peft import PeftModel
132        from transformers import AutoModelForCausalLM, AutoTokenizer
133    except ImportError as exc:  # pragma: no cover - environment-dependent
134        raise SystemExit(
135            f"merge_export: {exc} — run `pip install -r finetune/requirements.txt` first "
136            "(torch/transformers/peft). --dry-run needs none of them."
137        ) from exc
138
139    print(f"loading base {base_id} ...")
140    model = AutoModelForCausalLM.from_pretrained(base_id, torch_dtype=torch.float16)
141    print(f"applying adapter {adapter} ...")
142    model = PeftModel.from_pretrained(model, str(adapter))
143    print("merging adapter weights into the base ...")
144    model = model.merge_and_unload()
145    merged_out.mkdir(parents=True, exist_ok=True)
146    model.save_pretrained(str(merged_out))
147    AutoTokenizer.from_pretrained(base_id).save_pretrained(str(merged_out))
148    print(f"merged model written to {merged_out}")
149
150
151def self_test() -> int:
152    fails = 0
153
154    def ok(got, want, label):
155        nonlocal fails
156        if got == want:
157            print(f"  ok   {label}")
158        else:
159            fails += 1
160            print(f"  FAIL {label} — got {got!r}, want {want!r}")
161
162    ok(model_name("tiny"), "marola-sea-tiny", "model_name follows MIP-0025 §5.1(3)'s stem")
163    ok(
164        prefixed_name("HuggingFaceTB/SmolLM2-360M-Instruct", "marola-sea-tiny"),
165        "marola-sea-tiny",
166        "a SmolLM2-derived model needs no Llama- prefix",
167    )
168    ok(
169        prefixed_name("unsloth/Llama-3.2-1B-Instruct", "marola-sea-small"),
170        "Llama-marola-sea-small",
171        "a Llama-derived model MUST start with Llama- (Meta Community Licence)",
172    )
173    ok(
174        prefixed_name("meta-llama/Llama-3.2-3B-Instruct", "marola-sea-base"),
175        "Llama-marola-sea-base",
176        "the gated base preset is Llama-derived too",
177    )
178    ok(
179        prefixed_name("Qwen/Qwen2.5-7B-Instruct", "marola-sea-qwen-7b"),
180        "marola-sea-qwen-7b",
181        "an Apache-2.0 Qwen base carries no naming obligation",
182    )
183
184    class _Args:
185        preset, base, adapter, out = "qwen-7b", None, None, None
186
187    j = plan_json(plan(_Args()))
188    ok(j["name"], "marola-sea-qwen-7b", "the plan JSON carries the published name")
189    ok(j["licence"], "apache-2.0", "the plan JSON carries the base's licence id")
190    ok(
191        j["gguf"]["Q4_K_M"].endswith("out/qwen-7b/marola-sea-qwen-7b-Q4_K_M.gguf"),
192        True,
193        "the plan JSON's GGUF path is the one merge_export actually writes",
194    )
195    ok(all(isinstance(v, str) for v in j["gguf"].values()), True, "the plan JSON is serialisable")
196
197    names = sorted(p.name for p in gguf_paths(Path("/o"), "marola-sea-tiny").values())
198    ok(
199        names,
200        ["marola-sea-tiny-Q4_K_M.gguf", "marola-sea-tiny-Q8_0.gguf", "marola-sea-tiny-f16.gguf"],
201        "exports f16 plus both quantizations MIP-0025 §5.1 asks for",
202    )
203    argv = convert_argv(Path("/llama"), Path("/o/merged"), Path("/o/m-f16.gguf"))
204    ok(
205        argv[1],
206        "/llama/convert_hf_to_gguf.py",
207        "converts with convert_hf_to_gguf.py, not convert_lora_to_gguf.py",
208    )
209    ok(argv[-1], "f16", "converts at f16 before quantizing")
210    ok(
211        quantize_argv(Path("/nope"), Path("/o/a.gguf"), Path("/o/b.gguf"), "Q4_K_M")[-1],
212        "Q4_K_M",
213        "quantize_argv passes the quantization type through",
214    )
215    ok(
216        default_adapter(Path("/definitely/missing")).name,
217        "adapter",
218        "with no DPO adapter present, the SFT adapter is the one merged",
219    )
220    # merge_export shells out to llama.cpp's convert_hf_to_gguf.py, whose vocab probe catches only
221    # FileNotFoundError: a missing sentencepiece surfaces as ModuleNotFoundError and kills the run.
222    # setup-ml-venv (labs/cuda) installs the requirements file as given and knows nothing of this.
223    req = (Path(__file__).parent / "requirements.txt").read_text()
224    for dep in ("gguf", "sentencepiece", "protobuf"):
225        ok(
226            dep in req,
227            True,
228            f"{dep} is in finetune/requirements.txt — convert_hf_to_gguf.py needs it",
229        )
230
231    if fails:
232        print(f"merge_export self-test: {fails} failure(s)", file=sys.stderr)
233        return 1
234    print("merge_export self-test: ok")
235    return 0
236
237
238def main(argv: list[str] | None = None) -> int:
239    ap = argparse.ArgumentParser(
240        description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
241    )
242    ap.add_argument("--preset", choices=sorted(PRESETS), default="tiny")
243    ap.add_argument("--base", default=None, help="explicit HF model id; overrides --preset")
244    ap.add_argument(
245        "--adapter",
246        default=None,
247        help="LoRA adapter dir (default: out/<preset>/dpo-adapter, else out/<preset>/adapter)",
248    )
249    ap.add_argument(
250        "--out",
251        default=None,
252        help="the run directory holding the adapter and the GGUFs "
253        "(default: finetune/out/<preset>, matching train_lora.py)",
254    )
255    ap.add_argument(
256        "--llama-cpp", default=None, help="path to a llama.cpp checkout (for convert + quantize)"
257    )
258    ap.add_argument(
259        "--skip-convert", action="store_true", help="merge only; leave GGUF conversion to you"
260    )
261    ap.add_argument(
262        "--dry-run", action="store_true", help="print the plan and the exact commands, run nothing"
263    )
264    ap.add_argument(
265        "--plan-json",
266        default=None,
267        help="also write the plan (name, licence, base, GGUF paths) here, for the publish step to "
268        "consume instead of rebuilding those strings itself",
269    )
270    ap.add_argument("--self-test", action="store_true")
271    args = ap.parse_args(argv)
272
273    if args.self_test:
274        return self_test()
275
276    p = plan(args)
277    if args.plan_json:
278        dest = Path(args.plan_json)
279        dest.parent.mkdir(parents=True, exist_ok=True)
280        dest.write_text(json.dumps(plan_json(p), indent=2) + "\n")
281        print(f"plan    : {dest}")
282    llama_cpp = Path(args.llama_cpp) if args.llama_cpp else None
283    print(f"base    : {p['base']} ({p['licence']})")
284    print(f"adapter : {p['adapter']}")
285    print(f"merged  : {p['merged']}")
286    for k, v in p["gguf"].items():
287        print(f"gguf    : {k:7} {v}")
288
289    if args.dry_run:
290        print("\n--- commands this would run ---")
291        print(f"# merge: peft merge_and_unload({p['base']} + {p['adapter']}) -> {p['merged']}")
292        if not args.skip_convert:
293            cc = llama_cpp or Path("<--llama-cpp>")
294            print(" ".join(convert_argv(cc, p["merged"], p["gguf"]["f16"])))
295            for q in QUANTIZATIONS:
296                print(" ".join(quantize_argv(cc, p["gguf"]["f16"], p["gguf"][q], q)))
297        print("\n# then publish the QUANTIZED files (never the adapter):")
298        print(
299            f"just finetune-publish repo=<you>/{p['name']}-GGUF "
300            f"gguf={p['gguf']['Q4_K_M']} base={p['base']} --base-license {p['licence']}"
301        )
302        print("dry run: nothing was written")
303        return 0
304
305    if not p["adapter"].exists():
306        raise SystemExit(
307            f"merge_export: no adapter at {p['adapter']} — run `just finetune-train` (and "
308            "optionally `just finetune-train-dpo`) first"
309        )
310    check_adapter_base(p["adapter"], p["base"])
311    merge(p["base"], p["adapter"], p["merged"])
312
313    if args.skip_convert:
314        print("--skip-convert: merged model only, no GGUF written")
315        return 0
316    if llama_cpp is None or not (llama_cpp / "convert_hf_to_gguf.py").exists():
317        print(
318            f"\nmerged model is at {p['merged']}. Pass --llama-cpp <path to a llama.cpp checkout> "
319            "to convert and quantize it here, or run these yourself:",
320            file=sys.stderr,
321        )
322        cc = llama_cpp or Path("<llama.cpp>")
323        print(" ".join(convert_argv(cc, p["merged"], p["gguf"]["f16"])), file=sys.stderr)
324        return 1
325    subprocess.run(convert_argv(llama_cpp, p["merged"], p["gguf"]["f16"]), check=True)
326    for q in QUANTIZATIONS:
327        subprocess.run(quantize_argv(llama_cpp, p["gguf"]["f16"], p["gguf"][q], q), check=True)
328        print(f"quantized {p['gguf'][q].name}")
329    print(
330        f"\nnext: just finetune-publish repo=<you>/{p['name']}-GGUF gguf={p['gguf']['Q4_K_M']} "
331        f"base={p['base']} --base-license {p['licence']}"
332    )
333    return 0
334
335
336if __name__ == "__main__":
337    sys.exit(main())
QUANTIZATIONS = ('Q4_K_M', 'Q8_0')
def default_adapter(out_dir: pathlib.Path) -> pathlib.Path:
47def default_adapter(out_dir: Path) -> Path:
48    """The best adapter present: DPO continues training *from* the SFT adapter, so when both
49    exist `out/<preset>/dpo-adapter` already contains the SFT weights and is the one to publish."""
50    dpo, sft = out_dir / "dpo-adapter", out_dir / "adapter"
51    return dpo if dpo.exists() else sft

The best adapter present: DPO continues training from the SFT adapter, so when both exist out/<preset>/dpo-adapter already contains the SFT weights and is the one to publish.

def model_name(preset: str) -> str:
54def model_name(preset: str) -> str:
55    """`marola-sea-tiny` — the published model's stem, per MIP-0025 §5.1(3)."""
56    return f"marola-sea-{preset}"

marola-sea-tiny — the published model's stem, per MIP-0025 §5.1(3).

def prefixed_name(base_hf_id: str, name: str) -> str:
59def prefixed_name(base_hf_id: str, name: str) -> str:
60    """Meta's Community Licence requires a Llama-derived model's name to START with `Llama-`
61    (MIP-0025 §5.1(3)). The obligation is a property of the base, so it is declared once in
62    train_lora.PRESETS (`name_prefix`) rather than re-guessed from the id here."""
63    preset = PRESETS.get(preset_for(base_hf_id) or "", {})
64    return preset.get("name_prefix", "") + name

Meta's Community Licence requires a Llama-derived model's name to START with Llama- (MIP-0025 §5.1(3)). The obligation is a property of the base, so it is declared once in train_lora.PRESETS (name_prefix) rather than re-guessed from the id here.

def gguf_paths(out_dir: pathlib.Path, name: str) -> dict[str, pathlib.Path]:
67def gguf_paths(out_dir: Path, name: str) -> dict[str, Path]:
68    paths = {"f16": out_dir / f"{name}-f16.gguf"}
69    for q in QUANTIZATIONS:
70        paths[q] = out_dir / f"{name}-{q}.gguf"
71    return paths
def convert_argv( llama_cpp: pathlib.Path, merged: pathlib.Path, out_f16: pathlib.Path) -> list[str]:
74def convert_argv(llama_cpp: Path, merged: Path, out_f16: Path) -> list[str]:
75    return [
76        sys.executable,
77        str(llama_cpp / "convert_hf_to_gguf.py"),
78        str(merged),
79        "--outfile",
80        str(out_f16),
81        "--outtype",
82        "f16",
83    ]
def quantize_argv( llama_cpp: pathlib.Path, out_f16: pathlib.Path, out_q: pathlib.Path, quant: str) -> list[str]:
86def quantize_argv(llama_cpp: Path, out_f16: Path, out_q: Path, quant: str) -> list[str]:
87    """llama.cpp's quantizer moved from `./quantize` to `llama-quantize` and into build/bin; take
88    whichever exists so a user's checkout layout doesn't matter."""
89    for candidate in (
90        llama_cpp / "llama-quantize",
91        llama_cpp / "build" / "bin" / "llama-quantize",
92        llama_cpp / "quantize",
93    ):
94        if candidate.exists():
95            return [str(candidate), str(out_f16), str(out_q), quant]
96    return ["llama-quantize", str(out_f16), str(out_q), quant]

llama.cpp's quantizer moved from ./quantize to llama-quantize and into build/bin; take whichever exists so a user's checkout layout doesn't matter.

def plan(args) -> dict:
 99def plan(args) -> dict:
100    base = args.base or PRESETS[args.preset]["hf"]
101    # One directory per base — the same `out/<preset>/` layout train_lora.py writes into, so a
102    # Qwen merge can never pick up a SmolLM2 adapter or overwrite its GGUFs.
103    out_dir = Path(args.out) if args.out else run_dir(base)
104    name = prefixed_name(base, model_name(run_slug(base)))
105    return {
106        "base": base,
107        "adapter": Path(args.adapter) if args.adapter else default_adapter(out_dir),
108        "merged": out_dir / "merged",
109        "name": name,
110        "licence": PRESETS.get(preset_for(base) or "", {}).get("licence", "apache-2.0"),
111        "gguf": gguf_paths(out_dir, name),
112    }
def plan_json(p: dict) -> dict:
115def plan_json(p: dict) -> dict:
116    """The publish step reads this instead of rebuilding names, paths and licence itself."""
117    return {
118        "base": p["base"],
119        "name": p["name"],
120        "licence": p["licence"],
121        "adapter": str(p["adapter"]),
122        "merged": str(p["merged"]),
123        "gguf": {k: str(v) for k, v in p["gguf"].items()},
124    }

The publish step reads this instead of rebuilding names, paths and licence itself.

def merge(base_id: str, adapter: pathlib.Path, merged_out: pathlib.Path) -> None:
127def merge(base_id: str, adapter: Path, merged_out: Path) -> None:
128    """The only part that needs the heavy dependencies — imported here so --dry-run/--self-test
129    stay runnable on a machine with neither."""
130    try:
131        import torch
132        from peft import PeftModel
133        from transformers import AutoModelForCausalLM, AutoTokenizer
134    except ImportError as exc:  # pragma: no cover - environment-dependent
135        raise SystemExit(
136            f"merge_export: {exc} — run `pip install -r finetune/requirements.txt` first "
137            "(torch/transformers/peft). --dry-run needs none of them."
138        ) from exc
139
140    print(f"loading base {base_id} ...")
141    model = AutoModelForCausalLM.from_pretrained(base_id, torch_dtype=torch.float16)
142    print(f"applying adapter {adapter} ...")
143    model = PeftModel.from_pretrained(model, str(adapter))
144    print("merging adapter weights into the base ...")
145    model = model.merge_and_unload()
146    merged_out.mkdir(parents=True, exist_ok=True)
147    model.save_pretrained(str(merged_out))
148    AutoTokenizer.from_pretrained(base_id).save_pretrained(str(merged_out))
149    print(f"merged model written to {merged_out}")

The only part that needs the heavy dependencies — imported here so --dry-run/--self-test stay runnable on a machine with neither.

def self_test() -> int:
152def self_test() -> int:
153    fails = 0
154
155    def ok(got, want, label):
156        nonlocal fails
157        if got == want:
158            print(f"  ok   {label}")
159        else:
160            fails += 1
161            print(f"  FAIL {label} — got {got!r}, want {want!r}")
162
163    ok(model_name("tiny"), "marola-sea-tiny", "model_name follows MIP-0025 §5.1(3)'s stem")
164    ok(
165        prefixed_name("HuggingFaceTB/SmolLM2-360M-Instruct", "marola-sea-tiny"),
166        "marola-sea-tiny",
167        "a SmolLM2-derived model needs no Llama- prefix",
168    )
169    ok(
170        prefixed_name("unsloth/Llama-3.2-1B-Instruct", "marola-sea-small"),
171        "Llama-marola-sea-small",
172        "a Llama-derived model MUST start with Llama- (Meta Community Licence)",
173    )
174    ok(
175        prefixed_name("meta-llama/Llama-3.2-3B-Instruct", "marola-sea-base"),
176        "Llama-marola-sea-base",
177        "the gated base preset is Llama-derived too",
178    )
179    ok(
180        prefixed_name("Qwen/Qwen2.5-7B-Instruct", "marola-sea-qwen-7b"),
181        "marola-sea-qwen-7b",
182        "an Apache-2.0 Qwen base carries no naming obligation",
183    )
184
185    class _Args:
186        preset, base, adapter, out = "qwen-7b", None, None, None
187
188    j = plan_json(plan(_Args()))
189    ok(j["name"], "marola-sea-qwen-7b", "the plan JSON carries the published name")
190    ok(j["licence"], "apache-2.0", "the plan JSON carries the base's licence id")
191    ok(
192        j["gguf"]["Q4_K_M"].endswith("out/qwen-7b/marola-sea-qwen-7b-Q4_K_M.gguf"),
193        True,
194        "the plan JSON's GGUF path is the one merge_export actually writes",
195    )
196    ok(all(isinstance(v, str) for v in j["gguf"].values()), True, "the plan JSON is serialisable")
197
198    names = sorted(p.name for p in gguf_paths(Path("/o"), "marola-sea-tiny").values())
199    ok(
200        names,
201        ["marola-sea-tiny-Q4_K_M.gguf", "marola-sea-tiny-Q8_0.gguf", "marola-sea-tiny-f16.gguf"],
202        "exports f16 plus both quantizations MIP-0025 §5.1 asks for",
203    )
204    argv = convert_argv(Path("/llama"), Path("/o/merged"), Path("/o/m-f16.gguf"))
205    ok(
206        argv[1],
207        "/llama/convert_hf_to_gguf.py",
208        "converts with convert_hf_to_gguf.py, not convert_lora_to_gguf.py",
209    )
210    ok(argv[-1], "f16", "converts at f16 before quantizing")
211    ok(
212        quantize_argv(Path("/nope"), Path("/o/a.gguf"), Path("/o/b.gguf"), "Q4_K_M")[-1],
213        "Q4_K_M",
214        "quantize_argv passes the quantization type through",
215    )
216    ok(
217        default_adapter(Path("/definitely/missing")).name,
218        "adapter",
219        "with no DPO adapter present, the SFT adapter is the one merged",
220    )
221    # merge_export shells out to llama.cpp's convert_hf_to_gguf.py, whose vocab probe catches only
222    # FileNotFoundError: a missing sentencepiece surfaces as ModuleNotFoundError and kills the run.
223    # setup-ml-venv (labs/cuda) installs the requirements file as given and knows nothing of this.
224    req = (Path(__file__).parent / "requirements.txt").read_text()
225    for dep in ("gguf", "sentencepiece", "protobuf"):
226        ok(
227            dep in req,
228            True,
229            f"{dep} is in finetune/requirements.txt — convert_hf_to_gguf.py needs it",
230        )
231
232    if fails:
233        print(f"merge_export self-test: {fails} failure(s)", file=sys.stderr)
234        return 1
235    print("merge_export self-test: ok")
236    return 0
def main(argv: list[str] | None = None) -> int:
239def main(argv: list[str] | None = None) -> int:
240    ap = argparse.ArgumentParser(
241        description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
242    )
243    ap.add_argument("--preset", choices=sorted(PRESETS), default="tiny")
244    ap.add_argument("--base", default=None, help="explicit HF model id; overrides --preset")
245    ap.add_argument(
246        "--adapter",
247        default=None,
248        help="LoRA adapter dir (default: out/<preset>/dpo-adapter, else out/<preset>/adapter)",
249    )
250    ap.add_argument(
251        "--out",
252        default=None,
253        help="the run directory holding the adapter and the GGUFs "
254        "(default: finetune/out/<preset>, matching train_lora.py)",
255    )
256    ap.add_argument(
257        "--llama-cpp", default=None, help="path to a llama.cpp checkout (for convert + quantize)"
258    )
259    ap.add_argument(
260        "--skip-convert", action="store_true", help="merge only; leave GGUF conversion to you"
261    )
262    ap.add_argument(
263        "--dry-run", action="store_true", help="print the plan and the exact commands, run nothing"
264    )
265    ap.add_argument(
266        "--plan-json",
267        default=None,
268        help="also write the plan (name, licence, base, GGUF paths) here, for the publish step to "
269        "consume instead of rebuilding those strings itself",
270    )
271    ap.add_argument("--self-test", action="store_true")
272    args = ap.parse_args(argv)
273
274    if args.self_test:
275        return self_test()
276
277    p = plan(args)
278    if args.plan_json:
279        dest = Path(args.plan_json)
280        dest.parent.mkdir(parents=True, exist_ok=True)
281        dest.write_text(json.dumps(plan_json(p), indent=2) + "\n")
282        print(f"plan    : {dest}")
283    llama_cpp = Path(args.llama_cpp) if args.llama_cpp else None
284    print(f"base    : {p['base']} ({p['licence']})")
285    print(f"adapter : {p['adapter']}")
286    print(f"merged  : {p['merged']}")
287    for k, v in p["gguf"].items():
288        print(f"gguf    : {k:7} {v}")
289
290    if args.dry_run:
291        print("\n--- commands this would run ---")
292        print(f"# merge: peft merge_and_unload({p['base']} + {p['adapter']}) -> {p['merged']}")
293        if not args.skip_convert:
294            cc = llama_cpp or Path("<--llama-cpp>")
295            print(" ".join(convert_argv(cc, p["merged"], p["gguf"]["f16"])))
296            for q in QUANTIZATIONS:
297                print(" ".join(quantize_argv(cc, p["gguf"]["f16"], p["gguf"][q], q)))
298        print("\n# then publish the QUANTIZED files (never the adapter):")
299        print(
300            f"just finetune-publish repo=<you>/{p['name']}-GGUF "
301            f"gguf={p['gguf']['Q4_K_M']} base={p['base']} --base-license {p['licence']}"
302        )
303        print("dry run: nothing was written")
304        return 0
305
306    if not p["adapter"].exists():
307        raise SystemExit(
308            f"merge_export: no adapter at {p['adapter']} — run `just finetune-train` (and "
309            "optionally `just finetune-train-dpo`) first"
310        )
311    check_adapter_base(p["adapter"], p["base"])
312    merge(p["base"], p["adapter"], p["merged"])
313
314    if args.skip_convert:
315        print("--skip-convert: merged model only, no GGUF written")
316        return 0
317    if llama_cpp is None or not (llama_cpp / "convert_hf_to_gguf.py").exists():
318        print(
319            f"\nmerged model is at {p['merged']}. Pass --llama-cpp <path to a llama.cpp checkout> "
320            "to convert and quantize it here, or run these yourself:",
321            file=sys.stderr,
322        )
323        cc = llama_cpp or Path("<llama.cpp>")
324        print(" ".join(convert_argv(cc, p["merged"], p["gguf"]["f16"])), file=sys.stderr)
325        return 1
326    subprocess.run(convert_argv(llama_cpp, p["merged"], p["gguf"]["f16"]), check=True)
327    for q in QUANTIZATIONS:
328        subprocess.run(quantize_argv(llama_cpp, p["gguf"]["f16"], p["gguf"][q], q), check=True)
329        print(f"quantized {p['gguf'][q].name}")
330    print(
331        f"\nnext: just finetune-publish repo=<you>/{p['name']}-GGUF gguf={p['gguf']['Q4_K_M']} "
332        f"base={p['base']} --base-license {p['licence']}"
333    )
334    return 0