pr_label_nlp
pr_label_nlp — a cheap, local, marola-aware NLP classifier for the area/* PR labels.
scripts/pr_label_nlp.py --pr 168 # classify a real PR (needs `gh auth status` OK)
scripts/pr_label_nlp.py --title "..." --body "..." # classify arbitrary text, no `gh` call
scripts/pr_label_nlp.py --pr 168 --top 3 # print the top 3 candidates, not just the best
scripts/pr_label_nlp.py --self-test # offline check against fixed text, no network
scripts/lib/pr_labels.sh's own header explains why the taxonomy it drives is deterministic
("never guessed from prose, never an LLM call") — that stays true for what actually gets applied
to a real PR. This script exists to answer a different, narrower question: could a cheap, local
NLP method (no API, no network, no GPU — TF-IDF + cosine similarity, scikit-learn) do a
comparably good job at the one part of the taxonomy that genuinely comes from prose, area/*,
which today only resolves via a PR's MIP number and falls back to area/unscoped otherwise?
Method: TF-IDF vectorizes a real PR's title + body + commit subjects against a small per-label
corpus (grep-generated below, so it's marola vocabulary, not generic English — the actual jargon
this repo's own PRs use: PRÓPRIA/IMPRÓPRIA, Overpass, Kyo, DSPy, MLflow,
MCP, and so on) and reports the cosine-nearest label(s). It is used by
scripts/backfill-pr-labels.sh's --nlp flag strictly as a side-by-side comparison against the
deterministic result — never applied to a real PR unless --nlp-apply-unscoped is also passed,
and even then only to fill a genuine area/unscoped gap, never to override a confident deterministic
call. Same "judge, never veto" shape as llm/Reviewer.scala over Swimability.scala's score.
1#!/usr/bin/env python3 2"""pr_label_nlp — a cheap, local, marola-aware NLP classifier for the `area/*` PR labels. 3 4 scripts/pr_label_nlp.py --pr 168 # classify a real PR (needs `gh auth status` OK) 5 scripts/pr_label_nlp.py --title "..." --body "..." # classify arbitrary text, no `gh` call 6 scripts/pr_label_nlp.py --pr 168 --top 3 # print the top 3 candidates, not just the best 7 scripts/pr_label_nlp.py --self-test # offline check against fixed text, no network 8 9scripts/lib/pr_labels.sh's own header explains why the taxonomy it drives is deterministic 10("never guessed from prose, never an LLM call") — that stays true for what actually gets applied 11to a real PR. This script exists to answer a different, narrower question: could a cheap, local 12NLP method (no API, no network, no GPU — TF-IDF + cosine similarity, scikit-learn) do a 13comparably good job at the one part of the taxonomy that genuinely comes from prose, `area/*`, 14which today only resolves via a PR's MIP number and falls back to `area/unscoped` otherwise? 15 16Method: TF-IDF vectorizes a real PR's title + body + commit subjects against a small per-label 17corpus (grep-generated below, so it's marola vocabulary, not generic English — the actual jargon 18this repo's own PRs use: PRÓPRIA/IMPRÓPRIA, Overpass, Kyo, DSPy, MLflow, 19MCP, and so on) and reports the cosine-nearest label(s). It is used by 20scripts/backfill-pr-labels.sh's `--nlp` flag strictly as a side-by-side comparison against the 21deterministic result — never applied to a real PR unless `--nlp-apply-unscoped` is also passed, 22and even then only to fill a genuine area/unscoped gap, never to override a confident deterministic 23call. Same "judge, never veto" shape as llm/Reviewer.scala over Swimability.scala's score. 24""" 25 26import argparse 27import json 28import subprocess 29import sys 30from pathlib import Path 31 32REPO_ROOT = Path(__file__).resolve().parent.parent 33PR_LABELS_SH = REPO_ROOT / "scripts" / "lib" / "pr_labels.sh" 34 35# Marola-domain vocabulary per area label, on top of scripts/lib/pr_labels.sh's own one-line 36# taxonomy description — this is what makes the classifier "aware of marola context" rather than 37# a generic bag-of-words model. Kept here (not in the .sh file) since it's NLP-specific tuning, 38# not part of the deterministic taxonomy's own source of truth. 39AREA_CONTEXT_KEYWORDS = { 40 "area/conditions": "sea weather tide forecast open-meteo swell wind wave temperature cache latency swimability score", 41 "area/water-quality": "bathing water quality sampling point ima inea inema propria impropria enterococci veto unfit", 42 "area/sea-life": "jellyfish whale sighting heuristic sea lore corpus marine life season", 43 "area/safety": "safety footer hazard escalation rip current warning veto never overturn", 44 "area/accessibility": "parking toilets shower lifeguard facilities osm overpass amenity beach", 45 "area/map-site": "map static site marker leaflet tooltip legend index.html app.js style.css board", 46 "area/telegram-bot": "telegram bot reply digest subscription matching chat", 47 "area/outreach": "book exporter instagram waitlist promotion readme site copy marketing", 48 "area/ml-infra": "mlflow llm4s dspy fine-tuned finetune model forecasting research benchmark lora sft dpo", 49 "area/dev-tooling": "claude code opencode agentic tooling dev workflow skill hook justfile ci", 50 "area/positioning": "product naming positioning slogan brand", 51} 52 53 54def load_taxonomy_descriptions() -> dict[str, str]: 55 """Source scripts/lib/pr_labels.sh and print PR_LABEL_TAXONOMY so the two classifiers never 56 drift apart on label names/descriptions — this script only adds vocabulary, never invents a 57 label the deterministic taxonomy doesn't already define.""" 58 out = subprocess.run( 59 ["bash", "-c", f'source "{PR_LABELS_SH}" && printf "%s\\n" "${{PR_LABEL_TAXONOMY[@]}}"'], 60 capture_output=True, 61 text=True, 62 check=True, 63 ).stdout 64 descriptions = {} 65 for line in out.splitlines(): 66 parts = line.split(":", 2) 67 if len(parts) == 3: 68 name, _color, desc = parts 69 descriptions[name] = desc 70 return descriptions 71 72 73def build_area_corpus() -> dict[str, str]: 74 descriptions = load_taxonomy_descriptions() 75 corpus = {} 76 for label, keywords in AREA_CONTEXT_KEYWORDS.items(): 77 corpus[label] = f"{descriptions.get(label, '')} {keywords}" 78 return corpus 79 80 81def classify(text: str, top: int = 1) -> list[tuple[str, float]]: 82 """Returns up to `top` (label, cosine_similarity) pairs, highest similarity first. An empty 83 list means nothing scored above zero similarity — the caller decides what that means.""" 84 from sklearn.feature_extraction.text import TfidfVectorizer 85 from sklearn.metrics.pairwise import cosine_similarity 86 87 corpus = build_area_corpus() 88 labels = list(corpus.keys()) 89 documents = [corpus[label] for label in labels] + [text] 90 vectorizer = TfidfVectorizer(stop_words="english", lowercase=True) 91 matrix = vectorizer.fit_transform(documents) 92 pr_vector = matrix[-1] 93 label_vectors = matrix[:-1] 94 similarities = cosine_similarity(pr_vector, label_vectors)[0] 95 ranked = sorted(zip(labels, similarities, strict=True), key=lambda pair: pair[1], reverse=True) 96 return [(label, float(score)) for label, score in ranked[:top] if score > 0] 97 98 99def fetch_pr_text(pr_number: str) -> str: 100 out = subprocess.run( 101 ["gh", "pr", "view", pr_number, "--json", "title,body,commits"], 102 capture_output=True, 103 text=True, 104 check=True, 105 ).stdout 106 data = json.loads(out) 107 subjects = " ".join(c.get("messageHeadline", "") for c in data.get("commits", [])) 108 return f"{data.get('title', '')} {data.get('body', '') or ''} {subjects}" 109 110 111def self_test() -> int: 112 fails = 0 113 skipped = False 114 115 def ok(got, want, label): 116 nonlocal fails 117 status = "ok" if got == want else "FAIL" 118 if got != want: 119 fails += 1 120 print(f" {status} {label}" + ("" if got == want else f" — got '{got}', want '{want}'")) 121 122 cases = [ 123 ("Fix jellyfish sighting heuristic — whale season peak hour", "area/sea-life"), 124 ("Telegram bot: daily digest subscription for swim matching", "area/telegram-bot"), 125 ( 126 "INEA bathing water quality PDF parser — PRÓPRIA/IMPRÓPRIA sampling points", 127 "area/water-quality", 128 ), 129 ("Add parking and lifeguard facilities from Overpass OSM amenities", "area/accessibility"), 130 ("MLflow benchmark run for the fine-tuned DPO model on Ollama", "area/ml-infra"), 131 ("Claude Code hook: format-on-write, skills, agentic dev workflow", "area/dev-tooling"), 132 ("Safety footer copy: rip current hazard warning, veto text", "area/safety"), 133 ("Map marker tooltip and legend on the static site index.html", "area/map-site"), 134 ] 135 # This is a real, small test corpus check against the actual scikit-learn/vectorizer pipeline — 136 # requires scikit-learn importable, but no network and no `gh` call, matching every other 137 # script's --self-test convention in this repo (offline, deterministic given fixed input). 138 # scikit-learn is an optional dependency here: it is not in flake.nix's devShell and not on 139 # the CI runner, so a missing import is an environment fact, not a failure. Skip only the 140 # classifier cases and still run the parsing checks below — returning 1 for a skip used to 141 # abort `repo_stats.py`'s Python-coverage measurement (it runs every --self-test under 142 # `check=True`) and take the whole repo-stats job with it. 143 try: 144 for text, expected in cases: 145 ranked = classify(text, top=1) 146 got = ranked[0][0] if ranked else None 147 ok(got, expected, f"classify('{text[:40]}...') picks {expected}") 148 except ImportError as exc: 149 print( 150 f" SKIP scikit-learn not importable ({exc}) — classifier cases skipped, " 151 "parsing checks still run", 152 file=sys.stderr, 153 ) 154 skipped = True 155 156 # Pure parsing check, no scikit-learn needed. 157 descriptions = load_taxonomy_descriptions() 158 ok( 159 "area/water-quality" in descriptions, 160 True, 161 "load_taxonomy_descriptions finds area/water-quality from scripts/lib/pr_labels.sh", 162 ) 163 for label in AREA_CONTEXT_KEYWORDS: 164 if label not in descriptions: 165 fails += 1 166 print(f" FAIL AREA_CONTEXT_KEYWORDS has '{label}' which is not in the real taxonomy") 167 168 if fails == 0: 169 print("pr_label_nlp self-test: ok" + (" (classifier cases skipped)" if skipped else "")) 170 return 0 171 print(f"pr_label_nlp self-test: {fails} failure(s)", file=sys.stderr) 172 return 1 173 174 175def main() -> int: 176 parser = argparse.ArgumentParser( 177 description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter 178 ) 179 parser.add_argument("--pr", help="PR number to classify (needs gh auth status OK)") 180 parser.add_argument("--title", default="", help="classify arbitrary text instead of a real PR") 181 parser.add_argument("--body", default="", help="paired with --title") 182 parser.add_argument( 183 "--top", type=int, default=1, help="how many candidate labels to print (default 1)" 184 ) 185 parser.add_argument("--json", action="store_true", help="machine-readable output") 186 parser.add_argument( 187 "--self-test", action="store_true", help="offline check, no network, no gh call" 188 ) 189 args = parser.parse_args() 190 191 if args.self_test: 192 return self_test() 193 194 if args.pr: 195 text = fetch_pr_text(args.pr) 196 elif args.title or args.body: 197 text = f"{args.title} {args.body}" 198 else: 199 parser.error("need --pr, or --title/--body, or --self-test") 200 return 2 201 202 ranked = classify(text, top=args.top) 203 if args.json: 204 print( 205 json.dumps([{"label": label, "similarity": round(score, 4)} for label, score in ranked]) 206 ) 207 elif not ranked: 208 print( 209 "pr_label_nlp: no area label scored above zero similarity — text may be too short or off-domain" 210 ) 211 else: 212 for label, score in ranked: 213 print(f"{label}\t{score:.4f}") 214 return 0 215 216 217if __name__ == "__main__": 218 sys.exit(main())
55def load_taxonomy_descriptions() -> dict[str, str]: 56 """Source scripts/lib/pr_labels.sh and print PR_LABEL_TAXONOMY so the two classifiers never 57 drift apart on label names/descriptions — this script only adds vocabulary, never invents a 58 label the deterministic taxonomy doesn't already define.""" 59 out = subprocess.run( 60 ["bash", "-c", f'source "{PR_LABELS_SH}" && printf "%s\\n" "${{PR_LABEL_TAXONOMY[@]}}"'], 61 capture_output=True, 62 text=True, 63 check=True, 64 ).stdout 65 descriptions = {} 66 for line in out.splitlines(): 67 parts = line.split(":", 2) 68 if len(parts) == 3: 69 name, _color, desc = parts 70 descriptions[name] = desc 71 return descriptions
Source scripts/lib/pr_labels.sh and print PR_LABEL_TAXONOMY so the two classifiers never drift apart on label names/descriptions — this script only adds vocabulary, never invents a label the deterministic taxonomy doesn't already define.
82def classify(text: str, top: int = 1) -> list[tuple[str, float]]: 83 """Returns up to `top` (label, cosine_similarity) pairs, highest similarity first. An empty 84 list means nothing scored above zero similarity — the caller decides what that means.""" 85 from sklearn.feature_extraction.text import TfidfVectorizer 86 from sklearn.metrics.pairwise import cosine_similarity 87 88 corpus = build_area_corpus() 89 labels = list(corpus.keys()) 90 documents = [corpus[label] for label in labels] + [text] 91 vectorizer = TfidfVectorizer(stop_words="english", lowercase=True) 92 matrix = vectorizer.fit_transform(documents) 93 pr_vector = matrix[-1] 94 label_vectors = matrix[:-1] 95 similarities = cosine_similarity(pr_vector, label_vectors)[0] 96 ranked = sorted(zip(labels, similarities, strict=True), key=lambda pair: pair[1], reverse=True) 97 return [(label, float(score)) for label, score in ranked[:top] if score > 0]
Returns up to top (label, cosine_similarity) pairs, highest similarity first. An empty
list means nothing scored above zero similarity — the caller decides what that means.
100def fetch_pr_text(pr_number: str) -> str: 101 out = subprocess.run( 102 ["gh", "pr", "view", pr_number, "--json", "title,body,commits"], 103 capture_output=True, 104 text=True, 105 check=True, 106 ).stdout 107 data = json.loads(out) 108 subjects = " ".join(c.get("messageHeadline", "") for c in data.get("commits", [])) 109 return f"{data.get('title', '')} {data.get('body', '') or ''} {subjects}"
112def self_test() -> int: 113 fails = 0 114 skipped = False 115 116 def ok(got, want, label): 117 nonlocal fails 118 status = "ok" if got == want else "FAIL" 119 if got != want: 120 fails += 1 121 print(f" {status} {label}" + ("" if got == want else f" — got '{got}', want '{want}'")) 122 123 cases = [ 124 ("Fix jellyfish sighting heuristic — whale season peak hour", "area/sea-life"), 125 ("Telegram bot: daily digest subscription for swim matching", "area/telegram-bot"), 126 ( 127 "INEA bathing water quality PDF parser — PRÓPRIA/IMPRÓPRIA sampling points", 128 "area/water-quality", 129 ), 130 ("Add parking and lifeguard facilities from Overpass OSM amenities", "area/accessibility"), 131 ("MLflow benchmark run for the fine-tuned DPO model on Ollama", "area/ml-infra"), 132 ("Claude Code hook: format-on-write, skills, agentic dev workflow", "area/dev-tooling"), 133 ("Safety footer copy: rip current hazard warning, veto text", "area/safety"), 134 ("Map marker tooltip and legend on the static site index.html", "area/map-site"), 135 ] 136 # This is a real, small test corpus check against the actual scikit-learn/vectorizer pipeline — 137 # requires scikit-learn importable, but no network and no `gh` call, matching every other 138 # script's --self-test convention in this repo (offline, deterministic given fixed input). 139 # scikit-learn is an optional dependency here: it is not in flake.nix's devShell and not on 140 # the CI runner, so a missing import is an environment fact, not a failure. Skip only the 141 # classifier cases and still run the parsing checks below — returning 1 for a skip used to 142 # abort `repo_stats.py`'s Python-coverage measurement (it runs every --self-test under 143 # `check=True`) and take the whole repo-stats job with it. 144 try: 145 for text, expected in cases: 146 ranked = classify(text, top=1) 147 got = ranked[0][0] if ranked else None 148 ok(got, expected, f"classify('{text[:40]}...') picks {expected}") 149 except ImportError as exc: 150 print( 151 f" SKIP scikit-learn not importable ({exc}) — classifier cases skipped, " 152 "parsing checks still run", 153 file=sys.stderr, 154 ) 155 skipped = True 156 157 # Pure parsing check, no scikit-learn needed. 158 descriptions = load_taxonomy_descriptions() 159 ok( 160 "area/water-quality" in descriptions, 161 True, 162 "load_taxonomy_descriptions finds area/water-quality from scripts/lib/pr_labels.sh", 163 ) 164 for label in AREA_CONTEXT_KEYWORDS: 165 if label not in descriptions: 166 fails += 1 167 print(f" FAIL AREA_CONTEXT_KEYWORDS has '{label}' which is not in the real taxonomy") 168 169 if fails == 0: 170 print("pr_label_nlp self-test: ok" + (" (classifier cases skipped)" if skipped else "")) 171 return 0 172 print(f"pr_label_nlp self-test: {fails} failure(s)", file=sys.stderr) 173 return 1
176def main() -> int: 177 parser = argparse.ArgumentParser( 178 description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter 179 ) 180 parser.add_argument("--pr", help="PR number to classify (needs gh auth status OK)") 181 parser.add_argument("--title", default="", help="classify arbitrary text instead of a real PR") 182 parser.add_argument("--body", default="", help="paired with --title") 183 parser.add_argument( 184 "--top", type=int, default=1, help="how many candidate labels to print (default 1)" 185 ) 186 parser.add_argument("--json", action="store_true", help="machine-readable output") 187 parser.add_argument( 188 "--self-test", action="store_true", help="offline check, no network, no gh call" 189 ) 190 args = parser.parse_args() 191 192 if args.self_test: 193 return self_test() 194 195 if args.pr: 196 text = fetch_pr_text(args.pr) 197 elif args.title or args.body: 198 text = f"{args.title} {args.body}" 199 else: 200 parser.error("need --pr, or --title/--body, or --self-test") 201 return 2 202 203 ranked = classify(text, top=args.top) 204 if args.json: 205 print( 206 json.dumps([{"label": label, "similarity": round(score, 4)} for label, score in ranked]) 207 ) 208 elif not ranked: 209 print( 210 "pr_label_nlp: no area label scored above zero similarity — text may be too short or off-domain" 211 ) 212 else: 213 for label, score in ranked: 214 print(f"{label}\t{score:.4f}") 215 return 0