pr_label_nlp

pr_label_nlp — a cheap, local, marola-aware NLP classifier for the area/* PR labels.

scripts/pr_label_nlp.py --pr 168                    # classify a real PR (needs `gh auth status` OK)
scripts/pr_label_nlp.py --title "..." --body "..."   # classify arbitrary text, no `gh` call
scripts/pr_label_nlp.py --pr 168 --top 3             # print the top 3 candidates, not just the best
scripts/pr_label_nlp.py --self-test                  # offline check against fixed text, no network

scripts/lib/pr_labels.sh's own header explains why the taxonomy it drives is deterministic ("never guessed from prose, never an LLM call") — that stays true for what actually gets applied to a real PR. This script exists to answer a different, narrower question: could a cheap, local NLP method (no API, no network, no GPU — TF-IDF + cosine similarity, scikit-learn) do a comparably good job at the one part of the taxonomy that genuinely comes from prose, area/*, which today only resolves via a PR's MIP number and falls back to area/unscoped otherwise?

Method: TF-IDF vectorizes a real PR's title + body + commit subjects against a small per-label corpus (grep-generated below, so it's marola vocabulary, not generic English — the actual jargon this repo's own PRs use: PRÓPRIA/IMPRÓPRIA, Overpass, Kyo, DSPy, MLflow, MCP, and so on) and reports the cosine-nearest label(s). It is used by scripts/backfill-pr-labels.sh's --nlp flag strictly as a side-by-side comparison against the deterministic result — never applied to a real PR unless --nlp-apply-unscoped is also passed, and even then only to fill a genuine area/unscoped gap, never to override a confident deterministic call. Same "judge, never veto" shape as llm/Reviewer.scala over Swimability.scala's score.

  1#!/usr/bin/env python3
  2"""pr_label_nlp — a cheap, local, marola-aware NLP classifier for the `area/*` PR labels.
  3
  4    scripts/pr_label_nlp.py --pr 168                    # classify a real PR (needs `gh auth status` OK)
  5    scripts/pr_label_nlp.py --title "..." --body "..."   # classify arbitrary text, no `gh` call
  6    scripts/pr_label_nlp.py --pr 168 --top 3             # print the top 3 candidates, not just the best
  7    scripts/pr_label_nlp.py --self-test                  # offline check against fixed text, no network
  8
  9scripts/lib/pr_labels.sh's own header explains why the taxonomy it drives is deterministic
 10("never guessed from prose, never an LLM call") — that stays true for what actually gets applied
 11to a real PR. This script exists to answer a different, narrower question: could a cheap, local
 12NLP method (no API, no network, no GPU — TF-IDF + cosine similarity, scikit-learn) do a
 13comparably good job at the one part of the taxonomy that genuinely comes from prose, `area/*`,
 14which today only resolves via a PR's MIP number and falls back to `area/unscoped` otherwise?
 15
 16Method: TF-IDF vectorizes a real PR's title + body + commit subjects against a small per-label
 17corpus (grep-generated below, so it's marola vocabulary, not generic English — the actual jargon
 18this repo's own PRs use: PRÓPRIA/IMPRÓPRIA, Overpass, Kyo, DSPy, MLflow,
 19MCP, and so on) and reports the cosine-nearest label(s). It is used by
 20scripts/backfill-pr-labels.sh's `--nlp` flag strictly as a side-by-side comparison against the
 21deterministic result — never applied to a real PR unless `--nlp-apply-unscoped` is also passed,
 22and even then only to fill a genuine area/unscoped gap, never to override a confident deterministic
 23call. Same "judge, never veto" shape as llm/Reviewer.scala over Swimability.scala's score.
 24"""
 25
 26import argparse
 27import json
 28import subprocess
 29import sys
 30from pathlib import Path
 31
 32REPO_ROOT = Path(__file__).resolve().parent.parent
 33PR_LABELS_SH = REPO_ROOT / "scripts" / "lib" / "pr_labels.sh"
 34
 35# Marola-domain vocabulary per area label, on top of scripts/lib/pr_labels.sh's own one-line
 36# taxonomy description — this is what makes the classifier "aware of marola context" rather than
 37# a generic bag-of-words model. Kept here (not in the .sh file) since it's NLP-specific tuning,
 38# not part of the deterministic taxonomy's own source of truth.
 39AREA_CONTEXT_KEYWORDS = {
 40    "area/conditions": "sea weather tide forecast open-meteo swell wind wave temperature cache latency swimability score",
 41    "area/water-quality": "bathing water quality sampling point ima inea inema propria impropria enterococci veto unfit",
 42    "area/sea-life": "jellyfish whale sighting heuristic sea lore corpus marine life season",
 43    "area/safety": "safety footer hazard escalation rip current warning veto never overturn",
 44    "area/accessibility": "parking toilets shower lifeguard facilities osm overpass amenity beach",
 45    "area/map-site": "map static site marker leaflet tooltip legend index.html app.js style.css board",
 46    "area/telegram-bot": "telegram bot reply digest subscription matching chat",
 47    "area/outreach": "book exporter instagram waitlist promotion readme site copy marketing",
 48    "area/ml-infra": "mlflow llm4s dspy fine-tuned finetune model forecasting research benchmark lora sft dpo",
 49    "area/dev-tooling": "claude code opencode agentic tooling dev workflow skill hook justfile ci",
 50    "area/positioning": "product naming positioning slogan brand",
 51}
 52
 53
 54def load_taxonomy_descriptions() -> dict[str, str]:
 55    """Source scripts/lib/pr_labels.sh and print PR_LABEL_TAXONOMY so the two classifiers never
 56    drift apart on label names/descriptions — this script only adds vocabulary, never invents a
 57    label the deterministic taxonomy doesn't already define."""
 58    out = subprocess.run(
 59        ["bash", "-c", f'source "{PR_LABELS_SH}" && printf "%s\\n" "${{PR_LABEL_TAXONOMY[@]}}"'],
 60        capture_output=True,
 61        text=True,
 62        check=True,
 63    ).stdout
 64    descriptions = {}
 65    for line in out.splitlines():
 66        parts = line.split(":", 2)
 67        if len(parts) == 3:
 68            name, _color, desc = parts
 69            descriptions[name] = desc
 70    return descriptions
 71
 72
 73def build_area_corpus() -> dict[str, str]:
 74    descriptions = load_taxonomy_descriptions()
 75    corpus = {}
 76    for label, keywords in AREA_CONTEXT_KEYWORDS.items():
 77        corpus[label] = f"{descriptions.get(label, '')} {keywords}"
 78    return corpus
 79
 80
 81def classify(text: str, top: int = 1) -> list[tuple[str, float]]:
 82    """Returns up to `top` (label, cosine_similarity) pairs, highest similarity first. An empty
 83    list means nothing scored above zero similarity — the caller decides what that means."""
 84    from sklearn.feature_extraction.text import TfidfVectorizer
 85    from sklearn.metrics.pairwise import cosine_similarity
 86
 87    corpus = build_area_corpus()
 88    labels = list(corpus.keys())
 89    documents = [corpus[label] for label in labels] + [text]
 90    vectorizer = TfidfVectorizer(stop_words="english", lowercase=True)
 91    matrix = vectorizer.fit_transform(documents)
 92    pr_vector = matrix[-1]
 93    label_vectors = matrix[:-1]
 94    similarities = cosine_similarity(pr_vector, label_vectors)[0]
 95    ranked = sorted(zip(labels, similarities, strict=True), key=lambda pair: pair[1], reverse=True)
 96    return [(label, float(score)) for label, score in ranked[:top] if score > 0]
 97
 98
 99def fetch_pr_text(pr_number: str) -> str:
100    out = subprocess.run(
101        ["gh", "pr", "view", pr_number, "--json", "title,body,commits"],
102        capture_output=True,
103        text=True,
104        check=True,
105    ).stdout
106    data = json.loads(out)
107    subjects = " ".join(c.get("messageHeadline", "") for c in data.get("commits", []))
108    return f"{data.get('title', '')} {data.get('body', '') or ''} {subjects}"
109
110
111def self_test() -> int:
112    fails = 0
113    skipped = False
114
115    def ok(got, want, label):
116        nonlocal fails
117        status = "ok" if got == want else "FAIL"
118        if got != want:
119            fails += 1
120        print(f"  {status}   {label}" + ("" if got == want else f" — got '{got}', want '{want}'"))
121
122    cases = [
123        ("Fix jellyfish sighting heuristic — whale season peak hour", "area/sea-life"),
124        ("Telegram bot: daily digest subscription for swim matching", "area/telegram-bot"),
125        (
126            "INEA bathing water quality PDF parser — PRÓPRIA/IMPRÓPRIA sampling points",
127            "area/water-quality",
128        ),
129        ("Add parking and lifeguard facilities from Overpass OSM amenities", "area/accessibility"),
130        ("MLflow benchmark run for the fine-tuned DPO model on Ollama", "area/ml-infra"),
131        ("Claude Code hook: format-on-write, skills, agentic dev workflow", "area/dev-tooling"),
132        ("Safety footer copy: rip current hazard warning, veto text", "area/safety"),
133        ("Map marker tooltip and legend on the static site index.html", "area/map-site"),
134    ]
135    # This is a real, small test corpus check against the actual scikit-learn/vectorizer pipeline —
136    # requires scikit-learn importable, but no network and no `gh` call, matching every other
137    # script's --self-test convention in this repo (offline, deterministic given fixed input).
138    # scikit-learn is an optional dependency here: it is not in flake.nix's devShell and not on
139    # the CI runner, so a missing import is an environment fact, not a failure. Skip only the
140    # classifier cases and still run the parsing checks below — returning 1 for a skip used to
141    # abort `repo_stats.py`'s Python-coverage measurement (it runs every --self-test under
142    # `check=True`) and take the whole repo-stats job with it.
143    try:
144        for text, expected in cases:
145            ranked = classify(text, top=1)
146            got = ranked[0][0] if ranked else None
147            ok(got, expected, f"classify('{text[:40]}...') picks {expected}")
148    except ImportError as exc:
149        print(
150            f"  SKIP  scikit-learn not importable ({exc}) — classifier cases skipped, "
151            "parsing checks still run",
152            file=sys.stderr,
153        )
154        skipped = True
155
156    # Pure parsing check, no scikit-learn needed.
157    descriptions = load_taxonomy_descriptions()
158    ok(
159        "area/water-quality" in descriptions,
160        True,
161        "load_taxonomy_descriptions finds area/water-quality from scripts/lib/pr_labels.sh",
162    )
163    for label in AREA_CONTEXT_KEYWORDS:
164        if label not in descriptions:
165            fails += 1
166            print(f"  FAIL  AREA_CONTEXT_KEYWORDS has '{label}' which is not in the real taxonomy")
167
168    if fails == 0:
169        print("pr_label_nlp self-test: ok" + (" (classifier cases skipped)" if skipped else ""))
170        return 0
171    print(f"pr_label_nlp self-test: {fails} failure(s)", file=sys.stderr)
172    return 1
173
174
175def main() -> int:
176    parser = argparse.ArgumentParser(
177        description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
178    )
179    parser.add_argument("--pr", help="PR number to classify (needs gh auth status OK)")
180    parser.add_argument("--title", default="", help="classify arbitrary text instead of a real PR")
181    parser.add_argument("--body", default="", help="paired with --title")
182    parser.add_argument(
183        "--top", type=int, default=1, help="how many candidate labels to print (default 1)"
184    )
185    parser.add_argument("--json", action="store_true", help="machine-readable output")
186    parser.add_argument(
187        "--self-test", action="store_true", help="offline check, no network, no gh call"
188    )
189    args = parser.parse_args()
190
191    if args.self_test:
192        return self_test()
193
194    if args.pr:
195        text = fetch_pr_text(args.pr)
196    elif args.title or args.body:
197        text = f"{args.title} {args.body}"
198    else:
199        parser.error("need --pr, or --title/--body, or --self-test")
200        return 2
201
202    ranked = classify(text, top=args.top)
203    if args.json:
204        print(
205            json.dumps([{"label": label, "similarity": round(score, 4)} for label, score in ranked])
206        )
207    elif not ranked:
208        print(
209            "pr_label_nlp: no area label scored above zero similarity — text may be too short or off-domain"
210        )
211    else:
212        for label, score in ranked:
213            print(f"{label}\t{score:.4f}")
214    return 0
215
216
217if __name__ == "__main__":
218    sys.exit(main())
REPO_ROOT = PosixPath('/home/runner/work/marola/marola')
PR_LABELS_SH = PosixPath('/home/runner/work/marola/marola/scripts/lib/pr_labels.sh')
AREA_CONTEXT_KEYWORDS = {'area/conditions': 'sea weather tide forecast open-meteo swell wind wave temperature cache latency swimability score', 'area/water-quality': 'bathing water quality sampling point ima inea inema propria impropria enterococci veto unfit', 'area/sea-life': 'jellyfish whale sighting heuristic sea lore corpus marine life season', 'area/safety': 'safety footer hazard escalation rip current warning veto never overturn', 'area/accessibility': 'parking toilets shower lifeguard facilities osm overpass amenity beach', 'area/map-site': 'map static site marker leaflet tooltip legend index.html app.js style.css board', 'area/telegram-bot': 'telegram bot reply digest subscription matching chat', 'area/outreach': 'book exporter instagram waitlist promotion readme site copy marketing', 'area/ml-infra': 'mlflow llm4s dspy fine-tuned finetune model forecasting research benchmark lora sft dpo', 'area/dev-tooling': 'claude code opencode agentic tooling dev workflow skill hook justfile ci', 'area/positioning': 'product naming positioning slogan brand'}
def load_taxonomy_descriptions() -> dict[str, str]:
55def load_taxonomy_descriptions() -> dict[str, str]:
56    """Source scripts/lib/pr_labels.sh and print PR_LABEL_TAXONOMY so the two classifiers never
57    drift apart on label names/descriptions — this script only adds vocabulary, never invents a
58    label the deterministic taxonomy doesn't already define."""
59    out = subprocess.run(
60        ["bash", "-c", f'source "{PR_LABELS_SH}" && printf "%s\\n" "${{PR_LABEL_TAXONOMY[@]}}"'],
61        capture_output=True,
62        text=True,
63        check=True,
64    ).stdout
65    descriptions = {}
66    for line in out.splitlines():
67        parts = line.split(":", 2)
68        if len(parts) == 3:
69            name, _color, desc = parts
70            descriptions[name] = desc
71    return descriptions

Source scripts/lib/pr_labels.sh and print PR_LABEL_TAXONOMY so the two classifiers never drift apart on label names/descriptions — this script only adds vocabulary, never invents a label the deterministic taxonomy doesn't already define.

def build_area_corpus() -> dict[str, str]:
74def build_area_corpus() -> dict[str, str]:
75    descriptions = load_taxonomy_descriptions()
76    corpus = {}
77    for label, keywords in AREA_CONTEXT_KEYWORDS.items():
78        corpus[label] = f"{descriptions.get(label, '')} {keywords}"
79    return corpus
def classify(text: str, top: int = 1) -> list[tuple[str, float]]:
82def classify(text: str, top: int = 1) -> list[tuple[str, float]]:
83    """Returns up to `top` (label, cosine_similarity) pairs, highest similarity first. An empty
84    list means nothing scored above zero similarity — the caller decides what that means."""
85    from sklearn.feature_extraction.text import TfidfVectorizer
86    from sklearn.metrics.pairwise import cosine_similarity
87
88    corpus = build_area_corpus()
89    labels = list(corpus.keys())
90    documents = [corpus[label] for label in labels] + [text]
91    vectorizer = TfidfVectorizer(stop_words="english", lowercase=True)
92    matrix = vectorizer.fit_transform(documents)
93    pr_vector = matrix[-1]
94    label_vectors = matrix[:-1]
95    similarities = cosine_similarity(pr_vector, label_vectors)[0]
96    ranked = sorted(zip(labels, similarities, strict=True), key=lambda pair: pair[1], reverse=True)
97    return [(label, float(score)) for label, score in ranked[:top] if score > 0]

Returns up to top (label, cosine_similarity) pairs, highest similarity first. An empty list means nothing scored above zero similarity — the caller decides what that means.

def fetch_pr_text(pr_number: str) -> str:
100def fetch_pr_text(pr_number: str) -> str:
101    out = subprocess.run(
102        ["gh", "pr", "view", pr_number, "--json", "title,body,commits"],
103        capture_output=True,
104        text=True,
105        check=True,
106    ).stdout
107    data = json.loads(out)
108    subjects = " ".join(c.get("messageHeadline", "") for c in data.get("commits", []))
109    return f"{data.get('title', '')} {data.get('body', '') or ''} {subjects}"
def self_test() -> int:
112def self_test() -> int:
113    fails = 0
114    skipped = False
115
116    def ok(got, want, label):
117        nonlocal fails
118        status = "ok" if got == want else "FAIL"
119        if got != want:
120            fails += 1
121        print(f"  {status}   {label}" + ("" if got == want else f" — got '{got}', want '{want}'"))
122
123    cases = [
124        ("Fix jellyfish sighting heuristic — whale season peak hour", "area/sea-life"),
125        ("Telegram bot: daily digest subscription for swim matching", "area/telegram-bot"),
126        (
127            "INEA bathing water quality PDF parser — PRÓPRIA/IMPRÓPRIA sampling points",
128            "area/water-quality",
129        ),
130        ("Add parking and lifeguard facilities from Overpass OSM amenities", "area/accessibility"),
131        ("MLflow benchmark run for the fine-tuned DPO model on Ollama", "area/ml-infra"),
132        ("Claude Code hook: format-on-write, skills, agentic dev workflow", "area/dev-tooling"),
133        ("Safety footer copy: rip current hazard warning, veto text", "area/safety"),
134        ("Map marker tooltip and legend on the static site index.html", "area/map-site"),
135    ]
136    # This is a real, small test corpus check against the actual scikit-learn/vectorizer pipeline —
137    # requires scikit-learn importable, but no network and no `gh` call, matching every other
138    # script's --self-test convention in this repo (offline, deterministic given fixed input).
139    # scikit-learn is an optional dependency here: it is not in flake.nix's devShell and not on
140    # the CI runner, so a missing import is an environment fact, not a failure. Skip only the
141    # classifier cases and still run the parsing checks below — returning 1 for a skip used to
142    # abort `repo_stats.py`'s Python-coverage measurement (it runs every --self-test under
143    # `check=True`) and take the whole repo-stats job with it.
144    try:
145        for text, expected in cases:
146            ranked = classify(text, top=1)
147            got = ranked[0][0] if ranked else None
148            ok(got, expected, f"classify('{text[:40]}...') picks {expected}")
149    except ImportError as exc:
150        print(
151            f"  SKIP  scikit-learn not importable ({exc}) — classifier cases skipped, "
152            "parsing checks still run",
153            file=sys.stderr,
154        )
155        skipped = True
156
157    # Pure parsing check, no scikit-learn needed.
158    descriptions = load_taxonomy_descriptions()
159    ok(
160        "area/water-quality" in descriptions,
161        True,
162        "load_taxonomy_descriptions finds area/water-quality from scripts/lib/pr_labels.sh",
163    )
164    for label in AREA_CONTEXT_KEYWORDS:
165        if label not in descriptions:
166            fails += 1
167            print(f"  FAIL  AREA_CONTEXT_KEYWORDS has '{label}' which is not in the real taxonomy")
168
169    if fails == 0:
170        print("pr_label_nlp self-test: ok" + (" (classifier cases skipped)" if skipped else ""))
171        return 0
172    print(f"pr_label_nlp self-test: {fails} failure(s)", file=sys.stderr)
173    return 1
def main() -> int:
176def main() -> int:
177    parser = argparse.ArgumentParser(
178        description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
179    )
180    parser.add_argument("--pr", help="PR number to classify (needs gh auth status OK)")
181    parser.add_argument("--title", default="", help="classify arbitrary text instead of a real PR")
182    parser.add_argument("--body", default="", help="paired with --title")
183    parser.add_argument(
184        "--top", type=int, default=1, help="how many candidate labels to print (default 1)"
185    )
186    parser.add_argument("--json", action="store_true", help="machine-readable output")
187    parser.add_argument(
188        "--self-test", action="store_true", help="offline check, no network, no gh call"
189    )
190    args = parser.parse_args()
191
192    if args.self_test:
193        return self_test()
194
195    if args.pr:
196        text = fetch_pr_text(args.pr)
197    elif args.title or args.body:
198        text = f"{args.title} {args.body}"
199    else:
200        parser.error("need --pr, or --title/--body, or --self-test")
201        return 2
202
203    ranked = classify(text, top=args.top)
204    if args.json:
205        print(
206            json.dumps([{"label": label, "similarity": round(score, 4)} for label, score in ranked])
207        )
208    elif not ranked:
209        print(
210            "pr_label_nlp: no area label scored above zero similarity — text may be too short or off-domain"
211        )
212    else:
213        for label, score in ranked:
214            print(f"{label}\t{score:.4f}")
215    return 0