arxiv_digest
arxiv_digest — cache arXiv papers relevant to marola's forecasting domain.
scripts/arxiv_digest.py # fetch every query, cache new papers, print a summary
scripts/arxiv_digest.py --max-results 5 # results per query (default 5)
scripts/arxiv_digest.py --json # machine-readable summary on stdout
scripts/arxiv_digest.py --self-test # parser + cache self-check, no network (just quality-other)
Queries the real arXiv API (export.arxiv.org/api/query, Atom XML) for oceanography, sea/ocean
condition forecasting, jellyfish/marine-life prediction, and LLM-for-forecasting papers — the
research surface behind marola's own heuristics (docs/2-Building-marola/ARCHITECTURE.md) and a feeder for MIP
research (see MIP-0019). Caches the same way MadsLorentzen/ai-job-search's job tracker does
(surveyed in docs/3-Working-on-the-repo/SELF-DOCUMENTING.md): one flat JSON file per paper, keyed by arXiv id, so a
re-run never re-fetches or duplicates an already-seen paper, plus one flat JSONL index a human or
another script can grep/tail without touching the per-paper files.
Cache layout, under .tmp/arxiv_cache/ (gitignored — fetched data, not source):
papers/
Network is stdlib-only (urllib.request), matching this repo's other scripts (cost-split.py).
1#!/usr/bin/env python3 2"""arxiv_digest — cache arXiv papers relevant to marola's forecasting domain. 3 4 scripts/arxiv_digest.py # fetch every query, cache new papers, print a summary 5 scripts/arxiv_digest.py --max-results 5 # results per query (default 5) 6 scripts/arxiv_digest.py --json # machine-readable summary on stdout 7 scripts/arxiv_digest.py --self-test # parser + cache self-check, no network (just quality-other) 8 9Queries the real arXiv API (export.arxiv.org/api/query, Atom XML) for oceanography, sea/ocean 10condition forecasting, jellyfish/marine-life prediction, and LLM-for-forecasting papers — the 11research surface behind marola's own heuristics (docs/2-Building-marola/ARCHITECTURE.md) and a feeder for MIP 12research (see MIP-0019). Caches the same way `MadsLorentzen/ai-job-search`'s job tracker does 13(surveyed in docs/3-Working-on-the-repo/SELF-DOCUMENTING.md): one flat JSON file per paper, keyed by arXiv id, so a 14re-run never re-fetches or duplicates an already-seen paper, plus one flat JSONL index a human or 15another script can grep/tail without touching the per-paper files. 16 17Cache layout, under .tmp/arxiv_cache/ (gitignored — fetched data, not source): 18 papers/<arxiv-id-sans-version>.json one file per paper (id, title, summary, authors, 19 categories, published, matched_queries, relevance_score, 20 fetched_at) 21 index.jsonl one line per cached paper (id, title, published, 22 relevance_score, fetched_at) — rewritten each run from 23 the current papers/ contents, so it never drifts from them 24 25Network is stdlib-only (`urllib.request`), matching this repo's other scripts (cost-split.py). 26""" 27 28import argparse 29import datetime as dt 30import json 31import re 32import sys 33import urllib.error 34import urllib.parse 35import urllib.request 36from pathlib import Path 37from xml.etree import ElementTree 38 39ATOM_NS = "http://www.w3.org/2005/Atom" 40ARXIV_NS = "http://arxiv.org/schemas/atom" 41API_BASE = "https://export.arxiv.org/api/query" 42 43# Each query is (label, arXiv search_query, weight). Verified against the live API on 2026-09-06 — 44# every one of these returns real, on-topic results (some cross-domain noise is expected and 45# scored down, not filtered out, per MIP-0019's honesty about query precision). 46QUERIES: list[tuple[str, str, int]] = [ 47 ("ocean-physics", "cat:physics.ao-ph AND abs:ocean", 1), 48 ("sst-forecast", 'abs:"sea surface temperature" AND abs:forecast', 2), 49 ("llm-ocean", 'abs:"large language model" AND abs:ocean', 3), 50 ("llm-weather-forecast", 'abs:"large language model" AND abs:"weather forecasting"', 3), 51 ("jellyfish-prediction", "abs:jellyfish AND abs:prediction", 2), 52 ("marine-heatwave", 'abs:"marine heatwave"', 2), 53 ("ao-ph-ml", 'cat:physics.ao-ph AND abs:"machine learning"', 1), 54] 55 56 57def cache_dir(repo_root: Path) -> Path: 58 d = repo_root / ".tmp" / "arxiv_cache" 59 (d / "papers").mkdir(parents=True, exist_ok=True) 60 return d 61 62 63def arxiv_id_from_url(id_url: str) -> str: 64 """http://arxiv.org/abs/2609.03658v1 -> 2609.03658 (version-stripped, safe as a filename).""" 65 tail = id_url.rstrip("/").rsplit("/", 1)[-1] 66 return re.sub(r"v\d+$", "", tail) 67 68 69def parse_atom(xml_text: str) -> list[dict]: 70 """Atom feed text -> list of paper dicts (id/title/summary/authors/categories/published). 71 Raises on malformed XML — a caller decides whether that's fatal or skip-and-continue.""" 72 root = ElementTree.fromstring(xml_text) 73 entries = [] 74 for entry in root.findall(f"{{{ATOM_NS}}}entry"): 75 76 def text(tag: str, ns: str = ATOM_NS, _entry=entry) -> str: 77 el = _entry.find(f"{{{ns}}}{tag}") 78 return (el.text or "").strip() if el is not None else "" 79 80 id_url = text("id") 81 if not id_url: 82 continue 83 authors = [ 84 (a.find(f"{{{ATOM_NS}}}name").text or "").strip() 85 for a in entry.findall(f"{{{ATOM_NS}}}author") 86 if a.find(f"{{{ATOM_NS}}}name") is not None 87 ] 88 categories = [ 89 c.get("term", "") for c in entry.findall(f"{{{ATOM_NS}}}category") if c.get("term") 90 ] 91 entries.append( 92 { 93 "arxiv_id": arxiv_id_from_url(id_url), 94 "abs_url": id_url, 95 "title": re.sub(r"\s+", " ", text("title")), 96 "summary": re.sub(r"\s+", " ", text("summary")), 97 "authors": authors, 98 "categories": categories, 99 "published": text("published"), 100 } 101 ) 102 return entries 103 104 105def fetch_query(search_query: str, max_results: int, timeout: float = 20.0) -> str: 106 params = urllib.parse.urlencode( 107 { 108 "search_query": search_query, 109 "start": 0, 110 "max_results": max_results, 111 "sortBy": "submittedDate", 112 "sortOrder": "descending", 113 } 114 ) 115 req = urllib.request.Request( 116 f"{API_BASE}?{params}", headers={"User-Agent": "marola-arxiv-digest/1"} 117 ) 118 with urllib.request.urlopen(req, timeout=timeout) as r: # noqa: S310 (fixed http(s) API host) 119 return r.read().decode("utf-8") 120 121 122def relevance_score(paper: dict, matched: list[tuple[str, int]]) -> int: 123 """Sum of the weights of every query that matched this paper — a paper hit by both an 124 LLM-forecast query and a jellyfish query scores higher than one hit by a single broad query.""" 125 return sum(weight for _label, weight in matched) 126 127 128def load_cached_ids(store: Path) -> set[str]: 129 return {p.stem for p in (store / "papers").glob("*.json")} 130 131 132def write_index(store: Path) -> None: 133 rows = [] 134 for f in sorted((store / "papers").glob("*.json")): 135 data = json.loads(f.read_text()) 136 rows.append( 137 { 138 "arxiv_id": data["arxiv_id"], 139 "title": data["title"], 140 "published": data["published"], 141 "relevance_score": data["relevance_score"], 142 "fetched_at": data["fetched_at"], 143 } 144 ) 145 rows.sort(key=lambda r: r["relevance_score"], reverse=True) 146 with (store / "index.jsonl").open("w") as f: 147 for row in rows: 148 f.write(json.dumps(row, ensure_ascii=False) + "\n") 149 150 151def run(repo_root: Path, max_results: int) -> dict: 152 store = cache_dir(repo_root) 153 already = load_cached_ids(store) 154 matches_by_id: dict[str, list[tuple[str, int]]] = {} 155 papers_by_id: dict[str, dict] = {} 156 errors = [] 157 158 for label, query, weight in QUERIES: 159 try: 160 xml_text = fetch_query(query, max_results) 161 except (urllib.error.URLError, TimeoutError) as e: 162 errors.append(f"{label}: {e}") 163 continue 164 for paper in parse_atom(xml_text): 165 pid = paper["arxiv_id"] 166 papers_by_id.setdefault(pid, paper) 167 matches_by_id.setdefault(pid, []).append((label, weight)) 168 169 new_count = 0 170 for pid, paper in papers_by_id.items(): 171 if pid in already: 172 continue 173 matched = matches_by_id[pid] 174 record = { 175 **paper, 176 "matched_queries": [label for label, _w in matched], 177 "relevance_score": relevance_score(paper, matched), 178 "fetched_at": dt.datetime.now(dt.UTC).isoformat(), 179 } 180 (store / "papers" / f"{pid}.json").write_text( 181 json.dumps(record, indent=2, ensure_ascii=False) 182 ) 183 new_count += 1 184 185 write_index(store) 186 return { 187 "queries_run": len(QUERIES) - len(errors), 188 "queries_failed": errors, 189 "new_papers": new_count, 190 "total_cached": len(load_cached_ids(store)), 191 "index_path": str(store / "index.jsonl"), 192 } 193 194 195def self_test() -> None: 196 import tempfile 197 198 fails = 0 199 200 fixture = f"""<?xml version='1.0' encoding='UTF-8'?> 201<feed xmlns:opensearch="http://a9.com/-/spec/opensearch/1.1/" xmlns:arxiv="{ARXIV_NS}" xmlns="{ATOM_NS}"> 202 <entry> 203 <id>http://arxiv.org/abs/2609.03658v1</id> 204 <title> A Deep Learning Model for 205 Forecasting Sea Surface Temperature</title> 206 <updated>2026-09-03T10:56:02Z</updated> 207 <summary>We forecast sea surface temperature anomalies with a transformer.</summary> 208 <category term="physics.ao-ph" scheme="http://arxiv.org/schemas/atom"/> 209 <published>2026-09-03T10:56:02Z</published> 210 <author><name>Ada Lovelace</name></author> 211 <author><name>Grace Hopper</name></author> 212 </entry> 213 <entry> 214 <id>http://arxiv.org/abs/2601.00001v2</id> 215 <title>Notify About Jellyfish Blooms via Sonar</title> 216 <updated>2026-01-01T00:00:00Z</updated> 217 <summary>A sonar pipeline to detect jellyfish blooms near beaches.</summary> 218 <category term="cs.CV" scheme="http://arxiv.org/schemas/atom"/> 219 <published>2026-01-01T00:00:00Z</published> 220 <author><name>Marie Curie</name></author> 221 </entry> 222</feed>""" 223 224 entries = parse_atom(fixture) 225 if len(entries) == 2: 226 print(" ok parse_atom extracts both fixture entries") 227 else: 228 print(f" FAIL parse_atom got {len(entries)} entries, expected 2") 229 fails += 1 230 231 e0 = entries[0] 232 if e0["arxiv_id"] == "2609.03658": 233 print(" ok version suffix stripped from the arXiv id") 234 else: 235 print(f" FAIL arxiv_id was {e0['arxiv_id']!r}, expected '2609.03658'") 236 fails += 1 237 if e0["title"] == "A Deep Learning Model for Forecasting Sea Surface Temperature": 238 print(" ok multi-line title is whitespace-normalized") 239 else: 240 print(f" FAIL title was {e0['title']!r}") 241 fails += 1 242 if e0["authors"] == ["Ada Lovelace", "Grace Hopper"]: 243 print(" ok both authors extracted in order") 244 else: 245 print(f" FAIL authors were {e0['authors']!r}") 246 fails += 1 247 if e0["categories"] == ["physics.ao-ph"]: 248 print(" ok category term extracted") 249 else: 250 print(f" FAIL categories were {e0['categories']!r}") 251 fails += 1 252 253 e1 = entries[1] 254 if e1["arxiv_id"] == "2601.00001": 255 print(" ok a 'v2' suffix is also stripped correctly") 256 else: 257 print(f" FAIL arxiv_id was {e1['arxiv_id']!r}, expected '2601.00001'") 258 fails += 1 259 260 # relevance_score: multiple query matches outscore a single broad match. 261 single = relevance_score(e0, [("ocean-physics", 1)]) 262 double = relevance_score(e0, [("ocean-physics", 1), ("sst-forecast", 2)]) 263 if double > single and double == 3: 264 print(" ok relevance_score sums matched-query weights") 265 else: 266 print(f" FAIL relevance_score: single={single} double={double}") 267 fails += 1 268 269 # Cache round-trip: write both fixture papers, confirm dedup on a second write and a 270 # correctly sorted, rewritten index — entirely inside a temp dir, no network. 271 with tempfile.TemporaryDirectory() as tmp: 272 repo_root = Path(tmp) 273 store = cache_dir(repo_root) 274 if load_cached_ids(store) == set(): 275 print(" ok a fresh cache dir starts empty") 276 else: 277 print(" FAIL fresh cache dir was not empty") 278 fails += 1 279 280 for paper, matched in ((e0, [("sst-forecast", 2)]), (e1, [("jellyfish-prediction", 2)])): 281 record = { 282 **paper, 283 "matched_queries": [label for label, _w in matched], 284 "relevance_score": relevance_score(paper, matched), 285 "fetched_at": "2026-09-06T00:00:00+00:00", 286 } 287 (store / "papers" / f"{paper['arxiv_id']}.json").write_text(json.dumps(record)) 288 write_index(store) 289 290 cached = load_cached_ids(store) 291 if cached == {"2609.03658", "2601.00001"}: 292 print(" ok both papers land in the cache, keyed by arxiv id") 293 else: 294 print(f" FAIL cached ids were {cached!r}") 295 fails += 1 296 297 index_lines = (store / "index.jsonl").read_text().splitlines() 298 if len(index_lines) == 2: 299 print(" ok index.jsonl has one line per cached paper") 300 else: 301 print(f" FAIL index.jsonl had {len(index_lines)} lines, expected 2") 302 fails += 1 303 304 # Simulate re-fetching the same paper: caller-side dedup (run()'s own logic) must skip 305 # ids already in load_cached_ids() rather than re-write/duplicate. 306 already = load_cached_ids(store) 307 would_skip = "2609.03658" in already 308 if would_skip: 309 print(" ok an already-cached id is recognized for skip-on-refetch") 310 else: 311 print(" FAIL already-cached id was not recognized") 312 fails += 1 313 314 # Malformed XML raises rather than silently returning nothing — a caller must not mistake a 315 # parse failure for "no results this query". 316 try: 317 parse_atom("<not-xml") 318 print(" FAIL malformed XML did not raise") 319 fails += 1 320 except ElementTree.ParseError: 321 print(" ok malformed XML raises ParseError, not swallowed silently") 322 323 if fails == 0: 324 print("arxiv_digest self-test: ok") 325 else: 326 print(f"arxiv_digest self-test: {fails} failure(s)", file=sys.stderr) 327 sys.exit(1) 328 329 330def main() -> None: 331 ap = argparse.ArgumentParser( 332 description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter 333 ) 334 ap.add_argument("--max-results", type=int, default=5, help="results per query (default 5)") 335 ap.add_argument("--json", action="store_true", help="machine-readable summary on stdout") 336 ap.add_argument("--self-test", action="store_true") 337 args = ap.parse_args() 338 339 if args.self_test: 340 self_test() 341 return 342 343 repo_root = Path(__file__).resolve().parent.parent 344 summary = run(repo_root, args.max_results) 345 if args.json: 346 print(json.dumps(summary, indent=2)) 347 else: 348 print(f"queries run: {summary['queries_run']}/{len(QUERIES)}") 349 if summary["queries_failed"]: 350 print(f"queries failed: {summary['queries_failed']}") 351 print(f"new papers cached: {summary['new_papers']}") 352 print(f"total cached: {summary['total_cached']}") 353 print(f"index: {summary['index_path']}") 354 355 356if __name__ == "__main__": 357 main()
64def arxiv_id_from_url(id_url: str) -> str: 65 """http://arxiv.org/abs/2609.03658v1 -> 2609.03658 (version-stripped, safe as a filename).""" 66 tail = id_url.rstrip("/").rsplit("/", 1)[-1] 67 return re.sub(r"v\d+$", "", tail)
http://arxiv.org/abs/2609.03658v1 -> 2609.03658 (version-stripped, safe as a filename).
70def parse_atom(xml_text: str) -> list[dict]: 71 """Atom feed text -> list of paper dicts (id/title/summary/authors/categories/published). 72 Raises on malformed XML — a caller decides whether that's fatal or skip-and-continue.""" 73 root = ElementTree.fromstring(xml_text) 74 entries = [] 75 for entry in root.findall(f"{{{ATOM_NS}}}entry"): 76 77 def text(tag: str, ns: str = ATOM_NS, _entry=entry) -> str: 78 el = _entry.find(f"{{{ns}}}{tag}") 79 return (el.text or "").strip() if el is not None else "" 80 81 id_url = text("id") 82 if not id_url: 83 continue 84 authors = [ 85 (a.find(f"{{{ATOM_NS}}}name").text or "").strip() 86 for a in entry.findall(f"{{{ATOM_NS}}}author") 87 if a.find(f"{{{ATOM_NS}}}name") is not None 88 ] 89 categories = [ 90 c.get("term", "") for c in entry.findall(f"{{{ATOM_NS}}}category") if c.get("term") 91 ] 92 entries.append( 93 { 94 "arxiv_id": arxiv_id_from_url(id_url), 95 "abs_url": id_url, 96 "title": re.sub(r"\s+", " ", text("title")), 97 "summary": re.sub(r"\s+", " ", text("summary")), 98 "authors": authors, 99 "categories": categories, 100 "published": text("published"), 101 } 102 ) 103 return entries
Atom feed text -> list of paper dicts (id/title/summary/authors/categories/published). Raises on malformed XML — a caller decides whether that's fatal or skip-and-continue.
106def fetch_query(search_query: str, max_results: int, timeout: float = 20.0) -> str: 107 params = urllib.parse.urlencode( 108 { 109 "search_query": search_query, 110 "start": 0, 111 "max_results": max_results, 112 "sortBy": "submittedDate", 113 "sortOrder": "descending", 114 } 115 ) 116 req = urllib.request.Request( 117 f"{API_BASE}?{params}", headers={"User-Agent": "marola-arxiv-digest/1"} 118 ) 119 with urllib.request.urlopen(req, timeout=timeout) as r: # noqa: S310 (fixed http(s) API host) 120 return r.read().decode("utf-8")
123def relevance_score(paper: dict, matched: list[tuple[str, int]]) -> int: 124 """Sum of the weights of every query that matched this paper — a paper hit by both an 125 LLM-forecast query and a jellyfish query scores higher than one hit by a single broad query.""" 126 return sum(weight for _label, weight in matched)
Sum of the weights of every query that matched this paper — a paper hit by both an LLM-forecast query and a jellyfish query scores higher than one hit by a single broad query.
133def write_index(store: Path) -> None: 134 rows = [] 135 for f in sorted((store / "papers").glob("*.json")): 136 data = json.loads(f.read_text()) 137 rows.append( 138 { 139 "arxiv_id": data["arxiv_id"], 140 "title": data["title"], 141 "published": data["published"], 142 "relevance_score": data["relevance_score"], 143 "fetched_at": data["fetched_at"], 144 } 145 ) 146 rows.sort(key=lambda r: r["relevance_score"], reverse=True) 147 with (store / "index.jsonl").open("w") as f: 148 for row in rows: 149 f.write(json.dumps(row, ensure_ascii=False) + "\n")
152def run(repo_root: Path, max_results: int) -> dict: 153 store = cache_dir(repo_root) 154 already = load_cached_ids(store) 155 matches_by_id: dict[str, list[tuple[str, int]]] = {} 156 papers_by_id: dict[str, dict] = {} 157 errors = [] 158 159 for label, query, weight in QUERIES: 160 try: 161 xml_text = fetch_query(query, max_results) 162 except (urllib.error.URLError, TimeoutError) as e: 163 errors.append(f"{label}: {e}") 164 continue 165 for paper in parse_atom(xml_text): 166 pid = paper["arxiv_id"] 167 papers_by_id.setdefault(pid, paper) 168 matches_by_id.setdefault(pid, []).append((label, weight)) 169 170 new_count = 0 171 for pid, paper in papers_by_id.items(): 172 if pid in already: 173 continue 174 matched = matches_by_id[pid] 175 record = { 176 **paper, 177 "matched_queries": [label for label, _w in matched], 178 "relevance_score": relevance_score(paper, matched), 179 "fetched_at": dt.datetime.now(dt.UTC).isoformat(), 180 } 181 (store / "papers" / f"{pid}.json").write_text( 182 json.dumps(record, indent=2, ensure_ascii=False) 183 ) 184 new_count += 1 185 186 write_index(store) 187 return { 188 "queries_run": len(QUERIES) - len(errors), 189 "queries_failed": errors, 190 "new_papers": new_count, 191 "total_cached": len(load_cached_ids(store)), 192 "index_path": str(store / "index.jsonl"), 193 }
196def self_test() -> None: 197 import tempfile 198 199 fails = 0 200 201 fixture = f"""<?xml version='1.0' encoding='UTF-8'?> 202<feed xmlns:opensearch="http://a9.com/-/spec/opensearch/1.1/" xmlns:arxiv="{ARXIV_NS}" xmlns="{ATOM_NS}"> 203 <entry> 204 <id>http://arxiv.org/abs/2609.03658v1</id> 205 <title> A Deep Learning Model for 206 Forecasting Sea Surface Temperature</title> 207 <updated>2026-09-03T10:56:02Z</updated> 208 <summary>We forecast sea surface temperature anomalies with a transformer.</summary> 209 <category term="physics.ao-ph" scheme="http://arxiv.org/schemas/atom"/> 210 <published>2026-09-03T10:56:02Z</published> 211 <author><name>Ada Lovelace</name></author> 212 <author><name>Grace Hopper</name></author> 213 </entry> 214 <entry> 215 <id>http://arxiv.org/abs/2601.00001v2</id> 216 <title>Notify About Jellyfish Blooms via Sonar</title> 217 <updated>2026-01-01T00:00:00Z</updated> 218 <summary>A sonar pipeline to detect jellyfish blooms near beaches.</summary> 219 <category term="cs.CV" scheme="http://arxiv.org/schemas/atom"/> 220 <published>2026-01-01T00:00:00Z</published> 221 <author><name>Marie Curie</name></author> 222 </entry> 223</feed>""" 224 225 entries = parse_atom(fixture) 226 if len(entries) == 2: 227 print(" ok parse_atom extracts both fixture entries") 228 else: 229 print(f" FAIL parse_atom got {len(entries)} entries, expected 2") 230 fails += 1 231 232 e0 = entries[0] 233 if e0["arxiv_id"] == "2609.03658": 234 print(" ok version suffix stripped from the arXiv id") 235 else: 236 print(f" FAIL arxiv_id was {e0['arxiv_id']!r}, expected '2609.03658'") 237 fails += 1 238 if e0["title"] == "A Deep Learning Model for Forecasting Sea Surface Temperature": 239 print(" ok multi-line title is whitespace-normalized") 240 else: 241 print(f" FAIL title was {e0['title']!r}") 242 fails += 1 243 if e0["authors"] == ["Ada Lovelace", "Grace Hopper"]: 244 print(" ok both authors extracted in order") 245 else: 246 print(f" FAIL authors were {e0['authors']!r}") 247 fails += 1 248 if e0["categories"] == ["physics.ao-ph"]: 249 print(" ok category term extracted") 250 else: 251 print(f" FAIL categories were {e0['categories']!r}") 252 fails += 1 253 254 e1 = entries[1] 255 if e1["arxiv_id"] == "2601.00001": 256 print(" ok a 'v2' suffix is also stripped correctly") 257 else: 258 print(f" FAIL arxiv_id was {e1['arxiv_id']!r}, expected '2601.00001'") 259 fails += 1 260 261 # relevance_score: multiple query matches outscore a single broad match. 262 single = relevance_score(e0, [("ocean-physics", 1)]) 263 double = relevance_score(e0, [("ocean-physics", 1), ("sst-forecast", 2)]) 264 if double > single and double == 3: 265 print(" ok relevance_score sums matched-query weights") 266 else: 267 print(f" FAIL relevance_score: single={single} double={double}") 268 fails += 1 269 270 # Cache round-trip: write both fixture papers, confirm dedup on a second write and a 271 # correctly sorted, rewritten index — entirely inside a temp dir, no network. 272 with tempfile.TemporaryDirectory() as tmp: 273 repo_root = Path(tmp) 274 store = cache_dir(repo_root) 275 if load_cached_ids(store) == set(): 276 print(" ok a fresh cache dir starts empty") 277 else: 278 print(" FAIL fresh cache dir was not empty") 279 fails += 1 280 281 for paper, matched in ((e0, [("sst-forecast", 2)]), (e1, [("jellyfish-prediction", 2)])): 282 record = { 283 **paper, 284 "matched_queries": [label for label, _w in matched], 285 "relevance_score": relevance_score(paper, matched), 286 "fetched_at": "2026-09-06T00:00:00+00:00", 287 } 288 (store / "papers" / f"{paper['arxiv_id']}.json").write_text(json.dumps(record)) 289 write_index(store) 290 291 cached = load_cached_ids(store) 292 if cached == {"2609.03658", "2601.00001"}: 293 print(" ok both papers land in the cache, keyed by arxiv id") 294 else: 295 print(f" FAIL cached ids were {cached!r}") 296 fails += 1 297 298 index_lines = (store / "index.jsonl").read_text().splitlines() 299 if len(index_lines) == 2: 300 print(" ok index.jsonl has one line per cached paper") 301 else: 302 print(f" FAIL index.jsonl had {len(index_lines)} lines, expected 2") 303 fails += 1 304 305 # Simulate re-fetching the same paper: caller-side dedup (run()'s own logic) must skip 306 # ids already in load_cached_ids() rather than re-write/duplicate. 307 already = load_cached_ids(store) 308 would_skip = "2609.03658" in already 309 if would_skip: 310 print(" ok an already-cached id is recognized for skip-on-refetch") 311 else: 312 print(" FAIL already-cached id was not recognized") 313 fails += 1 314 315 # Malformed XML raises rather than silently returning nothing — a caller must not mistake a 316 # parse failure for "no results this query". 317 try: 318 parse_atom("<not-xml") 319 print(" FAIL malformed XML did not raise") 320 fails += 1 321 except ElementTree.ParseError: 322 print(" ok malformed XML raises ParseError, not swallowed silently") 323 324 if fails == 0: 325 print("arxiv_digest self-test: ok") 326 else: 327 print(f"arxiv_digest self-test: {fails} failure(s)", file=sys.stderr) 328 sys.exit(1)
331def main() -> None: 332 ap = argparse.ArgumentParser( 333 description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter 334 ) 335 ap.add_argument("--max-results", type=int, default=5, help="results per query (default 5)") 336 ap.add_argument("--json", action="store_true", help="machine-readable summary on stdout") 337 ap.add_argument("--self-test", action="store_true") 338 args = ap.parse_args() 339 340 if args.self_test: 341 self_test() 342 return 343 344 repo_root = Path(__file__).resolve().parent.parent 345 summary = run(repo_root, args.max_results) 346 if args.json: 347 print(json.dumps(summary, indent=2)) 348 else: 349 print(f"queries run: {summary['queries_run']}/{len(QUERIES)}") 350 if summary["queries_failed"]: 351 print(f"queries failed: {summary['queries_failed']}") 352 print(f"new papers cached: {summary['new_papers']}") 353 print(f"total cached: {summary['total_cached']}") 354 print(f"index: {summary['index_path']}")