arxiv_digest

arxiv_digest — cache arXiv papers relevant to marola's forecasting domain.

scripts/arxiv_digest.py                 # fetch every query, cache new papers, print a summary
scripts/arxiv_digest.py --max-results 5  # results per query (default 5)
scripts/arxiv_digest.py --json           # machine-readable summary on stdout
scripts/arxiv_digest.py --self-test      # parser + cache self-check, no network (just quality-other)

Queries the real arXiv API (export.arxiv.org/api/query, Atom XML) for oceanography, sea/ocean condition forecasting, jellyfish/marine-life prediction, and LLM-for-forecasting papers — the research surface behind marola's own heuristics (docs/2-Building-marola/ARCHITECTURE.md) and a feeder for MIP research (see MIP-0019). Caches the same way MadsLorentzen/ai-job-search's job tracker does (surveyed in docs/3-Working-on-the-repo/SELF-DOCUMENTING.md): one flat JSON file per paper, keyed by arXiv id, so a re-run never re-fetches or duplicates an already-seen paper, plus one flat JSONL index a human or another script can grep/tail without touching the per-paper files.

Cache layout, under .tmp/arxiv_cache/ (gitignored — fetched data, not source): papers/.json one file per paper (id, title, summary, authors, categories, published, matched_queries, relevance_score, fetched_at) index.jsonl one line per cached paper (id, title, published, relevance_score, fetched_at) — rewritten each run from the current papers/ contents, so it never drifts from them

Network is stdlib-only (urllib.request), matching this repo's other scripts (cost-split.py).

  1#!/usr/bin/env python3
  2"""arxiv_digest — cache arXiv papers relevant to marola's forecasting domain.
  3
  4    scripts/arxiv_digest.py                 # fetch every query, cache new papers, print a summary
  5    scripts/arxiv_digest.py --max-results 5  # results per query (default 5)
  6    scripts/arxiv_digest.py --json           # machine-readable summary on stdout
  7    scripts/arxiv_digest.py --self-test      # parser + cache self-check, no network (just quality-other)
  8
  9Queries the real arXiv API (export.arxiv.org/api/query, Atom XML) for oceanography, sea/ocean
 10condition forecasting, jellyfish/marine-life prediction, and LLM-for-forecasting papers — the
 11research surface behind marola's own heuristics (docs/2-Building-marola/ARCHITECTURE.md) and a feeder for MIP
 12research (see MIP-0019). Caches the same way `MadsLorentzen/ai-job-search`'s job tracker does
 13(surveyed in docs/3-Working-on-the-repo/SELF-DOCUMENTING.md): one flat JSON file per paper, keyed by arXiv id, so a
 14re-run never re-fetches or duplicates an already-seen paper, plus one flat JSONL index a human or
 15another script can grep/tail without touching the per-paper files.
 16
 17Cache layout, under .tmp/arxiv_cache/ (gitignored — fetched data, not source):
 18    papers/<arxiv-id-sans-version>.json   one file per paper (id, title, summary, authors,
 19                                           categories, published, matched_queries, relevance_score,
 20                                           fetched_at)
 21    index.jsonl                           one line per cached paper (id, title, published,
 22                                           relevance_score, fetched_at) — rewritten each run from
 23                                           the current papers/ contents, so it never drifts from them
 24
 25Network is stdlib-only (`urllib.request`), matching this repo's other scripts (cost-split.py).
 26"""
 27
 28import argparse
 29import datetime as dt
 30import json
 31import re
 32import sys
 33import urllib.error
 34import urllib.parse
 35import urllib.request
 36from pathlib import Path
 37from xml.etree import ElementTree
 38
 39ATOM_NS = "http://www.w3.org/2005/Atom"
 40ARXIV_NS = "http://arxiv.org/schemas/atom"
 41API_BASE = "https://export.arxiv.org/api/query"
 42
 43# Each query is (label, arXiv search_query, weight). Verified against the live API on 2026-09-06 —
 44# every one of these returns real, on-topic results (some cross-domain noise is expected and
 45# scored down, not filtered out, per MIP-0019's honesty about query precision).
 46QUERIES: list[tuple[str, str, int]] = [
 47    ("ocean-physics", "cat:physics.ao-ph AND abs:ocean", 1),
 48    ("sst-forecast", 'abs:"sea surface temperature" AND abs:forecast', 2),
 49    ("llm-ocean", 'abs:"large language model" AND abs:ocean', 3),
 50    ("llm-weather-forecast", 'abs:"large language model" AND abs:"weather forecasting"', 3),
 51    ("jellyfish-prediction", "abs:jellyfish AND abs:prediction", 2),
 52    ("marine-heatwave", 'abs:"marine heatwave"', 2),
 53    ("ao-ph-ml", 'cat:physics.ao-ph AND abs:"machine learning"', 1),
 54]
 55
 56
 57def cache_dir(repo_root: Path) -> Path:
 58    d = repo_root / ".tmp" / "arxiv_cache"
 59    (d / "papers").mkdir(parents=True, exist_ok=True)
 60    return d
 61
 62
 63def arxiv_id_from_url(id_url: str) -> str:
 64    """http://arxiv.org/abs/2609.03658v1 -> 2609.03658 (version-stripped, safe as a filename)."""
 65    tail = id_url.rstrip("/").rsplit("/", 1)[-1]
 66    return re.sub(r"v\d+$", "", tail)
 67
 68
 69def parse_atom(xml_text: str) -> list[dict]:
 70    """Atom feed text -> list of paper dicts (id/title/summary/authors/categories/published).
 71    Raises on malformed XML — a caller decides whether that's fatal or skip-and-continue."""
 72    root = ElementTree.fromstring(xml_text)
 73    entries = []
 74    for entry in root.findall(f"{{{ATOM_NS}}}entry"):
 75
 76        def text(tag: str, ns: str = ATOM_NS, _entry=entry) -> str:
 77            el = _entry.find(f"{{{ns}}}{tag}")
 78            return (el.text or "").strip() if el is not None else ""
 79
 80        id_url = text("id")
 81        if not id_url:
 82            continue
 83        authors = [
 84            (a.find(f"{{{ATOM_NS}}}name").text or "").strip()
 85            for a in entry.findall(f"{{{ATOM_NS}}}author")
 86            if a.find(f"{{{ATOM_NS}}}name") is not None
 87        ]
 88        categories = [
 89            c.get("term", "") for c in entry.findall(f"{{{ATOM_NS}}}category") if c.get("term")
 90        ]
 91        entries.append(
 92            {
 93                "arxiv_id": arxiv_id_from_url(id_url),
 94                "abs_url": id_url,
 95                "title": re.sub(r"\s+", " ", text("title")),
 96                "summary": re.sub(r"\s+", " ", text("summary")),
 97                "authors": authors,
 98                "categories": categories,
 99                "published": text("published"),
100            }
101        )
102    return entries
103
104
105def fetch_query(search_query: str, max_results: int, timeout: float = 20.0) -> str:
106    params = urllib.parse.urlencode(
107        {
108            "search_query": search_query,
109            "start": 0,
110            "max_results": max_results,
111            "sortBy": "submittedDate",
112            "sortOrder": "descending",
113        }
114    )
115    req = urllib.request.Request(
116        f"{API_BASE}?{params}", headers={"User-Agent": "marola-arxiv-digest/1"}
117    )
118    with urllib.request.urlopen(req, timeout=timeout) as r:  # noqa: S310 (fixed http(s) API host)
119        return r.read().decode("utf-8")
120
121
122def relevance_score(paper: dict, matched: list[tuple[str, int]]) -> int:
123    """Sum of the weights of every query that matched this paper — a paper hit by both an
124    LLM-forecast query and a jellyfish query scores higher than one hit by a single broad query."""
125    return sum(weight for _label, weight in matched)
126
127
128def load_cached_ids(store: Path) -> set[str]:
129    return {p.stem for p in (store / "papers").glob("*.json")}
130
131
132def write_index(store: Path) -> None:
133    rows = []
134    for f in sorted((store / "papers").glob("*.json")):
135        data = json.loads(f.read_text())
136        rows.append(
137            {
138                "arxiv_id": data["arxiv_id"],
139                "title": data["title"],
140                "published": data["published"],
141                "relevance_score": data["relevance_score"],
142                "fetched_at": data["fetched_at"],
143            }
144        )
145    rows.sort(key=lambda r: r["relevance_score"], reverse=True)
146    with (store / "index.jsonl").open("w") as f:
147        for row in rows:
148            f.write(json.dumps(row, ensure_ascii=False) + "\n")
149
150
151def run(repo_root: Path, max_results: int) -> dict:
152    store = cache_dir(repo_root)
153    already = load_cached_ids(store)
154    matches_by_id: dict[str, list[tuple[str, int]]] = {}
155    papers_by_id: dict[str, dict] = {}
156    errors = []
157
158    for label, query, weight in QUERIES:
159        try:
160            xml_text = fetch_query(query, max_results)
161        except (urllib.error.URLError, TimeoutError) as e:
162            errors.append(f"{label}: {e}")
163            continue
164        for paper in parse_atom(xml_text):
165            pid = paper["arxiv_id"]
166            papers_by_id.setdefault(pid, paper)
167            matches_by_id.setdefault(pid, []).append((label, weight))
168
169    new_count = 0
170    for pid, paper in papers_by_id.items():
171        if pid in already:
172            continue
173        matched = matches_by_id[pid]
174        record = {
175            **paper,
176            "matched_queries": [label for label, _w in matched],
177            "relevance_score": relevance_score(paper, matched),
178            "fetched_at": dt.datetime.now(dt.UTC).isoformat(),
179        }
180        (store / "papers" / f"{pid}.json").write_text(
181            json.dumps(record, indent=2, ensure_ascii=False)
182        )
183        new_count += 1
184
185    write_index(store)
186    return {
187        "queries_run": len(QUERIES) - len(errors),
188        "queries_failed": errors,
189        "new_papers": new_count,
190        "total_cached": len(load_cached_ids(store)),
191        "index_path": str(store / "index.jsonl"),
192    }
193
194
195def self_test() -> None:
196    import tempfile
197
198    fails = 0
199
200    fixture = f"""<?xml version='1.0' encoding='UTF-8'?>
201<feed xmlns:opensearch="http://a9.com/-/spec/opensearch/1.1/" xmlns:arxiv="{ARXIV_NS}" xmlns="{ATOM_NS}">
202  <entry>
203    <id>http://arxiv.org/abs/2609.03658v1</id>
204    <title>  A Deep Learning Model for
205      Forecasting  Sea Surface Temperature</title>
206    <updated>2026-09-03T10:56:02Z</updated>
207    <summary>We forecast sea surface temperature anomalies with a transformer.</summary>
208    <category term="physics.ao-ph" scheme="http://arxiv.org/schemas/atom"/>
209    <published>2026-09-03T10:56:02Z</published>
210    <author><name>Ada Lovelace</name></author>
211    <author><name>Grace Hopper</name></author>
212  </entry>
213  <entry>
214    <id>http://arxiv.org/abs/2601.00001v2</id>
215    <title>Notify About Jellyfish Blooms via Sonar</title>
216    <updated>2026-01-01T00:00:00Z</updated>
217    <summary>A sonar pipeline to detect jellyfish blooms near beaches.</summary>
218    <category term="cs.CV" scheme="http://arxiv.org/schemas/atom"/>
219    <published>2026-01-01T00:00:00Z</published>
220    <author><name>Marie Curie</name></author>
221  </entry>
222</feed>"""
223
224    entries = parse_atom(fixture)
225    if len(entries) == 2:
226        print("  ok   parse_atom extracts both fixture entries")
227    else:
228        print(f"  FAIL parse_atom got {len(entries)} entries, expected 2")
229        fails += 1
230
231    e0 = entries[0]
232    if e0["arxiv_id"] == "2609.03658":
233        print("  ok   version suffix stripped from the arXiv id")
234    else:
235        print(f"  FAIL arxiv_id was {e0['arxiv_id']!r}, expected '2609.03658'")
236        fails += 1
237    if e0["title"] == "A Deep Learning Model for Forecasting Sea Surface Temperature":
238        print("  ok   multi-line title is whitespace-normalized")
239    else:
240        print(f"  FAIL title was {e0['title']!r}")
241        fails += 1
242    if e0["authors"] == ["Ada Lovelace", "Grace Hopper"]:
243        print("  ok   both authors extracted in order")
244    else:
245        print(f"  FAIL authors were {e0['authors']!r}")
246        fails += 1
247    if e0["categories"] == ["physics.ao-ph"]:
248        print("  ok   category term extracted")
249    else:
250        print(f"  FAIL categories were {e0['categories']!r}")
251        fails += 1
252
253    e1 = entries[1]
254    if e1["arxiv_id"] == "2601.00001":
255        print("  ok   a 'v2' suffix is also stripped correctly")
256    else:
257        print(f"  FAIL arxiv_id was {e1['arxiv_id']!r}, expected '2601.00001'")
258        fails += 1
259
260    # relevance_score: multiple query matches outscore a single broad match.
261    single = relevance_score(e0, [("ocean-physics", 1)])
262    double = relevance_score(e0, [("ocean-physics", 1), ("sst-forecast", 2)])
263    if double > single and double == 3:
264        print("  ok   relevance_score sums matched-query weights")
265    else:
266        print(f"  FAIL relevance_score: single={single} double={double}")
267        fails += 1
268
269    # Cache round-trip: write both fixture papers, confirm dedup on a second write and a
270    # correctly sorted, rewritten index — entirely inside a temp dir, no network.
271    with tempfile.TemporaryDirectory() as tmp:
272        repo_root = Path(tmp)
273        store = cache_dir(repo_root)
274        if load_cached_ids(store) == set():
275            print("  ok   a fresh cache dir starts empty")
276        else:
277            print("  FAIL fresh cache dir was not empty")
278            fails += 1
279
280        for paper, matched in ((e0, [("sst-forecast", 2)]), (e1, [("jellyfish-prediction", 2)])):
281            record = {
282                **paper,
283                "matched_queries": [label for label, _w in matched],
284                "relevance_score": relevance_score(paper, matched),
285                "fetched_at": "2026-09-06T00:00:00+00:00",
286            }
287            (store / "papers" / f"{paper['arxiv_id']}.json").write_text(json.dumps(record))
288        write_index(store)
289
290        cached = load_cached_ids(store)
291        if cached == {"2609.03658", "2601.00001"}:
292            print("  ok   both papers land in the cache, keyed by arxiv id")
293        else:
294            print(f"  FAIL cached ids were {cached!r}")
295            fails += 1
296
297        index_lines = (store / "index.jsonl").read_text().splitlines()
298        if len(index_lines) == 2:
299            print("  ok   index.jsonl has one line per cached paper")
300        else:
301            print(f"  FAIL index.jsonl had {len(index_lines)} lines, expected 2")
302            fails += 1
303
304        # Simulate re-fetching the same paper: caller-side dedup (run()'s own logic) must skip
305        # ids already in load_cached_ids() rather than re-write/duplicate.
306        already = load_cached_ids(store)
307        would_skip = "2609.03658" in already
308        if would_skip:
309            print("  ok   an already-cached id is recognized for skip-on-refetch")
310        else:
311            print("  FAIL already-cached id was not recognized")
312            fails += 1
313
314    # Malformed XML raises rather than silently returning nothing — a caller must not mistake a
315    # parse failure for "no results this query".
316    try:
317        parse_atom("<not-xml")
318        print("  FAIL malformed XML did not raise")
319        fails += 1
320    except ElementTree.ParseError:
321        print("  ok   malformed XML raises ParseError, not swallowed silently")
322
323    if fails == 0:
324        print("arxiv_digest self-test: ok")
325    else:
326        print(f"arxiv_digest self-test: {fails} failure(s)", file=sys.stderr)
327        sys.exit(1)
328
329
330def main() -> None:
331    ap = argparse.ArgumentParser(
332        description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
333    )
334    ap.add_argument("--max-results", type=int, default=5, help="results per query (default 5)")
335    ap.add_argument("--json", action="store_true", help="machine-readable summary on stdout")
336    ap.add_argument("--self-test", action="store_true")
337    args = ap.parse_args()
338
339    if args.self_test:
340        self_test()
341        return
342
343    repo_root = Path(__file__).resolve().parent.parent
344    summary = run(repo_root, args.max_results)
345    if args.json:
346        print(json.dumps(summary, indent=2))
347    else:
348        print(f"queries run: {summary['queries_run']}/{len(QUERIES)}")
349        if summary["queries_failed"]:
350            print(f"queries failed: {summary['queries_failed']}")
351        print(f"new papers cached: {summary['new_papers']}")
352        print(f"total cached: {summary['total_cached']}")
353        print(f"index: {summary['index_path']}")
354
355
356if __name__ == "__main__":
357    main()
ATOM_NS = 'http://www.w3.org/2005/Atom'
ARXIV_NS = 'http://arxiv.org/schemas/atom'
API_BASE = 'https://export.arxiv.org/api/query'
QUERIES: list[tuple[str, str, int]] = [('ocean-physics', 'cat:physics.ao-ph AND abs:ocean', 1), ('sst-forecast', 'abs:"sea surface temperature" AND abs:forecast', 2), ('llm-ocean', 'abs:"large language model" AND abs:ocean', 3), ('llm-weather-forecast', 'abs:"large language model" AND abs:"weather forecasting"', 3), ('jellyfish-prediction', 'abs:jellyfish AND abs:prediction', 2), ('marine-heatwave', 'abs:"marine heatwave"', 2), ('ao-ph-ml', 'cat:physics.ao-ph AND abs:"machine learning"', 1)]
def cache_dir(repo_root: pathlib.Path) -> pathlib.Path:
58def cache_dir(repo_root: Path) -> Path:
59    d = repo_root / ".tmp" / "arxiv_cache"
60    (d / "papers").mkdir(parents=True, exist_ok=True)
61    return d
def arxiv_id_from_url(id_url: str) -> str:
64def arxiv_id_from_url(id_url: str) -> str:
65    """http://arxiv.org/abs/2609.03658v1 -> 2609.03658 (version-stripped, safe as a filename)."""
66    tail = id_url.rstrip("/").rsplit("/", 1)[-1]
67    return re.sub(r"v\d+$", "", tail)

http://arxiv.org/abs/2609.03658v1 -> 2609.03658 (version-stripped, safe as a filename).

def parse_atom(xml_text: str) -> list[dict]:
 70def parse_atom(xml_text: str) -> list[dict]:
 71    """Atom feed text -> list of paper dicts (id/title/summary/authors/categories/published).
 72    Raises on malformed XML — a caller decides whether that's fatal or skip-and-continue."""
 73    root = ElementTree.fromstring(xml_text)
 74    entries = []
 75    for entry in root.findall(f"{{{ATOM_NS}}}entry"):
 76
 77        def text(tag: str, ns: str = ATOM_NS, _entry=entry) -> str:
 78            el = _entry.find(f"{{{ns}}}{tag}")
 79            return (el.text or "").strip() if el is not None else ""
 80
 81        id_url = text("id")
 82        if not id_url:
 83            continue
 84        authors = [
 85            (a.find(f"{{{ATOM_NS}}}name").text or "").strip()
 86            for a in entry.findall(f"{{{ATOM_NS}}}author")
 87            if a.find(f"{{{ATOM_NS}}}name") is not None
 88        ]
 89        categories = [
 90            c.get("term", "") for c in entry.findall(f"{{{ATOM_NS}}}category") if c.get("term")
 91        ]
 92        entries.append(
 93            {
 94                "arxiv_id": arxiv_id_from_url(id_url),
 95                "abs_url": id_url,
 96                "title": re.sub(r"\s+", " ", text("title")),
 97                "summary": re.sub(r"\s+", " ", text("summary")),
 98                "authors": authors,
 99                "categories": categories,
100                "published": text("published"),
101            }
102        )
103    return entries

Atom feed text -> list of paper dicts (id/title/summary/authors/categories/published). Raises on malformed XML — a caller decides whether that's fatal or skip-and-continue.

def fetch_query(search_query: str, max_results: int, timeout: float = 20.0) -> str:
106def fetch_query(search_query: str, max_results: int, timeout: float = 20.0) -> str:
107    params = urllib.parse.urlencode(
108        {
109            "search_query": search_query,
110            "start": 0,
111            "max_results": max_results,
112            "sortBy": "submittedDate",
113            "sortOrder": "descending",
114        }
115    )
116    req = urllib.request.Request(
117        f"{API_BASE}?{params}", headers={"User-Agent": "marola-arxiv-digest/1"}
118    )
119    with urllib.request.urlopen(req, timeout=timeout) as r:  # noqa: S310 (fixed http(s) API host)
120        return r.read().decode("utf-8")
def relevance_score(paper: dict, matched: list[tuple[str, int]]) -> int:
123def relevance_score(paper: dict, matched: list[tuple[str, int]]) -> int:
124    """Sum of the weights of every query that matched this paper — a paper hit by both an
125    LLM-forecast query and a jellyfish query scores higher than one hit by a single broad query."""
126    return sum(weight for _label, weight in matched)

Sum of the weights of every query that matched this paper — a paper hit by both an LLM-forecast query and a jellyfish query scores higher than one hit by a single broad query.

def load_cached_ids(store: pathlib.Path) -> set[str]:
129def load_cached_ids(store: Path) -> set[str]:
130    return {p.stem for p in (store / "papers").glob("*.json")}
def write_index(store: pathlib.Path) -> None:
133def write_index(store: Path) -> None:
134    rows = []
135    for f in sorted((store / "papers").glob("*.json")):
136        data = json.loads(f.read_text())
137        rows.append(
138            {
139                "arxiv_id": data["arxiv_id"],
140                "title": data["title"],
141                "published": data["published"],
142                "relevance_score": data["relevance_score"],
143                "fetched_at": data["fetched_at"],
144            }
145        )
146    rows.sort(key=lambda r: r["relevance_score"], reverse=True)
147    with (store / "index.jsonl").open("w") as f:
148        for row in rows:
149            f.write(json.dumps(row, ensure_ascii=False) + "\n")
def run(repo_root: pathlib.Path, max_results: int) -> dict:
152def run(repo_root: Path, max_results: int) -> dict:
153    store = cache_dir(repo_root)
154    already = load_cached_ids(store)
155    matches_by_id: dict[str, list[tuple[str, int]]] = {}
156    papers_by_id: dict[str, dict] = {}
157    errors = []
158
159    for label, query, weight in QUERIES:
160        try:
161            xml_text = fetch_query(query, max_results)
162        except (urllib.error.URLError, TimeoutError) as e:
163            errors.append(f"{label}: {e}")
164            continue
165        for paper in parse_atom(xml_text):
166            pid = paper["arxiv_id"]
167            papers_by_id.setdefault(pid, paper)
168            matches_by_id.setdefault(pid, []).append((label, weight))
169
170    new_count = 0
171    for pid, paper in papers_by_id.items():
172        if pid in already:
173            continue
174        matched = matches_by_id[pid]
175        record = {
176            **paper,
177            "matched_queries": [label for label, _w in matched],
178            "relevance_score": relevance_score(paper, matched),
179            "fetched_at": dt.datetime.now(dt.UTC).isoformat(),
180        }
181        (store / "papers" / f"{pid}.json").write_text(
182            json.dumps(record, indent=2, ensure_ascii=False)
183        )
184        new_count += 1
185
186    write_index(store)
187    return {
188        "queries_run": len(QUERIES) - len(errors),
189        "queries_failed": errors,
190        "new_papers": new_count,
191        "total_cached": len(load_cached_ids(store)),
192        "index_path": str(store / "index.jsonl"),
193    }
def self_test() -> None:
196def self_test() -> None:
197    import tempfile
198
199    fails = 0
200
201    fixture = f"""<?xml version='1.0' encoding='UTF-8'?>
202<feed xmlns:opensearch="http://a9.com/-/spec/opensearch/1.1/" xmlns:arxiv="{ARXIV_NS}" xmlns="{ATOM_NS}">
203  <entry>
204    <id>http://arxiv.org/abs/2609.03658v1</id>
205    <title>  A Deep Learning Model for
206      Forecasting  Sea Surface Temperature</title>
207    <updated>2026-09-03T10:56:02Z</updated>
208    <summary>We forecast sea surface temperature anomalies with a transformer.</summary>
209    <category term="physics.ao-ph" scheme="http://arxiv.org/schemas/atom"/>
210    <published>2026-09-03T10:56:02Z</published>
211    <author><name>Ada Lovelace</name></author>
212    <author><name>Grace Hopper</name></author>
213  </entry>
214  <entry>
215    <id>http://arxiv.org/abs/2601.00001v2</id>
216    <title>Notify About Jellyfish Blooms via Sonar</title>
217    <updated>2026-01-01T00:00:00Z</updated>
218    <summary>A sonar pipeline to detect jellyfish blooms near beaches.</summary>
219    <category term="cs.CV" scheme="http://arxiv.org/schemas/atom"/>
220    <published>2026-01-01T00:00:00Z</published>
221    <author><name>Marie Curie</name></author>
222  </entry>
223</feed>"""
224
225    entries = parse_atom(fixture)
226    if len(entries) == 2:
227        print("  ok   parse_atom extracts both fixture entries")
228    else:
229        print(f"  FAIL parse_atom got {len(entries)} entries, expected 2")
230        fails += 1
231
232    e0 = entries[0]
233    if e0["arxiv_id"] == "2609.03658":
234        print("  ok   version suffix stripped from the arXiv id")
235    else:
236        print(f"  FAIL arxiv_id was {e0['arxiv_id']!r}, expected '2609.03658'")
237        fails += 1
238    if e0["title"] == "A Deep Learning Model for Forecasting Sea Surface Temperature":
239        print("  ok   multi-line title is whitespace-normalized")
240    else:
241        print(f"  FAIL title was {e0['title']!r}")
242        fails += 1
243    if e0["authors"] == ["Ada Lovelace", "Grace Hopper"]:
244        print("  ok   both authors extracted in order")
245    else:
246        print(f"  FAIL authors were {e0['authors']!r}")
247        fails += 1
248    if e0["categories"] == ["physics.ao-ph"]:
249        print("  ok   category term extracted")
250    else:
251        print(f"  FAIL categories were {e0['categories']!r}")
252        fails += 1
253
254    e1 = entries[1]
255    if e1["arxiv_id"] == "2601.00001":
256        print("  ok   a 'v2' suffix is also stripped correctly")
257    else:
258        print(f"  FAIL arxiv_id was {e1['arxiv_id']!r}, expected '2601.00001'")
259        fails += 1
260
261    # relevance_score: multiple query matches outscore a single broad match.
262    single = relevance_score(e0, [("ocean-physics", 1)])
263    double = relevance_score(e0, [("ocean-physics", 1), ("sst-forecast", 2)])
264    if double > single and double == 3:
265        print("  ok   relevance_score sums matched-query weights")
266    else:
267        print(f"  FAIL relevance_score: single={single} double={double}")
268        fails += 1
269
270    # Cache round-trip: write both fixture papers, confirm dedup on a second write and a
271    # correctly sorted, rewritten index — entirely inside a temp dir, no network.
272    with tempfile.TemporaryDirectory() as tmp:
273        repo_root = Path(tmp)
274        store = cache_dir(repo_root)
275        if load_cached_ids(store) == set():
276            print("  ok   a fresh cache dir starts empty")
277        else:
278            print("  FAIL fresh cache dir was not empty")
279            fails += 1
280
281        for paper, matched in ((e0, [("sst-forecast", 2)]), (e1, [("jellyfish-prediction", 2)])):
282            record = {
283                **paper,
284                "matched_queries": [label for label, _w in matched],
285                "relevance_score": relevance_score(paper, matched),
286                "fetched_at": "2026-09-06T00:00:00+00:00",
287            }
288            (store / "papers" / f"{paper['arxiv_id']}.json").write_text(json.dumps(record))
289        write_index(store)
290
291        cached = load_cached_ids(store)
292        if cached == {"2609.03658", "2601.00001"}:
293            print("  ok   both papers land in the cache, keyed by arxiv id")
294        else:
295            print(f"  FAIL cached ids were {cached!r}")
296            fails += 1
297
298        index_lines = (store / "index.jsonl").read_text().splitlines()
299        if len(index_lines) == 2:
300            print("  ok   index.jsonl has one line per cached paper")
301        else:
302            print(f"  FAIL index.jsonl had {len(index_lines)} lines, expected 2")
303            fails += 1
304
305        # Simulate re-fetching the same paper: caller-side dedup (run()'s own logic) must skip
306        # ids already in load_cached_ids() rather than re-write/duplicate.
307        already = load_cached_ids(store)
308        would_skip = "2609.03658" in already
309        if would_skip:
310            print("  ok   an already-cached id is recognized for skip-on-refetch")
311        else:
312            print("  FAIL already-cached id was not recognized")
313            fails += 1
314
315    # Malformed XML raises rather than silently returning nothing — a caller must not mistake a
316    # parse failure for "no results this query".
317    try:
318        parse_atom("<not-xml")
319        print("  FAIL malformed XML did not raise")
320        fails += 1
321    except ElementTree.ParseError:
322        print("  ok   malformed XML raises ParseError, not swallowed silently")
323
324    if fails == 0:
325        print("arxiv_digest self-test: ok")
326    else:
327        print(f"arxiv_digest self-test: {fails} failure(s)", file=sys.stderr)
328        sys.exit(1)
def main() -> None:
331def main() -> None:
332    ap = argparse.ArgumentParser(
333        description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
334    )
335    ap.add_argument("--max-results", type=int, default=5, help="results per query (default 5)")
336    ap.add_argument("--json", action="store_true", help="machine-readable summary on stdout")
337    ap.add_argument("--self-test", action="store_true")
338    args = ap.parse_args()
339
340    if args.self_test:
341        self_test()
342        return
343
344    repo_root = Path(__file__).resolve().parent.parent
345    summary = run(repo_root, args.max_results)
346    if args.json:
347        print(json.dumps(summary, indent=2))
348    else:
349        print(f"queries run: {summary['queries_run']}/{len(QUERIES)}")
350        if summary["queries_failed"]:
351            print(f"queries failed: {summary['queries_failed']}")
352        print(f"new papers cached: {summary['new_papers']}")
353        print(f"total cached: {summary['total_cached']}")
354        print(f"index: {summary['index_path']}")