awesome_agentic_digest
awesome_agentic_digest — cache GitHub repo candidates for docs/4-Research-and-plans/AWESOME-AGENTIC-ENGINEERING.md.
scripts/awesome_agentic_digest.py # fetch every query, cache new candidates, print a summary
scripts/awesome_agentic_digest.py --max-results 5 # results per query (default 5)
scripts/awesome_agentic_digest.py --json # machine-readable summary on stdout
scripts/awesome_agentic_digest.py --self-test # parser + cache self-check, no network (just quality-other)
Queries the real GitHub Search API (api.github.com/search/repositories) across agentic-engineering GitHub topics — verified live against real, non-zero results on 2026-09-07 (see MIP-0043 §4.2) — for candidate repos to hand-curate into docs/4-Research-and-plans/AWESOME-AGENTIC-ENGINEERING.md.
This script never writes to docs/4-Research-and-plans/AWESOME-AGENTIC-ENGINEERING.md. It only caches candidates and
prints a summary. A human reviews the cache (or the printed summary) and hand-writes any genuinely
good match into the curated doc, in that doc's own - [Title](URL) - Description. entry format.
This mirrors MIP-0041's book_digest.py: propose candidates, a human decides, nothing gets written
into the curated artifact unattended.
Cache layout, under .tmp/awesome_agentic_cache/ (gitignored — fetched data, not source):
repos/
Network is stdlib-only (urllib.request), matching this repo's other scripts (arxiv_digest.py,
cost-split.py). GitHub's search API needs no auth token for reasonable unauthenticated use
(confirmed live) but its rate limit is tight (empirically low tens of requests/minute for search
specifically) — this script runs a small, fixed query set once per invocation, no retry-hammering.
1#!/usr/bin/env python3 2"""awesome_agentic_digest — cache GitHub repo candidates for docs/4-Research-and-plans/AWESOME-AGENTIC-ENGINEERING.md. 3 4 scripts/awesome_agentic_digest.py # fetch every query, cache new candidates, print a summary 5 scripts/awesome_agentic_digest.py --max-results 5 # results per query (default 5) 6 scripts/awesome_agentic_digest.py --json # machine-readable summary on stdout 7 scripts/awesome_agentic_digest.py --self-test # parser + cache self-check, no network (just quality-other) 8 9Queries the real GitHub Search API (api.github.com/search/repositories) across agentic-engineering 10GitHub topics — verified live against real, non-zero results on 2026-09-07 (see MIP-0043 §4.2) — 11for candidate repos to hand-curate into docs/4-Research-and-plans/AWESOME-AGENTIC-ENGINEERING.md. 12 13**This script never writes to docs/4-Research-and-plans/AWESOME-AGENTIC-ENGINEERING.md.** It only caches candidates and 14prints a summary. A human reviews the cache (or the printed summary) and hand-writes any genuinely 15good match into the curated doc, in that doc's own `- [Title](URL) - Description.` entry format. 16This mirrors MIP-0041's book_digest.py: propose candidates, a human decides, nothing gets written 17into the curated artifact unattended. 18 19Cache layout, under .tmp/awesome_agentic_cache/ (gitignored — fetched data, not source): 20 repos/<owner>__<repo>.json one file per repo (full_name, html_url, description, stars, 21 pushed_at, topics, matched_queries, relevance_score, fetched_at) 22 index.jsonl one line per cached repo (full_name, html_url, stars, description, 23 fetched_at) — rewritten each run from the current repos/ contents, 24 so it never drifts from them 25 26Network is stdlib-only (`urllib.request`), matching this repo's other scripts (arxiv_digest.py, 27cost-split.py). GitHub's search API needs no auth token for reasonable unauthenticated use 28(confirmed live) but its rate limit is tight (empirically low tens of requests/minute for search 29specifically) — this script runs a small, fixed query set once per invocation, no retry-hammering. 30""" 31 32import argparse 33import datetime as dt 34import json 35import sys 36import urllib.error 37import urllib.parse 38import urllib.request 39from pathlib import Path 40 41API_BASE = "https://api.github.com/search/repositories" 42 43# Each query is (label, github search_query, weight). Verified live against the real GitHub Search 44# API on 2026-09-07 — every topic below returned real, non-zero, on-topic results (see MIP-0043 45# §4.2's total_count table). Two sort orders are queried per label (stars, then recently-updated), 46# matching the ask's "sorted by stars or recently-updated" — both feed the same cache/dedup. 47QUERIES: list[tuple[str, str, int]] = [ 48 ("agents", "topic:agents", 1), 49 ("llm-agents", "topic:llm-agents", 2), 50 ("ai-agents", "topic:ai-agents", 1), 51 ("mcp", "topic:mcp", 2), 52 ("multi-agent-systems", "topic:multi-agent-systems", 2), 53] 54SORTS: list[str] = ["stars", "updated"] 55 56 57def cache_dir(repo_root: Path) -> Path: 58 d = repo_root / ".tmp" / "awesome_agentic_cache" 59 (d / "repos").mkdir(parents=True, exist_ok=True) 60 return d 61 62 63def repo_key(full_name: str) -> str: 64 """'owner/repo' -> 'owner__repo', safe as a filename.""" 65 return full_name.replace("/", "__") 66 67 68def parse_search_response(json_text: str) -> list[dict]: 69 """GitHub Search API JSON text -> list of repo dicts (full_name/html_url/description/stars/ 70 pushed_at/topics). Raises on malformed JSON — a caller decides whether that's fatal or 71 skip-and-continue.""" 72 data = json.loads(json_text) 73 items = data.get("items") 74 if items is None: 75 raise ValueError("GitHub Search API response missing 'items'") 76 repos = [] 77 for item in items: 78 full_name = item.get("full_name", "") 79 if not full_name: 80 continue 81 repos.append( 82 { 83 "full_name": full_name, 84 "html_url": item.get("html_url", f"https://github.com/{full_name}"), 85 "description": (item.get("description") or "").strip(), 86 "stars": item.get("stargazers_count", 0), 87 "pushed_at": item.get("pushed_at", ""), 88 "topics": item.get("topics", []) or [], 89 } 90 ) 91 return repos 92 93 94def fetch_query(search_query: str, sort: str, max_results: int, timeout: float = 20.0) -> str: 95 params = urllib.parse.urlencode( 96 { 97 "q": search_query, 98 "sort": sort, 99 "order": "desc", 100 "per_page": max_results, 101 } 102 ) 103 req = urllib.request.Request( 104 f"{API_BASE}?{params}", 105 headers={ 106 "User-Agent": "marola-awesome-agentic-digest/1", 107 "Accept": "application/vnd.github+json", 108 }, 109 ) 110 with urllib.request.urlopen(req, timeout=timeout) as r: # noqa: S310 (fixed http(s) API host) 111 return r.read().decode("utf-8") 112 113 114def relevance_score(matched: list[tuple[str, int]]) -> int: 115 """Sum of the weights of every query that matched this repo — a repo hit by both a broad 116 'agents' query and a narrower 'mcp' query scores higher than one hit by a single broad query.""" 117 return sum(weight for _label, weight in matched) 118 119 120def load_cached_ids(store: Path) -> set[str]: 121 return {p.stem for p in (store / "repos").glob("*.json")} 122 123 124def write_index(store: Path) -> None: 125 rows = [] 126 for f in sorted((store / "repos").glob("*.json")): 127 data = json.loads(f.read_text()) 128 rows.append( 129 { 130 "full_name": data["full_name"], 131 "html_url": data["html_url"], 132 "stars": data["stars"], 133 "description": data["description"], 134 "relevance_score": data["relevance_score"], 135 "fetched_at": data["fetched_at"], 136 } 137 ) 138 rows.sort(key=lambda r: (r["relevance_score"], r["stars"]), reverse=True) 139 with (store / "index.jsonl").open("w") as f: 140 for row in rows: 141 f.write(json.dumps(row, ensure_ascii=False) + "\n") 142 143 144def run(repo_root: Path, max_results: int) -> dict: 145 store = cache_dir(repo_root) 146 already = load_cached_ids(store) 147 matches_by_id: dict[str, list[tuple[str, int]]] = {} 148 repos_by_id: dict[str, dict] = {} 149 errors = [] 150 151 for label, query, weight in QUERIES: 152 for sort in SORTS: 153 try: 154 json_text = fetch_query(query, sort, max_results) 155 except (urllib.error.URLError, TimeoutError, ValueError) as e: 156 errors.append(f"{label} ({sort}): {e}") 157 continue 158 for repo in parse_search_response(json_text): 159 rid = repo_key(repo["full_name"]) 160 repos_by_id.setdefault(rid, repo) 161 matches_by_id.setdefault(rid, []).append((label, weight)) 162 163 new_count = 0 164 for rid, repo in repos_by_id.items(): 165 if rid in already: 166 continue 167 matched = matches_by_id[rid] 168 record = { 169 **repo, 170 "matched_queries": [label for label, _w in matched], 171 "relevance_score": relevance_score(matched), 172 "fetched_at": dt.datetime.now(dt.UTC).isoformat(), 173 } 174 (store / "repos" / f"{rid}.json").write_text( 175 json.dumps(record, indent=2, ensure_ascii=False) 176 ) 177 new_count += 1 178 179 write_index(store) 180 return { 181 "queries_run": len(QUERIES) * len(SORTS) - len(errors), 182 "queries_failed": errors, 183 "new_candidates": new_count, 184 "total_cached": len(load_cached_ids(store)), 185 "index_path": str(store / "index.jsonl"), 186 } 187 188 189def self_test() -> None: 190 import tempfile 191 192 fails = 0 193 194 fixture = json.dumps( 195 { 196 "total_count": 2, 197 "incomplete_results": False, 198 "items": [ 199 { 200 "full_name": "example-org/agent-critic", 201 "html_url": "https://github.com/example-org/agent-critic", 202 "description": "A reviewer/critic pattern for LLM agent pipelines.", 203 "stargazers_count": 4200, 204 "pushed_at": "2026-09-01T00:00:00Z", 205 "topics": ["llm-agents", "mcp"], 206 }, 207 { 208 "full_name": "another-org/dspy-scala", 209 "html_url": "https://github.com/another-org/dspy-scala", 210 "description": "DSPy-style prompt compilation, replayed from Scala.", 211 "stargazers_count": 130, 212 "pushed_at": "2026-08-15T00:00:00Z", 213 "topics": ["ai-agents"], 214 }, 215 ], 216 } 217 ) 218 219 repos = parse_search_response(fixture) 220 if len(repos) == 2: 221 print(" ok parse_search_response extracts both fixture items") 222 else: 223 print(f" FAIL parse_search_response got {len(repos)} items, expected 2") 224 fails += 1 225 226 r0 = repos[0] 227 if r0["full_name"] == "example-org/agent-critic": 228 print(" ok full_name extracted") 229 else: 230 print(f" FAIL full_name was {r0['full_name']!r}") 231 fails += 1 232 if r0["stars"] == 4200: 233 print(" ok stargazers_count mapped to 'stars'") 234 else: 235 print(f" FAIL stars was {r0['stars']!r}, expected 4200") 236 fails += 1 237 if repo_key(r0["full_name"]) == "example-org__agent-critic": 238 print(" ok repo_key produces a filesystem-safe id") 239 else: 240 print(f" FAIL repo_key was {repo_key(r0['full_name'])!r}") 241 fails += 1 242 243 # relevance_score: multiple query matches outscore a single broad match. 244 single = relevance_score([("agents", 1)]) 245 double = relevance_score([("agents", 1), ("mcp", 2)]) 246 if double > single and double == 3: 247 print(" ok relevance_score sums matched-query weights") 248 else: 249 print(f" FAIL relevance_score: single={single} double={double}") 250 fails += 1 251 252 # Cache round-trip: write both fixture repos, confirm dedup on a second write and a correctly 253 # sorted, rewritten index — entirely inside a temp dir, no network. 254 with tempfile.TemporaryDirectory() as tmp: 255 repo_root = Path(tmp) 256 store = cache_dir(repo_root) 257 if load_cached_ids(store) == set(): 258 print(" ok a fresh cache dir starts empty") 259 else: 260 print(" FAIL fresh cache dir was not empty") 261 fails += 1 262 263 for repo, matched in ((repos[0], [("mcp", 2)]), (repos[1], [("ai-agents", 1)])): 264 record = { 265 **repo, 266 "matched_queries": [label for label, _w in matched], 267 "relevance_score": relevance_score(matched), 268 "fetched_at": "2026-09-07T00:00:00+00:00", 269 } 270 (store / "repos" / f"{repo_key(repo['full_name'])}.json").write_text(json.dumps(record)) 271 write_index(store) 272 273 cached = load_cached_ids(store) 274 if cached == {"example-org__agent-critic", "another-org__dspy-scala"}: 275 print(" ok both repos land in the cache, keyed by owner__repo") 276 else: 277 print(f" FAIL cached ids were {cached!r}") 278 fails += 1 279 280 index_lines = (store / "index.jsonl").read_text().splitlines() 281 if len(index_lines) == 2: 282 print(" ok index.jsonl has one line per cached repo") 283 else: 284 print(f" FAIL index.jsonl had {len(index_lines)} lines, expected 2") 285 fails += 1 286 287 # Simulate re-fetching the same repo: caller-side dedup (run()'s own logic) must skip ids 288 # already in load_cached_ids() rather than re-write/duplicate. 289 already = load_cached_ids(store) 290 would_skip = "example-org__agent-critic" in already 291 if would_skip: 292 print(" ok an already-cached id is recognized for skip-on-refetch") 293 else: 294 print(" FAIL already-cached id was not recognized") 295 fails += 1 296 297 # Malformed JSON raises rather than silently returning nothing — a caller must not mistake a 298 # parse failure for "no results this query". 299 try: 300 parse_search_response("{not json") 301 print(" FAIL malformed JSON did not raise") 302 fails += 1 303 except json.JSONDecodeError: 304 print(" ok malformed JSON raises JSONDecodeError, not swallowed silently") 305 306 # A response missing 'items' entirely (e.g. a rate-limit error body) raises rather than being 307 # mistaken for zero results. 308 try: 309 parse_search_response(json.dumps({"message": "API rate limit exceeded"})) 310 print(" FAIL a response missing 'items' did not raise") 311 fails += 1 312 except ValueError: 313 print(" ok a response missing 'items' raises ValueError, not swallowed silently") 314 315 if fails == 0: 316 print("awesome_agentic_digest self-test: ok") 317 else: 318 print(f"awesome_agentic_digest self-test: {fails} failure(s)", file=sys.stderr) 319 sys.exit(1) 320 321 322def main() -> None: 323 ap = argparse.ArgumentParser( 324 description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter 325 ) 326 ap.add_argument("--max-results", type=int, default=5, help="results per query (default 5)") 327 ap.add_argument("--json", action="store_true", help="machine-readable summary on stdout") 328 ap.add_argument("--self-test", action="store_true") 329 args = ap.parse_args() 330 331 if args.self_test: 332 self_test() 333 return 334 335 repo_root = Path(__file__).resolve().parent.parent 336 summary = run(repo_root, args.max_results) 337 if args.json: 338 print(json.dumps(summary, indent=2)) 339 else: 340 print(f"queries run: {summary['queries_run']}/{len(QUERIES) * len(SORTS)}") 341 if summary["queries_failed"]: 342 print(f"queries failed: {summary['queries_failed']}") 343 print(f"new candidates cached: {summary['new_candidates']}") 344 print(f"total cached: {summary['total_cached']}") 345 print(f"index: {summary['index_path']}") 346 print( 347 "review the candidates above, then hand-curate any of them into " 348 "docs/4-Research-and-plans/AWESOME-AGENTIC-ENGINEERING.md" 349 ) 350 print( 351 "nothing was written to docs/4-Research-and-plans/AWESOME-AGENTIC-ENGINEERING.md — " 352 "this script only caches candidates" 353 ) 354 355 356if __name__ == "__main__": 357 main()
64def repo_key(full_name: str) -> str: 65 """'owner/repo' -> 'owner__repo', safe as a filename.""" 66 return full_name.replace("/", "__")
'owner/repo' -> 'owner__repo', safe as a filename.
69def parse_search_response(json_text: str) -> list[dict]: 70 """GitHub Search API JSON text -> list of repo dicts (full_name/html_url/description/stars/ 71 pushed_at/topics). Raises on malformed JSON — a caller decides whether that's fatal or 72 skip-and-continue.""" 73 data = json.loads(json_text) 74 items = data.get("items") 75 if items is None: 76 raise ValueError("GitHub Search API response missing 'items'") 77 repos = [] 78 for item in items: 79 full_name = item.get("full_name", "") 80 if not full_name: 81 continue 82 repos.append( 83 { 84 "full_name": full_name, 85 "html_url": item.get("html_url", f"https://github.com/{full_name}"), 86 "description": (item.get("description") or "").strip(), 87 "stars": item.get("stargazers_count", 0), 88 "pushed_at": item.get("pushed_at", ""), 89 "topics": item.get("topics", []) or [], 90 } 91 ) 92 return repos
GitHub Search API JSON text -> list of repo dicts (full_name/html_url/description/stars/ pushed_at/topics). Raises on malformed JSON — a caller decides whether that's fatal or skip-and-continue.
95def fetch_query(search_query: str, sort: str, max_results: int, timeout: float = 20.0) -> str: 96 params = urllib.parse.urlencode( 97 { 98 "q": search_query, 99 "sort": sort, 100 "order": "desc", 101 "per_page": max_results, 102 } 103 ) 104 req = urllib.request.Request( 105 f"{API_BASE}?{params}", 106 headers={ 107 "User-Agent": "marola-awesome-agentic-digest/1", 108 "Accept": "application/vnd.github+json", 109 }, 110 ) 111 with urllib.request.urlopen(req, timeout=timeout) as r: # noqa: S310 (fixed http(s) API host) 112 return r.read().decode("utf-8")
115def relevance_score(matched: list[tuple[str, int]]) -> int: 116 """Sum of the weights of every query that matched this repo — a repo hit by both a broad 117 'agents' query and a narrower 'mcp' query scores higher than one hit by a single broad query.""" 118 return sum(weight for _label, weight in matched)
Sum of the weights of every query that matched this repo — a repo hit by both a broad 'agents' query and a narrower 'mcp' query scores higher than one hit by a single broad query.
125def write_index(store: Path) -> None: 126 rows = [] 127 for f in sorted((store / "repos").glob("*.json")): 128 data = json.loads(f.read_text()) 129 rows.append( 130 { 131 "full_name": data["full_name"], 132 "html_url": data["html_url"], 133 "stars": data["stars"], 134 "description": data["description"], 135 "relevance_score": data["relevance_score"], 136 "fetched_at": data["fetched_at"], 137 } 138 ) 139 rows.sort(key=lambda r: (r["relevance_score"], r["stars"]), reverse=True) 140 with (store / "index.jsonl").open("w") as f: 141 for row in rows: 142 f.write(json.dumps(row, ensure_ascii=False) + "\n")
145def run(repo_root: Path, max_results: int) -> dict: 146 store = cache_dir(repo_root) 147 already = load_cached_ids(store) 148 matches_by_id: dict[str, list[tuple[str, int]]] = {} 149 repos_by_id: dict[str, dict] = {} 150 errors = [] 151 152 for label, query, weight in QUERIES: 153 for sort in SORTS: 154 try: 155 json_text = fetch_query(query, sort, max_results) 156 except (urllib.error.URLError, TimeoutError, ValueError) as e: 157 errors.append(f"{label} ({sort}): {e}") 158 continue 159 for repo in parse_search_response(json_text): 160 rid = repo_key(repo["full_name"]) 161 repos_by_id.setdefault(rid, repo) 162 matches_by_id.setdefault(rid, []).append((label, weight)) 163 164 new_count = 0 165 for rid, repo in repos_by_id.items(): 166 if rid in already: 167 continue 168 matched = matches_by_id[rid] 169 record = { 170 **repo, 171 "matched_queries": [label for label, _w in matched], 172 "relevance_score": relevance_score(matched), 173 "fetched_at": dt.datetime.now(dt.UTC).isoformat(), 174 } 175 (store / "repos" / f"{rid}.json").write_text( 176 json.dumps(record, indent=2, ensure_ascii=False) 177 ) 178 new_count += 1 179 180 write_index(store) 181 return { 182 "queries_run": len(QUERIES) * len(SORTS) - len(errors), 183 "queries_failed": errors, 184 "new_candidates": new_count, 185 "total_cached": len(load_cached_ids(store)), 186 "index_path": str(store / "index.jsonl"), 187 }
190def self_test() -> None: 191 import tempfile 192 193 fails = 0 194 195 fixture = json.dumps( 196 { 197 "total_count": 2, 198 "incomplete_results": False, 199 "items": [ 200 { 201 "full_name": "example-org/agent-critic", 202 "html_url": "https://github.com/example-org/agent-critic", 203 "description": "A reviewer/critic pattern for LLM agent pipelines.", 204 "stargazers_count": 4200, 205 "pushed_at": "2026-09-01T00:00:00Z", 206 "topics": ["llm-agents", "mcp"], 207 }, 208 { 209 "full_name": "another-org/dspy-scala", 210 "html_url": "https://github.com/another-org/dspy-scala", 211 "description": "DSPy-style prompt compilation, replayed from Scala.", 212 "stargazers_count": 130, 213 "pushed_at": "2026-08-15T00:00:00Z", 214 "topics": ["ai-agents"], 215 }, 216 ], 217 } 218 ) 219 220 repos = parse_search_response(fixture) 221 if len(repos) == 2: 222 print(" ok parse_search_response extracts both fixture items") 223 else: 224 print(f" FAIL parse_search_response got {len(repos)} items, expected 2") 225 fails += 1 226 227 r0 = repos[0] 228 if r0["full_name"] == "example-org/agent-critic": 229 print(" ok full_name extracted") 230 else: 231 print(f" FAIL full_name was {r0['full_name']!r}") 232 fails += 1 233 if r0["stars"] == 4200: 234 print(" ok stargazers_count mapped to 'stars'") 235 else: 236 print(f" FAIL stars was {r0['stars']!r}, expected 4200") 237 fails += 1 238 if repo_key(r0["full_name"]) == "example-org__agent-critic": 239 print(" ok repo_key produces a filesystem-safe id") 240 else: 241 print(f" FAIL repo_key was {repo_key(r0['full_name'])!r}") 242 fails += 1 243 244 # relevance_score: multiple query matches outscore a single broad match. 245 single = relevance_score([("agents", 1)]) 246 double = relevance_score([("agents", 1), ("mcp", 2)]) 247 if double > single and double == 3: 248 print(" ok relevance_score sums matched-query weights") 249 else: 250 print(f" FAIL relevance_score: single={single} double={double}") 251 fails += 1 252 253 # Cache round-trip: write both fixture repos, confirm dedup on a second write and a correctly 254 # sorted, rewritten index — entirely inside a temp dir, no network. 255 with tempfile.TemporaryDirectory() as tmp: 256 repo_root = Path(tmp) 257 store = cache_dir(repo_root) 258 if load_cached_ids(store) == set(): 259 print(" ok a fresh cache dir starts empty") 260 else: 261 print(" FAIL fresh cache dir was not empty") 262 fails += 1 263 264 for repo, matched in ((repos[0], [("mcp", 2)]), (repos[1], [("ai-agents", 1)])): 265 record = { 266 **repo, 267 "matched_queries": [label for label, _w in matched], 268 "relevance_score": relevance_score(matched), 269 "fetched_at": "2026-09-07T00:00:00+00:00", 270 } 271 (store / "repos" / f"{repo_key(repo['full_name'])}.json").write_text(json.dumps(record)) 272 write_index(store) 273 274 cached = load_cached_ids(store) 275 if cached == {"example-org__agent-critic", "another-org__dspy-scala"}: 276 print(" ok both repos land in the cache, keyed by owner__repo") 277 else: 278 print(f" FAIL cached ids were {cached!r}") 279 fails += 1 280 281 index_lines = (store / "index.jsonl").read_text().splitlines() 282 if len(index_lines) == 2: 283 print(" ok index.jsonl has one line per cached repo") 284 else: 285 print(f" FAIL index.jsonl had {len(index_lines)} lines, expected 2") 286 fails += 1 287 288 # Simulate re-fetching the same repo: caller-side dedup (run()'s own logic) must skip ids 289 # already in load_cached_ids() rather than re-write/duplicate. 290 already = load_cached_ids(store) 291 would_skip = "example-org__agent-critic" in already 292 if would_skip: 293 print(" ok an already-cached id is recognized for skip-on-refetch") 294 else: 295 print(" FAIL already-cached id was not recognized") 296 fails += 1 297 298 # Malformed JSON raises rather than silently returning nothing — a caller must not mistake a 299 # parse failure for "no results this query". 300 try: 301 parse_search_response("{not json") 302 print(" FAIL malformed JSON did not raise") 303 fails += 1 304 except json.JSONDecodeError: 305 print(" ok malformed JSON raises JSONDecodeError, not swallowed silently") 306 307 # A response missing 'items' entirely (e.g. a rate-limit error body) raises rather than being 308 # mistaken for zero results. 309 try: 310 parse_search_response(json.dumps({"message": "API rate limit exceeded"})) 311 print(" FAIL a response missing 'items' did not raise") 312 fails += 1 313 except ValueError: 314 print(" ok a response missing 'items' raises ValueError, not swallowed silently") 315 316 if fails == 0: 317 print("awesome_agentic_digest self-test: ok") 318 else: 319 print(f"awesome_agentic_digest self-test: {fails} failure(s)", file=sys.stderr) 320 sys.exit(1)
323def main() -> None: 324 ap = argparse.ArgumentParser( 325 description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter 326 ) 327 ap.add_argument("--max-results", type=int, default=5, help="results per query (default 5)") 328 ap.add_argument("--json", action="store_true", help="machine-readable summary on stdout") 329 ap.add_argument("--self-test", action="store_true") 330 args = ap.parse_args() 331 332 if args.self_test: 333 self_test() 334 return 335 336 repo_root = Path(__file__).resolve().parent.parent 337 summary = run(repo_root, args.max_results) 338 if args.json: 339 print(json.dumps(summary, indent=2)) 340 else: 341 print(f"queries run: {summary['queries_run']}/{len(QUERIES) * len(SORTS)}") 342 if summary["queries_failed"]: 343 print(f"queries failed: {summary['queries_failed']}") 344 print(f"new candidates cached: {summary['new_candidates']}") 345 print(f"total cached: {summary['total_cached']}") 346 print(f"index: {summary['index_path']}") 347 print( 348 "review the candidates above, then hand-curate any of them into " 349 "docs/4-Research-and-plans/AWESOME-AGENTIC-ENGINEERING.md" 350 ) 351 print( 352 "nothing was written to docs/4-Research-and-plans/AWESOME-AGENTIC-ENGINEERING.md — " 353 "this script only caches candidates" 354 )