build_dataset
Build marola's fine-tuning dataset from what the repo already has — no new labelling.
Sources (all local files, no network, no model calls):
- core/src/main/resources/recommendation_prompt.json — the DSPy-compiled summarizer demos: structured conditions in, one-to-two-sentence summary out. The core "shape" marola wants.
- core/src/main/resources/review_prompt.json — the reviewer demos: conditions + draft in, compact JSON verdict out. Teaches the JSON-only discipline small models break most.
- core/src/main/resources/sea_lore.json — "tell me something about X" → the sourced paragraph, with the source URL kept in the answer so the habit of citing survives.
- knowledge/*.md (recursively) — one Q/A per chunk and per sentence, each asked through several question templates (MIP-0025 §4.3 Layer 1). Every answer is verbatim text from the file it cites; no model call generates anything.
- Tool-call SFT (MIP-0025 §4.3 Layer 2): questions → one JSON call to the four MCP tools in cli/src/main/scala/marola/agent/SwimConditionsMcpServer.scala. Teaches when to call, not facts.
Output: finetune/data/train.jsonl and eval.jsonl in the chat format most trainers accept: {"messages": [{"role": "system", ...}, {"role": "user", ...}, {"role": "assistant", ...}]}
Run: python build_dataset.py (or just finetune-dataset)
Self-test: python build_dataset.py --self-test (or just quality-other)
1"""Build marola's fine-tuning dataset from what the repo already has — no new labelling. 2 3Sources (all local files, no network, no model calls): 4 1. core/src/main/resources/recommendation_prompt.json — the DSPy-compiled summarizer demos: 5 structured conditions in, one-to-two-sentence summary out. The core "shape" marola wants. 6 2. core/src/main/resources/review_prompt.json — the reviewer demos: conditions + draft in, 7 compact JSON verdict out. Teaches the JSON-only discipline small models break most. 8 3. core/src/main/resources/sea_lore.json — "tell me something about X" → the sourced paragraph, 9 with the source URL kept in the answer so the habit of citing survives. 10 4. knowledge/*.md (recursively) — one Q/A per chunk and per sentence, each asked through several 11 question templates (MIP-0025 §4.3 Layer 1). Every answer is verbatim text from the file it 12 cites; no model call generates anything. 13 5. Tool-call SFT (MIP-0025 §4.3 Layer 2): questions → one JSON call to the four MCP tools in 14 cli/src/main/scala/marola/agent/SwimConditionsMcpServer.scala. Teaches when to call, not facts. 15 16Output: finetune/data/train.jsonl and eval.jsonl in the chat format most trainers accept: 17 {"messages": [{"role": "system", ...}, {"role": "user", ...}, {"role": "assistant", ...}]} 18 19Run: python build_dataset.py (or `just finetune-dataset`) 20Self-test: python build_dataset.py --self-test (or `just quality-other`) 21""" 22 23from __future__ import annotations 24 25import json 26import random 27import re 28from pathlib import Path 29 30REPO = Path(__file__).resolve().parent.parent 31RESOURCES = REPO / "core" / "src" / "main" / "resources" 32KNOWLEDGE = REPO / "knowledge" 33OUT = Path(__file__).resolve().parent / "data" 34 35SYSTEM = ( 36 "You are marola, a swim-conditions assistant for open-water swimmers in Brazil. Answer in one " 37 "or two plain sentences from the facts given; never invent conditions; cite a source URL when " 38 "you quote your notes; return compact JSON and nothing else when asked for JSON." 39) 40 41INPUT_FIELDS = ( 42 "beach_name", 43 "hour_local", 44 "sea_temp_c", 45 "wind_kmh", 46 "wave_height_m", 47 "jellyfish_risk", 48 "whale_sighting_likelihood", 49 "score", 50) 51 52 53def render_inputs(demo: dict, fields: tuple[str, ...]) -> str: 54 return "\n".join(f"{f.replace('_', ' ').title()}: {demo[f]}" for f in fields if f in demo) 55 56 57def example(user: str, assistant: str) -> dict: 58 return { 59 "messages": [ 60 {"role": "system", "content": SYSTEM}, 61 {"role": "user", "content": user}, 62 {"role": "assistant", "content": assistant}, 63 ] 64 } 65 66 67def from_compiled_prompt( 68 path: Path, output_field: str, extra_inputs: tuple[str, ...] = () 69) -> list[dict]: 70 doc = json.loads(path.read_text(encoding="utf-8")) 71 instructions = doc.get("signature", {}).get("instructions", "").strip() 72 rows = [] 73 for demo in doc.get("demos", []): 74 if output_field not in demo: 75 continue 76 user = (instructions + "\n\n" if instructions else "") + render_inputs( 77 demo, INPUT_FIELDS + extra_inputs 78 ) 79 rows.append(example(user, demo[output_field])) 80 return rows 81 82 83def from_sea_lore(path: Path) -> list[dict]: 84 rows = [] 85 for e in json.loads(path.read_text(encoding="utf-8")): 86 topic = e["id"].replace("-", " ") 87 user = f"Tell me something about the sea: {topic}." 88 rows.append(example(user, f"{e['text']} [source: {e['source']}]")) 89 return rows 90 91 92def chunk_markdown(md: str, max_chars: int = 700) -> tuple[str, str, list[str]]: 93 """Mirror of marola.knowledge.Corpus.chunkDocument: title, source, merged paragraphs.""" 94 lines = md.splitlines() 95 title = next((ln[2:].strip() for ln in lines if ln.startswith("# ")), "") 96 source = next((ln[7:].strip() for ln in lines if ln.lower().startswith("source:")), "") 97 body = "\n".join( 98 ln for ln in lines if not (ln.startswith("# ") or ln.lower().startswith("source:")) 99 ) 100 paras = [p.strip() for p in re.split(r"\n\s*\n", body) if p.strip()] 101 chunks: list[str] = [] 102 for p in paras: 103 if chunks and len(chunks[-1]) + len(p) + 2 <= max_chars: 104 chunks[-1] = chunks[-1] + "\n\n" + p 105 else: 106 chunks.append(p) 107 return title, source, chunks 108 109 110# Chunk-level: the question is asked several ways, the answer is always the full real chunk. 111CHUNK_QUESTION_TEMPLATES = ( 112 "What do your notes say about {title}? (part {part})", 113 "Tell me what you know about {title}, part {part}.", 114 "Summarize what your notes cover on {title} (part {part}).", 115 "Give me the part {part} details from your notes on {title}.", 116 "What's in section {part} of your {title} notes?", 117 "Explain {title} using what your notes say (part {part}).", 118 "I'm curious about {title} — what does part {part} of your notes cover?", 119 "Recap part {part} of your knowledge on {title}.", 120 "What have you got written down about {title}, specifically part {part}?", 121 "Walk me through part {part} of your {title} notes.", 122 "What's the part {part} summary of {title} in your notes?", 123 "According to your notes, what's true about {title} (part {part})?", 124 "Pull up part {part} of what you know about {title}.", 125 "What can you tell a swimmer about {title}? (part {part})", 126 "Notes check: {title}, part {part}.", 127 "Brief me on {title}, part {part}, from your sources.", 128 "What's documented about {title} in part {part} of your notes?", 129 "Give a plain-language rundown of {title}, part {part}.", 130 "What should I know about {title}? (see part {part} of your notes)", 131 "Show me part {part} of your {title} material.", 132) 133 134# Sentence-level: same idea, one real sentence at a time — finer-grained facts, still verbatim. 135SENTENCE_QUESTION_TEMPLATES = ( 136 "Give me one specific fact from your notes on {title} (part {part}).", 137 "What's a detail from your {title} notes, part {part}?", 138 "Pick one fact from part {part} of your {title} notes.", 139 "What's something true about {title} from part {part} of your notes?", 140 "Quote one fact from your notes on {title} (part {part}).", 141 "Name a specific point from your {title} notes, part {part}.", 142 "What's one thing your notes say about {title}? (part {part})", 143 "Give a short, specific fact about {title} (part {part}).", 144 "What detail from part {part} of your {title} notes stands out?", 145 "Tell me a single fact from your {title} notes (part {part}).", 146 "What's one takeaway from part {part} of your {title} notes?", 147 "Share one specific fact about {title} (part {part}).", 148 "What's a concrete detail about {title}? (from part {part} of your notes)", 149 "Give me a fact, not a summary, about {title} (part {part}).", 150 "What does part {part} of your {title} notes say, in one fact?", 151 "Point to one fact in your {title} notes (part {part}).", 152 "What's a specific claim in part {part} of your {title} notes?", 153 "Give one sentence of fact about {title} (part {part}).", 154 "What's a single detail worth knowing about {title}? (part {part})", 155 "Extract one fact from part {part} of your {title} notes.", 156) 157 158 159def _split_sentences(text: str) -> list[str]: 160 """Real sentences only (>20 chars) so a stray abbreviation dot doesn't yield a fragment.""" 161 return [s.strip() for s in re.split(r"(?<=[.!?])\s+", text) if len(s.strip()) > 20] 162 163 164def _chunk_examples(title: str, source: str, chunk: str, part: int) -> list[dict]: 165 answer = f"{chunk}\n\nSource: {source}" 166 return [ 167 example(t.format(title=title.lower(), part=part), answer) for t in CHUNK_QUESTION_TEMPLATES 168 ] 169 170 171def _sentence_examples(title: str, source: str, chunk: str, part: int) -> list[dict]: 172 rows = [] 173 for sentence in _split_sentences(chunk): 174 answer = f"{sentence}\n\nSource: {source}" 175 rows += [ 176 example(t.format(title=title.lower(), part=part), answer) 177 for t in SENTENCE_QUESTION_TEMPLATES 178 ] 179 return rows 180 181 182def from_knowledge(dir_: Path) -> list[dict]: 183 rows = [] 184 for path in sorted(dir_.rglob("*.md")): 185 title, source, chunks = chunk_markdown(path.read_text(encoding="utf-8")) 186 if not source: 187 continue # README.md and anything else without a citable source 188 for i, chunk in enumerate(chunks): 189 part = i + 1 190 rows += _chunk_examples(title, source, chunk, part) 191 rows += _sentence_examples(title, source, chunk, part) 192 return rows 193 194 195def _load_knowledge_sources(dir_: Path) -> dict[str, str]: 196 """Map each knowledge doc's cited source URL to its raw file text, for provenance checks.""" 197 out: dict[str, str] = {} 198 for path in sorted(dir_.rglob("*.md")): 199 raw = path.read_text(encoding="utf-8") 200 _, source, _ = chunk_markdown(raw) 201 if source: 202 out[source] = raw 203 return out 204 205 206KNOWLEDGE_EXAMPLE_FLOOR = 2000 207 208 209# --- Layer 2: tool-call SFT (MIP-0025 §4.3) -------------------------------------------------- 210# Transcribed from SwimConditionsMcpServer.scala's tool schemas; change both together. 211TOOL_CALL_SYSTEM = ( 212 "You are marola, a swim-conditions assistant for open-water swimmers in Brazil. When a " 213 "question needs live data you don't have, reply with exactly one JSON object of the shape " 214 '{"tool": "<name>", "arguments": {...}} and nothing else — never invent the result yourself.' 215) 216 217TOOL_SCHEMAS: dict[str, dict[str, tuple[str, ...]]] = { 218 "find_nearby_beaches": {"required": ("lat", "lon"), "optional": ("radius_km",)}, 219 "get_swim_recommendation": {"required": ("lat", "lon"), "optional": ("radius_km",)}, 220 "get_water_quality": {"required": ("lat", "lon"), "optional": ("radius_km",)}, 221 "ask_ocean_question": {"required": ("question",), "optional": ()}, 222} 223 224# site/fixtures/board.json's beaches. 225LOCATIONS: tuple[tuple[float, float], ...] = ((-27.6296, -48.4487), (-27.4021, -48.4157)) 226RADII: tuple[float | None, ...] = (None, 10.0, 20.0) 227 228FIND_BEACHES_TEMPLATES = ( 229 "Find swim beaches near {lat}, {lon}.", 230 "What open-water beaches are close to {lat}, {lon}?", 231 "Search for beaches around latitude {lat}, longitude {lon}.", 232 "Are there any beaches near {lat}, {lon}?", 233 "List the beaches close to {lat}, {lon}.", 234 "I'm at {lat}, {lon} — any beaches nearby?", 235 "Beaches near {lat}, {lon}, please.", 236 "Which beaches are around {lat}, {lon}?", 237 "Show me swim spots close to {lat}, {lon}.", 238 "Look up beaches near coordinate {lat}, {lon}.", 239) 240 241RECOMMENDATION_TEMPLATES = ( 242 "What's the best hour to swim tomorrow near {lat}, {lon}?", 243 "Give me tomorrow's swim conditions near {lat}, {lon}.", 244 "When should I swim tomorrow around {lat}, {lon}?", 245 "Best swim window tomorrow near {lat}, {lon}?", 246 "What are tomorrow's conditions like near {lat}, {lon}?", 247 "Recommend a swim time near {lat}, {lon} for tomorrow.", 248 "I want to swim tomorrow near {lat}, {lon} — when's best?", 249 "Tell me tomorrow's swimability near {lat}, {lon}.", 250 "Forecast tomorrow's swim conditions for {lat}, {lon}.", 251 "What hour has the best score near {lat}, {lon} tomorrow?", 252) 253 254WATER_QUALITY_TEMPLATES = ( 255 "Is the water clean near {lat}, {lon}?", 256 "What's the bathing-water quality near {lat}, {lon}?", 257 "Check water quality around {lat}, {lon}.", 258 "Any water quality warnings near {lat}, {lon}?", 259 "Is it safe to swim near {lat}, {lon} — water quality-wise?", 260 "Give me the enterococci readings near {lat}, {lon}.", 261 "PRÓPRIA or IMPRÓPRIA near {lat}, {lon}?", 262 "What do the sampling points say near {lat}, {lon}?", 263 "Water quality check for {lat}, {lon}.", 264 "Has water near {lat}, {lon} been tested recently?", 265) 266 267ASK_QUESTION_TEMPLATES = ( 268 "{question}", 269 "Hey marola, {question}", 270 "Quick question: {question}", 271 "{question} Please look it up.", 272 "I want to know: {question}", 273 "Can you answer this: {question}", 274 "{question} (from a swimmer prepping a trip)", 275 "Ocean question: {question}", 276 "{question} What does your knowledge base say?", 277 "Before I go for a swim, {question}", 278) 279 280 281def _tool_call_example(user: str, tool: str, arguments: dict) -> dict: 282 assistant = json.dumps({"tool": tool, "arguments": arguments}) 283 return { 284 "messages": [ 285 {"role": "system", "content": TOOL_CALL_SYSTEM}, 286 {"role": "user", "content": user}, 287 {"role": "assistant", "content": assistant}, 288 ] 289 } 290 291 292def _latlon_tool_examples(tool: str, templates: tuple[str, ...]) -> list[dict]: 293 rows = [] 294 for lat, lon in LOCATIONS: 295 for radius in RADII: 296 arguments: dict[str, float] = {"lat": lat, "lon": lon} 297 if radius is not None: 298 arguments["radius_km"] = radius 299 for t in templates: 300 user = t.format(lat=lat, lon=lon) 301 if radius is not None: 302 user += f" Search within {radius:g} km." 303 rows.append(_tool_call_example(user, tool, arguments)) 304 return rows 305 306 307def _ask_tool_examples(knowledge_dir: Path, sea_lore_path: Path) -> list[dict]: 308 questions = [] 309 for path in sorted(knowledge_dir.rglob("*.md")): 310 title, source, _ = chunk_markdown(path.read_text(encoding="utf-8")) 311 if source: 312 questions.append(f"What do you know about {title.lower()}?") 313 if sea_lore_path.exists(): 314 for e in json.loads(sea_lore_path.read_text(encoding="utf-8")): 315 questions.append(f"Tell me about {e['id'].replace('-', ' ')}.") 316 rows = [] 317 for question in questions: 318 for t in ASK_QUESTION_TEMPLATES: 319 user = t.format(question=question) 320 rows.append(_tool_call_example(user, "ask_ocean_question", {"question": question})) 321 return rows 322 323 324def from_tool_calls(knowledge_dir: Path, sea_lore_path: Path) -> list[dict]: 325 rows = [] 326 rows += _latlon_tool_examples("find_nearby_beaches", FIND_BEACHES_TEMPLATES) 327 rows += _latlon_tool_examples("get_swim_recommendation", RECOMMENDATION_TEMPLATES) 328 rows += _latlon_tool_examples("get_water_quality", WATER_QUALITY_TEMPLATES) 329 rows += _ask_tool_examples(knowledge_dir, sea_lore_path) 330 return rows 331 332 333def _self_test() -> None: 334 rows = from_knowledge(KNOWLEDGE) 335 assert len(rows) >= KNOWLEDGE_EXAMPLE_FLOOR, ( 336 f"knowledge-derived example count {len(rows)} below floor {KNOWLEDGE_EXAMPLE_FLOOR} " 337 "— MIP-0025 task 2 requires thousands, not dozens" 338 ) 339 sources = _load_knowledge_sources(KNOWLEDGE) 340 assert sources, "no knowledge sources found under " + str(KNOWLEDGE) 341 for row in rows: 342 answer = row["messages"][2]["content"] 343 assert "\n\nSource: " in answer, f"synthetic example missing a Source line: {answer!r}" 344 fact, _, url = answer.rpartition("\n\nSource: ") 345 raw = sources.get(url) 346 assert raw is not None, f"cited source is not a real knowledge/*.md source: {url!r}" 347 assert fact in raw, ( 348 f"synthetic fact not found verbatim in the file that cites {url} — looks invented: " 349 f"{fact[:80]!r}" 350 ) 351 352 tool_rows = from_tool_calls(KNOWLEDGE, RESOURCES / "sea_lore.json") 353 assert tool_rows, "no tool-call examples generated" 354 seen_tools: set[str] = set() 355 for row in tool_rows: 356 assistant = row["messages"][2]["content"] 357 parsed = json.loads(assistant) # raises if not syntactically valid JSON 358 assert set(parsed.keys()) == {"tool", "arguments"}, f"unexpected shape: {parsed!r}" 359 tool = parsed["tool"] 360 assert tool in TOOL_SCHEMAS, f"not a real MCP tool name: {tool!r}" 361 seen_tools.add(tool) 362 schema = TOOL_SCHEMAS[tool] 363 args = parsed["arguments"] 364 assert isinstance(args, dict), f"{tool} arguments must be an object: {args!r}" 365 for req in schema["required"]: 366 assert req in args, f"{tool} call is missing required argument {req!r}: {args!r}" 367 allowed = set(schema["required"]) | set(schema["optional"]) 368 for key in args: 369 assert key in allowed, f"{tool} call has an argument not in its real schema: {key!r}" 370 assert seen_tools == set(TOOL_SCHEMAS), ( 371 f"missing tool coverage: {set(TOOL_SCHEMAS) - seen_tools}" 372 ) 373 print( 374 f"self-test OK: {len(rows)} knowledge-derived examples " 375 f"(floor {KNOWLEDGE_EXAMPLE_FLOOR}), every fact verified verbatim against its cited " 376 f"source; {len(tool_rows)} tool-call examples covering all {len(TOOL_SCHEMAS)} real " 377 f"MCP tools with syntactically valid call shapes" 378 ) 379 380 381def main() -> None: 382 rows: list[dict] = [] 383 rows += from_compiled_prompt(RESOURCES / "recommendation_prompt.json", "summary") 384 rows += from_compiled_prompt( 385 RESOURCES / "review_prompt.json", "review_json", extra_inputs=("summary",) 386 ) 387 rows += from_sea_lore(RESOURCES / "sea_lore.json") 388 rows += from_knowledge(KNOWLEDGE) 389 rows += from_tool_calls(KNOWLEDGE, RESOURCES / "sea_lore.json") 390 391 random.Random(42).shuffle(rows) 392 n_eval = max(2, len(rows) // 10) 393 eval_rows, train_rows = rows[:n_eval], rows[n_eval:] 394 395 OUT.mkdir(parents=True, exist_ok=True) 396 for name, data in (("train.jsonl", train_rows), ("eval.jsonl", eval_rows)): 397 with (OUT / name).open("w", encoding="utf-8") as f: 398 for r in data: 399 f.write(json.dumps(r, ensure_ascii=False) + "\n") 400 print(f"wrote {len(train_rows)} train + {len(eval_rows)} eval examples to {OUT}") 401 402 403if __name__ == "__main__": 404 import sys 405 406 if "--self-test" in sys.argv: 407 _self_test() 408 else: 409 main()
REPO =
PosixPath('/home/runner/work/marola/marola')
RESOURCES =
PosixPath('/home/runner/work/marola/marola/core/src/main/resources')
KNOWLEDGE =
PosixPath('/home/runner/work/marola/marola/knowledge')
OUT =
PosixPath('/home/runner/work/marola/marola/finetune/data')
SYSTEM =
'You are marola, a swim-conditions assistant for open-water swimmers in Brazil. Answer in one or two plain sentences from the facts given; never invent conditions; cite a source URL when you quote your notes; return compact JSON and nothing else when asked for JSON.'
INPUT_FIELDS =
('beach_name', 'hour_local', 'sea_temp_c', 'wind_kmh', 'wave_height_m', 'jellyfish_risk', 'whale_sighting_likelihood', 'score')
def
render_inputs(demo: dict, fields: tuple[str, ...]) -> str:
def
example(user: str, assistant: str) -> dict:
def
from_compiled_prompt( path: pathlib.Path, output_field: str, extra_inputs: tuple[str, ...] = ()) -> list[dict]:
68def from_compiled_prompt( 69 path: Path, output_field: str, extra_inputs: tuple[str, ...] = () 70) -> list[dict]: 71 doc = json.loads(path.read_text(encoding="utf-8")) 72 instructions = doc.get("signature", {}).get("instructions", "").strip() 73 rows = [] 74 for demo in doc.get("demos", []): 75 if output_field not in demo: 76 continue 77 user = (instructions + "\n\n" if instructions else "") + render_inputs( 78 demo, INPUT_FIELDS + extra_inputs 79 ) 80 rows.append(example(user, demo[output_field])) 81 return rows
def
from_sea_lore(path: pathlib.Path) -> list[dict]:
def
chunk_markdown(md: str, max_chars: int = 700) -> tuple[str, str, list[str]]:
93def chunk_markdown(md: str, max_chars: int = 700) -> tuple[str, str, list[str]]: 94 """Mirror of marola.knowledge.Corpus.chunkDocument: title, source, merged paragraphs.""" 95 lines = md.splitlines() 96 title = next((ln[2:].strip() for ln in lines if ln.startswith("# ")), "") 97 source = next((ln[7:].strip() for ln in lines if ln.lower().startswith("source:")), "") 98 body = "\n".join( 99 ln for ln in lines if not (ln.startswith("# ") or ln.lower().startswith("source:")) 100 ) 101 paras = [p.strip() for p in re.split(r"\n\s*\n", body) if p.strip()] 102 chunks: list[str] = [] 103 for p in paras: 104 if chunks and len(chunks[-1]) + len(p) + 2 <= max_chars: 105 chunks[-1] = chunks[-1] + "\n\n" + p 106 else: 107 chunks.append(p) 108 return title, source, chunks
Mirror of marola.knowledge.Corpus.chunkDocument: title, source, merged paragraphs.
CHUNK_QUESTION_TEMPLATES =
('What do your notes say about {title}? (part {part})', 'Tell me what you know about {title}, part {part}.', 'Summarize what your notes cover on {title} (part {part}).', 'Give me the part {part} details from your notes on {title}.', "What's in section {part} of your {title} notes?", 'Explain {title} using what your notes say (part {part}).', "I'm curious about {title} — what does part {part} of your notes cover?", 'Recap part {part} of your knowledge on {title}.', 'What have you got written down about {title}, specifically part {part}?', 'Walk me through part {part} of your {title} notes.', "What's the part {part} summary of {title} in your notes?", "According to your notes, what's true about {title} (part {part})?", 'Pull up part {part} of what you know about {title}.', 'What can you tell a swimmer about {title}? (part {part})', 'Notes check: {title}, part {part}.', 'Brief me on {title}, part {part}, from your sources.', "What's documented about {title} in part {part} of your notes?", 'Give a plain-language rundown of {title}, part {part}.', 'What should I know about {title}? (see part {part} of your notes)', 'Show me part {part} of your {title} material.')
SENTENCE_QUESTION_TEMPLATES =
('Give me one specific fact from your notes on {title} (part {part}).', "What's a detail from your {title} notes, part {part}?", 'Pick one fact from part {part} of your {title} notes.', "What's something true about {title} from part {part} of your notes?", 'Quote one fact from your notes on {title} (part {part}).', 'Name a specific point from your {title} notes, part {part}.', "What's one thing your notes say about {title}? (part {part})", 'Give a short, specific fact about {title} (part {part}).', 'What detail from part {part} of your {title} notes stands out?', 'Tell me a single fact from your {title} notes (part {part}).', "What's one takeaway from part {part} of your {title} notes?", 'Share one specific fact about {title} (part {part}).', "What's a concrete detail about {title}? (from part {part} of your notes)", 'Give me a fact, not a summary, about {title} (part {part}).', 'What does part {part} of your {title} notes say, in one fact?', 'Point to one fact in your {title} notes (part {part}).', "What's a specific claim in part {part} of your {title} notes?", 'Give one sentence of fact about {title} (part {part}).', "What's a single detail worth knowing about {title}? (part {part})", 'Extract one fact from part {part} of your {title} notes.')
def
from_knowledge(dir_: pathlib.Path) -> list[dict]:
183def from_knowledge(dir_: Path) -> list[dict]: 184 rows = [] 185 for path in sorted(dir_.rglob("*.md")): 186 title, source, chunks = chunk_markdown(path.read_text(encoding="utf-8")) 187 if not source: 188 continue # README.md and anything else without a citable source 189 for i, chunk in enumerate(chunks): 190 part = i + 1 191 rows += _chunk_examples(title, source, chunk, part) 192 rows += _sentence_examples(title, source, chunk, part) 193 return rows
KNOWLEDGE_EXAMPLE_FLOOR =
2000
TOOL_CALL_SYSTEM =
'You are marola, a swim-conditions assistant for open-water swimmers in Brazil. When a question needs live data you don\'t have, reply with exactly one JSON object of the shape {"tool": "<name>", "arguments": {...}} and nothing else — never invent the result yourself.'
TOOL_SCHEMAS: dict[str, dict[str, tuple[str, ...]]] =
{'find_nearby_beaches': {'required': ('lat', 'lon'), 'optional': ('radius_km',)}, 'get_swim_recommendation': {'required': ('lat', 'lon'), 'optional': ('radius_km',)}, 'get_water_quality': {'required': ('lat', 'lon'), 'optional': ('radius_km',)}, 'ask_ocean_question': {'required': ('question',), 'optional': ()}}
LOCATIONS: tuple[tuple[float, float], ...] =
((-27.6296, -48.4487), (-27.4021, -48.4157))
RADII: tuple[float | None, ...] =
(None, 10.0, 20.0)
FIND_BEACHES_TEMPLATES =
('Find swim beaches near {lat}, {lon}.', 'What open-water beaches are close to {lat}, {lon}?', 'Search for beaches around latitude {lat}, longitude {lon}.', 'Are there any beaches near {lat}, {lon}?', 'List the beaches close to {lat}, {lon}.', "I'm at {lat}, {lon} — any beaches nearby?", 'Beaches near {lat}, {lon}, please.', 'Which beaches are around {lat}, {lon}?', 'Show me swim spots close to {lat}, {lon}.', 'Look up beaches near coordinate {lat}, {lon}.')
RECOMMENDATION_TEMPLATES =
("What's the best hour to swim tomorrow near {lat}, {lon}?", "Give me tomorrow's swim conditions near {lat}, {lon}.", 'When should I swim tomorrow around {lat}, {lon}?', 'Best swim window tomorrow near {lat}, {lon}?', "What are tomorrow's conditions like near {lat}, {lon}?", 'Recommend a swim time near {lat}, {lon} for tomorrow.', "I want to swim tomorrow near {lat}, {lon} — when's best?", "Tell me tomorrow's swimability near {lat}, {lon}.", "Forecast tomorrow's swim conditions for {lat}, {lon}.", 'What hour has the best score near {lat}, {lon} tomorrow?')
WATER_QUALITY_TEMPLATES =
('Is the water clean near {lat}, {lon}?', "What's the bathing-water quality near {lat}, {lon}?", 'Check water quality around {lat}, {lon}.', 'Any water quality warnings near {lat}, {lon}?', 'Is it safe to swim near {lat}, {lon} — water quality-wise?', 'Give me the enterococci readings near {lat}, {lon}.', 'PRÓPRIA or IMPRÓPRIA near {lat}, {lon}?', 'What do the sampling points say near {lat}, {lon}?', 'Water quality check for {lat}, {lon}.', 'Has water near {lat}, {lon} been tested recently?')
ASK_QUESTION_TEMPLATES =
('{question}', 'Hey marola, {question}', 'Quick question: {question}', '{question} Please look it up.', 'I want to know: {question}', 'Can you answer this: {question}', '{question} (from a swimmer prepping a trip)', 'Ocean question: {question}', '{question} What does your knowledge base say?', 'Before I go for a swim, {question}')
def
from_tool_calls(knowledge_dir: pathlib.Path, sea_lore_path: pathlib.Path) -> list[dict]:
325def from_tool_calls(knowledge_dir: Path, sea_lore_path: Path) -> list[dict]: 326 rows = [] 327 rows += _latlon_tool_examples("find_nearby_beaches", FIND_BEACHES_TEMPLATES) 328 rows += _latlon_tool_examples("get_swim_recommendation", RECOMMENDATION_TEMPLATES) 329 rows += _latlon_tool_examples("get_water_quality", WATER_QUALITY_TEMPLATES) 330 rows += _ask_tool_examples(knowledge_dir, sea_lore_path) 331 return rows
def
main() -> None:
382def main() -> None: 383 rows: list[dict] = [] 384 rows += from_compiled_prompt(RESOURCES / "recommendation_prompt.json", "summary") 385 rows += from_compiled_prompt( 386 RESOURCES / "review_prompt.json", "review_json", extra_inputs=("summary",) 387 ) 388 rows += from_sea_lore(RESOURCES / "sea_lore.json") 389 rows += from_knowledge(KNOWLEDGE) 390 rows += from_tool_calls(KNOWLEDGE, RESOURCES / "sea_lore.json") 391 392 random.Random(42).shuffle(rows) 393 n_eval = max(2, len(rows) // 10) 394 eval_rows, train_rows = rows[:n_eval], rows[n_eval:] 395 396 OUT.mkdir(parents=True, exist_ok=True) 397 for name, data in (("train.jsonl", train_rows), ("eval.jsonl", eval_rows)): 398 with (OUT / name).open("w", encoding="utf-8") as f: 399 for r in data: 400 f.write(json.dumps(r, ensure_ascii=False) + "\n") 401 print(f"wrote {len(train_rows)} train + {len(eval_rows)} eval examples to {OUT}")