build_dataset

Build marola's fine-tuning dataset from what the repo already has — no new labelling.

Sources (all local files, no network, no model calls):

  1. core/src/main/resources/recommendation_prompt.json — the DSPy-compiled summarizer demos: structured conditions in, one-to-two-sentence summary out. The core "shape" marola wants.
  2. core/src/main/resources/review_prompt.json — the reviewer demos: conditions + draft in, compact JSON verdict out. Teaches the JSON-only discipline small models break most.
  3. core/src/main/resources/sea_lore.json — "tell me something about X" → the sourced paragraph, with the source URL kept in the answer so the habit of citing survives.
  4. knowledge/*.md (recursively) — one Q/A per chunk and per sentence, each asked through several question templates (MIP-0025 §4.3 Layer 1). Every answer is verbatim text from the file it cites; no model call generates anything.
  5. Tool-call SFT (MIP-0025 §4.3 Layer 2): questions → one JSON call to the four MCP tools in cli/src/main/scala/marola/agent/SwimConditionsMcpServer.scala. Teaches when to call, not facts.

Output: finetune/data/train.jsonl and eval.jsonl in the chat format most trainers accept: {"messages": [{"role": "system", ...}, {"role": "user", ...}, {"role": "assistant", ...}]}

Run: python build_dataset.py (or just finetune-dataset) Self-test: python build_dataset.py --self-test (or just quality-other)

  1"""Build marola's fine-tuning dataset from what the repo already has — no new labelling.
  2
  3Sources (all local files, no network, no model calls):
  4  1. core/src/main/resources/recommendation_prompt.json — the DSPy-compiled summarizer demos:
  5     structured conditions in, one-to-two-sentence summary out. The core "shape" marola wants.
  6  2. core/src/main/resources/review_prompt.json — the reviewer demos: conditions + draft in,
  7     compact JSON verdict out. Teaches the JSON-only discipline small models break most.
  8  3. core/src/main/resources/sea_lore.json — "tell me something about X" → the sourced paragraph,
  9     with the source URL kept in the answer so the habit of citing survives.
 10  4. knowledge/*.md (recursively) — one Q/A per chunk and per sentence, each asked through several
 11     question templates (MIP-0025 §4.3 Layer 1). Every answer is verbatim text from the file it
 12     cites; no model call generates anything.
 13  5. Tool-call SFT (MIP-0025 §4.3 Layer 2): questions → one JSON call to the four MCP tools in
 14     cli/src/main/scala/marola/agent/SwimConditionsMcpServer.scala. Teaches when to call, not facts.
 15
 16Output: finetune/data/train.jsonl and eval.jsonl in the chat format most trainers accept:
 17  {"messages": [{"role": "system", ...}, {"role": "user", ...}, {"role": "assistant", ...}]}
 18
 19Run:  python build_dataset.py            (or `just finetune-dataset`)
 20Self-test:  python build_dataset.py --self-test   (or `just quality-other`)
 21"""
 22
 23from __future__ import annotations
 24
 25import json
 26import random
 27import re
 28from pathlib import Path
 29
 30REPO = Path(__file__).resolve().parent.parent
 31RESOURCES = REPO / "core" / "src" / "main" / "resources"
 32KNOWLEDGE = REPO / "knowledge"
 33OUT = Path(__file__).resolve().parent / "data"
 34
 35SYSTEM = (
 36    "You are marola, a swim-conditions assistant for open-water swimmers in Brazil. Answer in one "
 37    "or two plain sentences from the facts given; never invent conditions; cite a source URL when "
 38    "you quote your notes; return compact JSON and nothing else when asked for JSON."
 39)
 40
 41INPUT_FIELDS = (
 42    "beach_name",
 43    "hour_local",
 44    "sea_temp_c",
 45    "wind_kmh",
 46    "wave_height_m",
 47    "jellyfish_risk",
 48    "whale_sighting_likelihood",
 49    "score",
 50)
 51
 52
 53def render_inputs(demo: dict, fields: tuple[str, ...]) -> str:
 54    return "\n".join(f"{f.replace('_', ' ').title()}: {demo[f]}" for f in fields if f in demo)
 55
 56
 57def example(user: str, assistant: str) -> dict:
 58    return {
 59        "messages": [
 60            {"role": "system", "content": SYSTEM},
 61            {"role": "user", "content": user},
 62            {"role": "assistant", "content": assistant},
 63        ]
 64    }
 65
 66
 67def from_compiled_prompt(
 68    path: Path, output_field: str, extra_inputs: tuple[str, ...] = ()
 69) -> list[dict]:
 70    doc = json.loads(path.read_text(encoding="utf-8"))
 71    instructions = doc.get("signature", {}).get("instructions", "").strip()
 72    rows = []
 73    for demo in doc.get("demos", []):
 74        if output_field not in demo:
 75            continue
 76        user = (instructions + "\n\n" if instructions else "") + render_inputs(
 77            demo, INPUT_FIELDS + extra_inputs
 78        )
 79        rows.append(example(user, demo[output_field]))
 80    return rows
 81
 82
 83def from_sea_lore(path: Path) -> list[dict]:
 84    rows = []
 85    for e in json.loads(path.read_text(encoding="utf-8")):
 86        topic = e["id"].replace("-", " ")
 87        user = f"Tell me something about the sea: {topic}."
 88        rows.append(example(user, f"{e['text']} [source: {e['source']}]"))
 89    return rows
 90
 91
 92def chunk_markdown(md: str, max_chars: int = 700) -> tuple[str, str, list[str]]:
 93    """Mirror of marola.knowledge.Corpus.chunkDocument: title, source, merged paragraphs."""
 94    lines = md.splitlines()
 95    title = next((ln[2:].strip() for ln in lines if ln.startswith("# ")), "")
 96    source = next((ln[7:].strip() for ln in lines if ln.lower().startswith("source:")), "")
 97    body = "\n".join(
 98        ln for ln in lines if not (ln.startswith("# ") or ln.lower().startswith("source:"))
 99    )
100    paras = [p.strip() for p in re.split(r"\n\s*\n", body) if p.strip()]
101    chunks: list[str] = []
102    for p in paras:
103        if chunks and len(chunks[-1]) + len(p) + 2 <= max_chars:
104            chunks[-1] = chunks[-1] + "\n\n" + p
105        else:
106            chunks.append(p)
107    return title, source, chunks
108
109
110# Chunk-level: the question is asked several ways, the answer is always the full real chunk.
111CHUNK_QUESTION_TEMPLATES = (
112    "What do your notes say about {title}? (part {part})",
113    "Tell me what you know about {title}, part {part}.",
114    "Summarize what your notes cover on {title} (part {part}).",
115    "Give me the part {part} details from your notes on {title}.",
116    "What's in section {part} of your {title} notes?",
117    "Explain {title} using what your notes say (part {part}).",
118    "I'm curious about {title} — what does part {part} of your notes cover?",
119    "Recap part {part} of your knowledge on {title}.",
120    "What have you got written down about {title}, specifically part {part}?",
121    "Walk me through part {part} of your {title} notes.",
122    "What's the part {part} summary of {title} in your notes?",
123    "According to your notes, what's true about {title} (part {part})?",
124    "Pull up part {part} of what you know about {title}.",
125    "What can you tell a swimmer about {title}? (part {part})",
126    "Notes check: {title}, part {part}.",
127    "Brief me on {title}, part {part}, from your sources.",
128    "What's documented about {title} in part {part} of your notes?",
129    "Give a plain-language rundown of {title}, part {part}.",
130    "What should I know about {title}? (see part {part} of your notes)",
131    "Show me part {part} of your {title} material.",
132)
133
134# Sentence-level: same idea, one real sentence at a time — finer-grained facts, still verbatim.
135SENTENCE_QUESTION_TEMPLATES = (
136    "Give me one specific fact from your notes on {title} (part {part}).",
137    "What's a detail from your {title} notes, part {part}?",
138    "Pick one fact from part {part} of your {title} notes.",
139    "What's something true about {title} from part {part} of your notes?",
140    "Quote one fact from your notes on {title} (part {part}).",
141    "Name a specific point from your {title} notes, part {part}.",
142    "What's one thing your notes say about {title}? (part {part})",
143    "Give a short, specific fact about {title} (part {part}).",
144    "What detail from part {part} of your {title} notes stands out?",
145    "Tell me a single fact from your {title} notes (part {part}).",
146    "What's one takeaway from part {part} of your {title} notes?",
147    "Share one specific fact about {title} (part {part}).",
148    "What's a concrete detail about {title}? (from part {part} of your notes)",
149    "Give me a fact, not a summary, about {title} (part {part}).",
150    "What does part {part} of your {title} notes say, in one fact?",
151    "Point to one fact in your {title} notes (part {part}).",
152    "What's a specific claim in part {part} of your {title} notes?",
153    "Give one sentence of fact about {title} (part {part}).",
154    "What's a single detail worth knowing about {title}? (part {part})",
155    "Extract one fact from part {part} of your {title} notes.",
156)
157
158
159def _split_sentences(text: str) -> list[str]:
160    """Real sentences only (>20 chars) so a stray abbreviation dot doesn't yield a fragment."""
161    return [s.strip() for s in re.split(r"(?<=[.!?])\s+", text) if len(s.strip()) > 20]
162
163
164def _chunk_examples(title: str, source: str, chunk: str, part: int) -> list[dict]:
165    answer = f"{chunk}\n\nSource: {source}"
166    return [
167        example(t.format(title=title.lower(), part=part), answer) for t in CHUNK_QUESTION_TEMPLATES
168    ]
169
170
171def _sentence_examples(title: str, source: str, chunk: str, part: int) -> list[dict]:
172    rows = []
173    for sentence in _split_sentences(chunk):
174        answer = f"{sentence}\n\nSource: {source}"
175        rows += [
176            example(t.format(title=title.lower(), part=part), answer)
177            for t in SENTENCE_QUESTION_TEMPLATES
178        ]
179    return rows
180
181
182def from_knowledge(dir_: Path) -> list[dict]:
183    rows = []
184    for path in sorted(dir_.rglob("*.md")):
185        title, source, chunks = chunk_markdown(path.read_text(encoding="utf-8"))
186        if not source:
187            continue  # README.md and anything else without a citable source
188        for i, chunk in enumerate(chunks):
189            part = i + 1
190            rows += _chunk_examples(title, source, chunk, part)
191            rows += _sentence_examples(title, source, chunk, part)
192    return rows
193
194
195def _load_knowledge_sources(dir_: Path) -> dict[str, str]:
196    """Map each knowledge doc's cited source URL to its raw file text, for provenance checks."""
197    out: dict[str, str] = {}
198    for path in sorted(dir_.rglob("*.md")):
199        raw = path.read_text(encoding="utf-8")
200        _, source, _ = chunk_markdown(raw)
201        if source:
202            out[source] = raw
203    return out
204
205
206KNOWLEDGE_EXAMPLE_FLOOR = 2000
207
208
209# --- Layer 2: tool-call SFT (MIP-0025 §4.3) --------------------------------------------------
210# Transcribed from SwimConditionsMcpServer.scala's tool schemas; change both together.
211TOOL_CALL_SYSTEM = (
212    "You are marola, a swim-conditions assistant for open-water swimmers in Brazil. When a "
213    "question needs live data you don't have, reply with exactly one JSON object of the shape "
214    '{"tool": "<name>", "arguments": {...}} and nothing else — never invent the result yourself.'
215)
216
217TOOL_SCHEMAS: dict[str, dict[str, tuple[str, ...]]] = {
218    "find_nearby_beaches": {"required": ("lat", "lon"), "optional": ("radius_km",)},
219    "get_swim_recommendation": {"required": ("lat", "lon"), "optional": ("radius_km",)},
220    "get_water_quality": {"required": ("lat", "lon"), "optional": ("radius_km",)},
221    "ask_ocean_question": {"required": ("question",), "optional": ()},
222}
223
224# site/fixtures/board.json's beaches.
225LOCATIONS: tuple[tuple[float, float], ...] = ((-27.6296, -48.4487), (-27.4021, -48.4157))
226RADII: tuple[float | None, ...] = (None, 10.0, 20.0)
227
228FIND_BEACHES_TEMPLATES = (
229    "Find swim beaches near {lat}, {lon}.",
230    "What open-water beaches are close to {lat}, {lon}?",
231    "Search for beaches around latitude {lat}, longitude {lon}.",
232    "Are there any beaches near {lat}, {lon}?",
233    "List the beaches close to {lat}, {lon}.",
234    "I'm at {lat}, {lon} — any beaches nearby?",
235    "Beaches near {lat}, {lon}, please.",
236    "Which beaches are around {lat}, {lon}?",
237    "Show me swim spots close to {lat}, {lon}.",
238    "Look up beaches near coordinate {lat}, {lon}.",
239)
240
241RECOMMENDATION_TEMPLATES = (
242    "What's the best hour to swim tomorrow near {lat}, {lon}?",
243    "Give me tomorrow's swim conditions near {lat}, {lon}.",
244    "When should I swim tomorrow around {lat}, {lon}?",
245    "Best swim window tomorrow near {lat}, {lon}?",
246    "What are tomorrow's conditions like near {lat}, {lon}?",
247    "Recommend a swim time near {lat}, {lon} for tomorrow.",
248    "I want to swim tomorrow near {lat}, {lon} — when's best?",
249    "Tell me tomorrow's swimability near {lat}, {lon}.",
250    "Forecast tomorrow's swim conditions for {lat}, {lon}.",
251    "What hour has the best score near {lat}, {lon} tomorrow?",
252)
253
254WATER_QUALITY_TEMPLATES = (
255    "Is the water clean near {lat}, {lon}?",
256    "What's the bathing-water quality near {lat}, {lon}?",
257    "Check water quality around {lat}, {lon}.",
258    "Any water quality warnings near {lat}, {lon}?",
259    "Is it safe to swim near {lat}, {lon} — water quality-wise?",
260    "Give me the enterococci readings near {lat}, {lon}.",
261    "PRÓPRIA or IMPRÓPRIA near {lat}, {lon}?",
262    "What do the sampling points say near {lat}, {lon}?",
263    "Water quality check for {lat}, {lon}.",
264    "Has water near {lat}, {lon} been tested recently?",
265)
266
267ASK_QUESTION_TEMPLATES = (
268    "{question}",
269    "Hey marola, {question}",
270    "Quick question: {question}",
271    "{question} Please look it up.",
272    "I want to know: {question}",
273    "Can you answer this: {question}",
274    "{question} (from a swimmer prepping a trip)",
275    "Ocean question: {question}",
276    "{question} What does your knowledge base say?",
277    "Before I go for a swim, {question}",
278)
279
280
281def _tool_call_example(user: str, tool: str, arguments: dict) -> dict:
282    assistant = json.dumps({"tool": tool, "arguments": arguments})
283    return {
284        "messages": [
285            {"role": "system", "content": TOOL_CALL_SYSTEM},
286            {"role": "user", "content": user},
287            {"role": "assistant", "content": assistant},
288        ]
289    }
290
291
292def _latlon_tool_examples(tool: str, templates: tuple[str, ...]) -> list[dict]:
293    rows = []
294    for lat, lon in LOCATIONS:
295        for radius in RADII:
296            arguments: dict[str, float] = {"lat": lat, "lon": lon}
297            if radius is not None:
298                arguments["radius_km"] = radius
299            for t in templates:
300                user = t.format(lat=lat, lon=lon)
301                if radius is not None:
302                    user += f" Search within {radius:g} km."
303                rows.append(_tool_call_example(user, tool, arguments))
304    return rows
305
306
307def _ask_tool_examples(knowledge_dir: Path, sea_lore_path: Path) -> list[dict]:
308    questions = []
309    for path in sorted(knowledge_dir.rglob("*.md")):
310        title, source, _ = chunk_markdown(path.read_text(encoding="utf-8"))
311        if source:
312            questions.append(f"What do you know about {title.lower()}?")
313    if sea_lore_path.exists():
314        for e in json.loads(sea_lore_path.read_text(encoding="utf-8")):
315            questions.append(f"Tell me about {e['id'].replace('-', ' ')}.")
316    rows = []
317    for question in questions:
318        for t in ASK_QUESTION_TEMPLATES:
319            user = t.format(question=question)
320            rows.append(_tool_call_example(user, "ask_ocean_question", {"question": question}))
321    return rows
322
323
324def from_tool_calls(knowledge_dir: Path, sea_lore_path: Path) -> list[dict]:
325    rows = []
326    rows += _latlon_tool_examples("find_nearby_beaches", FIND_BEACHES_TEMPLATES)
327    rows += _latlon_tool_examples("get_swim_recommendation", RECOMMENDATION_TEMPLATES)
328    rows += _latlon_tool_examples("get_water_quality", WATER_QUALITY_TEMPLATES)
329    rows += _ask_tool_examples(knowledge_dir, sea_lore_path)
330    return rows
331
332
333def _self_test() -> None:
334    rows = from_knowledge(KNOWLEDGE)
335    assert len(rows) >= KNOWLEDGE_EXAMPLE_FLOOR, (
336        f"knowledge-derived example count {len(rows)} below floor {KNOWLEDGE_EXAMPLE_FLOOR} "
337        "— MIP-0025 task 2 requires thousands, not dozens"
338    )
339    sources = _load_knowledge_sources(KNOWLEDGE)
340    assert sources, "no knowledge sources found under " + str(KNOWLEDGE)
341    for row in rows:
342        answer = row["messages"][2]["content"]
343        assert "\n\nSource: " in answer, f"synthetic example missing a Source line: {answer!r}"
344        fact, _, url = answer.rpartition("\n\nSource: ")
345        raw = sources.get(url)
346        assert raw is not None, f"cited source is not a real knowledge/*.md source: {url!r}"
347        assert fact in raw, (
348            f"synthetic fact not found verbatim in the file that cites {url} — looks invented: "
349            f"{fact[:80]!r}"
350        )
351
352    tool_rows = from_tool_calls(KNOWLEDGE, RESOURCES / "sea_lore.json")
353    assert tool_rows, "no tool-call examples generated"
354    seen_tools: set[str] = set()
355    for row in tool_rows:
356        assistant = row["messages"][2]["content"]
357        parsed = json.loads(assistant)  # raises if not syntactically valid JSON
358        assert set(parsed.keys()) == {"tool", "arguments"}, f"unexpected shape: {parsed!r}"
359        tool = parsed["tool"]
360        assert tool in TOOL_SCHEMAS, f"not a real MCP tool name: {tool!r}"
361        seen_tools.add(tool)
362        schema = TOOL_SCHEMAS[tool]
363        args = parsed["arguments"]
364        assert isinstance(args, dict), f"{tool} arguments must be an object: {args!r}"
365        for req in schema["required"]:
366            assert req in args, f"{tool} call is missing required argument {req!r}: {args!r}"
367        allowed = set(schema["required"]) | set(schema["optional"])
368        for key in args:
369            assert key in allowed, f"{tool} call has an argument not in its real schema: {key!r}"
370    assert seen_tools == set(TOOL_SCHEMAS), (
371        f"missing tool coverage: {set(TOOL_SCHEMAS) - seen_tools}"
372    )
373    print(
374        f"self-test OK: {len(rows)} knowledge-derived examples "
375        f"(floor {KNOWLEDGE_EXAMPLE_FLOOR}), every fact verified verbatim against its cited "
376        f"source; {len(tool_rows)} tool-call examples covering all {len(TOOL_SCHEMAS)} real "
377        f"MCP tools with syntactically valid call shapes"
378    )
379
380
381def main() -> None:
382    rows: list[dict] = []
383    rows += from_compiled_prompt(RESOURCES / "recommendation_prompt.json", "summary")
384    rows += from_compiled_prompt(
385        RESOURCES / "review_prompt.json", "review_json", extra_inputs=("summary",)
386    )
387    rows += from_sea_lore(RESOURCES / "sea_lore.json")
388    rows += from_knowledge(KNOWLEDGE)
389    rows += from_tool_calls(KNOWLEDGE, RESOURCES / "sea_lore.json")
390
391    random.Random(42).shuffle(rows)
392    n_eval = max(2, len(rows) // 10)
393    eval_rows, train_rows = rows[:n_eval], rows[n_eval:]
394
395    OUT.mkdir(parents=True, exist_ok=True)
396    for name, data in (("train.jsonl", train_rows), ("eval.jsonl", eval_rows)):
397        with (OUT / name).open("w", encoding="utf-8") as f:
398            for r in data:
399                f.write(json.dumps(r, ensure_ascii=False) + "\n")
400    print(f"wrote {len(train_rows)} train + {len(eval_rows)} eval examples to {OUT}")
401
402
403if __name__ == "__main__":
404    import sys
405
406    if "--self-test" in sys.argv:
407        _self_test()
408    else:
409        main()
REPO = PosixPath('/home/runner/work/marola/marola')
RESOURCES = PosixPath('/home/runner/work/marola/marola/core/src/main/resources')
KNOWLEDGE = PosixPath('/home/runner/work/marola/marola/knowledge')
OUT = PosixPath('/home/runner/work/marola/marola/finetune/data')
SYSTEM = 'You are marola, a swim-conditions assistant for open-water swimmers in Brazil. Answer in one or two plain sentences from the facts given; never invent conditions; cite a source URL when you quote your notes; return compact JSON and nothing else when asked for JSON.'
INPUT_FIELDS = ('beach_name', 'hour_local', 'sea_temp_c', 'wind_kmh', 'wave_height_m', 'jellyfish_risk', 'whale_sighting_likelihood', 'score')
def render_inputs(demo: dict, fields: tuple[str, ...]) -> str:
54def render_inputs(demo: dict, fields: tuple[str, ...]) -> str:
55    return "\n".join(f"{f.replace('_', ' ').title()}: {demo[f]}" for f in fields if f in demo)
def example(user: str, assistant: str) -> dict:
58def example(user: str, assistant: str) -> dict:
59    return {
60        "messages": [
61            {"role": "system", "content": SYSTEM},
62            {"role": "user", "content": user},
63            {"role": "assistant", "content": assistant},
64        ]
65    }
def from_compiled_prompt( path: pathlib.Path, output_field: str, extra_inputs: tuple[str, ...] = ()) -> list[dict]:
68def from_compiled_prompt(
69    path: Path, output_field: str, extra_inputs: tuple[str, ...] = ()
70) -> list[dict]:
71    doc = json.loads(path.read_text(encoding="utf-8"))
72    instructions = doc.get("signature", {}).get("instructions", "").strip()
73    rows = []
74    for demo in doc.get("demos", []):
75        if output_field not in demo:
76            continue
77        user = (instructions + "\n\n" if instructions else "") + render_inputs(
78            demo, INPUT_FIELDS + extra_inputs
79        )
80        rows.append(example(user, demo[output_field]))
81    return rows
def from_sea_lore(path: pathlib.Path) -> list[dict]:
84def from_sea_lore(path: Path) -> list[dict]:
85    rows = []
86    for e in json.loads(path.read_text(encoding="utf-8")):
87        topic = e["id"].replace("-", " ")
88        user = f"Tell me something about the sea: {topic}."
89        rows.append(example(user, f"{e['text']} [source: {e['source']}]"))
90    return rows
def chunk_markdown(md: str, max_chars: int = 700) -> tuple[str, str, list[str]]:
 93def chunk_markdown(md: str, max_chars: int = 700) -> tuple[str, str, list[str]]:
 94    """Mirror of marola.knowledge.Corpus.chunkDocument: title, source, merged paragraphs."""
 95    lines = md.splitlines()
 96    title = next((ln[2:].strip() for ln in lines if ln.startswith("# ")), "")
 97    source = next((ln[7:].strip() for ln in lines if ln.lower().startswith("source:")), "")
 98    body = "\n".join(
 99        ln for ln in lines if not (ln.startswith("# ") or ln.lower().startswith("source:"))
100    )
101    paras = [p.strip() for p in re.split(r"\n\s*\n", body) if p.strip()]
102    chunks: list[str] = []
103    for p in paras:
104        if chunks and len(chunks[-1]) + len(p) + 2 <= max_chars:
105            chunks[-1] = chunks[-1] + "\n\n" + p
106        else:
107            chunks.append(p)
108    return title, source, chunks

Mirror of marola.knowledge.Corpus.chunkDocument: title, source, merged paragraphs.

CHUNK_QUESTION_TEMPLATES = ('What do your notes say about {title}? (part {part})', 'Tell me what you know about {title}, part {part}.', 'Summarize what your notes cover on {title} (part {part}).', 'Give me the part {part} details from your notes on {title}.', "What's in section {part} of your {title} notes?", 'Explain {title} using what your notes say (part {part}).', "I'm curious about {title} — what does part {part} of your notes cover?", 'Recap part {part} of your knowledge on {title}.', 'What have you got written down about {title}, specifically part {part}?', 'Walk me through part {part} of your {title} notes.', "What's the part {part} summary of {title} in your notes?", "According to your notes, what's true about {title} (part {part})?", 'Pull up part {part} of what you know about {title}.', 'What can you tell a swimmer about {title}? (part {part})', 'Notes check: {title}, part {part}.', 'Brief me on {title}, part {part}, from your sources.', "What's documented about {title} in part {part} of your notes?", 'Give a plain-language rundown of {title}, part {part}.', 'What should I know about {title}? (see part {part} of your notes)', 'Show me part {part} of your {title} material.')
SENTENCE_QUESTION_TEMPLATES = ('Give me one specific fact from your notes on {title} (part {part}).', "What's a detail from your {title} notes, part {part}?", 'Pick one fact from part {part} of your {title} notes.', "What's something true about {title} from part {part} of your notes?", 'Quote one fact from your notes on {title} (part {part}).', 'Name a specific point from your {title} notes, part {part}.', "What's one thing your notes say about {title}? (part {part})", 'Give a short, specific fact about {title} (part {part}).', 'What detail from part {part} of your {title} notes stands out?', 'Tell me a single fact from your {title} notes (part {part}).', "What's one takeaway from part {part} of your {title} notes?", 'Share one specific fact about {title} (part {part}).', "What's a concrete detail about {title}? (from part {part} of your notes)", 'Give me a fact, not a summary, about {title} (part {part}).', 'What does part {part} of your {title} notes say, in one fact?', 'Point to one fact in your {title} notes (part {part}).', "What's a specific claim in part {part} of your {title} notes?", 'Give one sentence of fact about {title} (part {part}).', "What's a single detail worth knowing about {title}? (part {part})", 'Extract one fact from part {part} of your {title} notes.')
def from_knowledge(dir_: pathlib.Path) -> list[dict]:
183def from_knowledge(dir_: Path) -> list[dict]:
184    rows = []
185    for path in sorted(dir_.rglob("*.md")):
186        title, source, chunks = chunk_markdown(path.read_text(encoding="utf-8"))
187        if not source:
188            continue  # README.md and anything else without a citable source
189        for i, chunk in enumerate(chunks):
190            part = i + 1
191            rows += _chunk_examples(title, source, chunk, part)
192            rows += _sentence_examples(title, source, chunk, part)
193    return rows
KNOWLEDGE_EXAMPLE_FLOOR = 2000
TOOL_CALL_SYSTEM = 'You are marola, a swim-conditions assistant for open-water swimmers in Brazil. When a question needs live data you don\'t have, reply with exactly one JSON object of the shape {"tool": "<name>", "arguments": {...}} and nothing else — never invent the result yourself.'
TOOL_SCHEMAS: dict[str, dict[str, tuple[str, ...]]] = {'find_nearby_beaches': {'required': ('lat', 'lon'), 'optional': ('radius_km',)}, 'get_swim_recommendation': {'required': ('lat', 'lon'), 'optional': ('radius_km',)}, 'get_water_quality': {'required': ('lat', 'lon'), 'optional': ('radius_km',)}, 'ask_ocean_question': {'required': ('question',), 'optional': ()}}
LOCATIONS: tuple[tuple[float, float], ...] = ((-27.6296, -48.4487), (-27.4021, -48.4157))
RADII: tuple[float | None, ...] = (None, 10.0, 20.0)
FIND_BEACHES_TEMPLATES = ('Find swim beaches near {lat}, {lon}.', 'What open-water beaches are close to {lat}, {lon}?', 'Search for beaches around latitude {lat}, longitude {lon}.', 'Are there any beaches near {lat}, {lon}?', 'List the beaches close to {lat}, {lon}.', "I'm at {lat}, {lon} — any beaches nearby?", 'Beaches near {lat}, {lon}, please.', 'Which beaches are around {lat}, {lon}?', 'Show me swim spots close to {lat}, {lon}.', 'Look up beaches near coordinate {lat}, {lon}.')
RECOMMENDATION_TEMPLATES = ("What's the best hour to swim tomorrow near {lat}, {lon}?", "Give me tomorrow's swim conditions near {lat}, {lon}.", 'When should I swim tomorrow around {lat}, {lon}?', 'Best swim window tomorrow near {lat}, {lon}?', "What are tomorrow's conditions like near {lat}, {lon}?", 'Recommend a swim time near {lat}, {lon} for tomorrow.', "I want to swim tomorrow near {lat}, {lon} — when's best?", "Tell me tomorrow's swimability near {lat}, {lon}.", "Forecast tomorrow's swim conditions for {lat}, {lon}.", 'What hour has the best score near {lat}, {lon} tomorrow?')
WATER_QUALITY_TEMPLATES = ('Is the water clean near {lat}, {lon}?', "What's the bathing-water quality near {lat}, {lon}?", 'Check water quality around {lat}, {lon}.', 'Any water quality warnings near {lat}, {lon}?', 'Is it safe to swim near {lat}, {lon} — water quality-wise?', 'Give me the enterococci readings near {lat}, {lon}.', 'PRÓPRIA or IMPRÓPRIA near {lat}, {lon}?', 'What do the sampling points say near {lat}, {lon}?', 'Water quality check for {lat}, {lon}.', 'Has water near {lat}, {lon} been tested recently?')
ASK_QUESTION_TEMPLATES = ('{question}', 'Hey marola, {question}', 'Quick question: {question}', '{question} Please look it up.', 'I want to know: {question}', 'Can you answer this: {question}', '{question} (from a swimmer prepping a trip)', 'Ocean question: {question}', '{question} What does your knowledge base say?', 'Before I go for a swim, {question}')
def from_tool_calls(knowledge_dir: pathlib.Path, sea_lore_path: pathlib.Path) -> list[dict]:
325def from_tool_calls(knowledge_dir: Path, sea_lore_path: Path) -> list[dict]:
326    rows = []
327    rows += _latlon_tool_examples("find_nearby_beaches", FIND_BEACHES_TEMPLATES)
328    rows += _latlon_tool_examples("get_swim_recommendation", RECOMMENDATION_TEMPLATES)
329    rows += _latlon_tool_examples("get_water_quality", WATER_QUALITY_TEMPLATES)
330    rows += _ask_tool_examples(knowledge_dir, sea_lore_path)
331    return rows
def main() -> None:
382def main() -> None:
383    rows: list[dict] = []
384    rows += from_compiled_prompt(RESOURCES / "recommendation_prompt.json", "summary")
385    rows += from_compiled_prompt(
386        RESOURCES / "review_prompt.json", "review_json", extra_inputs=("summary",)
387    )
388    rows += from_sea_lore(RESOURCES / "sea_lore.json")
389    rows += from_knowledge(KNOWLEDGE)
390    rows += from_tool_calls(KNOWLEDGE, RESOURCES / "sea_lore.json")
391
392    random.Random(42).shuffle(rows)
393    n_eval = max(2, len(rows) // 10)
394    eval_rows, train_rows = rows[:n_eval], rows[n_eval:]
395
396    OUT.mkdir(parents=True, exist_ok=True)
397    for name, data in (("train.jsonl", train_rows), ("eval.jsonl", eval_rows)):
398        with (OUT / name).open("w", encoding="utf-8") as f:
399            for r in data:
400                f.write(json.dumps(r, ensure_ascii=False) + "\n")
401    print(f"wrote {len(train_rows)} train + {len(eval_rows)} eval examples to {OUT}")