build_dataset

Build marola's fine-tuning dataset from what the repo already has — no new labelling.

Sources (all local files, no network, no model calls):

  1. core/src/main/resources/recommendation_prompt.json — the DSPy-compiled summarizer demos: structured conditions in, one-to-two-sentence summary out. The core "shape" marola wants.
  2. core/src/main/resources/review_prompt.json — the reviewer demos: conditions + draft in, compact JSON verdict out. Teaches the JSON-only discipline small models break most.
  3. core/src/main/resources/sea_lore.json — "tell me something about X" → the sourced paragraph, with the source URL kept in the answer so the habit of citing survives.
  4. knowledge/*.md (recursively) — one Q/A per chunk and per sentence, each asked through several question templates (MIP-0025 §4.3 Layer 1). Every answer is verbatim text from the file it cites; no model call generates anything.
  5. Tool-call SFT (MIP-0025 §4.3 Layer 2): questions → one JSON call to the four MCP tools in cli/src/main/scala/marola/agent/SwimConditionsMcpServer.scala. Teaches when to call, not facts.

Output: finetune/data/train.jsonl and eval.jsonl in the chat format most trainers accept: {"messages": [{"role": "system", ...}, {"role": "user", ...}, {"role": "assistant", ...}]}

Run: python build_dataset.py (or just finetune-dataset) Self-test: python build_dataset.py --self-test (or just quality-other)

--resources DIR / --knowledge DIR override where 1-3 and 4-5 above are read from — the app -> ml contract (MIP-0070 §5.4): once marola-ml is a separate repo, --resources points at the unpacked resources tarball ci.yml publishes, not ../core. --knowledge defaults to $MAROLA_KNOWLEDGE_DIR if set, else this repo's .tmp/knowledge from just corpus-fetch (anchored like --resources, not the cwd — this file is documented to run from finetune/).

  1"""Build marola's fine-tuning dataset from what the repo already has — no new labelling.
  2
  3Sources (all local files, no network, no model calls):
  4  1. core/src/main/resources/recommendation_prompt.json — the DSPy-compiled summarizer demos:
  5     structured conditions in, one-to-two-sentence summary out. The core "shape" marola wants.
  6  2. core/src/main/resources/review_prompt.json — the reviewer demos: conditions + draft in,
  7     compact JSON verdict out. Teaches the JSON-only discipline small models break most.
  8  3. core/src/main/resources/sea_lore.json — "tell me something about X" → the sourced paragraph,
  9     with the source URL kept in the answer so the habit of citing survives.
 10  4. knowledge/*.md (recursively) — one Q/A per chunk and per sentence, each asked through several
 11     question templates (MIP-0025 §4.3 Layer 1). Every answer is verbatim text from the file it
 12     cites; no model call generates anything.
 13  5. Tool-call SFT (MIP-0025 §4.3 Layer 2): questions → one JSON call to the four MCP tools in
 14     cli/src/main/scala/marola/agent/SwimConditionsMcpServer.scala. Teaches when to call, not facts.
 15
 16Output: finetune/data/train.jsonl and eval.jsonl in the chat format most trainers accept:
 17  {"messages": [{"role": "system", ...}, {"role": "user", ...}, {"role": "assistant", ...}]}
 18
 19Run:  python build_dataset.py            (or `just finetune-dataset`)
 20Self-test:  python build_dataset.py --self-test   (or `just quality-other`)
 21
 22`--resources DIR` / `--knowledge DIR` override where 1-3 and 4-5 above are read from — the app ->
 23ml contract (MIP-0070 §5.4): once marola-ml is a separate repo, `--resources` points at the
 24unpacked resources tarball ci.yml publishes, not `../core`. `--knowledge` defaults to
 25$MAROLA_KNOWLEDGE_DIR if set, else this repo's `.tmp/knowledge` from `just corpus-fetch` (anchored
 26like `--resources`, not the cwd — this file is documented to run from `finetune/`).
 27"""
 28
 29from __future__ import annotations
 30
 31import argparse
 32import json
 33import os
 34import random
 35import re
 36import subprocess
 37import tempfile
 38from pathlib import Path
 39
 40REPO = Path(__file__).resolve().parent.parent
 41RESOURCES = REPO / "core" / "src" / "main" / "resources"
 42OUT = Path(__file__).resolve().parent / "data"
 43
 44
 45def default_knowledge_dir() -> Path:
 46    env = os.environ.get("MAROLA_KNOWLEDGE_DIR")
 47    return Path(env) if env is not None else REPO / ".tmp" / "knowledge"
 48
 49
 50SYSTEM = (
 51    "You are marola, a swim-conditions assistant for open-water swimmers in Brazil. Answer in one "
 52    "or two plain sentences from the facts given; never invent conditions; cite a source URL when "
 53    "you quote your notes; return compact JSON and nothing else when asked for JSON."
 54)
 55
 56INPUT_FIELDS = (
 57    "beach_name",
 58    "hour_local",
 59    "sea_temp_c",
 60    "wind_kmh",
 61    "wave_height_m",
 62    "jellyfish_risk",
 63    "whale_sighting_likelihood",
 64    "score",
 65)
 66
 67
 68def render_inputs(demo: dict, fields: tuple[str, ...]) -> str:
 69    return "\n".join(f"{f.replace('_', ' ').title()}: {demo[f]}" for f in fields if f in demo)
 70
 71
 72def example(user: str, assistant: str) -> dict:
 73    return {
 74        "messages": [
 75            {"role": "system", "content": SYSTEM},
 76            {"role": "user", "content": user},
 77            {"role": "assistant", "content": assistant},
 78        ]
 79    }
 80
 81
 82def from_compiled_prompt(
 83    path: Path, output_field: str, extra_inputs: tuple[str, ...] = ()
 84) -> list[dict]:
 85    doc = json.loads(path.read_text(encoding="utf-8"))
 86    instructions = doc.get("signature", {}).get("instructions", "").strip()
 87    rows = []
 88    for demo in doc.get("demos", []):
 89        if output_field not in demo:
 90            continue
 91        user = (instructions + "\n\n" if instructions else "") + render_inputs(
 92            demo, INPUT_FIELDS + extra_inputs
 93        )
 94        rows.append(example(user, demo[output_field]))
 95    return rows
 96
 97
 98def from_sea_lore(path: Path) -> list[dict]:
 99    rows = []
100    for e in json.loads(path.read_text(encoding="utf-8")):
101        topic = e["id"].replace("-", " ")
102        user = f"Tell me something about the sea: {topic}."
103        rows.append(example(user, f"{e['text']} [source: {e['source']}]"))
104    return rows
105
106
107def chunk_markdown(md: str, max_chars: int = 700) -> tuple[str, str, list[str]]:
108    """Mirror of marola.knowledge.Corpus.chunkDocument: title, source, merged paragraphs."""
109    lines = md.splitlines()
110    title = next((ln[2:].strip() for ln in lines if ln.startswith("# ")), "")
111    source = next((ln[7:].strip() for ln in lines if ln.lower().startswith("source:")), "")
112    body = "\n".join(
113        ln for ln in lines if not (ln.startswith("# ") or ln.lower().startswith("source:"))
114    )
115    paras = [p.strip() for p in re.split(r"\n\s*\n", body) if p.strip()]
116    chunks: list[str] = []
117    for p in paras:
118        if chunks and len(chunks[-1]) + len(p) + 2 <= max_chars:
119            chunks[-1] = chunks[-1] + "\n\n" + p
120        else:
121            chunks.append(p)
122    return title, source, chunks
123
124
125# Chunk-level: the question is asked several ways, the answer is always the full real chunk.
126CHUNK_QUESTION_TEMPLATES = (
127    "What do your notes say about {title}? (part {part})",
128    "Tell me what you know about {title}, part {part}.",
129    "Summarize what your notes cover on {title} (part {part}).",
130    "Give me the part {part} details from your notes on {title}.",
131    "What's in section {part} of your {title} notes?",
132    "Explain {title} using what your notes say (part {part}).",
133    "I'm curious about {title} — what does part {part} of your notes cover?",
134    "Recap part {part} of your knowledge on {title}.",
135    "What have you got written down about {title}, specifically part {part}?",
136    "Walk me through part {part} of your {title} notes.",
137    "What's the part {part} summary of {title} in your notes?",
138    "According to your notes, what's true about {title} (part {part})?",
139    "Pull up part {part} of what you know about {title}.",
140    "What can you tell a swimmer about {title}? (part {part})",
141    "Notes check: {title}, part {part}.",
142    "Brief me on {title}, part {part}, from your sources.",
143    "What's documented about {title} in part {part} of your notes?",
144    "Give a plain-language rundown of {title}, part {part}.",
145    "What should I know about {title}? (see part {part} of your notes)",
146    "Show me part {part} of your {title} material.",
147)
148
149# Sentence-level: same idea, one real sentence at a time — finer-grained facts, still verbatim.
150SENTENCE_QUESTION_TEMPLATES = (
151    "Give me one specific fact from your notes on {title} (part {part}).",
152    "What's a detail from your {title} notes, part {part}?",
153    "Pick one fact from part {part} of your {title} notes.",
154    "What's something true about {title} from part {part} of your notes?",
155    "Quote one fact from your notes on {title} (part {part}).",
156    "Name a specific point from your {title} notes, part {part}.",
157    "What's one thing your notes say about {title}? (part {part})",
158    "Give a short, specific fact about {title} (part {part}).",
159    "What detail from part {part} of your {title} notes stands out?",
160    "Tell me a single fact from your {title} notes (part {part}).",
161    "What's one takeaway from part {part} of your {title} notes?",
162    "Share one specific fact about {title} (part {part}).",
163    "What's a concrete detail about {title}? (from part {part} of your notes)",
164    "Give me a fact, not a summary, about {title} (part {part}).",
165    "What does part {part} of your {title} notes say, in one fact?",
166    "Point to one fact in your {title} notes (part {part}).",
167    "What's a specific claim in part {part} of your {title} notes?",
168    "Give one sentence of fact about {title} (part {part}).",
169    "What's a single detail worth knowing about {title}? (part {part})",
170    "Extract one fact from part {part} of your {title} notes.",
171)
172
173
174def _split_sentences(text: str) -> list[str]:
175    """Real sentences only (>20 chars) so a stray abbreviation dot doesn't yield a fragment."""
176    return [s.strip() for s in re.split(r"(?<=[.!?])\s+", text) if len(s.strip()) > 20]
177
178
179def _chunk_examples(title: str, source: str, chunk: str, part: int) -> list[dict]:
180    answer = f"{chunk}\n\nSource: {source}"
181    return [
182        example(t.format(title=title.lower(), part=part), answer) for t in CHUNK_QUESTION_TEMPLATES
183    ]
184
185
186def _sentence_examples(title: str, source: str, chunk: str, part: int) -> list[dict]:
187    rows = []
188    for sentence in _split_sentences(chunk):
189        answer = f"{sentence}\n\nSource: {source}"
190        rows += [
191            example(t.format(title=title.lower(), part=part), answer)
192            for t in SENTENCE_QUESTION_TEMPLATES
193        ]
194    return rows
195
196
197def from_knowledge(dir_: Path) -> list[dict]:
198    rows = []
199    for path in sorted(dir_.rglob("*.md")):
200        title, source, chunks = chunk_markdown(path.read_text(encoding="utf-8"))
201        if not source:
202            continue  # README.md and anything else without a citable source
203        for i, chunk in enumerate(chunks):
204            part = i + 1
205            rows += _chunk_examples(title, source, chunk, part)
206            rows += _sentence_examples(title, source, chunk, part)
207    return rows
208
209
210def _load_knowledge_sources(dir_: Path) -> dict[str, str]:
211    """Map each knowledge doc's cited source URL to its raw file text, for provenance checks."""
212    out: dict[str, str] = {}
213    for path in sorted(dir_.rglob("*.md")):
214        raw = path.read_text(encoding="utf-8")
215        _, source, _ = chunk_markdown(raw)
216        if source:
217            out[source] = raw
218    return out
219
220
221KNOWLEDGE_EXAMPLE_FLOOR = 2000
222
223
224# --- Layer 2: tool-call SFT (MIP-0025 §4.3) --------------------------------------------------
225# Transcribed from SwimConditionsMcpServer.scala's tool schemas; change both together.
226TOOL_CALL_SYSTEM = (
227    "You are marola, a swim-conditions assistant for open-water swimmers in Brazil. When a "
228    "question needs live data you don't have, reply with exactly one JSON object of the shape "
229    '{"tool": "<name>", "arguments": {...}} and nothing else — never invent the result yourself.'
230)
231
232TOOL_SCHEMAS: dict[str, dict[str, tuple[str, ...]]] = {
233    "find_nearby_beaches": {"required": ("lat", "lon"), "optional": ("radius_km",)},
234    "get_swim_recommendation": {"required": ("lat", "lon"), "optional": ("radius_km",)},
235    "get_water_quality": {"required": ("lat", "lon"), "optional": ("radius_km",)},
236    "ask_ocean_question": {"required": ("question",), "optional": ()},
237}
238
239# cli/src/test/resources/site/board.json's beaches.
240LOCATIONS: tuple[tuple[float, float], ...] = ((-27.6296, -48.4487), (-27.4021, -48.4157))
241RADII: tuple[float | None, ...] = (None, 10.0, 20.0)
242
243FIND_BEACHES_TEMPLATES = (
244    "Find swim beaches near {lat}, {lon}.",
245    "What open-water beaches are close to {lat}, {lon}?",
246    "Search for beaches around latitude {lat}, longitude {lon}.",
247    "Are there any beaches near {lat}, {lon}?",
248    "List the beaches close to {lat}, {lon}.",
249    "I'm at {lat}, {lon} — any beaches nearby?",
250    "Beaches near {lat}, {lon}, please.",
251    "Which beaches are around {lat}, {lon}?",
252    "Show me swim spots close to {lat}, {lon}.",
253    "Look up beaches near coordinate {lat}, {lon}.",
254)
255
256RECOMMENDATION_TEMPLATES = (
257    "What's the best hour to swim tomorrow near {lat}, {lon}?",
258    "Give me tomorrow's swim conditions near {lat}, {lon}.",
259    "When should I swim tomorrow around {lat}, {lon}?",
260    "Best swim window tomorrow near {lat}, {lon}?",
261    "What are tomorrow's conditions like near {lat}, {lon}?",
262    "Recommend a swim time near {lat}, {lon} for tomorrow.",
263    "I want to swim tomorrow near {lat}, {lon} — when's best?",
264    "Tell me tomorrow's swimability near {lat}, {lon}.",
265    "Forecast tomorrow's swim conditions for {lat}, {lon}.",
266    "What hour has the best score near {lat}, {lon} tomorrow?",
267)
268
269WATER_QUALITY_TEMPLATES = (
270    "Is the water clean near {lat}, {lon}?",
271    "What's the bathing-water quality near {lat}, {lon}?",
272    "Check water quality around {lat}, {lon}.",
273    "Any water quality warnings near {lat}, {lon}?",
274    "Is it safe to swim near {lat}, {lon} — water quality-wise?",
275    "Give me the enterococci readings near {lat}, {lon}.",
276    "PRÓPRIA or IMPRÓPRIA near {lat}, {lon}?",
277    "What do the sampling points say near {lat}, {lon}?",
278    "Water quality check for {lat}, {lon}.",
279    "Has water near {lat}, {lon} been tested recently?",
280)
281
282ASK_QUESTION_TEMPLATES = (
283    "{question}",
284    "Hey marola, {question}",
285    "Quick question: {question}",
286    "{question} Please look it up.",
287    "I want to know: {question}",
288    "Can you answer this: {question}",
289    "{question} (from a swimmer prepping a trip)",
290    "Ocean question: {question}",
291    "{question} What does your knowledge base say?",
292    "Before I go for a swim, {question}",
293)
294
295
296def _tool_call_example(user: str, tool: str, arguments: dict) -> dict:
297    assistant = json.dumps({"tool": tool, "arguments": arguments})
298    return {
299        "messages": [
300            {"role": "system", "content": TOOL_CALL_SYSTEM},
301            {"role": "user", "content": user},
302            {"role": "assistant", "content": assistant},
303        ]
304    }
305
306
307def _latlon_tool_examples(tool: str, templates: tuple[str, ...]) -> list[dict]:
308    rows = []
309    for lat, lon in LOCATIONS:
310        for radius in RADII:
311            arguments: dict[str, float] = {"lat": lat, "lon": lon}
312            if radius is not None:
313                arguments["radius_km"] = radius
314            for t in templates:
315                user = t.format(lat=lat, lon=lon)
316                if radius is not None:
317                    user += f" Search within {radius:g} km."
318                rows.append(_tool_call_example(user, tool, arguments))
319    return rows
320
321
322def _ask_tool_examples(knowledge_dir: Path, sea_lore_path: Path) -> list[dict]:
323    questions = []
324    for path in sorted(knowledge_dir.rglob("*.md")):
325        title, source, _ = chunk_markdown(path.read_text(encoding="utf-8"))
326        if source:
327            questions.append(f"What do you know about {title.lower()}?")
328    if sea_lore_path.exists():
329        for e in json.loads(sea_lore_path.read_text(encoding="utf-8")):
330            questions.append(f"Tell me about {e['id'].replace('-', ' ')}.")
331    rows = []
332    for question in questions:
333        for t in ASK_QUESTION_TEMPLATES:
334            user = t.format(question=question)
335            rows.append(_tool_call_example(user, "ask_ocean_question", {"question": question}))
336    return rows
337
338
339def from_tool_calls(knowledge_dir: Path, sea_lore_path: Path) -> list[dict]:
340    rows = []
341    rows += _latlon_tool_examples("find_nearby_beaches", FIND_BEACHES_TEMPLATES)
342    rows += _latlon_tool_examples("get_swim_recommendation", RECOMMENDATION_TEMPLATES)
343    rows += _latlon_tool_examples("get_water_quality", WATER_QUALITY_TEMPLATES)
344    rows += _ask_tool_examples(knowledge_dir, sea_lore_path)
345    return rows
346
347
348def _self_test(knowledge: Path) -> None:
349    # A missing or empty corpus must stop main() before it writes a dataset without Layer 1.
350    global OUT
351    real_out = OUT
352    with tempfile.TemporaryDirectory() as tmp:
353        OUT = Path(tmp) / "out"
354        (Path(tmp) / "empty").mkdir()
355        for bad in (Path(tmp) / "missing", Path(tmp) / "empty"):
356            try:
357                main(RESOURCES, bad)
358            except SystemExit as e:
359                assert e.code not in (None, 0), f"main() exited 0 on corpus {bad}"
360            else:
361                raise AssertionError(f"main() built a dataset from corpus {bad}")
362        assert not OUT.exists(), "main() wrote a dataset before checking the corpus"
363    OUT = real_out
364
365    rows = from_knowledge(knowledge)
366    assert len(rows) >= KNOWLEDGE_EXAMPLE_FLOOR, (
367        f"knowledge-derived example count {len(rows)} below floor {KNOWLEDGE_EXAMPLE_FLOOR} "
368        "— MIP-0025 task 2 requires thousands, not dozens"
369    )
370    sources = _load_knowledge_sources(knowledge)
371    assert sources, "no knowledge sources found under " + str(knowledge)
372    for row in rows:
373        answer = row["messages"][2]["content"]
374        assert "\n\nSource: " in answer, f"synthetic example missing a Source line: {answer!r}"
375        fact, _, url = answer.rpartition("\n\nSource: ")
376        raw = sources.get(url)
377        assert raw is not None, f"cited source is not a real knowledge/*.md source: {url!r}"
378        assert fact in raw, (
379            f"synthetic fact not found verbatim in the file that cites {url} — looks invented: "
380            f"{fact[:80]!r}"
381        )
382
383    # The real app -> ml contract artifact (scripts/build-resources-tarball.sh), unpacked into a
384    # dir with no core/ sibling at all: proves --resources isn't secretly hardcoded to
385    # REPO/core/src/main/resources, and exercises the real sea_lore.json branch of
386    # _ask_tool_examples (MIP-0070 §5.4, task 3's own acceptance check).
387    with tempfile.TemporaryDirectory() as tmp:
388        tmp_path = Path(tmp)
389        tar_path = tmp_path / "resources.tar.gz"
390        subprocess.run(
391            [str(REPO / "scripts" / "build-resources-tarball.sh"), str(tar_path)],
392            cwd=REPO,
393            check=True,
394            capture_output=True,
395        )
396        unpacked = tmp_path / "unpacked"
397        unpacked.mkdir()
398        subprocess.run(["tar", "-xzf", str(tar_path), "-C", str(unpacked)], check=True)
399        sea_lore_path = unpacked / "sea_lore.json"
400        assert sea_lore_path.exists(), (
401            "build-resources-tarball.sh's output isn't flat at its root — "
402            "build_dataset.py --resources can't read it (MIP-0070 §5.4)"
403        )
404        no_lore_count = len(from_tool_calls(knowledge, tmp_path / "does-not-exist.json"))
405        tool_rows = from_tool_calls(knowledge, sea_lore_path)
406    assert len(tool_rows) > no_lore_count, (
407        f"the real sea_lore.json added no tool-call examples: {no_lore_count} without it, "
408        f"{len(tool_rows)} with it"
409    )
410    assert tool_rows, "no tool-call examples generated"
411    seen_tools: set[str] = set()
412    for row in tool_rows:
413        assistant = row["messages"][2]["content"]
414        parsed = json.loads(assistant)  # raises if not syntactically valid JSON
415        assert set(parsed.keys()) == {"tool", "arguments"}, f"unexpected shape: {parsed!r}"
416        tool = parsed["tool"]
417        assert tool in TOOL_SCHEMAS, f"not a real MCP tool name: {tool!r}"
418        seen_tools.add(tool)
419        schema = TOOL_SCHEMAS[tool]
420        args = parsed["arguments"]
421        assert isinstance(args, dict), f"{tool} arguments must be an object: {args!r}"
422        for req in schema["required"]:
423            assert req in args, f"{tool} call is missing required argument {req!r}: {args!r}"
424        allowed = set(schema["required"]) | set(schema["optional"])
425        for key in args:
426            assert key in allowed, f"{tool} call has an argument not in its real schema: {key!r}"
427    assert seen_tools == set(TOOL_SCHEMAS), (
428        f"missing tool coverage: {set(TOOL_SCHEMAS) - seen_tools}"
429    )
430    print(
431        f"self-test OK: {len(rows)} knowledge-derived examples "
432        f"(floor {KNOWLEDGE_EXAMPLE_FLOOR}), every fact verified verbatim against its cited "
433        f"source; {len(tool_rows)} tool-call examples ({no_lore_count} without sea_lore.json, "
434        f"{len(tool_rows)} with it) covering all {len(TOOL_SCHEMAS)} real MCP tools with "
435        f"syntactically valid call shapes"
436    )
437
438
439def main(resources: Path, knowledge: Path) -> None:
440    # from_knowledge() yields nothing for a missing dir: fail rather than train without Layer 1.
441    if not knowledge.is_dir() or not any(knowledge.rglob("*.md")):
442        raise SystemExit(
443            f"build_dataset: no knowledge/*.md under {knowledge} — run `just corpus-fetch`, "
444            "or point --knowledge / MAROLA_KNOWLEDGE_DIR at a corpus checkout"
445        )
446    rows: list[dict] = []
447    rows += from_compiled_prompt(resources / "recommendation_prompt.json", "summary")
448    rows += from_compiled_prompt(
449        resources / "review_prompt.json", "review_json", extra_inputs=("summary",)
450    )
451    rows += from_sea_lore(resources / "sea_lore.json")
452    rows += from_knowledge(knowledge)
453    rows += from_tool_calls(knowledge, resources / "sea_lore.json")
454
455    random.Random(42).shuffle(rows)
456    n_eval = max(2, len(rows) // 10)
457    eval_rows, train_rows = rows[:n_eval], rows[n_eval:]
458
459    OUT.mkdir(parents=True, exist_ok=True)
460    for name, data in (("train.jsonl", train_rows), ("eval.jsonl", eval_rows)):
461        with (OUT / name).open("w", encoding="utf-8") as f:
462            for r in data:
463                f.write(json.dumps(r, ensure_ascii=False) + "\n")
464    print(f"wrote {len(train_rows)} train + {len(eval_rows)} eval examples to {OUT}")
465
466
467def parse_args(argv: list[str] | None = None) -> argparse.Namespace:
468    ap = argparse.ArgumentParser(
469        description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
470    )
471    ap.add_argument(
472        "--resources",
473        type=Path,
474        default=RESOURCES,
475        help="dir with recommendation_prompt.json / review_prompt.json / sea_lore.json "
476        "(default: core/src/main/resources)",
477    )
478    ap.add_argument(
479        "--knowledge",
480        type=Path,
481        default=default_knowledge_dir(),
482        help="knowledge/*.md corpus dir (default: $MAROLA_KNOWLEDGE_DIR if set, else this "
483        "repo's .tmp/knowledge, anchored like --resources)",
484    )
485    ap.add_argument("--self-test", action="store_true")
486    return ap.parse_args(argv)
487
488
489if __name__ == "__main__":
490    args = parse_args()
491    if args.self_test:
492        _self_test(args.knowledge)
493    else:
494        main(args.resources, args.knowledge)
REPO = PosixPath('/home/runner/work/marola/marola')
RESOURCES = PosixPath('/home/runner/work/marola/marola/core/src/main/resources')
OUT = PosixPath('/home/runner/work/marola/marola/finetune/data')
def default_knowledge_dir() -> pathlib.Path:
46def default_knowledge_dir() -> Path:
47    env = os.environ.get("MAROLA_KNOWLEDGE_DIR")
48    return Path(env) if env is not None else REPO / ".tmp" / "knowledge"
SYSTEM = 'You are marola, a swim-conditions assistant for open-water swimmers in Brazil. Answer in one or two plain sentences from the facts given; never invent conditions; cite a source URL when you quote your notes; return compact JSON and nothing else when asked for JSON.'
INPUT_FIELDS = ('beach_name', 'hour_local', 'sea_temp_c', 'wind_kmh', 'wave_height_m', 'jellyfish_risk', 'whale_sighting_likelihood', 'score')
def render_inputs(demo: dict, fields: tuple[str, ...]) -> str:
69def render_inputs(demo: dict, fields: tuple[str, ...]) -> str:
70    return "\n".join(f"{f.replace('_', ' ').title()}: {demo[f]}" for f in fields if f in demo)
def example(user: str, assistant: str) -> dict:
73def example(user: str, assistant: str) -> dict:
74    return {
75        "messages": [
76            {"role": "system", "content": SYSTEM},
77            {"role": "user", "content": user},
78            {"role": "assistant", "content": assistant},
79        ]
80    }
def from_compiled_prompt( path: pathlib.Path, output_field: str, extra_inputs: tuple[str, ...] = ()) -> list[dict]:
83def from_compiled_prompt(
84    path: Path, output_field: str, extra_inputs: tuple[str, ...] = ()
85) -> list[dict]:
86    doc = json.loads(path.read_text(encoding="utf-8"))
87    instructions = doc.get("signature", {}).get("instructions", "").strip()
88    rows = []
89    for demo in doc.get("demos", []):
90        if output_field not in demo:
91            continue
92        user = (instructions + "\n\n" if instructions else "") + render_inputs(
93            demo, INPUT_FIELDS + extra_inputs
94        )
95        rows.append(example(user, demo[output_field]))
96    return rows
def from_sea_lore(path: pathlib.Path) -> list[dict]:
 99def from_sea_lore(path: Path) -> list[dict]:
100    rows = []
101    for e in json.loads(path.read_text(encoding="utf-8")):
102        topic = e["id"].replace("-", " ")
103        user = f"Tell me something about the sea: {topic}."
104        rows.append(example(user, f"{e['text']} [source: {e['source']}]"))
105    return rows
def chunk_markdown(md: str, max_chars: int = 700) -> tuple[str, str, list[str]]:
108def chunk_markdown(md: str, max_chars: int = 700) -> tuple[str, str, list[str]]:
109    """Mirror of marola.knowledge.Corpus.chunkDocument: title, source, merged paragraphs."""
110    lines = md.splitlines()
111    title = next((ln[2:].strip() for ln in lines if ln.startswith("# ")), "")
112    source = next((ln[7:].strip() for ln in lines if ln.lower().startswith("source:")), "")
113    body = "\n".join(
114        ln for ln in lines if not (ln.startswith("# ") or ln.lower().startswith("source:"))
115    )
116    paras = [p.strip() for p in re.split(r"\n\s*\n", body) if p.strip()]
117    chunks: list[str] = []
118    for p in paras:
119        if chunks and len(chunks[-1]) + len(p) + 2 <= max_chars:
120            chunks[-1] = chunks[-1] + "\n\n" + p
121        else:
122            chunks.append(p)
123    return title, source, chunks

Mirror of marola.knowledge.Corpus.chunkDocument: title, source, merged paragraphs.

CHUNK_QUESTION_TEMPLATES = ('What do your notes say about {title}? (part {part})', 'Tell me what you know about {title}, part {part}.', 'Summarize what your notes cover on {title} (part {part}).', 'Give me the part {part} details from your notes on {title}.', "What's in section {part} of your {title} notes?", 'Explain {title} using what your notes say (part {part}).', "I'm curious about {title} — what does part {part} of your notes cover?", 'Recap part {part} of your knowledge on {title}.', 'What have you got written down about {title}, specifically part {part}?', 'Walk me through part {part} of your {title} notes.', "What's the part {part} summary of {title} in your notes?", "According to your notes, what's true about {title} (part {part})?", 'Pull up part {part} of what you know about {title}.', 'What can you tell a swimmer about {title}? (part {part})', 'Notes check: {title}, part {part}.', 'Brief me on {title}, part {part}, from your sources.', "What's documented about {title} in part {part} of your notes?", 'Give a plain-language rundown of {title}, part {part}.', 'What should I know about {title}? (see part {part} of your notes)', 'Show me part {part} of your {title} material.')
SENTENCE_QUESTION_TEMPLATES = ('Give me one specific fact from your notes on {title} (part {part}).', "What's a detail from your {title} notes, part {part}?", 'Pick one fact from part {part} of your {title} notes.', "What's something true about {title} from part {part} of your notes?", 'Quote one fact from your notes on {title} (part {part}).', 'Name a specific point from your {title} notes, part {part}.', "What's one thing your notes say about {title}? (part {part})", 'Give a short, specific fact about {title} (part {part}).', 'What detail from part {part} of your {title} notes stands out?', 'Tell me a single fact from your {title} notes (part {part}).', "What's one takeaway from part {part} of your {title} notes?", 'Share one specific fact about {title} (part {part}).', "What's a concrete detail about {title}? (from part {part} of your notes)", 'Give me a fact, not a summary, about {title} (part {part}).', 'What does part {part} of your {title} notes say, in one fact?', 'Point to one fact in your {title} notes (part {part}).', "What's a specific claim in part {part} of your {title} notes?", 'Give one sentence of fact about {title} (part {part}).', "What's a single detail worth knowing about {title}? (part {part})", 'Extract one fact from part {part} of your {title} notes.')
def from_knowledge(dir_: pathlib.Path) -> list[dict]:
198def from_knowledge(dir_: Path) -> list[dict]:
199    rows = []
200    for path in sorted(dir_.rglob("*.md")):
201        title, source, chunks = chunk_markdown(path.read_text(encoding="utf-8"))
202        if not source:
203            continue  # README.md and anything else without a citable source
204        for i, chunk in enumerate(chunks):
205            part = i + 1
206            rows += _chunk_examples(title, source, chunk, part)
207            rows += _sentence_examples(title, source, chunk, part)
208    return rows
KNOWLEDGE_EXAMPLE_FLOOR = 2000
TOOL_CALL_SYSTEM = 'You are marola, a swim-conditions assistant for open-water swimmers in Brazil. When a question needs live data you don\'t have, reply with exactly one JSON object of the shape {"tool": "<name>", "arguments": {...}} and nothing else — never invent the result yourself.'
TOOL_SCHEMAS: dict[str, dict[str, tuple[str, ...]]] = {'find_nearby_beaches': {'required': ('lat', 'lon'), 'optional': ('radius_km',)}, 'get_swim_recommendation': {'required': ('lat', 'lon'), 'optional': ('radius_km',)}, 'get_water_quality': {'required': ('lat', 'lon'), 'optional': ('radius_km',)}, 'ask_ocean_question': {'required': ('question',), 'optional': ()}}
LOCATIONS: tuple[tuple[float, float], ...] = ((-27.6296, -48.4487), (-27.4021, -48.4157))
RADII: tuple[float | None, ...] = (None, 10.0, 20.0)
FIND_BEACHES_TEMPLATES = ('Find swim beaches near {lat}, {lon}.', 'What open-water beaches are close to {lat}, {lon}?', 'Search for beaches around latitude {lat}, longitude {lon}.', 'Are there any beaches near {lat}, {lon}?', 'List the beaches close to {lat}, {lon}.', "I'm at {lat}, {lon} — any beaches nearby?", 'Beaches near {lat}, {lon}, please.', 'Which beaches are around {lat}, {lon}?', 'Show me swim spots close to {lat}, {lon}.', 'Look up beaches near coordinate {lat}, {lon}.')
RECOMMENDATION_TEMPLATES = ("What's the best hour to swim tomorrow near {lat}, {lon}?", "Give me tomorrow's swim conditions near {lat}, {lon}.", 'When should I swim tomorrow around {lat}, {lon}?', 'Best swim window tomorrow near {lat}, {lon}?', "What are tomorrow's conditions like near {lat}, {lon}?", 'Recommend a swim time near {lat}, {lon} for tomorrow.', "I want to swim tomorrow near {lat}, {lon} — when's best?", "Tell me tomorrow's swimability near {lat}, {lon}.", "Forecast tomorrow's swim conditions for {lat}, {lon}.", 'What hour has the best score near {lat}, {lon} tomorrow?')
WATER_QUALITY_TEMPLATES = ('Is the water clean near {lat}, {lon}?', "What's the bathing-water quality near {lat}, {lon}?", 'Check water quality around {lat}, {lon}.', 'Any water quality warnings near {lat}, {lon}?', 'Is it safe to swim near {lat}, {lon} — water quality-wise?', 'Give me the enterococci readings near {lat}, {lon}.', 'PRÓPRIA or IMPRÓPRIA near {lat}, {lon}?', 'What do the sampling points say near {lat}, {lon}?', 'Water quality check for {lat}, {lon}.', 'Has water near {lat}, {lon} been tested recently?')
ASK_QUESTION_TEMPLATES = ('{question}', 'Hey marola, {question}', 'Quick question: {question}', '{question} Please look it up.', 'I want to know: {question}', 'Can you answer this: {question}', '{question} (from a swimmer prepping a trip)', 'Ocean question: {question}', '{question} What does your knowledge base say?', 'Before I go for a swim, {question}')
def from_tool_calls(knowledge_dir: pathlib.Path, sea_lore_path: pathlib.Path) -> list[dict]:
340def from_tool_calls(knowledge_dir: Path, sea_lore_path: Path) -> list[dict]:
341    rows = []
342    rows += _latlon_tool_examples("find_nearby_beaches", FIND_BEACHES_TEMPLATES)
343    rows += _latlon_tool_examples("get_swim_recommendation", RECOMMENDATION_TEMPLATES)
344    rows += _latlon_tool_examples("get_water_quality", WATER_QUALITY_TEMPLATES)
345    rows += _ask_tool_examples(knowledge_dir, sea_lore_path)
346    return rows
def main(resources: pathlib.Path, knowledge: pathlib.Path) -> None:
440def main(resources: Path, knowledge: Path) -> None:
441    # from_knowledge() yields nothing for a missing dir: fail rather than train without Layer 1.
442    if not knowledge.is_dir() or not any(knowledge.rglob("*.md")):
443        raise SystemExit(
444            f"build_dataset: no knowledge/*.md under {knowledge} — run `just corpus-fetch`, "
445            "or point --knowledge / MAROLA_KNOWLEDGE_DIR at a corpus checkout"
446        )
447    rows: list[dict] = []
448    rows += from_compiled_prompt(resources / "recommendation_prompt.json", "summary")
449    rows += from_compiled_prompt(
450        resources / "review_prompt.json", "review_json", extra_inputs=("summary",)
451    )
452    rows += from_sea_lore(resources / "sea_lore.json")
453    rows += from_knowledge(knowledge)
454    rows += from_tool_calls(knowledge, resources / "sea_lore.json")
455
456    random.Random(42).shuffle(rows)
457    n_eval = max(2, len(rows) // 10)
458    eval_rows, train_rows = rows[:n_eval], rows[n_eval:]
459
460    OUT.mkdir(parents=True, exist_ok=True)
461    for name, data in (("train.jsonl", train_rows), ("eval.jsonl", eval_rows)):
462        with (OUT / name).open("w", encoding="utf-8") as f:
463            for r in data:
464                f.write(json.dumps(r, ensure_ascii=False) + "\n")
465    print(f"wrote {len(train_rows)} train + {len(eval_rows)} eval examples to {OUT}")
def parse_args(argv: list[str] | None = None) -> argparse.Namespace:
468def parse_args(argv: list[str] | None = None) -> argparse.Namespace:
469    ap = argparse.ArgumentParser(
470        description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
471    )
472    ap.add_argument(
473        "--resources",
474        type=Path,
475        default=RESOURCES,
476        help="dir with recommendation_prompt.json / review_prompt.json / sea_lore.json "
477        "(default: core/src/main/resources)",
478    )
479    ap.add_argument(
480        "--knowledge",
481        type=Path,
482        default=default_knowledge_dir(),
483        help="knowledge/*.md corpus dir (default: $MAROLA_KNOWLEDGE_DIR if set, else this "
484        "repo's .tmp/knowledge, anchored like --resources)",
485    )
486    ap.add_argument("--self-test", action="store_true")
487    return ap.parse_args(argv)