build_dataset
Build marola's fine-tuning dataset from what the repo already has — no new labelling.
Sources (all local files, no network, no model calls):
- core/src/main/resources/recommendation_prompt.json — the DSPy-compiled summarizer demos: structured conditions in, one-to-two-sentence summary out. The core "shape" marola wants.
- core/src/main/resources/review_prompt.json — the reviewer demos: conditions + draft in, compact JSON verdict out. Teaches the JSON-only discipline small models break most.
- core/src/main/resources/sea_lore.json — "tell me something about X" → the sourced paragraph, with the source URL kept in the answer so the habit of citing survives.
- knowledge/*.md (recursively) — one Q/A per chunk and per sentence, each asked through several question templates (MIP-0025 §4.3 Layer 1). Every answer is verbatim text from the file it cites; no model call generates anything.
- Tool-call SFT (MIP-0025 §4.3 Layer 2): questions → one JSON call to the four MCP tools in cli/src/main/scala/marola/agent/SwimConditionsMcpServer.scala. Teaches when to call, not facts.
Output: finetune/data/train.jsonl and eval.jsonl in the chat format most trainers accept: {"messages": [{"role": "system", ...}, {"role": "user", ...}, {"role": "assistant", ...}]}
Run: python build_dataset.py (or just finetune-dataset)
Self-test: python build_dataset.py --self-test (or just quality-other)
--resources DIR / --knowledge DIR override where 1-3 and 4-5 above are read from — the app ->
ml contract (MIP-0070 §5.4): once marola-ml is a separate repo, --resources points at the
unpacked resources tarball ci.yml publishes, not ../core. --knowledge defaults to
$MAROLA_KNOWLEDGE_DIR if set, else this repo's .tmp/knowledge from just corpus-fetch (anchored
like --resources, not the cwd — this file is documented to run from finetune/).
1"""Build marola's fine-tuning dataset from what the repo already has — no new labelling. 2 3Sources (all local files, no network, no model calls): 4 1. core/src/main/resources/recommendation_prompt.json — the DSPy-compiled summarizer demos: 5 structured conditions in, one-to-two-sentence summary out. The core "shape" marola wants. 6 2. core/src/main/resources/review_prompt.json — the reviewer demos: conditions + draft in, 7 compact JSON verdict out. Teaches the JSON-only discipline small models break most. 8 3. core/src/main/resources/sea_lore.json — "tell me something about X" → the sourced paragraph, 9 with the source URL kept in the answer so the habit of citing survives. 10 4. knowledge/*.md (recursively) — one Q/A per chunk and per sentence, each asked through several 11 question templates (MIP-0025 §4.3 Layer 1). Every answer is verbatim text from the file it 12 cites; no model call generates anything. 13 5. Tool-call SFT (MIP-0025 §4.3 Layer 2): questions → one JSON call to the four MCP tools in 14 cli/src/main/scala/marola/agent/SwimConditionsMcpServer.scala. Teaches when to call, not facts. 15 16Output: finetune/data/train.jsonl and eval.jsonl in the chat format most trainers accept: 17 {"messages": [{"role": "system", ...}, {"role": "user", ...}, {"role": "assistant", ...}]} 18 19Run: python build_dataset.py (or `just finetune-dataset`) 20Self-test: python build_dataset.py --self-test (or `just quality-other`) 21 22`--resources DIR` / `--knowledge DIR` override where 1-3 and 4-5 above are read from — the app -> 23ml contract (MIP-0070 §5.4): once marola-ml is a separate repo, `--resources` points at the 24unpacked resources tarball ci.yml publishes, not `../core`. `--knowledge` defaults to 25$MAROLA_KNOWLEDGE_DIR if set, else this repo's `.tmp/knowledge` from `just corpus-fetch` (anchored 26like `--resources`, not the cwd — this file is documented to run from `finetune/`). 27""" 28 29from __future__ import annotations 30 31import argparse 32import json 33import os 34import random 35import re 36import subprocess 37import tempfile 38from pathlib import Path 39 40REPO = Path(__file__).resolve().parent.parent 41RESOURCES = REPO / "core" / "src" / "main" / "resources" 42OUT = Path(__file__).resolve().parent / "data" 43 44 45def default_knowledge_dir() -> Path: 46 env = os.environ.get("MAROLA_KNOWLEDGE_DIR") 47 return Path(env) if env is not None else REPO / ".tmp" / "knowledge" 48 49 50SYSTEM = ( 51 "You are marola, a swim-conditions assistant for open-water swimmers in Brazil. Answer in one " 52 "or two plain sentences from the facts given; never invent conditions; cite a source URL when " 53 "you quote your notes; return compact JSON and nothing else when asked for JSON." 54) 55 56INPUT_FIELDS = ( 57 "beach_name", 58 "hour_local", 59 "sea_temp_c", 60 "wind_kmh", 61 "wave_height_m", 62 "jellyfish_risk", 63 "whale_sighting_likelihood", 64 "score", 65) 66 67 68def render_inputs(demo: dict, fields: tuple[str, ...]) -> str: 69 return "\n".join(f"{f.replace('_', ' ').title()}: {demo[f]}" for f in fields if f in demo) 70 71 72def example(user: str, assistant: str) -> dict: 73 return { 74 "messages": [ 75 {"role": "system", "content": SYSTEM}, 76 {"role": "user", "content": user}, 77 {"role": "assistant", "content": assistant}, 78 ] 79 } 80 81 82def from_compiled_prompt( 83 path: Path, output_field: str, extra_inputs: tuple[str, ...] = () 84) -> list[dict]: 85 doc = json.loads(path.read_text(encoding="utf-8")) 86 instructions = doc.get("signature", {}).get("instructions", "").strip() 87 rows = [] 88 for demo in doc.get("demos", []): 89 if output_field not in demo: 90 continue 91 user = (instructions + "\n\n" if instructions else "") + render_inputs( 92 demo, INPUT_FIELDS + extra_inputs 93 ) 94 rows.append(example(user, demo[output_field])) 95 return rows 96 97 98def from_sea_lore(path: Path) -> list[dict]: 99 rows = [] 100 for e in json.loads(path.read_text(encoding="utf-8")): 101 topic = e["id"].replace("-", " ") 102 user = f"Tell me something about the sea: {topic}." 103 rows.append(example(user, f"{e['text']} [source: {e['source']}]")) 104 return rows 105 106 107def chunk_markdown(md: str, max_chars: int = 700) -> tuple[str, str, list[str]]: 108 """Mirror of marola.knowledge.Corpus.chunkDocument: title, source, merged paragraphs.""" 109 lines = md.splitlines() 110 title = next((ln[2:].strip() for ln in lines if ln.startswith("# ")), "") 111 source = next((ln[7:].strip() for ln in lines if ln.lower().startswith("source:")), "") 112 body = "\n".join( 113 ln for ln in lines if not (ln.startswith("# ") or ln.lower().startswith("source:")) 114 ) 115 paras = [p.strip() for p in re.split(r"\n\s*\n", body) if p.strip()] 116 chunks: list[str] = [] 117 for p in paras: 118 if chunks and len(chunks[-1]) + len(p) + 2 <= max_chars: 119 chunks[-1] = chunks[-1] + "\n\n" + p 120 else: 121 chunks.append(p) 122 return title, source, chunks 123 124 125# Chunk-level: the question is asked several ways, the answer is always the full real chunk. 126CHUNK_QUESTION_TEMPLATES = ( 127 "What do your notes say about {title}? (part {part})", 128 "Tell me what you know about {title}, part {part}.", 129 "Summarize what your notes cover on {title} (part {part}).", 130 "Give me the part {part} details from your notes on {title}.", 131 "What's in section {part} of your {title} notes?", 132 "Explain {title} using what your notes say (part {part}).", 133 "I'm curious about {title} — what does part {part} of your notes cover?", 134 "Recap part {part} of your knowledge on {title}.", 135 "What have you got written down about {title}, specifically part {part}?", 136 "Walk me through part {part} of your {title} notes.", 137 "What's the part {part} summary of {title} in your notes?", 138 "According to your notes, what's true about {title} (part {part})?", 139 "Pull up part {part} of what you know about {title}.", 140 "What can you tell a swimmer about {title}? (part {part})", 141 "Notes check: {title}, part {part}.", 142 "Brief me on {title}, part {part}, from your sources.", 143 "What's documented about {title} in part {part} of your notes?", 144 "Give a plain-language rundown of {title}, part {part}.", 145 "What should I know about {title}? (see part {part} of your notes)", 146 "Show me part {part} of your {title} material.", 147) 148 149# Sentence-level: same idea, one real sentence at a time — finer-grained facts, still verbatim. 150SENTENCE_QUESTION_TEMPLATES = ( 151 "Give me one specific fact from your notes on {title} (part {part}).", 152 "What's a detail from your {title} notes, part {part}?", 153 "Pick one fact from part {part} of your {title} notes.", 154 "What's something true about {title} from part {part} of your notes?", 155 "Quote one fact from your notes on {title} (part {part}).", 156 "Name a specific point from your {title} notes, part {part}.", 157 "What's one thing your notes say about {title}? (part {part})", 158 "Give a short, specific fact about {title} (part {part}).", 159 "What detail from part {part} of your {title} notes stands out?", 160 "Tell me a single fact from your {title} notes (part {part}).", 161 "What's one takeaway from part {part} of your {title} notes?", 162 "Share one specific fact about {title} (part {part}).", 163 "What's a concrete detail about {title}? (from part {part} of your notes)", 164 "Give me a fact, not a summary, about {title} (part {part}).", 165 "What does part {part} of your {title} notes say, in one fact?", 166 "Point to one fact in your {title} notes (part {part}).", 167 "What's a specific claim in part {part} of your {title} notes?", 168 "Give one sentence of fact about {title} (part {part}).", 169 "What's a single detail worth knowing about {title}? (part {part})", 170 "Extract one fact from part {part} of your {title} notes.", 171) 172 173 174def _split_sentences(text: str) -> list[str]: 175 """Real sentences only (>20 chars) so a stray abbreviation dot doesn't yield a fragment.""" 176 return [s.strip() for s in re.split(r"(?<=[.!?])\s+", text) if len(s.strip()) > 20] 177 178 179def _chunk_examples(title: str, source: str, chunk: str, part: int) -> list[dict]: 180 answer = f"{chunk}\n\nSource: {source}" 181 return [ 182 example(t.format(title=title.lower(), part=part), answer) for t in CHUNK_QUESTION_TEMPLATES 183 ] 184 185 186def _sentence_examples(title: str, source: str, chunk: str, part: int) -> list[dict]: 187 rows = [] 188 for sentence in _split_sentences(chunk): 189 answer = f"{sentence}\n\nSource: {source}" 190 rows += [ 191 example(t.format(title=title.lower(), part=part), answer) 192 for t in SENTENCE_QUESTION_TEMPLATES 193 ] 194 return rows 195 196 197def from_knowledge(dir_: Path) -> list[dict]: 198 rows = [] 199 for path in sorted(dir_.rglob("*.md")): 200 title, source, chunks = chunk_markdown(path.read_text(encoding="utf-8")) 201 if not source: 202 continue # README.md and anything else without a citable source 203 for i, chunk in enumerate(chunks): 204 part = i + 1 205 rows += _chunk_examples(title, source, chunk, part) 206 rows += _sentence_examples(title, source, chunk, part) 207 return rows 208 209 210def _load_knowledge_sources(dir_: Path) -> dict[str, str]: 211 """Map each knowledge doc's cited source URL to its raw file text, for provenance checks.""" 212 out: dict[str, str] = {} 213 for path in sorted(dir_.rglob("*.md")): 214 raw = path.read_text(encoding="utf-8") 215 _, source, _ = chunk_markdown(raw) 216 if source: 217 out[source] = raw 218 return out 219 220 221KNOWLEDGE_EXAMPLE_FLOOR = 2000 222 223 224# --- Layer 2: tool-call SFT (MIP-0025 §4.3) -------------------------------------------------- 225# Transcribed from SwimConditionsMcpServer.scala's tool schemas; change both together. 226TOOL_CALL_SYSTEM = ( 227 "You are marola, a swim-conditions assistant for open-water swimmers in Brazil. When a " 228 "question needs live data you don't have, reply with exactly one JSON object of the shape " 229 '{"tool": "<name>", "arguments": {...}} and nothing else — never invent the result yourself.' 230) 231 232TOOL_SCHEMAS: dict[str, dict[str, tuple[str, ...]]] = { 233 "find_nearby_beaches": {"required": ("lat", "lon"), "optional": ("radius_km",)}, 234 "get_swim_recommendation": {"required": ("lat", "lon"), "optional": ("radius_km",)}, 235 "get_water_quality": {"required": ("lat", "lon"), "optional": ("radius_km",)}, 236 "ask_ocean_question": {"required": ("question",), "optional": ()}, 237} 238 239# cli/src/test/resources/site/board.json's beaches. 240LOCATIONS: tuple[tuple[float, float], ...] = ((-27.6296, -48.4487), (-27.4021, -48.4157)) 241RADII: tuple[float | None, ...] = (None, 10.0, 20.0) 242 243FIND_BEACHES_TEMPLATES = ( 244 "Find swim beaches near {lat}, {lon}.", 245 "What open-water beaches are close to {lat}, {lon}?", 246 "Search for beaches around latitude {lat}, longitude {lon}.", 247 "Are there any beaches near {lat}, {lon}?", 248 "List the beaches close to {lat}, {lon}.", 249 "I'm at {lat}, {lon} — any beaches nearby?", 250 "Beaches near {lat}, {lon}, please.", 251 "Which beaches are around {lat}, {lon}?", 252 "Show me swim spots close to {lat}, {lon}.", 253 "Look up beaches near coordinate {lat}, {lon}.", 254) 255 256RECOMMENDATION_TEMPLATES = ( 257 "What's the best hour to swim tomorrow near {lat}, {lon}?", 258 "Give me tomorrow's swim conditions near {lat}, {lon}.", 259 "When should I swim tomorrow around {lat}, {lon}?", 260 "Best swim window tomorrow near {lat}, {lon}?", 261 "What are tomorrow's conditions like near {lat}, {lon}?", 262 "Recommend a swim time near {lat}, {lon} for tomorrow.", 263 "I want to swim tomorrow near {lat}, {lon} — when's best?", 264 "Tell me tomorrow's swimability near {lat}, {lon}.", 265 "Forecast tomorrow's swim conditions for {lat}, {lon}.", 266 "What hour has the best score near {lat}, {lon} tomorrow?", 267) 268 269WATER_QUALITY_TEMPLATES = ( 270 "Is the water clean near {lat}, {lon}?", 271 "What's the bathing-water quality near {lat}, {lon}?", 272 "Check water quality around {lat}, {lon}.", 273 "Any water quality warnings near {lat}, {lon}?", 274 "Is it safe to swim near {lat}, {lon} — water quality-wise?", 275 "Give me the enterococci readings near {lat}, {lon}.", 276 "PRÓPRIA or IMPRÓPRIA near {lat}, {lon}?", 277 "What do the sampling points say near {lat}, {lon}?", 278 "Water quality check for {lat}, {lon}.", 279 "Has water near {lat}, {lon} been tested recently?", 280) 281 282ASK_QUESTION_TEMPLATES = ( 283 "{question}", 284 "Hey marola, {question}", 285 "Quick question: {question}", 286 "{question} Please look it up.", 287 "I want to know: {question}", 288 "Can you answer this: {question}", 289 "{question} (from a swimmer prepping a trip)", 290 "Ocean question: {question}", 291 "{question} What does your knowledge base say?", 292 "Before I go for a swim, {question}", 293) 294 295 296def _tool_call_example(user: str, tool: str, arguments: dict) -> dict: 297 assistant = json.dumps({"tool": tool, "arguments": arguments}) 298 return { 299 "messages": [ 300 {"role": "system", "content": TOOL_CALL_SYSTEM}, 301 {"role": "user", "content": user}, 302 {"role": "assistant", "content": assistant}, 303 ] 304 } 305 306 307def _latlon_tool_examples(tool: str, templates: tuple[str, ...]) -> list[dict]: 308 rows = [] 309 for lat, lon in LOCATIONS: 310 for radius in RADII: 311 arguments: dict[str, float] = {"lat": lat, "lon": lon} 312 if radius is not None: 313 arguments["radius_km"] = radius 314 for t in templates: 315 user = t.format(lat=lat, lon=lon) 316 if radius is not None: 317 user += f" Search within {radius:g} km." 318 rows.append(_tool_call_example(user, tool, arguments)) 319 return rows 320 321 322def _ask_tool_examples(knowledge_dir: Path, sea_lore_path: Path) -> list[dict]: 323 questions = [] 324 for path in sorted(knowledge_dir.rglob("*.md")): 325 title, source, _ = chunk_markdown(path.read_text(encoding="utf-8")) 326 if source: 327 questions.append(f"What do you know about {title.lower()}?") 328 if sea_lore_path.exists(): 329 for e in json.loads(sea_lore_path.read_text(encoding="utf-8")): 330 questions.append(f"Tell me about {e['id'].replace('-', ' ')}.") 331 rows = [] 332 for question in questions: 333 for t in ASK_QUESTION_TEMPLATES: 334 user = t.format(question=question) 335 rows.append(_tool_call_example(user, "ask_ocean_question", {"question": question})) 336 return rows 337 338 339def from_tool_calls(knowledge_dir: Path, sea_lore_path: Path) -> list[dict]: 340 rows = [] 341 rows += _latlon_tool_examples("find_nearby_beaches", FIND_BEACHES_TEMPLATES) 342 rows += _latlon_tool_examples("get_swim_recommendation", RECOMMENDATION_TEMPLATES) 343 rows += _latlon_tool_examples("get_water_quality", WATER_QUALITY_TEMPLATES) 344 rows += _ask_tool_examples(knowledge_dir, sea_lore_path) 345 return rows 346 347 348def _self_test(knowledge: Path) -> None: 349 # A missing or empty corpus must stop main() before it writes a dataset without Layer 1. 350 global OUT 351 real_out = OUT 352 with tempfile.TemporaryDirectory() as tmp: 353 OUT = Path(tmp) / "out" 354 (Path(tmp) / "empty").mkdir() 355 for bad in (Path(tmp) / "missing", Path(tmp) / "empty"): 356 try: 357 main(RESOURCES, bad) 358 except SystemExit as e: 359 assert e.code not in (None, 0), f"main() exited 0 on corpus {bad}" 360 else: 361 raise AssertionError(f"main() built a dataset from corpus {bad}") 362 assert not OUT.exists(), "main() wrote a dataset before checking the corpus" 363 OUT = real_out 364 365 rows = from_knowledge(knowledge) 366 assert len(rows) >= KNOWLEDGE_EXAMPLE_FLOOR, ( 367 f"knowledge-derived example count {len(rows)} below floor {KNOWLEDGE_EXAMPLE_FLOOR} " 368 "— MIP-0025 task 2 requires thousands, not dozens" 369 ) 370 sources = _load_knowledge_sources(knowledge) 371 assert sources, "no knowledge sources found under " + str(knowledge) 372 for row in rows: 373 answer = row["messages"][2]["content"] 374 assert "\n\nSource: " in answer, f"synthetic example missing a Source line: {answer!r}" 375 fact, _, url = answer.rpartition("\n\nSource: ") 376 raw = sources.get(url) 377 assert raw is not None, f"cited source is not a real knowledge/*.md source: {url!r}" 378 assert fact in raw, ( 379 f"synthetic fact not found verbatim in the file that cites {url} — looks invented: " 380 f"{fact[:80]!r}" 381 ) 382 383 # The real app -> ml contract artifact (scripts/build-resources-tarball.sh), unpacked into a 384 # dir with no core/ sibling at all: proves --resources isn't secretly hardcoded to 385 # REPO/core/src/main/resources, and exercises the real sea_lore.json branch of 386 # _ask_tool_examples (MIP-0070 §5.4, task 3's own acceptance check). 387 with tempfile.TemporaryDirectory() as tmp: 388 tmp_path = Path(tmp) 389 tar_path = tmp_path / "resources.tar.gz" 390 subprocess.run( 391 [str(REPO / "scripts" / "build-resources-tarball.sh"), str(tar_path)], 392 cwd=REPO, 393 check=True, 394 capture_output=True, 395 ) 396 unpacked = tmp_path / "unpacked" 397 unpacked.mkdir() 398 subprocess.run(["tar", "-xzf", str(tar_path), "-C", str(unpacked)], check=True) 399 sea_lore_path = unpacked / "sea_lore.json" 400 assert sea_lore_path.exists(), ( 401 "build-resources-tarball.sh's output isn't flat at its root — " 402 "build_dataset.py --resources can't read it (MIP-0070 §5.4)" 403 ) 404 no_lore_count = len(from_tool_calls(knowledge, tmp_path / "does-not-exist.json")) 405 tool_rows = from_tool_calls(knowledge, sea_lore_path) 406 assert len(tool_rows) > no_lore_count, ( 407 f"the real sea_lore.json added no tool-call examples: {no_lore_count} without it, " 408 f"{len(tool_rows)} with it" 409 ) 410 assert tool_rows, "no tool-call examples generated" 411 seen_tools: set[str] = set() 412 for row in tool_rows: 413 assistant = row["messages"][2]["content"] 414 parsed = json.loads(assistant) # raises if not syntactically valid JSON 415 assert set(parsed.keys()) == {"tool", "arguments"}, f"unexpected shape: {parsed!r}" 416 tool = parsed["tool"] 417 assert tool in TOOL_SCHEMAS, f"not a real MCP tool name: {tool!r}" 418 seen_tools.add(tool) 419 schema = TOOL_SCHEMAS[tool] 420 args = parsed["arguments"] 421 assert isinstance(args, dict), f"{tool} arguments must be an object: {args!r}" 422 for req in schema["required"]: 423 assert req in args, f"{tool} call is missing required argument {req!r}: {args!r}" 424 allowed = set(schema["required"]) | set(schema["optional"]) 425 for key in args: 426 assert key in allowed, f"{tool} call has an argument not in its real schema: {key!r}" 427 assert seen_tools == set(TOOL_SCHEMAS), ( 428 f"missing tool coverage: {set(TOOL_SCHEMAS) - seen_tools}" 429 ) 430 print( 431 f"self-test OK: {len(rows)} knowledge-derived examples " 432 f"(floor {KNOWLEDGE_EXAMPLE_FLOOR}), every fact verified verbatim against its cited " 433 f"source; {len(tool_rows)} tool-call examples ({no_lore_count} without sea_lore.json, " 434 f"{len(tool_rows)} with it) covering all {len(TOOL_SCHEMAS)} real MCP tools with " 435 f"syntactically valid call shapes" 436 ) 437 438 439def main(resources: Path, knowledge: Path) -> None: 440 # from_knowledge() yields nothing for a missing dir: fail rather than train without Layer 1. 441 if not knowledge.is_dir() or not any(knowledge.rglob("*.md")): 442 raise SystemExit( 443 f"build_dataset: no knowledge/*.md under {knowledge} — run `just corpus-fetch`, " 444 "or point --knowledge / MAROLA_KNOWLEDGE_DIR at a corpus checkout" 445 ) 446 rows: list[dict] = [] 447 rows += from_compiled_prompt(resources / "recommendation_prompt.json", "summary") 448 rows += from_compiled_prompt( 449 resources / "review_prompt.json", "review_json", extra_inputs=("summary",) 450 ) 451 rows += from_sea_lore(resources / "sea_lore.json") 452 rows += from_knowledge(knowledge) 453 rows += from_tool_calls(knowledge, resources / "sea_lore.json") 454 455 random.Random(42).shuffle(rows) 456 n_eval = max(2, len(rows) // 10) 457 eval_rows, train_rows = rows[:n_eval], rows[n_eval:] 458 459 OUT.mkdir(parents=True, exist_ok=True) 460 for name, data in (("train.jsonl", train_rows), ("eval.jsonl", eval_rows)): 461 with (OUT / name).open("w", encoding="utf-8") as f: 462 for r in data: 463 f.write(json.dumps(r, ensure_ascii=False) + "\n") 464 print(f"wrote {len(train_rows)} train + {len(eval_rows)} eval examples to {OUT}") 465 466 467def parse_args(argv: list[str] | None = None) -> argparse.Namespace: 468 ap = argparse.ArgumentParser( 469 description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter 470 ) 471 ap.add_argument( 472 "--resources", 473 type=Path, 474 default=RESOURCES, 475 help="dir with recommendation_prompt.json / review_prompt.json / sea_lore.json " 476 "(default: core/src/main/resources)", 477 ) 478 ap.add_argument( 479 "--knowledge", 480 type=Path, 481 default=default_knowledge_dir(), 482 help="knowledge/*.md corpus dir (default: $MAROLA_KNOWLEDGE_DIR if set, else this " 483 "repo's .tmp/knowledge, anchored like --resources)", 484 ) 485 ap.add_argument("--self-test", action="store_true") 486 return ap.parse_args(argv) 487 488 489if __name__ == "__main__": 490 args = parse_args() 491 if args.self_test: 492 _self_test(args.knowledge) 493 else: 494 main(args.resources, args.knowledge)
83def from_compiled_prompt( 84 path: Path, output_field: str, extra_inputs: tuple[str, ...] = () 85) -> list[dict]: 86 doc = json.loads(path.read_text(encoding="utf-8")) 87 instructions = doc.get("signature", {}).get("instructions", "").strip() 88 rows = [] 89 for demo in doc.get("demos", []): 90 if output_field not in demo: 91 continue 92 user = (instructions + "\n\n" if instructions else "") + render_inputs( 93 demo, INPUT_FIELDS + extra_inputs 94 ) 95 rows.append(example(user, demo[output_field])) 96 return rows
99def from_sea_lore(path: Path) -> list[dict]: 100 rows = [] 101 for e in json.loads(path.read_text(encoding="utf-8")): 102 topic = e["id"].replace("-", " ") 103 user = f"Tell me something about the sea: {topic}." 104 rows.append(example(user, f"{e['text']} [source: {e['source']}]")) 105 return rows
108def chunk_markdown(md: str, max_chars: int = 700) -> tuple[str, str, list[str]]: 109 """Mirror of marola.knowledge.Corpus.chunkDocument: title, source, merged paragraphs.""" 110 lines = md.splitlines() 111 title = next((ln[2:].strip() for ln in lines if ln.startswith("# ")), "") 112 source = next((ln[7:].strip() for ln in lines if ln.lower().startswith("source:")), "") 113 body = "\n".join( 114 ln for ln in lines if not (ln.startswith("# ") or ln.lower().startswith("source:")) 115 ) 116 paras = [p.strip() for p in re.split(r"\n\s*\n", body) if p.strip()] 117 chunks: list[str] = [] 118 for p in paras: 119 if chunks and len(chunks[-1]) + len(p) + 2 <= max_chars: 120 chunks[-1] = chunks[-1] + "\n\n" + p 121 else: 122 chunks.append(p) 123 return title, source, chunks
Mirror of marola.knowledge.Corpus.chunkDocument: title, source, merged paragraphs.
198def from_knowledge(dir_: Path) -> list[dict]: 199 rows = [] 200 for path in sorted(dir_.rglob("*.md")): 201 title, source, chunks = chunk_markdown(path.read_text(encoding="utf-8")) 202 if not source: 203 continue # README.md and anything else without a citable source 204 for i, chunk in enumerate(chunks): 205 part = i + 1 206 rows += _chunk_examples(title, source, chunk, part) 207 rows += _sentence_examples(title, source, chunk, part) 208 return rows
340def from_tool_calls(knowledge_dir: Path, sea_lore_path: Path) -> list[dict]: 341 rows = [] 342 rows += _latlon_tool_examples("find_nearby_beaches", FIND_BEACHES_TEMPLATES) 343 rows += _latlon_tool_examples("get_swim_recommendation", RECOMMENDATION_TEMPLATES) 344 rows += _latlon_tool_examples("get_water_quality", WATER_QUALITY_TEMPLATES) 345 rows += _ask_tool_examples(knowledge_dir, sea_lore_path) 346 return rows
440def main(resources: Path, knowledge: Path) -> None: 441 # from_knowledge() yields nothing for a missing dir: fail rather than train without Layer 1. 442 if not knowledge.is_dir() or not any(knowledge.rglob("*.md")): 443 raise SystemExit( 444 f"build_dataset: no knowledge/*.md under {knowledge} — run `just corpus-fetch`, " 445 "or point --knowledge / MAROLA_KNOWLEDGE_DIR at a corpus checkout" 446 ) 447 rows: list[dict] = [] 448 rows += from_compiled_prompt(resources / "recommendation_prompt.json", "summary") 449 rows += from_compiled_prompt( 450 resources / "review_prompt.json", "review_json", extra_inputs=("summary",) 451 ) 452 rows += from_sea_lore(resources / "sea_lore.json") 453 rows += from_knowledge(knowledge) 454 rows += from_tool_calls(knowledge, resources / "sea_lore.json") 455 456 random.Random(42).shuffle(rows) 457 n_eval = max(2, len(rows) // 10) 458 eval_rows, train_rows = rows[:n_eval], rows[n_eval:] 459 460 OUT.mkdir(parents=True, exist_ok=True) 461 for name, data in (("train.jsonl", train_rows), ("eval.jsonl", eval_rows)): 462 with (OUT / name).open("w", encoding="utf-8") as f: 463 for r in data: 464 f.write(json.dumps(r, ensure_ascii=False) + "\n") 465 print(f"wrote {len(train_rows)} train + {len(eval_rows)} eval examples to {OUT}")
468def parse_args(argv: list[str] | None = None) -> argparse.Namespace: 469 ap = argparse.ArgumentParser( 470 description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter 471 ) 472 ap.add_argument( 473 "--resources", 474 type=Path, 475 default=RESOURCES, 476 help="dir with recommendation_prompt.json / review_prompt.json / sea_lore.json " 477 "(default: core/src/main/resources)", 478 ) 479 ap.add_argument( 480 "--knowledge", 481 type=Path, 482 default=default_knowledge_dir(), 483 help="knowledge/*.md corpus dir (default: $MAROLA_KNOWLEDGE_DIR if set, else this " 484 "repo's .tmp/knowledge, anchored like --resources)", 485 ) 486 ap.add_argument("--self-test", action="store_true") 487 return ap.parse_args(argv)