"""Parse and re-emit the Skill Bible prose that lives in skill_bible.json. The `notes` field of each skill is a plain-text sheet with a fixed shape. This module converts between that string and a structured dict, losslessly, so the prose can live in writing/voices/*.md and still be folded back into the JSON. Round-tripping is the whole contract: notes -> dict -> notes must be identity for every skill in the bible, and check_roundtrip() asserts exactly that. """ import re PROSE_KEYS = [ "name", "domain", "what it wants for you", "what it's wrong about", "how it addresses you", "register", ] # "verbal tics" is handled separately (it carries bullets) TAIL_KEYS = ["what it never does", "opposed to", "allied with"] VOICE_RE = re.compile(r'^(SUCCESS|FAILURE|PASSIVE) VOICE(?: — (.*?))?: (.*)$') FIRING_RE = re.compile(r'^Firing frequency target \(per hour\): (\d+)$') FORBID_RE = re.compile(r'^Domains it must NOT comment on: (.*)$') BULLET_RE = re.compile(r'^\* \((\d+)\) (.*)$') def parse_notes(notes): """notes string -> dict. Raises on anything unrecognised.""" out = {"fields": {}, "tics": [], "voices": {}, "voice_gloss": {}} lines = notes.split("\n") i = 0 if lines[i].strip() != "SKILL": raise ValueError("expected a leading SKILL line") out["_header"] = lines[i] i += 1 while i < len(lines): line = lines[i] if line.strip() == "": i += 1 continue m = BULLET_RE.match(line) if m: out["tics"].append((int(m.group(1)), m.group(2))) i += 1 continue m = VOICE_RE.match(line) if m: kind = m.group(1).lower() out["voices"][kind] = m.group(3) if m.group(2): out["voice_gloss"][kind] = m.group(2) i += 1 continue m = FIRING_RE.match(line) if m: out["firing"] = int(m.group(1)) i += 1 continue m = FORBID_RE.match(line) if m: out["forbidden"] = m.group(1) i += 1 continue m = re.match(r"^([a-z][a-z '\-]*): ?(.*)$", line) if m and m.group(1) in PROSE_KEYS + TAIL_KEYS + ["verbal tics"]: out["fields"][m.group(1)] = m.group(2) i += 1 continue raise ValueError(f"unrecognised line: {line!r}") return out def emit_notes(d): """dict -> notes string. Inverse of parse_notes.""" L = [d.get("_header", "SKILL")] for k in PROSE_KEYS: L.append(f"{k}: {d['fields'][k]}") L.append("verbal tics:") for n, text in d["tics"]: L.append(f"* ({n}) {text}") for k in TAIL_KEYS: L.append(f"{k}: {d['fields'][k]}") L.append("") for kind in ("success", "failure", "passive"): gloss = d["voice_gloss"].get(kind) label = f"{kind.upper()} VOICE" if gloss: L.append(f"{label} — {gloss}: {d['voices'][kind]}") else: L.append(f"{label}: {d['voices'][kind]}") L.append("") L.append(f"Firing frequency target (per hour): {d['firing']}") L.append(f"Domains it must NOT comment on: {d['forbidden']}") return "\n".join(L) def check_roundtrip(skills): """Assert notes -> dict -> notes is identity for every skill.""" bad = [] for s in skills: try: again = emit_notes(parse_notes(s["notes"])) except Exception as e: bad.append((s["id"], f"parse error: {e}")) continue if again != s["notes"]: bad.append((s["id"], "round-trip differs")) return bad