"""Build per-character voice wikis from English_JSON dialogue via a local Ollama model. For each character indexed in character_data/{id}.json, processes the character's "mentions" files one at a time. Each file gets its own stateless LLM call (a fresh context window): the model receives the current wiki plus one file and returns the updated wiki. The wiki is saved after every file and progress is tracked in character_wikis/.progress.json, so an interrupted run resumes where it left off. Usage: python build_character_wikis.py # all characters (resumes) python build_character_wikis.py --character brad # single character python build_character_wikis.py --force # rebuild ignoring progress """ import argparse import json import sys import time import urllib.error import urllib.request from pathlib import Path ROOT = Path(__file__).parent CHAR_DIR = ROOT / "character_data" OUT_DIR = ROOT / "character_wikis" PROGRESS_PATH = OUT_DIR / ".progress.json" OLLAMA_URL = "http://localhost:11434/api/chat" MODEL = "gemma4:12b-mlx" # Gemma's recommended sampling parameters. 32k ctx comfortably fits the largest # mention file (~13k tokens incl. prompt + wiki) without over-allocating KV cache. MODEL_OPTIONS = {"temperature": 1, "top_k": 64, "top_p": 0.95, "num_ctx": 32768} REQUEST_TIMEOUT_S = 1800 MAX_ATTEMPTS = 3 RETRY_BACKOFF_S = 5 SYSTEM_PROMPT = """\ You are building a character voice wiki for {name}{aka}, a character in "Riverstone", \ a narrative mystery mobile game told through phone chats and calls. Purpose of the \ wiki: a translator LLM will later receive a translated line of {name}'s dialogue plus \ this wiki, and adjust the translation to sound like {name} rather than a generic, \ robotic rendering. Record only what serves that goal. You are given the current wiki and one game file. Update the wiki with any NEW \ evidence the file provides about: - Tone & style: formal/casual, warmth, sarcasm, energy, politeness. - Speech patterns: slang, catchphrases, filler words, abbreviations, \ emoji/punctuation habits, typical message length, grammar quirks. - Personality & relationships, insofar as they shape how {name} speaks and to whom. - Example lines: a few short verbatim quotes that best capture the voice. Rules: - Use only evidence from the file. Never invent details. - Strings starting with "system::" are engine keys, and "#0"-style keys, "next" and \ "options_next" values are navigation IDs - not dialogue. Ignore them. - If the file adds nothing new about {name}, return the wiki unchanged. - Keep the wiki under 400 words, in markdown, under exactly these headings: # {name} ## Role & Relationships ## Personality ## Tone & Style ## Speech Patterns ## Example Lines - Output ONLY the complete updated wiki. No commentary before or after. """ USER_TEMPLATE = """\ CURRENT WIKI: {wiki} FILE: {path} {hint} FILE CONTENT: {content} """ EMPTY_WIKI_PLACEHOLDER = "(empty - this is the first file)" def attribution_hint(rel_path: str, name: str) -> str: """Tell the model who speaks which fields, based on the file's location.""" parts = Path(rel_path).parts if "Adam_s Phone" in parts: owner = "Detective Adam (the player)" elif "Zoey_s Phone" in parts: owner = "Zoey" else: owner = "the player" if "Conversations" in parts: return ( f'In this branching dialogue, "message" values are spoken by {name}; ' f'"options" values are reply choices spoken by {owner}, the phone\'s ' f"owner - not {name}." ) if "Filler Chats" in parts: return ( f"This is a chat log on {owner}'s phone between {owner} and {name} " f'(group chats may include others). Different "type" values mark ' f"different speakers; attribute a line to {name} only when context " f"makes the speaker clear." ) return ( f"Extract only what this file reveals about {name}; do not attribute " f"other speakers' lines to them." ) def call_ollama(system_prompt: str, user_msg: str) -> str: payload = json.dumps( { "model": MODEL, "messages": [ {"role": "system", "content": system_prompt}, {"role": "user", "content": user_msg}, ], "stream": False, # gemma4 has thinking enabled by default, which multiplies generation # time several-fold per call; the wiki task doesn't need it. "think": False, "options": MODEL_OPTIONS, } ).encode("utf-8") last_error: Exception | None = None for attempt in range(1, MAX_ATTEMPTS + 1): request = urllib.request.Request( OLLAMA_URL, data=payload, headers={"Content-Type": "application/json"} ) try: with urllib.request.urlopen(request, timeout=REQUEST_TIMEOUT_S) as resp: data = json.loads(resp.read().decode("utf-8")) return data["message"]["content"] except (urllib.error.URLError, TimeoutError, KeyError, json.JSONDecodeError) as exc: last_error = exc if attempt < MAX_ATTEMPTS: print(f" request failed ({exc}); retrying in {RETRY_BACKOFF_S}s...") time.sleep(RETRY_BACKOFF_S) raise RuntimeError( f"Ollama request failed after {MAX_ATTEMPTS} attempts: {last_error}. " f"Is the server running at {OLLAMA_URL}? Progress is saved; rerun to resume." ) def clean_reply(text: str) -> str: """Strip whitespace and an optional wrapping markdown code fence.""" text = text.strip() if text.startswith("```") and text.endswith("```"): first_newline = text.find("\n") if first_newline != -1: text = text[first_newline + 1 : -3].strip() return text def load_progress() -> dict: if PROGRESS_PATH.exists(): return json.loads(PROGRESS_PATH.read_text(encoding="utf-8")) return {} def save_progress(progress: dict) -> None: PROGRESS_PATH.write_text( json.dumps(progress, indent=2, ensure_ascii=False), encoding="utf-8" ) def process_character(char_data: dict, progress: dict, force: bool) -> dict: """Run the per-file wiki-update loop for one character; returns new progress.""" char_id = char_data["id"] names = char_data.get("name") or [] display_name = names[0] if names else char_id aka = f' (also appearing as {", ".join(names[1:])})' if len(names) > 1 else "" mentions = char_data.get("mentions", []) if not mentions: print(f"[{char_id}] no mention files - skipped") return progress wiki_path = OUT_DIR / f"{char_id}.md" done = [] if force else list(progress.get(char_id, [])) wiki = "" if not force and wiki_path.exists(): wiki = wiki_path.read_text(encoding="utf-8").strip() pending = [m for m in mentions if m not in done] if not pending: print(f"[{char_id}] already complete ({len(mentions)} files)") return progress system_prompt = SYSTEM_PROMPT.format(name=display_name, aka=aka) print(f"[{char_id}] {len(pending)} file(s) to process") for i, rel_path in enumerate(pending, 1): file_path = ROOT / rel_path try: data = json.loads(file_path.read_text(encoding="utf-8")) except (OSError, json.JSONDecodeError) as exc: print(f" ({i}/{len(pending)}) WARNING: cannot read {rel_path}: {exc}") done = done + [rel_path] progress = {**progress, char_id: done} save_progress(progress) continue user_msg = USER_TEMPLATE.format( wiki=wiki or EMPTY_WIKI_PLACEHOLDER, path=rel_path.removeprefix("English_JSON/"), hint=attribution_hint(rel_path, display_name), content=json.dumps(data, indent=2, ensure_ascii=False), ) started = time.monotonic() reply = clean_reply(call_ollama(system_prompt, user_msg)) elapsed = time.monotonic() - started if reply.startswith("#"): wiki = reply wiki_path.write_text(wiki + "\n", encoding="utf-8") print(f" ({i}/{len(pending)}) {rel_path} - updated ({elapsed:.0f}s)") else: print( f" ({i}/{len(pending)}) {rel_path} - WARNING: malformed reply, " f"keeping previous wiki ({elapsed:.0f}s)" ) done = done + [rel_path] progress = {**progress, char_id: done} save_progress(progress) return progress def main() -> None: parser = argparse.ArgumentParser(description=__doc__.split("\n")[0]) parser.add_argument("--character", help="process a single character id (e.g. brad)") parser.add_argument( "--force", action="store_true", help="rebuild from scratch, ignoring saved progress" ) args = parser.parse_args() sys.stdout.reconfigure(line_buffering=True) char_files = sorted(CHAR_DIR.glob("*.json")) if args.character: char_files = [CHAR_DIR / f"{args.character}.json"] if not char_files[0].exists(): sys.exit(f"No such character: {args.character} (expected {char_files[0]})") if not char_files: sys.exit(f"No character files found in {CHAR_DIR}; run build_character_data.py first.") OUT_DIR.mkdir(exist_ok=True) progress = load_progress() try: for char_file in char_files: char_data = json.loads(char_file.read_text(encoding="utf-8")) progress = process_character(char_data, progress, args.force) except KeyboardInterrupt: print("\nInterrupted - progress saved; rerun to resume.") sys.exit(130) print(f"Done. Wikis written to {OUT_DIR}/") if __name__ == "__main__": main()