| """Build per-character voice wikis from English_JSON dialogue via a local Ollama model. |
| |
| For each character indexed in character_data/{id}.json, processes the character's |
| "mentions" files one at a time. Each file gets its own stateless LLM call (a fresh |
| context window): the model receives the current wiki plus one file and returns the |
| updated wiki. The wiki is saved after every file and progress is tracked in |
| character_wikis/.progress.json, so an interrupted run resumes where it left off. |
| |
| Usage: |
| python build_character_wikis.py # all characters (resumes) |
| python build_character_wikis.py --character brad # single character |
| python build_character_wikis.py --force # rebuild ignoring progress |
| """ |
|
|
| import argparse |
| import json |
| import sys |
| import time |
| import urllib.error |
| import urllib.request |
| from pathlib import Path |
|
|
| ROOT = Path(__file__).parent |
| CHAR_DIR = ROOT / "character_data" |
| OUT_DIR = ROOT / "character_wikis" |
| PROGRESS_PATH = OUT_DIR / ".progress.json" |
|
|
| OLLAMA_URL = "http://localhost:11434/api/chat" |
| MODEL = "gemma4:12b-mlx" |
| |
| |
| MODEL_OPTIONS = {"temperature": 1, "top_k": 64, "top_p": 0.95, "num_ctx": 32768} |
| REQUEST_TIMEOUT_S = 1800 |
| MAX_ATTEMPTS = 3 |
| RETRY_BACKOFF_S = 5 |
|
|
| SYSTEM_PROMPT = """\ |
| You are building a character voice wiki for {name}{aka}, a character in "Riverstone", \ |
| a narrative mystery mobile game told through phone chats and calls. Purpose of the \ |
| wiki: a translator LLM will later receive a translated line of {name}'s dialogue plus \ |
| this wiki, and adjust the translation to sound like {name} rather than a generic, \ |
| robotic rendering. Record only what serves that goal. |
| |
| You are given the current wiki and one game file. Update the wiki with any NEW \ |
| evidence the file provides about: |
| - Tone & style: formal/casual, warmth, sarcasm, energy, politeness. |
| - Speech patterns: slang, catchphrases, filler words, abbreviations, \ |
| emoji/punctuation habits, typical message length, grammar quirks. |
| - Personality & relationships, insofar as they shape how {name} speaks and to whom. |
| - Example lines: a few short verbatim quotes that best capture the voice. |
| |
| Rules: |
| - Use only evidence from the file. Never invent details. |
| - Strings starting with "system::" are engine keys, and "#0"-style keys, "next" and \ |
| "options_next" values are navigation IDs - not dialogue. Ignore them. |
| - If the file adds nothing new about {name}, return the wiki unchanged. |
| - Keep the wiki under 400 words, in markdown, under exactly these headings: |
| # {name} |
| ## Role & Relationships |
| ## Personality |
| ## Tone & Style |
| ## Speech Patterns |
| ## Example Lines |
| - Output ONLY the complete updated wiki. No commentary before or after. |
| """ |
|
|
| USER_TEMPLATE = """\ |
| CURRENT WIKI: |
| {wiki} |
| |
| FILE: {path} |
| {hint} |
| |
| FILE CONTENT: |
| {content} |
| """ |
|
|
| EMPTY_WIKI_PLACEHOLDER = "(empty - this is the first file)" |
|
|
|
|
| def attribution_hint(rel_path: str, name: str) -> str: |
| """Tell the model who speaks which fields, based on the file's location.""" |
| parts = Path(rel_path).parts |
| if "Adam_s Phone" in parts: |
| owner = "Detective Adam (the player)" |
| elif "Zoey_s Phone" in parts: |
| owner = "Zoey" |
| else: |
| owner = "the player" |
|
|
| if "Conversations" in parts: |
| return ( |
| f'In this branching dialogue, "message" values are spoken by {name}; ' |
| f'"options" values are reply choices spoken by {owner}, the phone\'s ' |
| f"owner - not {name}." |
| ) |
| if "Filler Chats" in parts: |
| return ( |
| f"This is a chat log on {owner}'s phone between {owner} and {name} " |
| f'(group chats may include others). Different "type" values mark ' |
| f"different speakers; attribute a line to {name} only when context " |
| f"makes the speaker clear." |
| ) |
| return ( |
| f"Extract only what this file reveals about {name}; do not attribute " |
| f"other speakers' lines to them." |
| ) |
|
|
|
|
| def call_ollama(system_prompt: str, user_msg: str) -> str: |
| payload = json.dumps( |
| { |
| "model": MODEL, |
| "messages": [ |
| {"role": "system", "content": system_prompt}, |
| {"role": "user", "content": user_msg}, |
| ], |
| "stream": False, |
| |
| |
| "think": False, |
| "options": MODEL_OPTIONS, |
| } |
| ).encode("utf-8") |
|
|
| last_error: Exception | None = None |
| for attempt in range(1, MAX_ATTEMPTS + 1): |
| request = urllib.request.Request( |
| OLLAMA_URL, data=payload, headers={"Content-Type": "application/json"} |
| ) |
| try: |
| with urllib.request.urlopen(request, timeout=REQUEST_TIMEOUT_S) as resp: |
| data = json.loads(resp.read().decode("utf-8")) |
| return data["message"]["content"] |
| except (urllib.error.URLError, TimeoutError, KeyError, json.JSONDecodeError) as exc: |
| last_error = exc |
| if attempt < MAX_ATTEMPTS: |
| print(f" request failed ({exc}); retrying in {RETRY_BACKOFF_S}s...") |
| time.sleep(RETRY_BACKOFF_S) |
| raise RuntimeError( |
| f"Ollama request failed after {MAX_ATTEMPTS} attempts: {last_error}. " |
| f"Is the server running at {OLLAMA_URL}? Progress is saved; rerun to resume." |
| ) |
|
|
|
|
| def clean_reply(text: str) -> str: |
| """Strip whitespace and an optional wrapping markdown code fence.""" |
| text = text.strip() |
| if text.startswith("```") and text.endswith("```"): |
| first_newline = text.find("\n") |
| if first_newline != -1: |
| text = text[first_newline + 1 : -3].strip() |
| return text |
|
|
|
|
| def load_progress() -> dict: |
| if PROGRESS_PATH.exists(): |
| return json.loads(PROGRESS_PATH.read_text(encoding="utf-8")) |
| return {} |
|
|
|
|
| def save_progress(progress: dict) -> None: |
| PROGRESS_PATH.write_text( |
| json.dumps(progress, indent=2, ensure_ascii=False), encoding="utf-8" |
| ) |
|
|
|
|
| def process_character(char_data: dict, progress: dict, force: bool) -> dict: |
| """Run the per-file wiki-update loop for one character; returns new progress.""" |
| char_id = char_data["id"] |
| names = char_data.get("name") or [] |
| display_name = names[0] if names else char_id |
| aka = f' (also appearing as {", ".join(names[1:])})' if len(names) > 1 else "" |
| mentions = char_data.get("mentions", []) |
|
|
| if not mentions: |
| print(f"[{char_id}] no mention files - skipped") |
| return progress |
|
|
| wiki_path = OUT_DIR / f"{char_id}.md" |
| done = [] if force else list(progress.get(char_id, [])) |
| wiki = "" |
| if not force and wiki_path.exists(): |
| wiki = wiki_path.read_text(encoding="utf-8").strip() |
|
|
| pending = [m for m in mentions if m not in done] |
| if not pending: |
| print(f"[{char_id}] already complete ({len(mentions)} files)") |
| return progress |
|
|
| system_prompt = SYSTEM_PROMPT.format(name=display_name, aka=aka) |
| print(f"[{char_id}] {len(pending)} file(s) to process") |
|
|
| for i, rel_path in enumerate(pending, 1): |
| file_path = ROOT / rel_path |
| try: |
| data = json.loads(file_path.read_text(encoding="utf-8")) |
| except (OSError, json.JSONDecodeError) as exc: |
| print(f" ({i}/{len(pending)}) WARNING: cannot read {rel_path}: {exc}") |
| done = done + [rel_path] |
| progress = {**progress, char_id: done} |
| save_progress(progress) |
| continue |
|
|
| user_msg = USER_TEMPLATE.format( |
| wiki=wiki or EMPTY_WIKI_PLACEHOLDER, |
| path=rel_path.removeprefix("English_JSON/"), |
| hint=attribution_hint(rel_path, display_name), |
| content=json.dumps(data, indent=2, ensure_ascii=False), |
| ) |
|
|
| started = time.monotonic() |
| reply = clean_reply(call_ollama(system_prompt, user_msg)) |
| elapsed = time.monotonic() - started |
|
|
| if reply.startswith("#"): |
| wiki = reply |
| wiki_path.write_text(wiki + "\n", encoding="utf-8") |
| print(f" ({i}/{len(pending)}) {rel_path} - updated ({elapsed:.0f}s)") |
| else: |
| print( |
| f" ({i}/{len(pending)}) {rel_path} - WARNING: malformed reply, " |
| f"keeping previous wiki ({elapsed:.0f}s)" |
| ) |
|
|
| done = done + [rel_path] |
| progress = {**progress, char_id: done} |
| save_progress(progress) |
|
|
| return progress |
|
|
|
|
| def main() -> None: |
| parser = argparse.ArgumentParser(description=__doc__.split("\n")[0]) |
| parser.add_argument("--character", help="process a single character id (e.g. brad)") |
| parser.add_argument( |
| "--force", action="store_true", help="rebuild from scratch, ignoring saved progress" |
| ) |
| args = parser.parse_args() |
| sys.stdout.reconfigure(line_buffering=True) |
|
|
| char_files = sorted(CHAR_DIR.glob("*.json")) |
| if args.character: |
| char_files = [CHAR_DIR / f"{args.character}.json"] |
| if not char_files[0].exists(): |
| sys.exit(f"No such character: {args.character} (expected {char_files[0]})") |
| if not char_files: |
| sys.exit(f"No character files found in {CHAR_DIR}; run build_character_data.py first.") |
|
|
| OUT_DIR.mkdir(exist_ok=True) |
| progress = load_progress() |
|
|
| try: |
| for char_file in char_files: |
| char_data = json.loads(char_file.read_text(encoding="utf-8")) |
| progress = process_character(char_data, progress, args.force) |
| except KeyboardInterrupt: |
| print("\nInterrupted - progress saved; rerun to resume.") |
| sys.exit(130) |
|
|
| print(f"Done. Wikis written to {OUT_DIR}/") |
|
|
|
|
| if __name__ == "__main__": |
| main() |
|
|