Spaces:
Running on Zero
Running on Zero
| """Build per-character voice wikis from English_JSON dialogue via a local Ollama model. | |
| For each character indexed in character_data/{id}.json, processes the character's | |
| "mentions" files one at a time. Each file gets its own stateless LLM call (a fresh | |
| context window): the model receives the current wiki plus one file and returns the | |
| updated wiki. The wiki is saved after every file and progress is tracked in | |
| character_wikis/.progress.json, so an interrupted run resumes where it left off. | |
| Usage: | |
| python build_character_wikis.py # all characters (resumes) | |
| python build_character_wikis.py --character brad # single character | |
| python build_character_wikis.py --force # rebuild ignoring progress | |
| """ | |
| import argparse | |
| import json | |
| import sys | |
| import time | |
| import urllib.error | |
| import urllib.request | |
| from pathlib import Path | |
| ROOT = Path(__file__).parent | |
| CHAR_DIR = ROOT / "character_data" | |
| OUT_DIR = ROOT / "character_wikis" | |
| PROGRESS_PATH = OUT_DIR / ".progress.json" | |
| OLLAMA_URL = "http://localhost:11434/api/chat" | |
| MODEL = "gemma4:12b-mlx" | |
| # Gemma's recommended sampling parameters. 32k ctx comfortably fits the largest | |
| # mention file (~13k tokens incl. prompt + wiki) without over-allocating KV cache. | |
| MODEL_OPTIONS = {"temperature": 1, "top_k": 64, "top_p": 0.95, "num_ctx": 32768} | |
| REQUEST_TIMEOUT_S = 1800 | |
| MAX_ATTEMPTS = 3 | |
| RETRY_BACKOFF_S = 5 | |
| SYSTEM_PROMPT = """\ | |
| You are building a character voice wiki for {name}{aka}, a character in "Riverstone", \ | |
| a narrative mystery mobile game told through phone chats and calls. Purpose of the \ | |
| wiki: a translator LLM will later receive a translated line of {name}'s dialogue plus \ | |
| this wiki, and adjust the translation to sound like {name} rather than a generic, \ | |
| robotic rendering. Record only what serves that goal. | |
| You are given the current wiki and one game file. Update the wiki with any NEW \ | |
| evidence the file provides about: | |
| - Tone & style: formal/casual, warmth, sarcasm, energy, politeness. | |
| - Speech patterns: slang, catchphrases, filler words, abbreviations, \ | |
| emoji/punctuation habits, typical message length, grammar quirks. | |
| - Personality & relationships, insofar as they shape how {name} speaks and to whom. | |
| - Example lines: a few short verbatim quotes that best capture the voice. | |
| Rules: | |
| - Use only evidence from the file. Never invent details. | |
| - Strings starting with "system::" are engine keys, and "#0"-style keys, "next" and \ | |
| "options_next" values are navigation IDs - not dialogue. Ignore them. | |
| - If the file adds nothing new about {name}, return the wiki unchanged. | |
| - Keep the wiki under 400 words, in markdown, under exactly these headings: | |
| # {name} | |
| ## Role & Relationships | |
| ## Personality | |
| ## Tone & Style | |
| ## Speech Patterns | |
| ## Example Lines | |
| - Output ONLY the complete updated wiki. No commentary before or after. | |
| """ | |
| USER_TEMPLATE = """\ | |
| CURRENT WIKI: | |
| {wiki} | |
| FILE: {path} | |
| {hint} | |
| FILE CONTENT: | |
| {content} | |
| """ | |
| EMPTY_WIKI_PLACEHOLDER = "(empty - this is the first file)" | |
| def attribution_hint(rel_path: str, name: str) -> str: | |
| """Tell the model who speaks which fields, based on the file's location.""" | |
| parts = Path(rel_path).parts | |
| if "Adam_s Phone" in parts: | |
| owner = "Detective Adam (the player)" | |
| elif "Zoey_s Phone" in parts: | |
| owner = "Zoey" | |
| else: | |
| owner = "the player" | |
| if "Conversations" in parts: | |
| return ( | |
| f'In this branching dialogue, "message" values are spoken by {name}; ' | |
| f'"options" values are reply choices spoken by {owner}, the phone\'s ' | |
| f"owner - not {name}." | |
| ) | |
| if "Filler Chats" in parts: | |
| return ( | |
| f"This is a chat log on {owner}'s phone between {owner} and {name} " | |
| f'(group chats may include others). Different "type" values mark ' | |
| f"different speakers; attribute a line to {name} only when context " | |
| f"makes the speaker clear." | |
| ) | |
| return ( | |
| f"Extract only what this file reveals about {name}; do not attribute " | |
| f"other speakers' lines to them." | |
| ) | |
| def call_ollama(system_prompt: str, user_msg: str) -> str: | |
| payload = json.dumps( | |
| { | |
| "model": MODEL, | |
| "messages": [ | |
| {"role": "system", "content": system_prompt}, | |
| {"role": "user", "content": user_msg}, | |
| ], | |
| "stream": False, | |
| # gemma4 has thinking enabled by default, which multiplies generation | |
| # time several-fold per call; the wiki task doesn't need it. | |
| "think": False, | |
| "options": MODEL_OPTIONS, | |
| } | |
| ).encode("utf-8") | |
| last_error: Exception | None = None | |
| for attempt in range(1, MAX_ATTEMPTS + 1): | |
| request = urllib.request.Request( | |
| OLLAMA_URL, data=payload, headers={"Content-Type": "application/json"} | |
| ) | |
| try: | |
| with urllib.request.urlopen(request, timeout=REQUEST_TIMEOUT_S) as resp: | |
| data = json.loads(resp.read().decode("utf-8")) | |
| return data["message"]["content"] | |
| except (urllib.error.URLError, TimeoutError, KeyError, json.JSONDecodeError) as exc: | |
| last_error = exc | |
| if attempt < MAX_ATTEMPTS: | |
| print(f" request failed ({exc}); retrying in {RETRY_BACKOFF_S}s...") | |
| time.sleep(RETRY_BACKOFF_S) | |
| raise RuntimeError( | |
| f"Ollama request failed after {MAX_ATTEMPTS} attempts: {last_error}. " | |
| f"Is the server running at {OLLAMA_URL}? Progress is saved; rerun to resume." | |
| ) | |
| def clean_reply(text: str) -> str: | |
| """Strip whitespace and an optional wrapping markdown code fence.""" | |
| text = text.strip() | |
| if text.startswith("```") and text.endswith("```"): | |
| first_newline = text.find("\n") | |
| if first_newline != -1: | |
| text = text[first_newline + 1 : -3].strip() | |
| return text | |
| def load_progress() -> dict: | |
| if PROGRESS_PATH.exists(): | |
| return json.loads(PROGRESS_PATH.read_text(encoding="utf-8")) | |
| return {} | |
| def save_progress(progress: dict) -> None: | |
| PROGRESS_PATH.write_text( | |
| json.dumps(progress, indent=2, ensure_ascii=False), encoding="utf-8" | |
| ) | |
| def process_character(char_data: dict, progress: dict, force: bool) -> dict: | |
| """Run the per-file wiki-update loop for one character; returns new progress.""" | |
| char_id = char_data["id"] | |
| names = char_data.get("name") or [] | |
| display_name = names[0] if names else char_id | |
| aka = f' (also appearing as {", ".join(names[1:])})' if len(names) > 1 else "" | |
| mentions = char_data.get("mentions", []) | |
| if not mentions: | |
| print(f"[{char_id}] no mention files - skipped") | |
| return progress | |
| wiki_path = OUT_DIR / f"{char_id}.md" | |
| done = [] if force else list(progress.get(char_id, [])) | |
| wiki = "" | |
| if not force and wiki_path.exists(): | |
| wiki = wiki_path.read_text(encoding="utf-8").strip() | |
| pending = [m for m in mentions if m not in done] | |
| if not pending: | |
| print(f"[{char_id}] already complete ({len(mentions)} files)") | |
| return progress | |
| system_prompt = SYSTEM_PROMPT.format(name=display_name, aka=aka) | |
| print(f"[{char_id}] {len(pending)} file(s) to process") | |
| for i, rel_path in enumerate(pending, 1): | |
| file_path = ROOT / rel_path | |
| try: | |
| data = json.loads(file_path.read_text(encoding="utf-8")) | |
| except (OSError, json.JSONDecodeError) as exc: | |
| print(f" ({i}/{len(pending)}) WARNING: cannot read {rel_path}: {exc}") | |
| done = done + [rel_path] | |
| progress = {**progress, char_id: done} | |
| save_progress(progress) | |
| continue | |
| user_msg = USER_TEMPLATE.format( | |
| wiki=wiki or EMPTY_WIKI_PLACEHOLDER, | |
| path=rel_path.removeprefix("English_JSON/"), | |
| hint=attribution_hint(rel_path, display_name), | |
| content=json.dumps(data, indent=2, ensure_ascii=False), | |
| ) | |
| started = time.monotonic() | |
| reply = clean_reply(call_ollama(system_prompt, user_msg)) | |
| elapsed = time.monotonic() - started | |
| if reply.startswith("#"): | |
| wiki = reply | |
| wiki_path.write_text(wiki + "\n", encoding="utf-8") | |
| print(f" ({i}/{len(pending)}) {rel_path} - updated ({elapsed:.0f}s)") | |
| else: | |
| print( | |
| f" ({i}/{len(pending)}) {rel_path} - WARNING: malformed reply, " | |
| f"keeping previous wiki ({elapsed:.0f}s)" | |
| ) | |
| done = done + [rel_path] | |
| progress = {**progress, char_id: done} | |
| save_progress(progress) | |
| return progress | |
| def main() -> None: | |
| parser = argparse.ArgumentParser(description=__doc__.split("\n")[0]) | |
| parser.add_argument("--character", help="process a single character id (e.g. brad)") | |
| parser.add_argument( | |
| "--force", action="store_true", help="rebuild from scratch, ignoring saved progress" | |
| ) | |
| args = parser.parse_args() | |
| sys.stdout.reconfigure(line_buffering=True) | |
| char_files = sorted(CHAR_DIR.glob("*.json")) | |
| if args.character: | |
| char_files = [CHAR_DIR / f"{args.character}.json"] | |
| if not char_files[0].exists(): | |
| sys.exit(f"No such character: {args.character} (expected {char_files[0]})") | |
| if not char_files: | |
| sys.exit(f"No character files found in {CHAR_DIR}; run build_character_data.py first.") | |
| OUT_DIR.mkdir(exist_ok=True) | |
| progress = load_progress() | |
| try: | |
| for char_file in char_files: | |
| char_data = json.loads(char_file.read_text(encoding="utf-8")) | |
| progress = process_character(char_data, progress, args.force) | |
| except KeyboardInterrupt: | |
| print("\nInterrupted - progress saved; rerun to resume.") | |
| sys.exit(130) | |
| print(f"Done. Wikis written to {OUT_DIR}/") | |
| if __name__ == "__main__": | |
| main() | |