hacktest / build_character_wikis.py
bhardwaj08sarthak's picture
Upload 13 files
cbe0298 verified
Raw
History Blame Contribute Delete
9.87 kB
"""Build per-character voice wikis from English_JSON dialogue via a local Ollama model.
For each character indexed in character_data/{id}.json, processes the character's
"mentions" files one at a time. Each file gets its own stateless LLM call (a fresh
context window): the model receives the current wiki plus one file and returns the
updated wiki. The wiki is saved after every file and progress is tracked in
character_wikis/.progress.json, so an interrupted run resumes where it left off.
Usage:
python build_character_wikis.py # all characters (resumes)
python build_character_wikis.py --character brad # single character
python build_character_wikis.py --force # rebuild ignoring progress
"""
import argparse
import json
import sys
import time
import urllib.error
import urllib.request
from pathlib import Path
ROOT = Path(__file__).parent
CHAR_DIR = ROOT / "character_data"
OUT_DIR = ROOT / "character_wikis"
PROGRESS_PATH = OUT_DIR / ".progress.json"
OLLAMA_URL = "http://localhost:11434/api/chat"
MODEL = "gemma4:12b-mlx"
# Gemma's recommended sampling parameters. 32k ctx comfortably fits the largest
# mention file (~13k tokens incl. prompt + wiki) without over-allocating KV cache.
MODEL_OPTIONS = {"temperature": 1, "top_k": 64, "top_p": 0.95, "num_ctx": 32768}
REQUEST_TIMEOUT_S = 1800
MAX_ATTEMPTS = 3
RETRY_BACKOFF_S = 5
SYSTEM_PROMPT = """\
You are building a character voice wiki for {name}{aka}, a character in "Riverstone", \
a narrative mystery mobile game told through phone chats and calls. Purpose of the \
wiki: a translator LLM will later receive a translated line of {name}'s dialogue plus \
this wiki, and adjust the translation to sound like {name} rather than a generic, \
robotic rendering. Record only what serves that goal.
You are given the current wiki and one game file. Update the wiki with any NEW \
evidence the file provides about:
- Tone & style: formal/casual, warmth, sarcasm, energy, politeness.
- Speech patterns: slang, catchphrases, filler words, abbreviations, \
emoji/punctuation habits, typical message length, grammar quirks.
- Personality & relationships, insofar as they shape how {name} speaks and to whom.
- Example lines: a few short verbatim quotes that best capture the voice.
Rules:
- Use only evidence from the file. Never invent details.
- Strings starting with "system::" are engine keys, and "#0"-style keys, "next" and \
"options_next" values are navigation IDs - not dialogue. Ignore them.
- If the file adds nothing new about {name}, return the wiki unchanged.
- Keep the wiki under 400 words, in markdown, under exactly these headings:
# {name}
## Role & Relationships
## Personality
## Tone & Style
## Speech Patterns
## Example Lines
- Output ONLY the complete updated wiki. No commentary before or after.
"""
USER_TEMPLATE = """\
CURRENT WIKI:
{wiki}
FILE: {path}
{hint}
FILE CONTENT:
{content}
"""
EMPTY_WIKI_PLACEHOLDER = "(empty - this is the first file)"
def attribution_hint(rel_path: str, name: str) -> str:
"""Tell the model who speaks which fields, based on the file's location."""
parts = Path(rel_path).parts
if "Adam_s Phone" in parts:
owner = "Detective Adam (the player)"
elif "Zoey_s Phone" in parts:
owner = "Zoey"
else:
owner = "the player"
if "Conversations" in parts:
return (
f'In this branching dialogue, "message" values are spoken by {name}; '
f'"options" values are reply choices spoken by {owner}, the phone\'s '
f"owner - not {name}."
)
if "Filler Chats" in parts:
return (
f"This is a chat log on {owner}'s phone between {owner} and {name} "
f'(group chats may include others). Different "type" values mark '
f"different speakers; attribute a line to {name} only when context "
f"makes the speaker clear."
)
return (
f"Extract only what this file reveals about {name}; do not attribute "
f"other speakers' lines to them."
)
def call_ollama(system_prompt: str, user_msg: str) -> str:
payload = json.dumps(
{
"model": MODEL,
"messages": [
{"role": "system", "content": system_prompt},
{"role": "user", "content": user_msg},
],
"stream": False,
# gemma4 has thinking enabled by default, which multiplies generation
# time several-fold per call; the wiki task doesn't need it.
"think": False,
"options": MODEL_OPTIONS,
}
).encode("utf-8")
last_error: Exception | None = None
for attempt in range(1, MAX_ATTEMPTS + 1):
request = urllib.request.Request(
OLLAMA_URL, data=payload, headers={"Content-Type": "application/json"}
)
try:
with urllib.request.urlopen(request, timeout=REQUEST_TIMEOUT_S) as resp:
data = json.loads(resp.read().decode("utf-8"))
return data["message"]["content"]
except (urllib.error.URLError, TimeoutError, KeyError, json.JSONDecodeError) as exc:
last_error = exc
if attempt < MAX_ATTEMPTS:
print(f" request failed ({exc}); retrying in {RETRY_BACKOFF_S}s...")
time.sleep(RETRY_BACKOFF_S)
raise RuntimeError(
f"Ollama request failed after {MAX_ATTEMPTS} attempts: {last_error}. "
f"Is the server running at {OLLAMA_URL}? Progress is saved; rerun to resume."
)
def clean_reply(text: str) -> str:
"""Strip whitespace and an optional wrapping markdown code fence."""
text = text.strip()
if text.startswith("```") and text.endswith("```"):
first_newline = text.find("\n")
if first_newline != -1:
text = text[first_newline + 1 : -3].strip()
return text
def load_progress() -> dict:
if PROGRESS_PATH.exists():
return json.loads(PROGRESS_PATH.read_text(encoding="utf-8"))
return {}
def save_progress(progress: dict) -> None:
PROGRESS_PATH.write_text(
json.dumps(progress, indent=2, ensure_ascii=False), encoding="utf-8"
)
def process_character(char_data: dict, progress: dict, force: bool) -> dict:
"""Run the per-file wiki-update loop for one character; returns new progress."""
char_id = char_data["id"]
names = char_data.get("name") or []
display_name = names[0] if names else char_id
aka = f' (also appearing as {", ".join(names[1:])})' if len(names) > 1 else ""
mentions = char_data.get("mentions", [])
if not mentions:
print(f"[{char_id}] no mention files - skipped")
return progress
wiki_path = OUT_DIR / f"{char_id}.md"
done = [] if force else list(progress.get(char_id, []))
wiki = ""
if not force and wiki_path.exists():
wiki = wiki_path.read_text(encoding="utf-8").strip()
pending = [m for m in mentions if m not in done]
if not pending:
print(f"[{char_id}] already complete ({len(mentions)} files)")
return progress
system_prompt = SYSTEM_PROMPT.format(name=display_name, aka=aka)
print(f"[{char_id}] {len(pending)} file(s) to process")
for i, rel_path in enumerate(pending, 1):
file_path = ROOT / rel_path
try:
data = json.loads(file_path.read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError) as exc:
print(f" ({i}/{len(pending)}) WARNING: cannot read {rel_path}: {exc}")
done = done + [rel_path]
progress = {**progress, char_id: done}
save_progress(progress)
continue
user_msg = USER_TEMPLATE.format(
wiki=wiki or EMPTY_WIKI_PLACEHOLDER,
path=rel_path.removeprefix("English_JSON/"),
hint=attribution_hint(rel_path, display_name),
content=json.dumps(data, indent=2, ensure_ascii=False),
)
started = time.monotonic()
reply = clean_reply(call_ollama(system_prompt, user_msg))
elapsed = time.monotonic() - started
if reply.startswith("#"):
wiki = reply
wiki_path.write_text(wiki + "\n", encoding="utf-8")
print(f" ({i}/{len(pending)}) {rel_path} - updated ({elapsed:.0f}s)")
else:
print(
f" ({i}/{len(pending)}) {rel_path} - WARNING: malformed reply, "
f"keeping previous wiki ({elapsed:.0f}s)"
)
done = done + [rel_path]
progress = {**progress, char_id: done}
save_progress(progress)
return progress
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
parser.add_argument("--character", help="process a single character id (e.g. brad)")
parser.add_argument(
"--force", action="store_true", help="rebuild from scratch, ignoring saved progress"
)
args = parser.parse_args()
sys.stdout.reconfigure(line_buffering=True)
char_files = sorted(CHAR_DIR.glob("*.json"))
if args.character:
char_files = [CHAR_DIR / f"{args.character}.json"]
if not char_files[0].exists():
sys.exit(f"No such character: {args.character} (expected {char_files[0]})")
if not char_files:
sys.exit(f"No character files found in {CHAR_DIR}; run build_character_data.py first.")
OUT_DIR.mkdir(exist_ok=True)
progress = load_progress()
try:
for char_file in char_files:
char_data = json.loads(char_file.read_text(encoding="utf-8"))
progress = process_character(char_data, progress, args.force)
except KeyboardInterrupt:
print("\nInterrupted - progress saved; rerun to resume.")
sys.exit(130)
print(f"Done. Wikis written to {OUT_DIR}/")
if __name__ == "__main__":
main()