File size: 9,871 Bytes
7ebb6d1 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 | """Build per-character voice wikis from English_JSON dialogue via a local Ollama model.
For each character indexed in character_data/{id}.json, processes the character's
"mentions" files one at a time. Each file gets its own stateless LLM call (a fresh
context window): the model receives the current wiki plus one file and returns the
updated wiki. The wiki is saved after every file and progress is tracked in
character_wikis/.progress.json, so an interrupted run resumes where it left off.
Usage:
python build_character_wikis.py # all characters (resumes)
python build_character_wikis.py --character brad # single character
python build_character_wikis.py --force # rebuild ignoring progress
"""
import argparse
import json
import sys
import time
import urllib.error
import urllib.request
from pathlib import Path
ROOT = Path(__file__).parent
CHAR_DIR = ROOT / "character_data"
OUT_DIR = ROOT / "character_wikis"
PROGRESS_PATH = OUT_DIR / ".progress.json"
OLLAMA_URL = "http://localhost:11434/api/chat"
MODEL = "gemma4:12b-mlx"
# Gemma's recommended sampling parameters. 32k ctx comfortably fits the largest
# mention file (~13k tokens incl. prompt + wiki) without over-allocating KV cache.
MODEL_OPTIONS = {"temperature": 1, "top_k": 64, "top_p": 0.95, "num_ctx": 32768}
REQUEST_TIMEOUT_S = 1800
MAX_ATTEMPTS = 3
RETRY_BACKOFF_S = 5
SYSTEM_PROMPT = """\
You are building a character voice wiki for {name}{aka}, a character in "Riverstone", \
a narrative mystery mobile game told through phone chats and calls. Purpose of the \
wiki: a translator LLM will later receive a translated line of {name}'s dialogue plus \
this wiki, and adjust the translation to sound like {name} rather than a generic, \
robotic rendering. Record only what serves that goal.
You are given the current wiki and one game file. Update the wiki with any NEW \
evidence the file provides about:
- Tone & style: formal/casual, warmth, sarcasm, energy, politeness.
- Speech patterns: slang, catchphrases, filler words, abbreviations, \
emoji/punctuation habits, typical message length, grammar quirks.
- Personality & relationships, insofar as they shape how {name} speaks and to whom.
- Example lines: a few short verbatim quotes that best capture the voice.
Rules:
- Use only evidence from the file. Never invent details.
- Strings starting with "system::" are engine keys, and "#0"-style keys, "next" and \
"options_next" values are navigation IDs - not dialogue. Ignore them.
- If the file adds nothing new about {name}, return the wiki unchanged.
- Keep the wiki under 400 words, in markdown, under exactly these headings:
# {name}
## Role & Relationships
## Personality
## Tone & Style
## Speech Patterns
## Example Lines
- Output ONLY the complete updated wiki. No commentary before or after.
"""
USER_TEMPLATE = """\
CURRENT WIKI:
{wiki}
FILE: {path}
{hint}
FILE CONTENT:
{content}
"""
EMPTY_WIKI_PLACEHOLDER = "(empty - this is the first file)"
def attribution_hint(rel_path: str, name: str) -> str:
"""Tell the model who speaks which fields, based on the file's location."""
parts = Path(rel_path).parts
if "Adam_s Phone" in parts:
owner = "Detective Adam (the player)"
elif "Zoey_s Phone" in parts:
owner = "Zoey"
else:
owner = "the player"
if "Conversations" in parts:
return (
f'In this branching dialogue, "message" values are spoken by {name}; '
f'"options" values are reply choices spoken by {owner}, the phone\'s '
f"owner - not {name}."
)
if "Filler Chats" in parts:
return (
f"This is a chat log on {owner}'s phone between {owner} and {name} "
f'(group chats may include others). Different "type" values mark '
f"different speakers; attribute a line to {name} only when context "
f"makes the speaker clear."
)
return (
f"Extract only what this file reveals about {name}; do not attribute "
f"other speakers' lines to them."
)
def call_ollama(system_prompt: str, user_msg: str) -> str:
payload = json.dumps(
{
"model": MODEL,
"messages": [
{"role": "system", "content": system_prompt},
{"role": "user", "content": user_msg},
],
"stream": False,
# gemma4 has thinking enabled by default, which multiplies generation
# time several-fold per call; the wiki task doesn't need it.
"think": False,
"options": MODEL_OPTIONS,
}
).encode("utf-8")
last_error: Exception | None = None
for attempt in range(1, MAX_ATTEMPTS + 1):
request = urllib.request.Request(
OLLAMA_URL, data=payload, headers={"Content-Type": "application/json"}
)
try:
with urllib.request.urlopen(request, timeout=REQUEST_TIMEOUT_S) as resp:
data = json.loads(resp.read().decode("utf-8"))
return data["message"]["content"]
except (urllib.error.URLError, TimeoutError, KeyError, json.JSONDecodeError) as exc:
last_error = exc
if attempt < MAX_ATTEMPTS:
print(f" request failed ({exc}); retrying in {RETRY_BACKOFF_S}s...")
time.sleep(RETRY_BACKOFF_S)
raise RuntimeError(
f"Ollama request failed after {MAX_ATTEMPTS} attempts: {last_error}. "
f"Is the server running at {OLLAMA_URL}? Progress is saved; rerun to resume."
)
def clean_reply(text: str) -> str:
"""Strip whitespace and an optional wrapping markdown code fence."""
text = text.strip()
if text.startswith("```") and text.endswith("```"):
first_newline = text.find("\n")
if first_newline != -1:
text = text[first_newline + 1 : -3].strip()
return text
def load_progress() -> dict:
if PROGRESS_PATH.exists():
return json.loads(PROGRESS_PATH.read_text(encoding="utf-8"))
return {}
def save_progress(progress: dict) -> None:
PROGRESS_PATH.write_text(
json.dumps(progress, indent=2, ensure_ascii=False), encoding="utf-8"
)
def process_character(char_data: dict, progress: dict, force: bool) -> dict:
"""Run the per-file wiki-update loop for one character; returns new progress."""
char_id = char_data["id"]
names = char_data.get("name") or []
display_name = names[0] if names else char_id
aka = f' (also appearing as {", ".join(names[1:])})' if len(names) > 1 else ""
mentions = char_data.get("mentions", [])
if not mentions:
print(f"[{char_id}] no mention files - skipped")
return progress
wiki_path = OUT_DIR / f"{char_id}.md"
done = [] if force else list(progress.get(char_id, []))
wiki = ""
if not force and wiki_path.exists():
wiki = wiki_path.read_text(encoding="utf-8").strip()
pending = [m for m in mentions if m not in done]
if not pending:
print(f"[{char_id}] already complete ({len(mentions)} files)")
return progress
system_prompt = SYSTEM_PROMPT.format(name=display_name, aka=aka)
print(f"[{char_id}] {len(pending)} file(s) to process")
for i, rel_path in enumerate(pending, 1):
file_path = ROOT / rel_path
try:
data = json.loads(file_path.read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError) as exc:
print(f" ({i}/{len(pending)}) WARNING: cannot read {rel_path}: {exc}")
done = done + [rel_path]
progress = {**progress, char_id: done}
save_progress(progress)
continue
user_msg = USER_TEMPLATE.format(
wiki=wiki or EMPTY_WIKI_PLACEHOLDER,
path=rel_path.removeprefix("English_JSON/"),
hint=attribution_hint(rel_path, display_name),
content=json.dumps(data, indent=2, ensure_ascii=False),
)
started = time.monotonic()
reply = clean_reply(call_ollama(system_prompt, user_msg))
elapsed = time.monotonic() - started
if reply.startswith("#"):
wiki = reply
wiki_path.write_text(wiki + "\n", encoding="utf-8")
print(f" ({i}/{len(pending)}) {rel_path} - updated ({elapsed:.0f}s)")
else:
print(
f" ({i}/{len(pending)}) {rel_path} - WARNING: malformed reply, "
f"keeping previous wiki ({elapsed:.0f}s)"
)
done = done + [rel_path]
progress = {**progress, char_id: done}
save_progress(progress)
return progress
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
parser.add_argument("--character", help="process a single character id (e.g. brad)")
parser.add_argument(
"--force", action="store_true", help="rebuild from scratch, ignoring saved progress"
)
args = parser.parse_args()
sys.stdout.reconfigure(line_buffering=True)
char_files = sorted(CHAR_DIR.glob("*.json"))
if args.character:
char_files = [CHAR_DIR / f"{args.character}.json"]
if not char_files[0].exists():
sys.exit(f"No such character: {args.character} (expected {char_files[0]})")
if not char_files:
sys.exit(f"No character files found in {CHAR_DIR}; run build_character_data.py first.")
OUT_DIR.mkdir(exist_ok=True)
progress = load_progress()
try:
for char_file in char_files:
char_data = json.loads(char_file.read_text(encoding="utf-8"))
progress = process_character(char_data, progress, args.force)
except KeyboardInterrupt:
print("\nInterrupted - progress saved; rerun to resume.")
sys.exit(130)
print(f"Done. Wikis written to {OUT_DIR}/")
if __name__ == "__main__":
main()
|