File size: 9,871 Bytes
7ebb6d1
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
"""Build per-character voice wikis from English_JSON dialogue via a local Ollama model.

For each character indexed in character_data/{id}.json, processes the character's
"mentions" files one at a time. Each file gets its own stateless LLM call (a fresh
context window): the model receives the current wiki plus one file and returns the
updated wiki. The wiki is saved after every file and progress is tracked in
character_wikis/.progress.json, so an interrupted run resumes where it left off.

Usage:
    python build_character_wikis.py                  # all characters (resumes)
    python build_character_wikis.py --character brad # single character
    python build_character_wikis.py --force          # rebuild ignoring progress
"""

import argparse
import json
import sys
import time
import urllib.error
import urllib.request
from pathlib import Path

ROOT = Path(__file__).parent
CHAR_DIR = ROOT / "character_data"
OUT_DIR = ROOT / "character_wikis"
PROGRESS_PATH = OUT_DIR / ".progress.json"

OLLAMA_URL = "http://localhost:11434/api/chat"
MODEL = "gemma4:12b-mlx"
# Gemma's recommended sampling parameters. 32k ctx comfortably fits the largest
# mention file (~13k tokens incl. prompt + wiki) without over-allocating KV cache.
MODEL_OPTIONS = {"temperature": 1, "top_k": 64, "top_p": 0.95, "num_ctx": 32768}
REQUEST_TIMEOUT_S = 1800
MAX_ATTEMPTS = 3
RETRY_BACKOFF_S = 5

SYSTEM_PROMPT = """\
You are building a character voice wiki for {name}{aka}, a character in "Riverstone", \
a narrative mystery mobile game told through phone chats and calls. Purpose of the \
wiki: a translator LLM will later receive a translated line of {name}'s dialogue plus \
this wiki, and adjust the translation to sound like {name} rather than a generic, \
robotic rendering. Record only what serves that goal.

You are given the current wiki and one game file. Update the wiki with any NEW \
evidence the file provides about:
- Tone & style: formal/casual, warmth, sarcasm, energy, politeness.
- Speech patterns: slang, catchphrases, filler words, abbreviations, \
emoji/punctuation habits, typical message length, grammar quirks.
- Personality & relationships, insofar as they shape how {name} speaks and to whom.
- Example lines: a few short verbatim quotes that best capture the voice.

Rules:
- Use only evidence from the file. Never invent details.
- Strings starting with "system::" are engine keys, and "#0"-style keys, "next" and \
"options_next" values are navigation IDs - not dialogue. Ignore them.
- If the file adds nothing new about {name}, return the wiki unchanged.
- Keep the wiki under 400 words, in markdown, under exactly these headings:
  # {name}
  ## Role & Relationships
  ## Personality
  ## Tone & Style
  ## Speech Patterns
  ## Example Lines
- Output ONLY the complete updated wiki. No commentary before or after.
"""

USER_TEMPLATE = """\
CURRENT WIKI:
{wiki}

FILE: {path}
{hint}

FILE CONTENT:
{content}
"""

EMPTY_WIKI_PLACEHOLDER = "(empty - this is the first file)"


def attribution_hint(rel_path: str, name: str) -> str:
    """Tell the model who speaks which fields, based on the file's location."""
    parts = Path(rel_path).parts
    if "Adam_s Phone" in parts:
        owner = "Detective Adam (the player)"
    elif "Zoey_s Phone" in parts:
        owner = "Zoey"
    else:
        owner = "the player"

    if "Conversations" in parts:
        return (
            f'In this branching dialogue, "message" values are spoken by {name}; '
            f'"options" values are reply choices spoken by {owner}, the phone\'s '
            f"owner - not {name}."
        )
    if "Filler Chats" in parts:
        return (
            f"This is a chat log on {owner}'s phone between {owner} and {name} "
            f'(group chats may include others). Different "type" values mark '
            f"different speakers; attribute a line to {name} only when context "
            f"makes the speaker clear."
        )
    return (
        f"Extract only what this file reveals about {name}; do not attribute "
        f"other speakers' lines to them."
    )


def call_ollama(system_prompt: str, user_msg: str) -> str:
    payload = json.dumps(
        {
            "model": MODEL,
            "messages": [
                {"role": "system", "content": system_prompt},
                {"role": "user", "content": user_msg},
            ],
            "stream": False,
            # gemma4 has thinking enabled by default, which multiplies generation
            # time several-fold per call; the wiki task doesn't need it.
            "think": False,
            "options": MODEL_OPTIONS,
        }
    ).encode("utf-8")

    last_error: Exception | None = None
    for attempt in range(1, MAX_ATTEMPTS + 1):
        request = urllib.request.Request(
            OLLAMA_URL, data=payload, headers={"Content-Type": "application/json"}
        )
        try:
            with urllib.request.urlopen(request, timeout=REQUEST_TIMEOUT_S) as resp:
                data = json.loads(resp.read().decode("utf-8"))
            return data["message"]["content"]
        except (urllib.error.URLError, TimeoutError, KeyError, json.JSONDecodeError) as exc:
            last_error = exc
            if attempt < MAX_ATTEMPTS:
                print(f"    request failed ({exc}); retrying in {RETRY_BACKOFF_S}s...")
                time.sleep(RETRY_BACKOFF_S)
    raise RuntimeError(
        f"Ollama request failed after {MAX_ATTEMPTS} attempts: {last_error}. "
        f"Is the server running at {OLLAMA_URL}? Progress is saved; rerun to resume."
    )


def clean_reply(text: str) -> str:
    """Strip whitespace and an optional wrapping markdown code fence."""
    text = text.strip()
    if text.startswith("```") and text.endswith("```"):
        first_newline = text.find("\n")
        if first_newline != -1:
            text = text[first_newline + 1 : -3].strip()
    return text


def load_progress() -> dict:
    if PROGRESS_PATH.exists():
        return json.loads(PROGRESS_PATH.read_text(encoding="utf-8"))
    return {}


def save_progress(progress: dict) -> None:
    PROGRESS_PATH.write_text(
        json.dumps(progress, indent=2, ensure_ascii=False), encoding="utf-8"
    )


def process_character(char_data: dict, progress: dict, force: bool) -> dict:
    """Run the per-file wiki-update loop for one character; returns new progress."""
    char_id = char_data["id"]
    names = char_data.get("name") or []
    display_name = names[0] if names else char_id
    aka = f' (also appearing as {", ".join(names[1:])})' if len(names) > 1 else ""
    mentions = char_data.get("mentions", [])

    if not mentions:
        print(f"[{char_id}] no mention files - skipped")
        return progress

    wiki_path = OUT_DIR / f"{char_id}.md"
    done = [] if force else list(progress.get(char_id, []))
    wiki = ""
    if not force and wiki_path.exists():
        wiki = wiki_path.read_text(encoding="utf-8").strip()

    pending = [m for m in mentions if m not in done]
    if not pending:
        print(f"[{char_id}] already complete ({len(mentions)} files)")
        return progress

    system_prompt = SYSTEM_PROMPT.format(name=display_name, aka=aka)
    print(f"[{char_id}] {len(pending)} file(s) to process")

    for i, rel_path in enumerate(pending, 1):
        file_path = ROOT / rel_path
        try:
            data = json.loads(file_path.read_text(encoding="utf-8"))
        except (OSError, json.JSONDecodeError) as exc:
            print(f"  ({i}/{len(pending)}) WARNING: cannot read {rel_path}: {exc}")
            done = done + [rel_path]
            progress = {**progress, char_id: done}
            save_progress(progress)
            continue

        user_msg = USER_TEMPLATE.format(
            wiki=wiki or EMPTY_WIKI_PLACEHOLDER,
            path=rel_path.removeprefix("English_JSON/"),
            hint=attribution_hint(rel_path, display_name),
            content=json.dumps(data, indent=2, ensure_ascii=False),
        )

        started = time.monotonic()
        reply = clean_reply(call_ollama(system_prompt, user_msg))
        elapsed = time.monotonic() - started

        if reply.startswith("#"):
            wiki = reply
            wiki_path.write_text(wiki + "\n", encoding="utf-8")
            print(f"  ({i}/{len(pending)}) {rel_path} - updated ({elapsed:.0f}s)")
        else:
            print(
                f"  ({i}/{len(pending)}) {rel_path} - WARNING: malformed reply, "
                f"keeping previous wiki ({elapsed:.0f}s)"
            )

        done = done + [rel_path]
        progress = {**progress, char_id: done}
        save_progress(progress)

    return progress


def main() -> None:
    parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
    parser.add_argument("--character", help="process a single character id (e.g. brad)")
    parser.add_argument(
        "--force", action="store_true", help="rebuild from scratch, ignoring saved progress"
    )
    args = parser.parse_args()
    sys.stdout.reconfigure(line_buffering=True)

    char_files = sorted(CHAR_DIR.glob("*.json"))
    if args.character:
        char_files = [CHAR_DIR / f"{args.character}.json"]
        if not char_files[0].exists():
            sys.exit(f"No such character: {args.character} (expected {char_files[0]})")
    if not char_files:
        sys.exit(f"No character files found in {CHAR_DIR}; run build_character_data.py first.")

    OUT_DIR.mkdir(exist_ok=True)
    progress = load_progress()

    try:
        for char_file in char_files:
            char_data = json.loads(char_file.read_text(encoding="utf-8"))
            progress = process_character(char_data, progress, args.force)
    except KeyboardInterrupt:
        print("\nInterrupted - progress saved; rerun to resume.")
        sys.exit(130)

    print(f"Done. Wikis written to {OUT_DIR}/")


if __name__ == "__main__":
    main()