bhardwaj08sarthak commited on
Commit
7ebb6d1
·
verified ·
1 Parent(s): fc616f6

Upload 2 files

Browse files
Files changed (2) hide show
  1. build_character_data.py +126 -0
  2. build_character_wikis.py +265 -0
build_character_data.py ADDED
@@ -0,0 +1,126 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Build per-character file index from English_JSON.
2
+
3
+ For each character in Contacts.json, produces character_data/{id}.json with:
4
+ mentions - direct conversation/chat files (filename matches character)
5
+ references - other files where the character's name appears in content
6
+ """
7
+
8
+ import json
9
+ import os
10
+ import re
11
+ from pathlib import Path
12
+
13
+ ROOT = Path(__file__).parent
14
+ JSON_DIR = ROOT / "English_JSON"
15
+ OUT_DIR = ROOT / "character_data"
16
+
17
+
18
+ def _normalise_names(info: dict) -> list[str]:
19
+ names = info.get("name", [])
20
+ if isinstance(names, str):
21
+ names = [names] if names else []
22
+ detective = info.get("detective_name") or ""
23
+ if detective:
24
+ names = list(names) + [detective]
25
+ return [n for n in names if n]
26
+
27
+
28
+ def load_characters() -> dict:
29
+ """Merge all Contacts.json files into {id: {name, search_patterns, file_ids}}."""
30
+ chars: dict = {}
31
+ for contacts_path in sorted(JSON_DIR.rglob("Contacts.json")):
32
+ data = json.loads(contacts_path.read_text(encoding="utf-8"))
33
+ for char_id, info in data.items():
34
+ if char_id in chars:
35
+ continue
36
+ names = _normalise_names(info)
37
+ chars[char_id] = {
38
+ "name": names,
39
+ # Patterns for content search (word-boundary for plain words, substring otherwise)
40
+ "search_patterns": [_make_pattern(n) for n in names],
41
+ # Tokens to look for in filenames (ID + name variants, all lowercase)
42
+ "file_ids": {char_id.lower()} | {n.lower() for n in names},
43
+ }
44
+ return chars
45
+
46
+
47
+ def _make_pattern(name: str) -> re.Pattern:
48
+ if re.match(r"^[A-Za-z]+$", name):
49
+ return re.compile(r"\b" + re.escape(name) + r"\b", re.IGNORECASE)
50
+ return re.compile(re.escape(name), re.IGNORECASE)
51
+
52
+
53
+ def _is_chat_file(path: Path) -> bool:
54
+ return any(p in ("Conversations", "Filler Chats") for p in path.parts)
55
+
56
+
57
+ def _filename_matches(stem: str, file_ids: set[str]) -> bool:
58
+ stem_lower = stem.lower()
59
+ # Try each identifier: exact match, ends-with, or space-separated token
60
+ tokens = set(re.split(r"[^a-z0-9]+", stem_lower))
61
+ for fid in file_ids:
62
+ if fid in tokens or stem_lower == fid or stem_lower.endswith(fid):
63
+ return True
64
+ return False
65
+
66
+
67
+ def _flatten_strings(obj) -> list[str]:
68
+ if isinstance(obj, str):
69
+ return [obj]
70
+ if isinstance(obj, list):
71
+ out = []
72
+ for item in obj:
73
+ out.extend(_flatten_strings(item))
74
+ return out
75
+ if isinstance(obj, dict):
76
+ out = []
77
+ for k, v in obj.items():
78
+ out.append(k)
79
+ out.extend(_flatten_strings(v))
80
+ return out
81
+ return []
82
+
83
+
84
+ def _content_matches(text: str, patterns: list[re.Pattern]) -> bool:
85
+ return any(p.search(text) for p in patterns)
86
+
87
+
88
+ def main() -> None:
89
+ OUT_DIR.mkdir(exist_ok=True)
90
+ chars = load_characters()
91
+
92
+ results = {
93
+ char_id: {"id": char_id, "name": info["name"], "mentions": [], "references": []}
94
+ for char_id, info in chars.items()
95
+ }
96
+
97
+ for json_file in sorted(JSON_DIR.rglob("*.json")):
98
+ rel_path = str(json_file.relative_to(ROOT))
99
+ is_chat = _is_chat_file(json_file)
100
+ stem = json_file.stem
101
+
102
+ try:
103
+ data = json.loads(json_file.read_text(encoding="utf-8"))
104
+ except Exception:
105
+ continue
106
+
107
+ # Flatten content once per file; reuse across all character checks
108
+ flat_text = " ".join(_flatten_strings(data))
109
+
110
+ for char_id, info in chars.items():
111
+ if is_chat and _filename_matches(stem, info["file_ids"]):
112
+ results[char_id]["mentions"].append(rel_path)
113
+ elif info["search_patterns"] and _content_matches(flat_text, info["search_patterns"]):
114
+ results[char_id]["references"].append(rel_path)
115
+
116
+ for char_id, char_data in results.items():
117
+ out_path = OUT_DIR / f"{char_id}.json"
118
+ out_path.write_text(
119
+ json.dumps(char_data, indent=2, ensure_ascii=False), encoding="utf-8"
120
+ )
121
+
122
+ print(f"Written {len(results)} character files to {OUT_DIR}/")
123
+
124
+
125
+ if __name__ == "__main__":
126
+ main()
build_character_wikis.py ADDED
@@ -0,0 +1,265 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Build per-character voice wikis from English_JSON dialogue via a local Ollama model.
2
+
3
+ For each character indexed in character_data/{id}.json, processes the character's
4
+ "mentions" files one at a time. Each file gets its own stateless LLM call (a fresh
5
+ context window): the model receives the current wiki plus one file and returns the
6
+ updated wiki. The wiki is saved after every file and progress is tracked in
7
+ character_wikis/.progress.json, so an interrupted run resumes where it left off.
8
+
9
+ Usage:
10
+ python build_character_wikis.py # all characters (resumes)
11
+ python build_character_wikis.py --character brad # single character
12
+ python build_character_wikis.py --force # rebuild ignoring progress
13
+ """
14
+
15
+ import argparse
16
+ import json
17
+ import sys
18
+ import time
19
+ import urllib.error
20
+ import urllib.request
21
+ from pathlib import Path
22
+
23
+ ROOT = Path(__file__).parent
24
+ CHAR_DIR = ROOT / "character_data"
25
+ OUT_DIR = ROOT / "character_wikis"
26
+ PROGRESS_PATH = OUT_DIR / ".progress.json"
27
+
28
+ OLLAMA_URL = "http://localhost:11434/api/chat"
29
+ MODEL = "gemma4:12b-mlx"
30
+ # Gemma's recommended sampling parameters. 32k ctx comfortably fits the largest
31
+ # mention file (~13k tokens incl. prompt + wiki) without over-allocating KV cache.
32
+ MODEL_OPTIONS = {"temperature": 1, "top_k": 64, "top_p": 0.95, "num_ctx": 32768}
33
+ REQUEST_TIMEOUT_S = 1800
34
+ MAX_ATTEMPTS = 3
35
+ RETRY_BACKOFF_S = 5
36
+
37
+ SYSTEM_PROMPT = """\
38
+ You are building a character voice wiki for {name}{aka}, a character in "Riverstone", \
39
+ a narrative mystery mobile game told through phone chats and calls. Purpose of the \
40
+ wiki: a translator LLM will later receive a translated line of {name}'s dialogue plus \
41
+ this wiki, and adjust the translation to sound like {name} rather than a generic, \
42
+ robotic rendering. Record only what serves that goal.
43
+
44
+ You are given the current wiki and one game file. Update the wiki with any NEW \
45
+ evidence the file provides about:
46
+ - Tone & style: formal/casual, warmth, sarcasm, energy, politeness.
47
+ - Speech patterns: slang, catchphrases, filler words, abbreviations, \
48
+ emoji/punctuation habits, typical message length, grammar quirks.
49
+ - Personality & relationships, insofar as they shape how {name} speaks and to whom.
50
+ - Example lines: a few short verbatim quotes that best capture the voice.
51
+
52
+ Rules:
53
+ - Use only evidence from the file. Never invent details.
54
+ - Strings starting with "system::" are engine keys, and "#0"-style keys, "next" and \
55
+ "options_next" values are navigation IDs - not dialogue. Ignore them.
56
+ - If the file adds nothing new about {name}, return the wiki unchanged.
57
+ - Keep the wiki under 400 words, in markdown, under exactly these headings:
58
+ # {name}
59
+ ## Role & Relationships
60
+ ## Personality
61
+ ## Tone & Style
62
+ ## Speech Patterns
63
+ ## Example Lines
64
+ - Output ONLY the complete updated wiki. No commentary before or after.
65
+ """
66
+
67
+ USER_TEMPLATE = """\
68
+ CURRENT WIKI:
69
+ {wiki}
70
+
71
+ FILE: {path}
72
+ {hint}
73
+
74
+ FILE CONTENT:
75
+ {content}
76
+ """
77
+
78
+ EMPTY_WIKI_PLACEHOLDER = "(empty - this is the first file)"
79
+
80
+
81
+ def attribution_hint(rel_path: str, name: str) -> str:
82
+ """Tell the model who speaks which fields, based on the file's location."""
83
+ parts = Path(rel_path).parts
84
+ if "Adam_s Phone" in parts:
85
+ owner = "Detective Adam (the player)"
86
+ elif "Zoey_s Phone" in parts:
87
+ owner = "Zoey"
88
+ else:
89
+ owner = "the player"
90
+
91
+ if "Conversations" in parts:
92
+ return (
93
+ f'In this branching dialogue, "message" values are spoken by {name}; '
94
+ f'"options" values are reply choices spoken by {owner}, the phone\'s '
95
+ f"owner - not {name}."
96
+ )
97
+ if "Filler Chats" in parts:
98
+ return (
99
+ f"This is a chat log on {owner}'s phone between {owner} and {name} "
100
+ f'(group chats may include others). Different "type" values mark '
101
+ f"different speakers; attribute a line to {name} only when context "
102
+ f"makes the speaker clear."
103
+ )
104
+ return (
105
+ f"Extract only what this file reveals about {name}; do not attribute "
106
+ f"other speakers' lines to them."
107
+ )
108
+
109
+
110
+ def call_ollama(system_prompt: str, user_msg: str) -> str:
111
+ payload = json.dumps(
112
+ {
113
+ "model": MODEL,
114
+ "messages": [
115
+ {"role": "system", "content": system_prompt},
116
+ {"role": "user", "content": user_msg},
117
+ ],
118
+ "stream": False,
119
+ # gemma4 has thinking enabled by default, which multiplies generation
120
+ # time several-fold per call; the wiki task doesn't need it.
121
+ "think": False,
122
+ "options": MODEL_OPTIONS,
123
+ }
124
+ ).encode("utf-8")
125
+
126
+ last_error: Exception | None = None
127
+ for attempt in range(1, MAX_ATTEMPTS + 1):
128
+ request = urllib.request.Request(
129
+ OLLAMA_URL, data=payload, headers={"Content-Type": "application/json"}
130
+ )
131
+ try:
132
+ with urllib.request.urlopen(request, timeout=REQUEST_TIMEOUT_S) as resp:
133
+ data = json.loads(resp.read().decode("utf-8"))
134
+ return data["message"]["content"]
135
+ except (urllib.error.URLError, TimeoutError, KeyError, json.JSONDecodeError) as exc:
136
+ last_error = exc
137
+ if attempt < MAX_ATTEMPTS:
138
+ print(f" request failed ({exc}); retrying in {RETRY_BACKOFF_S}s...")
139
+ time.sleep(RETRY_BACKOFF_S)
140
+ raise RuntimeError(
141
+ f"Ollama request failed after {MAX_ATTEMPTS} attempts: {last_error}. "
142
+ f"Is the server running at {OLLAMA_URL}? Progress is saved; rerun to resume."
143
+ )
144
+
145
+
146
+ def clean_reply(text: str) -> str:
147
+ """Strip whitespace and an optional wrapping markdown code fence."""
148
+ text = text.strip()
149
+ if text.startswith("```") and text.endswith("```"):
150
+ first_newline = text.find("\n")
151
+ if first_newline != -1:
152
+ text = text[first_newline + 1 : -3].strip()
153
+ return text
154
+
155
+
156
+ def load_progress() -> dict:
157
+ if PROGRESS_PATH.exists():
158
+ return json.loads(PROGRESS_PATH.read_text(encoding="utf-8"))
159
+ return {}
160
+
161
+
162
+ def save_progress(progress: dict) -> None:
163
+ PROGRESS_PATH.write_text(
164
+ json.dumps(progress, indent=2, ensure_ascii=False), encoding="utf-8"
165
+ )
166
+
167
+
168
+ def process_character(char_data: dict, progress: dict, force: bool) -> dict:
169
+ """Run the per-file wiki-update loop for one character; returns new progress."""
170
+ char_id = char_data["id"]
171
+ names = char_data.get("name") or []
172
+ display_name = names[0] if names else char_id
173
+ aka = f' (also appearing as {", ".join(names[1:])})' if len(names) > 1 else ""
174
+ mentions = char_data.get("mentions", [])
175
+
176
+ if not mentions:
177
+ print(f"[{char_id}] no mention files - skipped")
178
+ return progress
179
+
180
+ wiki_path = OUT_DIR / f"{char_id}.md"
181
+ done = [] if force else list(progress.get(char_id, []))
182
+ wiki = ""
183
+ if not force and wiki_path.exists():
184
+ wiki = wiki_path.read_text(encoding="utf-8").strip()
185
+
186
+ pending = [m for m in mentions if m not in done]
187
+ if not pending:
188
+ print(f"[{char_id}] already complete ({len(mentions)} files)")
189
+ return progress
190
+
191
+ system_prompt = SYSTEM_PROMPT.format(name=display_name, aka=aka)
192
+ print(f"[{char_id}] {len(pending)} file(s) to process")
193
+
194
+ for i, rel_path in enumerate(pending, 1):
195
+ file_path = ROOT / rel_path
196
+ try:
197
+ data = json.loads(file_path.read_text(encoding="utf-8"))
198
+ except (OSError, json.JSONDecodeError) as exc:
199
+ print(f" ({i}/{len(pending)}) WARNING: cannot read {rel_path}: {exc}")
200
+ done = done + [rel_path]
201
+ progress = {**progress, char_id: done}
202
+ save_progress(progress)
203
+ continue
204
+
205
+ user_msg = USER_TEMPLATE.format(
206
+ wiki=wiki or EMPTY_WIKI_PLACEHOLDER,
207
+ path=rel_path.removeprefix("English_JSON/"),
208
+ hint=attribution_hint(rel_path, display_name),
209
+ content=json.dumps(data, indent=2, ensure_ascii=False),
210
+ )
211
+
212
+ started = time.monotonic()
213
+ reply = clean_reply(call_ollama(system_prompt, user_msg))
214
+ elapsed = time.monotonic() - started
215
+
216
+ if reply.startswith("#"):
217
+ wiki = reply
218
+ wiki_path.write_text(wiki + "\n", encoding="utf-8")
219
+ print(f" ({i}/{len(pending)}) {rel_path} - updated ({elapsed:.0f}s)")
220
+ else:
221
+ print(
222
+ f" ({i}/{len(pending)}) {rel_path} - WARNING: malformed reply, "
223
+ f"keeping previous wiki ({elapsed:.0f}s)"
224
+ )
225
+
226
+ done = done + [rel_path]
227
+ progress = {**progress, char_id: done}
228
+ save_progress(progress)
229
+
230
+ return progress
231
+
232
+
233
+ def main() -> None:
234
+ parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
235
+ parser.add_argument("--character", help="process a single character id (e.g. brad)")
236
+ parser.add_argument(
237
+ "--force", action="store_true", help="rebuild from scratch, ignoring saved progress"
238
+ )
239
+ args = parser.parse_args()
240
+ sys.stdout.reconfigure(line_buffering=True)
241
+
242
+ char_files = sorted(CHAR_DIR.glob("*.json"))
243
+ if args.character:
244
+ char_files = [CHAR_DIR / f"{args.character}.json"]
245
+ if not char_files[0].exists():
246
+ sys.exit(f"No such character: {args.character} (expected {char_files[0]})")
247
+ if not char_files:
248
+ sys.exit(f"No character files found in {CHAR_DIR}; run build_character_data.py first.")
249
+
250
+ OUT_DIR.mkdir(exist_ok=True)
251
+ progress = load_progress()
252
+
253
+ try:
254
+ for char_file in char_files:
255
+ char_data = json.loads(char_file.read_text(encoding="utf-8"))
256
+ progress = process_character(char_data, progress, args.force)
257
+ except KeyboardInterrupt:
258
+ print("\nInterrupted - progress saved; rerun to resume.")
259
+ sys.exit(130)
260
+
261
+ print(f"Done. Wikis written to {OUT_DIR}/")
262
+
263
+
264
+ if __name__ == "__main__":
265
+ main()