bhardwaj08sarthak commited on
Commit
478fb0c
·
verified ·
1 Parent(s): 21d8b09

Upload folder using huggingface_hub

Browse files
main/pipeline/__init__.py ADDED
File without changes
main/pipeline/build_file_context.py ADDED
@@ -0,0 +1,67 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Parse FILE_MAPPING.md into file_context.json.
2
+
3
+ FILE_MAPPING.md (teammate-generated) has one "### N. filename" section per
4
+ source file with bullet fields. We extract per file: content label, inferred
5
+ role, and detailed summary, keyed by the file's path inside the JSON pack
6
+ (e.g. "Initial/Gameplay/Adam_s Phone/Messages/General.json").
7
+
8
+ Usage:
9
+ python -m pipeline.build_file_context
10
+ """
11
+
12
+ import json
13
+ import re
14
+ import sys
15
+ from pathlib import Path
16
+
17
+ sys.path.insert(0, str(Path(__file__).parent.parent))
18
+ import config
19
+
20
+ _SECTION_RE = re.compile(r"^### \d+\. `", re.MULTILINE)
21
+ _FIELD_RES = {
22
+ "rel_path": re.compile(r"^- Relative path: `(.+?)`", re.MULTILINE),
23
+ "label": re.compile(r"^- Content label: `(.+?)`", re.MULTILINE),
24
+ "role": re.compile(r"^- Inferred role: (.+?)$", re.MULTILINE),
25
+ "summary": re.compile(r"^- Detailed summary: (.+?)$", re.MULTILINE),
26
+ }
27
+
28
+
29
+ def normalize_key(rel_path: str) -> str:
30
+ """Canonical lookup key: strip stray spaces around each path component."""
31
+ return "/".join(part.strip() for part in rel_path.replace("\\", "/").split("/"))
32
+
33
+
34
+ def _pack_key(rel_path: str) -> str:
35
+ """'English/Initial/.../General.docx' -> 'Initial/.../General.json'."""
36
+ path = rel_path.replace("\\", "/").removeprefix("English/")
37
+ return normalize_key(str(Path(path).with_suffix(".json")))
38
+
39
+
40
+ def parse_file_mapping(markdown: str) -> dict[str, dict[str, str]]:
41
+ context: dict[str, dict[str, str]] = {}
42
+ for section in _SECTION_RE.split(markdown)[1:]:
43
+ fields = {}
44
+ for name, pattern in _FIELD_RES.items():
45
+ match = pattern.search(section)
46
+ fields[name] = match.group(1).strip() if match else ""
47
+ if not fields["rel_path"]:
48
+ continue
49
+ context[_pack_key(fields["rel_path"])] = {
50
+ "label": fields["label"],
51
+ "role": fields["role"],
52
+ "summary": fields["summary"],
53
+ }
54
+ return context
55
+
56
+
57
+ def main() -> None:
58
+ markdown = config.FILE_MAPPING_PATH.read_text(encoding="utf-8")
59
+ context = parse_file_mapping(markdown)
60
+ config.FILE_CONTEXT_PATH.write_text(
61
+ json.dumps(context, indent=2, ensure_ascii=False), encoding="utf-8"
62
+ )
63
+ print(f"Wrote context for {len(context)} files to {config.FILE_CONTEXT_PATH.name}")
64
+
65
+
66
+ if __name__ == "__main__":
67
+ main()
main/pipeline/clients.py ADDED
@@ -0,0 +1,414 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """In-process Hugging Face clients for the two translation models.
2
+
3
+ Stage 1 - `translate`: google/translategemma-12b-it via its official
4
+ TranslateGemma chat template.
5
+
6
+ Stage 2 - `adjust_tone`: google/gemma-4-12B-it via the normal Gemma 4 chat
7
+ template, rewriting a draft translation in a character's voice using their
8
+ wiki.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import gc
14
+ import re
15
+ import sys
16
+ from dataclasses import dataclass
17
+ from pathlib import Path
18
+ from typing import Any
19
+
20
+ sys.path.insert(0, str(Path(__file__).parent.parent))
21
+ import config
22
+
23
+ try:
24
+ import spaces
25
+ except ImportError: # Local/dev installs do not need the Spaces runtime.
26
+ spaces = None
27
+
28
+
29
+ # Runtime-injected tokens like "Person1" that must survive translation verbatim.
30
+ _PLACEHOLDER_RE = re.compile(r"\b([A-Za-z]+)(\d+)\b")
31
+
32
+ TONE_SYSTEM_TEMPLATE = """\
33
+ You are a localization editor for "Riverstone", a narrative mystery mobile game \
34
+ told through phone chats. Below is the voice wiki for {name}, a game character.
35
+
36
+ {wiki}
37
+
38
+ You will receive one of {name}'s chat lines in English and a draft {language} \
39
+ translation. Rewrite the {language} draft so it reads like {name} texting in \
40
+ {language}: match the wiki's tone, register, slang level, emoji and punctuation \
41
+ habits.
42
+
43
+ Rules:
44
+ - Keep the exact meaning of the English line; never add or drop information.
45
+ - Keep placeholder tokens (e.g. Person1), proper names, emoji and special \
46
+ symbols exactly as written.
47
+ - Use the formality level the wiki implies for {name} (casual characters use \
48
+ informal address).
49
+ - If the draft already sounds right, return it unchanged.
50
+ - Output ONLY the final {language} line - no quotes, no commentary.
51
+ """
52
+
53
+ TONE_USER_TEMPLATE = """\
54
+ {context}English line: {source}
55
+ Draft {language} translation: {draft}
56
+ Final {language} line:"""
57
+
58
+ WIKI_MAX_NEW_TOKENS = 1200
59
+
60
+
61
+ @dataclass
62
+ class _LoadedModel:
63
+ processor: Any
64
+ model: Any
65
+
66
+
67
+ _MODELS: dict[str, _LoadedModel] = {}
68
+ _MODEL_IDS = {
69
+ "translate": config.TRANSLATE_MODEL_ID,
70
+ "tone": config.TONE_MODEL_ID,
71
+ }
72
+
73
+
74
+ def _gpu(fn):
75
+ if spaces is None:
76
+ return fn
77
+ return spaces.GPU(
78
+ duration=config.ZERO_GPU_DURATION_S,
79
+ size=config.ZERO_GPU_SIZE,
80
+ )(fn)
81
+
82
+
83
+ def _torch():
84
+ try:
85
+ import torch
86
+ except ImportError as exc:
87
+ raise RuntimeError(
88
+ "Hugging Face inference requires torch. Install the Space/runtime "
89
+ "dependencies from requirements.txt."
90
+ ) from exc
91
+ return torch
92
+
93
+
94
+ def _hf_classes():
95
+ try:
96
+ from transformers import AutoModelForMultimodalLM, AutoProcessor
97
+
98
+ return AutoProcessor, AutoModelForMultimodalLM
99
+ except ImportError:
100
+ try:
101
+ from transformers import AutoModelForImageTextToText, AutoProcessor
102
+
103
+ return AutoProcessor, AutoModelForImageTextToText
104
+ except ImportError as exc:
105
+ raise RuntimeError(
106
+ "Hugging Face inference requires a recent transformers release "
107
+ "with Gemma multimodal model support."
108
+ ) from exc
109
+
110
+
111
+ def _from_pretrained(model_id: str):
112
+ AutoProcessor, AutoModel = _hf_classes()
113
+ kwargs: dict[str, Any] = {}
114
+ if config.HF_DEVICE_MAP is not None:
115
+ kwargs["device_map"] = config.HF_DEVICE_MAP
116
+ if config.HF_DTYPE is not None:
117
+ kwargs["dtype"] = config.HF_DTYPE
118
+ if config.HF_ATTN_IMPLEMENTATION:
119
+ kwargs["attn_implementation"] = config.HF_ATTN_IMPLEMENTATION
120
+
121
+ processor = AutoProcessor.from_pretrained(model_id)
122
+ try:
123
+ model = AutoModel.from_pretrained(model_id, **kwargs)
124
+ except TypeError:
125
+ # Older Transformers used torch_dtype instead of dtype.
126
+ if "dtype" in kwargs:
127
+ kwargs["torch_dtype"] = kwargs.pop("dtype")
128
+ model = AutoModel.from_pretrained(model_id, **kwargs)
129
+ model.eval()
130
+ return _LoadedModel(processor=processor, model=model)
131
+
132
+
133
+ def release_models(except_key: str | None = None) -> None:
134
+ """Free loaded HF models, optionally keeping one active model resident."""
135
+ for key in list(_MODELS):
136
+ if key != except_key:
137
+ del _MODELS[key]
138
+ gc.collect()
139
+ try:
140
+ torch = _torch()
141
+ if torch.cuda.is_available():
142
+ torch.cuda.empty_cache()
143
+ except RuntimeError:
144
+ pass
145
+
146
+
147
+ def warm_models(*keys: str) -> None:
148
+ """Preload selected models, useful from a Space module at startup."""
149
+ for key in keys:
150
+ _get_model(key)
151
+
152
+
153
+ def _get_model(key: str) -> _LoadedModel:
154
+ if key not in _MODEL_IDS:
155
+ raise ValueError(f"Unknown model key: {key}")
156
+ if key not in _MODELS:
157
+ if not config.HF_KEEP_BOTH_MODELS:
158
+ release_models(except_key=key)
159
+ _MODELS[key] = _from_pretrained(_MODEL_IDS[key])
160
+ return _MODELS[key]
161
+
162
+
163
+ def _model_device(model):
164
+ device = getattr(model, "device", None)
165
+ if device is not None:
166
+ return device
167
+ try:
168
+ return next(model.parameters()).device
169
+ except StopIteration:
170
+ return None
171
+
172
+
173
+ def _model_dtype(model):
174
+ try:
175
+ for parameter in model.parameters():
176
+ if parameter.is_floating_point():
177
+ return parameter.dtype
178
+ except StopIteration:
179
+ return None
180
+ return None
181
+
182
+
183
+ def _move_inputs(inputs, model):
184
+ device = _model_device(model)
185
+ dtype = _model_dtype(model)
186
+ if device is None:
187
+ return inputs
188
+ if dtype is not None:
189
+ try:
190
+ return inputs.to(device, dtype=dtype)
191
+ except TypeError:
192
+ pass
193
+ return inputs.to(device)
194
+
195
+
196
+ def _apply_chat_template(processor, messages, *, enable_thinking: bool = False):
197
+ kwargs = {
198
+ "tokenize": True,
199
+ "add_generation_prompt": True,
200
+ "return_dict": True,
201
+ "return_tensors": "pt",
202
+ }
203
+ try:
204
+ return processor.apply_chat_template(
205
+ messages,
206
+ enable_thinking=enable_thinking,
207
+ **kwargs,
208
+ )
209
+ except TypeError:
210
+ return processor.apply_chat_template(messages, **kwargs)
211
+
212
+
213
+ def _generate_text(
214
+ key: str,
215
+ messages: list[dict],
216
+ *,
217
+ max_new_tokens: int,
218
+ enable_thinking: bool = False,
219
+ **generation_kwargs,
220
+ ) -> str:
221
+ loaded = _get_model(key)
222
+ torch = _torch()
223
+ inputs = _apply_chat_template(
224
+ loaded.processor,
225
+ messages,
226
+ enable_thinking=enable_thinking,
227
+ )
228
+ inputs = _move_inputs(inputs, loaded.model)
229
+ input_len = inputs["input_ids"].shape[-1]
230
+
231
+ with torch.inference_mode():
232
+ output = loaded.model.generate(
233
+ **inputs,
234
+ max_new_tokens=max_new_tokens,
235
+ **generation_kwargs,
236
+ )
237
+
238
+ generated = output[0][input_len:]
239
+ decoded = loaded.processor.decode(generated, skip_special_tokens=False)
240
+ return _parse_response(loaded.processor, decoded)
241
+
242
+
243
+ def _parse_response(processor, decoded: str) -> str:
244
+ if hasattr(processor, "parse_response"):
245
+ try:
246
+ parsed = processor.parse_response(decoded)
247
+ if isinstance(parsed, str):
248
+ return parsed.strip()
249
+ if isinstance(parsed, dict):
250
+ for key in ("content", "text", "response", "answer"):
251
+ value = parsed.get(key)
252
+ if isinstance(value, str):
253
+ return value.strip()
254
+ except Exception:
255
+ pass
256
+
257
+ text = decoded
258
+ text = re.sub(r"<\|channel\>thought.*?<channel\|>", "", text, flags=re.DOTALL)
259
+ text = re.sub(r"<[^>]+>", "", text)
260
+ return text.strip()
261
+
262
+
263
+ def restore_placeholders(source: str, translated: str) -> str:
264
+ """Undo model damage to PersonN-style tokens (e.g. 'Person 1' -> 'Person1')."""
265
+ result = translated
266
+ for word, num in _PLACEHOLDER_RE.findall(source):
267
+ token = f"{word}{num}"
268
+ if token in result:
269
+ continue
270
+ spaced = re.compile(rf"\b{re.escape(word)}\s+{num}\b", re.IGNORECASE)
271
+ result = spaced.sub(token, result)
272
+ return result
273
+
274
+
275
+ def _translate_text(text: str) -> str:
276
+ messages = [
277
+ {
278
+ "role": "user",
279
+ "content": [
280
+ {
281
+ "type": "text",
282
+ "source_lang_code": config.SOURCE_LANG_CODE,
283
+ "target_lang_code": config.TARGET_LANG_CODE,
284
+ "text": text,
285
+ }
286
+ ],
287
+ }
288
+ ]
289
+ reply = _generate_text(
290
+ "translate",
291
+ messages,
292
+ max_new_tokens=config.TRANSLATE_MAX_NEW_TOKENS,
293
+ do_sample=False,
294
+ )
295
+ return restore_placeholders(text, reply)
296
+
297
+
298
+ def _adjust_tone_text(
299
+ source: str,
300
+ draft: str,
301
+ wiki: str,
302
+ char_name: str,
303
+ context_lines: list[str] | None = None,
304
+ ) -> str:
305
+ """Rewrite a draft translation in the character's voice via Gemma 4."""
306
+ context = ""
307
+ if context_lines:
308
+ joined = "\n".join(f" {line}" for line in context_lines)
309
+ context = f"Preceding lines in this chat (English, for context only):\n{joined}\n\n"
310
+ messages = [
311
+ {
312
+ "role": "system",
313
+ "content": TONE_SYSTEM_TEMPLATE.format(
314
+ name=char_name, wiki=wiki, language=config.TARGET_LANG_NAME
315
+ ),
316
+ },
317
+ {
318
+ "role": "user",
319
+ "content": TONE_USER_TEMPLATE.format(
320
+ context=context,
321
+ source=source,
322
+ draft=draft,
323
+ language=config.TARGET_LANG_NAME,
324
+ ),
325
+ },
326
+ ]
327
+ reply = _generate_text(
328
+ "tone",
329
+ messages,
330
+ max_new_tokens=config.TONE_MAX_NEW_TOKENS,
331
+ enable_thinking=False,
332
+ **config.TONE_GENERATION_KWARGS,
333
+ )
334
+ reply = reply.strip().strip('"').strip()
335
+ return restore_placeholders(source, reply) if reply else draft
336
+
337
+
338
+ @_gpu
339
+ def translate(text: str) -> str:
340
+ """Translate one string with TranslateGemma (greedy)."""
341
+ return _translate_text(text)
342
+
343
+
344
+ @_gpu
345
+ def adjust_tone(
346
+ source: str,
347
+ draft: str,
348
+ wiki: str,
349
+ char_name: str,
350
+ context_lines: list[str] | None = None,
351
+ ) -> str:
352
+ """Rewrite a draft translation in the character's voice via Gemma 4."""
353
+ return _adjust_tone_text(source, draft, wiki, char_name, context_lines)
354
+
355
+
356
+ @_gpu
357
+ def translate_and_tone_items(
358
+ items: list[dict],
359
+ cache: dict,
360
+ wiki: str | None,
361
+ char_name: str | None,
362
+ context_window: int = 2,
363
+ ) -> tuple[list[dict], dict, dict]:
364
+ """Translate and tone a full review record in one GPU allocation."""
365
+ cache = dict(cache)
366
+ translated = 0
367
+ toned = 0
368
+
369
+ for item in items:
370
+ if item.get("mt") is not None:
371
+ continue
372
+ source = item["source"]
373
+ cached = cache.get(source)
374
+ if cached is not None:
375
+ item["mt"] = cached
376
+ else:
377
+ item["mt"] = _translate_text(source)
378
+ cache[source] = item["mt"]
379
+ translated += 1
380
+
381
+ if wiki and char_name:
382
+ sources = [item["source"] for item in items]
383
+ for i, item in enumerate(items):
384
+ if item.get("kind") != "dialogue" or item.get("toned") is not None:
385
+ continue
386
+ if item.get("mt") is None:
387
+ continue
388
+ context = sources[max(0, i - context_window) : i]
389
+ item["toned"] = _adjust_tone_text(
390
+ item["source"], item["mt"], wiki, char_name, context
391
+ )
392
+ toned += 1
393
+
394
+ return items, cache, {"translated": translated, "toned": toned}
395
+
396
+
397
+ @_gpu
398
+ def update_character_wiki(system_prompt: str, user_msg: str) -> str:
399
+ """Update a character wiki via the same in-process Gemma runtime as tone pass."""
400
+ messages = [
401
+ {"role": "system", "content": system_prompt},
402
+ {"role": "user", "content": user_msg},
403
+ ]
404
+ return _generate_text(
405
+ "tone",
406
+ messages,
407
+ max_new_tokens=WIKI_MAX_NEW_TOKENS,
408
+ enable_thinking=False,
409
+ **config.TONE_GENERATION_KWARGS,
410
+ ).strip()
411
+
412
+
413
+ if getattr(config, "HF_PRELOAD_MODELS", ()):
414
+ warm_models(*config.HF_PRELOAD_MODELS)
main/pipeline/dialogue_map.py ADDED
@@ -0,0 +1,68 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Map chat files to the character (or group) whose lines they contain.
2
+
3
+ Built from character_data/{id}.json "mentions" (produced by
4
+ build_character_data.py), keeping only characters that have a voice wiki in
5
+ character_wikis/. When several characters claim the same file (group chats,
6
+ flashback variants), the character whose id matches the filename stem wins,
7
+ then the longest id as tie-break.
8
+
9
+ Speaker attribution within a mapped file (see build_character_wikis.py):
10
+ - Conversations: "message" = the mapped character, "options" = player/owner.
11
+ - Filler Chats: type "1" = player/owner, type "-1" = system, anything else =
12
+ the mapped character/group.
13
+ """
14
+
15
+ import json
16
+ import re
17
+ import sys
18
+ from pathlib import Path
19
+
20
+ sys.path.insert(0, str(Path(__file__).parent.parent))
21
+ import config
22
+
23
+
24
+ def _stem_key(rel_path: str) -> str:
25
+ """Normalised filename stem: 'Flashback 1 Ralph.json' -> 'flashback1ralph'."""
26
+ return re.sub(r"[^a-z0-9]", "", Path(rel_path).stem.lower())
27
+
28
+
29
+ def build_dialogue_map() -> dict[str, str]:
30
+ """Return {rel_path inside the source pack: character_id}."""
31
+ claims: dict[str, list[str]] = {}
32
+ for char_file in sorted(config.CHAR_DATA_DIR.glob("*.json")):
33
+ char_id = char_file.stem
34
+ if not (config.CHAR_WIKI_DIR / f"{char_id}.md").exists():
35
+ continue
36
+ data = json.loads(char_file.read_text(encoding="utf-8"))
37
+ for mention in data.get("mentions", []):
38
+ rel = mention.removeprefix("English_JSON/")
39
+ claims.setdefault(rel, []).append(char_id)
40
+
41
+ mapping: dict[str, str] = {}
42
+ for rel, char_ids in claims.items():
43
+ stem = _stem_key(rel)
44
+ exact = [c for c in char_ids if stem.endswith(c)]
45
+ candidates = exact or char_ids
46
+ mapping[rel] = max(candidates, key=len)
47
+ return mapping
48
+
49
+
50
+ def load_wiki(char_id: str) -> str:
51
+ return (config.CHAR_WIKI_DIR / f"{char_id}.md").read_text(encoding="utf-8").strip()
52
+
53
+
54
+ def display_name(char_id: str) -> str:
55
+ """Character display name from character_data (falls back to the id)."""
56
+ char_file = config.CHAR_DATA_DIR / f"{char_id}.json"
57
+ if char_file.exists():
58
+ names = json.loads(char_file.read_text(encoding="utf-8")).get("name") or []
59
+ if names:
60
+ return names[0]
61
+ return char_id
62
+
63
+
64
+ if __name__ == "__main__":
65
+ dmap = build_dialogue_map()
66
+ print(f"{len(dmap)} chat files mapped to {len(set(dmap.values()))} characters")
67
+ for rel, cid in sorted(dmap.items())[:10]:
68
+ print(f" {cid:15s} {rel}")
main/pipeline/export_pack.py ADDED
@@ -0,0 +1,83 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Export the translated language pack (e.g. German_JSON/) from review records.
2
+
3
+ Every file from the source pack is deep-copied with translations substituted at
4
+ the recorded key paths; structure, IDs, and untranslatable values stay intact.
5
+ Per string the best available text is used: reviewer-final > toned > machine
6
+ translation > English source. A review-status summary is printed at the end.
7
+
8
+ Usage:
9
+ python -m pipeline.export_pack
10
+ """
11
+
12
+ import json
13
+ import sys
14
+ from pathlib import Path
15
+
16
+ sys.path.insert(0, str(Path(__file__).parent.parent))
17
+ import config
18
+ from pipeline.rules import get_at, set_many
19
+
20
+
21
+ def best_text(item: dict) -> str | None:
22
+ return item.get("final") or item.get("toned") or item.get("mt")
23
+
24
+
25
+ def export_file(source_data, record: dict) -> tuple[dict, dict]:
26
+ """Return (translated copy of source_data, status counts)."""
27
+ counts = {"approved": 0, "edited": 0, "rejected": 0, "pending": 0, "untranslated": 0}
28
+ updates = []
29
+ for item in record["items"]:
30
+ path = tuple(item["path"])
31
+ text = best_text(item)
32
+ if text is None:
33
+ counts["untranslated"] += 1
34
+ continue
35
+ if get_at(source_data, path) != item["source"]:
36
+ raise ValueError(f"source drift at {item['key']}: rerun translate_pack")
37
+ updates.append((path, text))
38
+ counts[item["status"]] += 1
39
+ return set_many(source_data, updates), counts
40
+
41
+
42
+ def main() -> None:
43
+ if not config.TRANSLATIONS_DIR.exists():
44
+ sys.exit("No review records found - run pipeline/translate_pack.py first.")
45
+
46
+ totals = {"approved": 0, "edited": 0, "rejected": 0, "pending": 0, "untranslated": 0}
47
+ exported = 0
48
+ for source_file in sorted(config.SOURCE_DIR.rglob("*.json")):
49
+ rel = str(source_file.relative_to(config.SOURCE_DIR))
50
+ source_data = json.loads(source_file.read_text(encoding="utf-8"))
51
+ record_path = config.TRANSLATIONS_DIR / rel
52
+
53
+ if record_path.exists():
54
+ record = json.loads(record_path.read_text(encoding="utf-8"))
55
+ translated, counts = export_file(source_data, record)
56
+ for key, n in counts.items():
57
+ totals[key] += n
58
+ else:
59
+ translated = source_data # nothing translatable or not yet processed
60
+
61
+ out_path = config.PACK_DIR / rel
62
+ out_path.parent.mkdir(parents=True, exist_ok=True)
63
+ out_path.write_text(
64
+ json.dumps(translated, indent=2, ensure_ascii=False), encoding="utf-8"
65
+ )
66
+ exported += 1
67
+
68
+ reviewed = totals["approved"] + totals["edited"]
69
+ translated_n = sum(totals.values()) - totals["untranslated"]
70
+ print(f"Exported {exported} files to {config.PACK_DIR.name}/")
71
+ print(
72
+ f"Strings: {translated_n} translated "
73
+ f"({totals['approved']} approved, {totals['edited']} edited, "
74
+ f"{totals['rejected']} rejected*, {totals['pending']} pending*, "
75
+ f"{totals['untranslated']} still English)."
76
+ )
77
+ print("* rejected/pending strings ship with the unreviewed machine translation.")
78
+ if translated_n:
79
+ print(f"Human-reviewed: {reviewed / translated_n:.0%}")
80
+
81
+
82
+ if __name__ == "__main__":
83
+ main()
main/pipeline/rules.py ADDED
@@ -0,0 +1,147 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Translatability rules: which strings in a language-pack JSON file get translated.
2
+
3
+ The public API is `iter_translatable(data, rel_path)`, which yields `Item`s with a
4
+ key path addressing the string inside the file, the source string, and a `kind`:
5
+
6
+ dialogue - a line spoken by the file's character (eligible for the tone pass)
7
+ player - a line spoken by the phone owner / player (plain translation)
8
+ other - UI copy, web text, notes, etc. (plain translation)
9
+
10
+ `get_at` / `set_at` address values by the same key paths and are used by the
11
+ exporter to substitute translations without touching structure.
12
+ """
13
+
14
+ import copy
15
+ import re
16
+ from collections.abc import Iterator
17
+ from typing import Any, NamedTuple
18
+
19
+ # Field names that hold structural data, never prose.
20
+ SKIP_FIELDS = {"id", "audio", "next", "options_next", "time", "type"}
21
+
22
+ # Speaker codes in Filler Chats ("Schema B").
23
+ TYPE_PLAYER = "1"
24
+ TYPE_SYSTEM = "-1"
25
+
26
+ _LETTER_RE = re.compile(r"[A-Za-z]")
27
+ _URL_OR_FILE_RE = re.compile(r"^[\w.-]+\.[a-z][a-z0-9]{1,5}(/\S*)?$", re.IGNORECASE)
28
+
29
+
30
+ class Item(NamedTuple):
31
+ path: tuple # key path into the file's JSON structure
32
+ source: str
33
+ kind: str # "dialogue" | "player" | "other"
34
+
35
+
36
+ def is_translatable_value(value: Any) -> bool:
37
+ """Value-level filter applied to every candidate string."""
38
+ if not isinstance(value, str):
39
+ return False
40
+ text = value.strip()
41
+ if not text or text == "-":
42
+ return False
43
+ if text.startswith("system::"):
44
+ return False
45
+ if text.startswith(("http://", "https://")) or _URL_OR_FILE_RE.match(text):
46
+ return False
47
+ if not _LETTER_RE.search(text): # numbers, emoji, punctuation only
48
+ return False
49
+ return True
50
+
51
+
52
+ def detect_schema(data: Any, rel_path: str) -> str:
53
+ """Classify a file as 'conversation', 'chat_log', or 'generic'."""
54
+ parts = rel_path.replace("\\", "/").split("/")
55
+ if "Conversations" in parts and isinstance(data, dict):
56
+ if any(isinstance(v, dict) and "message" in v for v in data.values()):
57
+ return "conversation"
58
+ if "Filler Chats" in parts:
59
+ rows = data if isinstance(data, list) else list(data.values()) if isinstance(data, dict) else []
60
+ if any(isinstance(r, dict) and "text" in r for r in rows):
61
+ return "chat_log"
62
+ return "generic"
63
+
64
+
65
+ def iter_translatable(data: Any, rel_path: str) -> Iterator[Item]:
66
+ schema = detect_schema(data, rel_path)
67
+ if schema == "conversation":
68
+ yield from _iter_conversation(data)
69
+ elif schema == "chat_log":
70
+ yield from _iter_chat_log(data)
71
+ else:
72
+ yield from _iter_generic(data, ())
73
+
74
+
75
+ def _iter_conversation(data: dict) -> Iterator[Item]:
76
+ for msg_id, node in data.items():
77
+ if not isinstance(node, dict):
78
+ continue
79
+ message = node.get("message")
80
+ if is_translatable_value(message):
81
+ yield Item((msg_id, "message"), message, "dialogue")
82
+ options = node.get("options")
83
+ if isinstance(options, str):
84
+ if is_translatable_value(options):
85
+ yield Item((msg_id, "options"), options, "player")
86
+ elif isinstance(options, list):
87
+ for i, opt in enumerate(options):
88
+ if is_translatable_value(opt):
89
+ yield Item((msg_id, "options", i), opt, "player")
90
+
91
+
92
+ def _iter_chat_log(data: Any) -> Iterator[Item]:
93
+ rows = enumerate(data) if isinstance(data, list) else data.items()
94
+ for key, row in rows:
95
+ if not isinstance(row, dict):
96
+ continue
97
+ text = row.get("text")
98
+ if not is_translatable_value(text):
99
+ continue
100
+ row_type = str(row.get("type", ""))
101
+ if row_type == TYPE_PLAYER:
102
+ kind = "player"
103
+ elif row_type == TYPE_SYSTEM:
104
+ kind = "other"
105
+ else: # "0" and any other code = a non-player chat participant
106
+ kind = "dialogue"
107
+ yield Item((key, "text"), text, kind)
108
+
109
+
110
+ def _iter_generic(node: Any, path: tuple) -> Iterator[Item]:
111
+ if isinstance(node, dict):
112
+ for key, value in node.items():
113
+ if key in SKIP_FIELDS:
114
+ continue
115
+ yield from _iter_generic(value, path + (key,))
116
+ elif isinstance(node, list):
117
+ for i, value in enumerate(node):
118
+ yield from _iter_generic(value, path + (i,))
119
+ elif is_translatable_value(node):
120
+ yield Item(path, node, "other")
121
+
122
+
123
+ def get_at(data: Any, path: tuple) -> Any:
124
+ node = data
125
+ for key in path:
126
+ node = node[key]
127
+ return node
128
+
129
+
130
+ def set_at(data: Any, path: tuple, value: Any) -> Any:
131
+ """Return a deep copy of `data` with the value at `path` replaced."""
132
+ return set_many(data, [(path, value)])
133
+
134
+
135
+ def set_many(data: Any, updates: list[tuple[tuple, Any]]) -> Any:
136
+ """Return a deep copy of `data` with every (path, value) update applied."""
137
+ result = copy.deepcopy(data)
138
+ for path, value in updates:
139
+ node = result
140
+ for key in path[:-1]:
141
+ node = node[key]
142
+ node[path[-1]] = value
143
+ return result
144
+
145
+
146
+ def key_to_str(path: tuple) -> str:
147
+ return ".".join(str(p) for p in path)
main/pipeline/translate_pack.py ADDED
@@ -0,0 +1,196 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Batch translation driver: English_JSON -> translations/<lang>/ review records.
2
+
3
+ Stage 1 (translate): every translatable string through TranslateGemma.
4
+ Stage 2 (tone): dialogue lines of mapped characters through the Ollama tone
5
+ model using the character's voice wiki.
6
+
7
+ One review record file is written per source file, mirroring the pack layout.
8
+ Runs are resumable: existing records are merged by item key and only missing
9
+ work is done. Repeated strings hit a shared translation cache.
10
+
11
+ Usage:
12
+ python -m pipeline.translate_pack # full run, both stages
13
+ python -m pipeline.translate_pack --filter "Filler Chats/Brad"
14
+ python -m pipeline.translate_pack --stage translate # stage 1 only
15
+ python -m pipeline.translate_pack --limit 5 # first 5 files
16
+ python -m pipeline.translate_pack --dry-run # plan only, no LLM calls
17
+ """
18
+
19
+ import argparse
20
+ import json
21
+ import sys
22
+ import time
23
+ from pathlib import Path
24
+
25
+ sys.path.insert(0, str(Path(__file__).parent.parent))
26
+ import config
27
+ from pipeline import clients
28
+ from pipeline.dialogue_map import build_dialogue_map, display_name, load_wiki
29
+ from pipeline.rules import iter_translatable, key_to_str
30
+
31
+ SAVE_EVERY = 25 # persist record/cache after this many new translations
32
+ CONTEXT_LINES = 2 # preceding source lines passed to the tone model
33
+
34
+
35
+ def load_record(record_path: Path) -> dict:
36
+ if record_path.exists():
37
+ return json.loads(record_path.read_text(encoding="utf-8"))
38
+ return {}
39
+
40
+
41
+ def save_json(path: Path, data: dict) -> None:
42
+ path.parent.mkdir(parents=True, exist_ok=True)
43
+ path.write_text(json.dumps(data, indent=2, ensure_ascii=False), encoding="utf-8")
44
+
45
+
46
+ def build_items(data, rel: str, char_id: str | None, existing: dict) -> list[dict]:
47
+ """Fresh item list from the source file, merged with prior record state."""
48
+ prior = {item["key"]: item for item in existing.get("items", [])}
49
+ items = []
50
+ for found in iter_translatable(data, rel):
51
+ key = key_to_str(found.path)
52
+ old = prior.get(key)
53
+ if old and old["source"] == found.source:
54
+ items.append(old)
55
+ continue
56
+ items.append(
57
+ {
58
+ "key": key,
59
+ "path": list(found.path),
60
+ "source": found.source,
61
+ "kind": found.kind,
62
+ "character": char_id if found.kind == "dialogue" else None,
63
+ "mt": None,
64
+ "toned": None,
65
+ "status": "pending",
66
+ "final": None,
67
+ }
68
+ )
69
+ return items
70
+
71
+
72
+ def run_translate_stage(record: dict, record_path: Path, cache: dict) -> int:
73
+ done = 0
74
+ for item in record["items"]:
75
+ if item["mt"] is not None:
76
+ continue
77
+ cached = cache.get(item["source"])
78
+ if cached is not None:
79
+ item["mt"] = cached
80
+ else:
81
+ item["mt"] = clients.translate(item["source"])
82
+ cache[item["source"]] = item["mt"]
83
+ done += 1
84
+ if done % SAVE_EVERY == 0:
85
+ save_json(record_path, record)
86
+ save_json(config.CACHE_PATH, cache)
87
+ return done
88
+
89
+
90
+ def run_tone_stage(record: dict, record_path: Path) -> int:
91
+ char_id = record.get("character")
92
+ if not char_id:
93
+ return 0
94
+ wiki = load_wiki(char_id)
95
+ name = display_name(char_id)
96
+ sources = [item["source"] for item in record["items"]]
97
+
98
+ done = 0
99
+ for i, item in enumerate(record["items"]):
100
+ if item["kind"] != "dialogue" or item["toned"] is not None or item["mt"] is None:
101
+ continue
102
+ context = sources[max(0, i - CONTEXT_LINES) : i]
103
+ item["toned"] = clients.adjust_tone(item["source"], item["mt"], wiki, name, context)
104
+ done += 1
105
+ if done % SAVE_EVERY == 0:
106
+ save_json(record_path, record)
107
+ return done
108
+
109
+
110
+ def needs_work(record: dict, stage: str) -> bool:
111
+ for item in record.get("items", []):
112
+ if stage in ("translate", "all") and item["mt"] is None:
113
+ return True
114
+ if (
115
+ stage in ("tone", "all")
116
+ and item["kind"] == "dialogue"
117
+ and record.get("character")
118
+ and item["mt"] is not None
119
+ and item["toned"] is None
120
+ ):
121
+ return True
122
+ return False
123
+
124
+
125
+ def main() -> None:
126
+ parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
127
+ parser.add_argument("--filter", default="", help="only files whose path contains this substring")
128
+ parser.add_argument("--limit", type=int, default=0, help="max number of files to process")
129
+ parser.add_argument("--stage", choices=["translate", "tone", "all"], default="all")
130
+ parser.add_argument("--dry-run", action="store_true", help="report planned work, no LLM calls")
131
+ args = parser.parse_args()
132
+ sys.stdout.reconfigure(line_buffering=True)
133
+
134
+ dmap = build_dialogue_map()
135
+ cache = json.loads(config.CACHE_PATH.read_text(encoding="utf-8")) if config.CACHE_PATH.exists() else {}
136
+
137
+ source_files = sorted(config.SOURCE_DIR.rglob("*.json"))
138
+ processed = 0
139
+ totals = {"translated": 0, "toned": 0, "files": 0}
140
+
141
+ try:
142
+ for source_file in source_files:
143
+ rel = str(source_file.relative_to(config.SOURCE_DIR))
144
+ if args.filter and args.filter not in rel:
145
+ continue
146
+ if args.limit and processed >= args.limit:
147
+ break
148
+
149
+ data = json.loads(source_file.read_text(encoding="utf-8"))
150
+ char_id = dmap.get(rel)
151
+ record_path = config.TRANSLATIONS_DIR / rel
152
+ record = load_record(record_path)
153
+ record = {
154
+ "file": rel,
155
+ "character": char_id,
156
+ "items": build_items(data, rel, char_id, record),
157
+ }
158
+ if not record["items"]:
159
+ continue
160
+ processed += 1
161
+ if not needs_work(record, args.stage):
162
+ continue
163
+
164
+ pending_mt = sum(1 for i in record["items"] if i["mt"] is None)
165
+ pending_tone = sum(
166
+ 1
167
+ for i in record["items"]
168
+ if i["kind"] == "dialogue" and char_id and i["toned"] is None
169
+ )
170
+ print(f"[{rel}] strings={len(record['items'])} mt-pending={pending_mt} "
171
+ f"tone-pending={pending_tone if char_id else 0}"
172
+ + (f" character={char_id}" if char_id else ""))
173
+ if args.dry_run:
174
+ continue
175
+
176
+ started = time.monotonic()
177
+ if args.stage in ("translate", "all"):
178
+ totals["translated"] += run_translate_stage(record, record_path, cache)
179
+ if args.stage in ("tone", "all"):
180
+ totals["toned"] += run_tone_stage(record, record_path)
181
+ save_json(record_path, record)
182
+ save_json(config.CACHE_PATH, cache)
183
+ totals["files"] += 1
184
+ print(f" done in {time.monotonic() - started:.0f}s")
185
+ except KeyboardInterrupt:
186
+ print("\nInterrupted - progress saved; rerun to resume.")
187
+ sys.exit(130)
188
+
189
+ print(
190
+ f"Finished: {totals['files']} file(s) updated, "
191
+ f"{totals['translated']} new translations, {totals['toned']} tone passes."
192
+ )
193
+
194
+
195
+ if __name__ == "__main__":
196
+ main()