bhardwaj08sarthak commited on
Commit
956e59c
·
verified ·
1 Parent(s): 5d2c4c7

Upload 3 files

Browse files
Files changed (3) hide show
  1. app.py +726 -0
  2. config.py +70 -0
  3. requirements.txt +9 -0
app.py ADDED
@@ -0,0 +1,726 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Vernacular - translation review tool for the Riverstone language pack.
2
+
3
+ Reads the review records produced by pipeline/translate_pack.py
4
+ (translations/<lang>/**), shows each string with its context and proposed
5
+ translation (character-toned for dialogue), and lets a reviewer approve,
6
+ reject, or correct it. Decisions are persisted straight back into the record
7
+ files, which pipeline/export_pack.py turns into the final language pack.
8
+ """
9
+
10
+ import json
11
+ import re
12
+ import zipfile
13
+ from pathlib import Path
14
+ from xml.etree import ElementTree as ET
15
+
16
+ import gradio as gr
17
+
18
+ import config
19
+ import build_character_wikis
20
+ from converter.docs import readLocalDoc
21
+ from converter.sheets import readLocalSheet
22
+ from converter.utils import toSafeEntityName
23
+ from pipeline import clients
24
+ from pipeline import translate_pack
25
+ from pipeline.build_file_context import normalize_key
26
+ from pipeline.rules import iter_translatable
27
+ from pipeline.dialogue_map import build_dialogue_map, display_name, load_wiki
28
+
29
+ _context_path = (
30
+ config.FILE_CONTEXT_PATH
31
+ if config.FILE_CONTEXT_PATH.exists()
32
+ else config.FILE_CONTEXT_SAMPLE_PATH
33
+ )
34
+ FILE_CONTEXT = (
35
+ json.loads(_context_path.read_text(encoding="utf-8")) if _context_path.exists() else {}
36
+ )
37
+ NEIGHBOR_LINES = 2
38
+ UPLOAD_SOURCE_DIR = config.ROOT / "uploaded_wiki_sources"
39
+ UPLOAD_REVIEW_DIR = config.SOURCE_DIR / "Uploaded" / "Filler Chats"
40
+ WIKI_BUNDLE_PATH = config.CHAR_WIKI_DIR / "character_wikis.json"
41
+
42
+
43
+ # --- character wiki upload/build/download -----------------------------------
44
+
45
+
46
+ def _safe_id(name: str) -> str:
47
+ stem = Path(name).stem.lower()
48
+ stem = re.sub(r"[^a-z0-9]+", "_", stem).strip("_")
49
+ return stem or "uploaded_character"
50
+
51
+
52
+ def _unique_path(path: Path) -> Path:
53
+ if not path.exists():
54
+ return path
55
+ for i in range(2, 1000):
56
+ candidate = path.with_name(f"{path.stem}_{i}{path.suffix}")
57
+ if not candidate.exists():
58
+ return candidate
59
+ raise RuntimeError(f"Could not find a free filename for {path.name}")
60
+
61
+
62
+ def _uploaded_path(uploaded) -> Path:
63
+ return Path(uploaded.name if hasattr(uploaded, "name") else uploaded)
64
+
65
+
66
+ def _text_from_docx(path: Path) -> str:
67
+ with zipfile.ZipFile(path) as archive:
68
+ xml = archive.read("word/document.xml")
69
+ root = ET.fromstring(xml)
70
+ ns = {"w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main"}
71
+ paragraphs = []
72
+ for paragraph in root.findall(".//w:p", ns):
73
+ text = "".join(node.text or "" for node in paragraph.findall(".//w:t", ns))
74
+ if text.strip():
75
+ paragraphs.append(text)
76
+ return "\n".join(paragraphs)
77
+
78
+
79
+ def _xlsx_shared_strings(archive: zipfile.ZipFile) -> list[str]:
80
+ try:
81
+ root = ET.fromstring(archive.read("xl/sharedStrings.xml"))
82
+ except KeyError:
83
+ return []
84
+ ns = {"x": "http://schemas.openxmlformats.org/spreadsheetml/2006/main"}
85
+ strings = []
86
+ for item in root.findall(".//x:si", ns):
87
+ strings.append("".join(node.text or "" for node in item.findall(".//x:t", ns)))
88
+ return strings
89
+
90
+
91
+ def _text_from_xlsx(path: Path) -> str:
92
+ ns = {"x": "http://schemas.openxmlformats.org/spreadsheetml/2006/main"}
93
+ lines = []
94
+ with zipfile.ZipFile(path) as archive:
95
+ shared = _xlsx_shared_strings(archive)
96
+ sheet_names = sorted(
97
+ name for name in archive.namelist()
98
+ if name.startswith("xl/worksheets/sheet") and name.endswith(".xml")
99
+ )
100
+ for sheet_name in sheet_names:
101
+ lines.append(f"[{Path(sheet_name).stem}]")
102
+ root = ET.fromstring(archive.read(sheet_name))
103
+ for row in root.findall(".//x:row", ns):
104
+ values = []
105
+ for cell in row.findall("x:c", ns):
106
+ value = cell.find("x:v", ns)
107
+ if value is None or value.text is None:
108
+ values.append("")
109
+ elif cell.attrib.get("t") == "s":
110
+ index = int(value.text)
111
+ values.append(shared[index] if index < len(shared) else "")
112
+ else:
113
+ values.append(value.text)
114
+ if any(v.strip() for v in values):
115
+ lines.append(" | ".join(values))
116
+ return "\n".join(lines)
117
+
118
+
119
+ def _text_from_upload(path: Path) -> str:
120
+ suffix = path.suffix.lower()
121
+ if suffix == ".docx":
122
+ return _text_from_docx(path)
123
+ if suffix == ".xlsx":
124
+ return _text_from_xlsx(path)
125
+ if suffix in {".txt", ".csv", ".json"}:
126
+ return path.read_text(encoding="utf-8", errors="replace")
127
+ if suffix in {".doc", ".xls"}:
128
+ raise ValueError(
129
+ f"{path.name}: old binary {suffix} files are not supported; save as "
130
+ f"{'.docx' if suffix == '.doc' else '.xlsx'} and upload again."
131
+ )
132
+ raise ValueError(f"{path.name}: unsupported file type.")
133
+
134
+
135
+ def _converter_file_path_key(path: Path) -> str:
136
+ parts = [toSafeEntityName(part) for part in path.with_suffix("").parts]
137
+ return "/".join(parts)
138
+
139
+
140
+ def _fallback_chat_payload(text: str) -> list[dict]:
141
+ return [
142
+ {"id": i + 1, "type": "0", "text": line}
143
+ for i, line in enumerate(text.splitlines())
144
+ if line.strip()
145
+ ]
146
+
147
+
148
+ def _converted_upload_payload(source_path: Path, text: str, review_rel: str):
149
+ suffix = source_path.suffix.lower()
150
+ file_path = _converter_file_path_key(Path(review_rel).with_suffix(source_path.suffix))
151
+ if suffix == ".docx":
152
+ converted = readLocalDoc(source_path, file_path)
153
+ elif suffix == ".xlsx":
154
+ converted = readLocalSheet(source_path, file_path)
155
+ else:
156
+ return _fallback_chat_payload(text)
157
+
158
+ if any(iter_translatable(converted, review_rel)):
159
+ return converted
160
+ return _fallback_chat_payload(text)
161
+
162
+
163
+ def _write_upload_source(uploaded) -> tuple[str, dict, str]:
164
+ source_path = _uploaded_path(uploaded)
165
+ char_id = _safe_id(source_path.name)
166
+ text = _text_from_upload(source_path)
167
+ if not text.strip():
168
+ raise ValueError(f"{source_path.name}: no readable text found.")
169
+ wiki_target = _unique_path(UPLOAD_SOURCE_DIR / f"{char_id}.json")
170
+ wiki_payload = {
171
+ "source_file": source_path.name,
172
+ "content": text,
173
+ }
174
+ wiki_target.write_text(
175
+ json.dumps(wiki_payload, indent=2, ensure_ascii=False), encoding="utf-8"
176
+ )
177
+ wiki_rel = str(wiki_target.relative_to(config.ROOT)).replace("\\", "/")
178
+
179
+ review_target = _unique_path(UPLOAD_REVIEW_DIR / f"{char_id}.json")
180
+ review_rel = str(review_target.relative_to(config.SOURCE_DIR)).replace("\\", "/")
181
+ review_payload = _converted_upload_payload(source_path, text, review_rel)
182
+ review_target.write_text(
183
+ json.dumps(review_payload, indent=2, ensure_ascii=False), encoding="utf-8"
184
+ )
185
+ char_data_path = config.CHAR_DATA_DIR / f"{char_id}.json"
186
+ char_data = {
187
+ "id": char_id,
188
+ "name": [Path(source_path.name).stem],
189
+ "mentions": [f"English_JSON/{review_rel}"],
190
+ "wiki_mentions": [wiki_rel],
191
+ "references": [],
192
+ }
193
+ if char_data_path.exists():
194
+ existing = json.loads(char_data_path.read_text(encoding="utf-8"))
195
+ mentions = existing.get("mentions", [])
196
+ wiki_mentions = existing.get("wiki_mentions", [])
197
+ char_data = {
198
+ **existing,
199
+ "id": char_id,
200
+ "name": existing.get("name") or char_data["name"],
201
+ "mentions": mentions
202
+ + ([] if f"English_JSON/{review_rel}" in mentions else [f"English_JSON/{review_rel}"]),
203
+ "wiki_mentions": wiki_mentions
204
+ + ([] if wiki_rel in wiki_mentions else [wiki_rel]),
205
+ "references": existing.get("references", []),
206
+ }
207
+ char_data_path.write_text(
208
+ json.dumps(char_data, indent=2, ensure_ascii=False), encoding="utf-8"
209
+ )
210
+ return char_id, char_data, review_rel
211
+
212
+
213
+ def _export_wiki_bundle() -> Path:
214
+ config.CHAR_WIKI_DIR.mkdir(exist_ok=True)
215
+ bundle = {}
216
+ for wiki_path in sorted(config.CHAR_WIKI_DIR.glob("*.md")):
217
+ if wiki_path.name.endswith(".sample.md"):
218
+ continue
219
+ char_id = wiki_path.stem
220
+ bundle[char_id] = {
221
+ "name": display_name(char_id),
222
+ "wiki": wiki_path.read_text(encoding="utf-8").strip(),
223
+ }
224
+ WIKI_BUNDLE_PATH.write_text(
225
+ json.dumps(bundle, indent=2, ensure_ascii=False), encoding="utf-8"
226
+ )
227
+ return WIKI_BUNDLE_PATH
228
+
229
+
230
+ def _import_wiki_bundle(uploaded) -> int:
231
+ if not uploaded:
232
+ return 0
233
+ path = _uploaded_path(uploaded)
234
+ data = json.loads(path.read_text(encoding="utf-8"))
235
+ if not isinstance(data, dict):
236
+ raise ValueError("Uploaded wiki JSON must be an object keyed by character id.")
237
+ config.CHAR_WIKI_DIR.mkdir(exist_ok=True)
238
+ imported = 0
239
+ for raw_id, value in data.items():
240
+ char_id = _safe_id(raw_id)
241
+ wiki = value.get("wiki") if isinstance(value, dict) else value
242
+ name = value.get("name") if isinstance(value, dict) else raw_id
243
+ if not isinstance(wiki, str) or not wiki.strip():
244
+ continue
245
+ (config.CHAR_WIKI_DIR / f"{char_id}.md").write_text(
246
+ wiki.strip() + "\n", encoding="utf-8"
247
+ )
248
+ char_data_path = config.CHAR_DATA_DIR / f"{char_id}.json"
249
+ if not char_data_path.exists():
250
+ char_data_path.write_text(
251
+ json.dumps(
252
+ {"id": char_id, "name": [str(name or raw_id)], "mentions": [], "references": []},
253
+ indent=2,
254
+ ensure_ascii=False,
255
+ ),
256
+ encoding="utf-8",
257
+ )
258
+ imported += 1
259
+ _export_wiki_bundle()
260
+ return imported
261
+
262
+
263
+ def _process_uploaded_character(char_data: dict, progress: dict) -> dict:
264
+ char_id = char_data["id"]
265
+ names = char_data.get("name") or []
266
+ name = names[0] if names else char_id
267
+ aka = f' (also appearing as {", ".join(names[1:])})' if len(names) > 1 else ""
268
+ mentions = char_data.get("wiki_mentions") or char_data.get("mentions", [])
269
+ wiki_path = config.CHAR_WIKI_DIR / f"{char_id}.md"
270
+ done = list(progress.get(char_id, []))
271
+ wiki = wiki_path.read_text(encoding="utf-8").strip() if wiki_path.exists() else ""
272
+ pending = [mention for mention in mentions if mention not in done]
273
+
274
+ if not pending:
275
+ return progress
276
+
277
+ system_prompt = build_character_wikis.SYSTEM_PROMPT.format(name=name, aka=aka)
278
+ for rel_path in pending:
279
+ file_path = config.ROOT / rel_path
280
+ data = json.loads(file_path.read_text(encoding="utf-8"))
281
+ user_msg = build_character_wikis.USER_TEMPLATE.format(
282
+ wiki=wiki or build_character_wikis.EMPTY_WIKI_PLACEHOLDER,
283
+ path=rel_path.removeprefix("uploaded_wiki_sources/"),
284
+ hint=build_character_wikis.attribution_hint(rel_path, name),
285
+ content=json.dumps(data, indent=2, ensure_ascii=False),
286
+ )
287
+ reply = build_character_wikis.clean_reply(
288
+ clients.update_character_wiki(system_prompt, user_msg)
289
+ )
290
+ if reply.startswith("#"):
291
+ wiki = reply
292
+ wiki_path.write_text(wiki + "\n", encoding="utf-8")
293
+ else:
294
+ raise RuntimeError(
295
+ f"Model returned a malformed wiki for {name}; previous wiki was kept."
296
+ )
297
+ done = done + [rel_path]
298
+ progress = {**progress, char_id: done}
299
+ build_character_wikis.save_progress(progress)
300
+ return progress
301
+
302
+
303
+ def build_uploaded_wikis(uploaded_files, existing_wiki_json):
304
+ uploaded_files = uploaded_files or []
305
+ try:
306
+ UPLOAD_SOURCE_DIR.mkdir(exist_ok=True)
307
+ UPLOAD_REVIEW_DIR.mkdir(parents=True, exist_ok=True)
308
+ config.CHAR_DATA_DIR.mkdir(exist_ok=True)
309
+ config.CHAR_WIKI_DIR.mkdir(exist_ok=True)
310
+
311
+ imported = _import_wiki_bundle(existing_wiki_json)
312
+ progress = build_character_wikis.load_progress()
313
+ built = []
314
+ review_files = []
315
+ for uploaded in uploaded_files:
316
+ char_id, char_data, review_rel = _write_upload_source(uploaded)
317
+ progress = _process_uploaded_character(char_data, progress)
318
+ built.append(char_id)
319
+ review_files.append(review_rel)
320
+ bundle_path = _export_wiki_bundle()
321
+ except Exception as exc:
322
+ return (
323
+ f"Wiki update failed: {type(exc).__name__}: {exc}",
324
+ gr.update(value=str(WIKI_BUNDLE_PATH) if WIKI_BUNDLE_PATH.exists() else None),
325
+ [],
326
+ )
327
+
328
+ parts = []
329
+ if imported:
330
+ parts.append(f"imported {imported} existing wiki entr{'y' if imported == 1 else 'ies'}")
331
+ if built:
332
+ parts.append(f"updated {len(built)} wiki entr{'y' if len(built) == 1 else 'ies'}")
333
+ if not parts:
334
+ parts.append("no files selected; refreshed the downloadable wiki JSON")
335
+ return (
336
+ "Wiki JSON ready: " + ", ".join(parts) + ".",
337
+ gr.update(value=str(bundle_path)),
338
+ sorted(set(review_files)),
339
+ )
340
+
341
+
342
+ # --- record access -----------------------------------------------------------
343
+
344
+
345
+ def record_path(rel: str) -> Path:
346
+ return config.TRANSLATIONS_DIR / rel
347
+
348
+
349
+ def load_record(rel: str) -> dict:
350
+ return json.loads(record_path(rel).read_text(encoding="utf-8"))
351
+
352
+
353
+ def save_record(rel: str, record: dict) -> None:
354
+ record_path(rel).write_text(
355
+ json.dumps(record, indent=2, ensure_ascii=False), encoding="utf-8"
356
+ )
357
+
358
+
359
+ def list_record_files() -> list[str]:
360
+ if not config.TRANSLATIONS_DIR.exists():
361
+ return []
362
+ return sorted(
363
+ str(p.relative_to(config.TRANSLATIONS_DIR))
364
+ for p in config.TRANSLATIONS_DIR.rglob("*.json")
365
+ if not p.name.startswith(".")
366
+ )
367
+
368
+
369
+ def proposed_text(item: dict) -> str:
370
+ return item.get("toned") or item.get("mt") or ""
371
+
372
+
373
+ def reviewed(item: dict) -> bool:
374
+ return item["status"] in ("approved", "edited")
375
+
376
+
377
+ def file_progress(rel: str) -> tuple[int, int]:
378
+ items = load_record(rel)["items"]
379
+ return sum(1 for i in items if reviewed(i)), len(items)
380
+
381
+
382
+ def file_choices() -> list[tuple[str, str]]:
383
+ choices = []
384
+ for rel in list_record_files():
385
+ done, total = file_progress(rel)
386
+ marker = "✅" if done == total else f"{done}/{total} reviewed"
387
+ choices.append((f"{rel} · {marker}", rel))
388
+ return choices
389
+
390
+
391
+ def overall_stats() -> str:
392
+ counts = {"approved": 0, "edited": 0, "rejected": 0, "pending": 0}
393
+ total = 0
394
+ for rel in list_record_files():
395
+ for item in load_record(rel)["items"]:
396
+ counts[item["status"]] = counts.get(item["status"], 0) + 1
397
+ total += 1
398
+ if not total:
399
+ return "No review records found - run `python -m pipeline.translate_pack` first."
400
+ done = counts["approved"] + counts["edited"]
401
+ return (
402
+ f"**{config.SOURCE_LANG_NAME} → {config.TARGET_LANG_NAME}** · "
403
+ f"{total} strings · ✅ {counts['approved']} approved · "
404
+ f"✏️ {counts['edited']} edited · ❌ {counts['rejected']} rejected · "
405
+ f"⏳ {counts['pending']} pending · **{done / total:.0%} reviewed**"
406
+ )
407
+
408
+
409
+ # --- rendering ---------------------------------------------------------------
410
+
411
+
412
+ def context_panel(record: dict, index: int) -> str:
413
+ rel = record["file"]
414
+ item = record["items"][index]
415
+ meta = FILE_CONTEXT.get(normalize_key(rel), {})
416
+ lines = [f"**File:** `{rel}` · **Key:** `{item['key']}`"]
417
+
418
+ if meta:
419
+ lines.append(f"**Content type:** {meta.get('label', '?')} — {meta.get('role', '')}")
420
+ if item["character"]:
421
+ lines.append(
422
+ f"**Speaker:** 🎭 {display_name(item['character'])} "
423
+ f"(dialogue - translation was adjusted to the character's voice)"
424
+ )
425
+ elif item["kind"] == "player":
426
+ lines.append("**Speaker:** 🕵️ player / phone owner (plain translation)")
427
+ if meta.get("summary"):
428
+ lines.append(f"<small>{meta['summary']}</small>")
429
+
430
+ neighbors = []
431
+ start = max(0, index - NEIGHBOR_LINES)
432
+ for i in range(start, min(len(record["items"]), index + NEIGHBOR_LINES + 1)):
433
+ other = record["items"][i]
434
+ marker = "→" if i == index else "&nbsp;&nbsp;"
435
+ who = display_name(other["character"]) if other["character"] else other["kind"]
436
+ neighbors.append(f"{marker} **{who}:** {other['source']}")
437
+ lines.append("**Surrounding lines:**<br>" + "<br>".join(neighbors))
438
+ return "\n\n".join(lines)
439
+
440
+
441
+ def wiki_markdown(item: dict) -> str:
442
+ if not item["character"]:
443
+ return "*No character wiki for this line.*"
444
+ # Full wiki locally; trimmed public sample (.sample.md) on HF Spaces.
445
+ for name in (f"{item['character']}.md", f"{item['character']}.sample.md"):
446
+ wiki_path = config.CHAR_WIKI_DIR / name
447
+ if wiki_path.exists():
448
+ return wiki_path.read_text(encoding="utf-8")
449
+ return "*Wiki missing.*"
450
+
451
+
452
+ STATUS_ICONS = {"pending": "⏳", "approved": "✅", "edited": "✏️", "rejected": "❌"}
453
+
454
+
455
+ def status_badge(item: dict) -> str:
456
+ extra = f" — final: *{item['final']}*" if item.get("final") else ""
457
+ return f"### {STATUS_ICONS[item['status']]} {item['status'].upper()}{extra}"
458
+
459
+
460
+ def _truncate(text: str, limit: int = 80) -> str:
461
+ return text if len(text) <= limit else text[: limit - 1] + "…"
462
+
463
+
464
+ def items_table(record: dict) -> list[list]:
465
+ """Overview of every string in the file, for the jump-to table."""
466
+ rows = []
467
+ for i, item in enumerate(record["items"]):
468
+ speaker = display_name(item["character"]) if item["character"] else item["kind"]
469
+ rows.append(
470
+ [
471
+ i + 1,
472
+ STATUS_ICONS[item["status"]],
473
+ speaker,
474
+ _truncate(item["source"]),
475
+ _truncate(item.get("final") or proposed_text(item)),
476
+ ]
477
+ )
478
+ return rows
479
+
480
+
481
+ def render(rel: str | None, index: int):
482
+ """Outputs: context, source, translation box, status, item label, wiki, stats, table."""
483
+ if not rel:
484
+ return ("*Select a file to start reviewing.*", "", "", "", "0 / 0", "", overall_stats(), [])
485
+ if not record_path(rel).exists():
486
+ msg = (
487
+ f"*This file has not been translated yet. Run:*\n\n"
488
+ f'```\npython -m pipeline.translate_pack --filter "{rel}"\n```'
489
+ )
490
+ return (msg, "", "", "", "0 / 0", "", overall_stats(), [])
491
+ record = load_record(rel)
492
+ items = record["items"]
493
+ index = max(0, min(index, len(items) - 1))
494
+ item = items[index]
495
+ translation = item["final"] if item.get("final") else proposed_text(item)
496
+ return (
497
+ context_panel(record, index),
498
+ item["source"],
499
+ translation,
500
+ status_badge(item),
501
+ f"{index + 1} / {len(items)}",
502
+ wiki_markdown(item),
503
+ overall_stats(),
504
+ items_table(record),
505
+ )
506
+
507
+
508
+ # --- actions -----------------------------------------------------------------
509
+
510
+
511
+ def next_pending(items: list[dict], after: int) -> int:
512
+ order = list(range(after + 1, len(items))) + list(range(0, after + 1))
513
+ for i in order:
514
+ if items[i]["status"] == "pending":
515
+ return i
516
+ return min(after + 1, len(items) - 1)
517
+
518
+
519
+ def select_file(rel: str):
520
+ index = 0
521
+ if rel and record_path(rel).exists():
522
+ index = next_pending(load_record(rel)["items"], -1)
523
+ return (rel, index, *render(rel, index))
524
+
525
+
526
+ def jump_to(rel: str, evt: gr.SelectData):
527
+ index = evt.index[0]
528
+ return (index, *render(rel, index))
529
+
530
+
531
+ def navigate(rel: str, index: int, delta: int):
532
+ if not rel:
533
+ return (index, *render(rel, index))
534
+ new_index = index + delta
535
+ return (new_index, *render(rel, new_index))
536
+
537
+
538
+ def decide(rel: str, index: int, text: str, decision: str):
539
+ if not rel or not record_path(rel).exists():
540
+ return (index, *render(rel, index), gr.update())
541
+ record = load_record(rel)
542
+ item = record["items"][index]
543
+ text = text.strip()
544
+ if decision == "reject":
545
+ item["status"], item["final"] = "rejected", None
546
+ elif text and text != proposed_text(item):
547
+ item["status"], item["final"] = "edited", text
548
+ else:
549
+ item["status"], item["final"] = "approved", proposed_text(item)
550
+ save_record(rel, record)
551
+ new_index = next_pending(record["items"], index)
552
+ return (new_index, *render(rel, new_index), gr.update(choices=file_choices(), value=rel))
553
+
554
+
555
+ def _translate_uploaded_records(uploaded_review_files: list[str]) -> str:
556
+ uploaded_review_files = sorted(set(uploaded_review_files or []))
557
+ if not uploaded_review_files and UPLOAD_REVIEW_DIR.exists():
558
+ uploaded_review_files = sorted(
559
+ str(path.relative_to(config.SOURCE_DIR)).replace("\\", "/")
560
+ for path in UPLOAD_REVIEW_DIR.glob("*.json")
561
+ )
562
+ if not uploaded_review_files:
563
+ return "No uploaded source files are waiting for translation."
564
+
565
+ dmap = build_dialogue_map()
566
+ cache = (
567
+ json.loads(config.CACHE_PATH.read_text(encoding="utf-8"))
568
+ if config.CACHE_PATH.exists()
569
+ else {}
570
+ )
571
+ updated = 0
572
+ translated = 0
573
+ toned = 0
574
+ skipped = 0
575
+ for rel in uploaded_review_files:
576
+ source_path = config.SOURCE_DIR / rel
577
+ if not source_path.exists():
578
+ skipped += 1
579
+ continue
580
+ data = json.loads(source_path.read_text(encoding="utf-8"))
581
+ char_id = dmap.get(rel)
582
+ record_path_for_rel = config.TRANSLATIONS_DIR / rel
583
+ existing = translate_pack.load_record(record_path_for_rel)
584
+ record = {
585
+ "file": rel,
586
+ "character": char_id,
587
+ "items": translate_pack.build_items(data, rel, char_id, existing),
588
+ }
589
+ if not record["items"]:
590
+ skipped += 1
591
+ continue
592
+ if translate_pack.needs_work(record, "all"):
593
+ wiki = load_wiki(char_id) if char_id else None
594
+ name = display_name(char_id) if char_id else None
595
+ items, cache, counts = clients.translate_and_tone_items(
596
+ record["items"],
597
+ cache,
598
+ wiki,
599
+ name,
600
+ translate_pack.CONTEXT_LINES,
601
+ )
602
+ record["items"] = items
603
+ translated += counts["translated"]
604
+ toned += counts["toned"]
605
+ updated += 1
606
+ translate_pack.save_json(record_path_for_rel, record)
607
+ translate_pack.save_json(config.CACHE_PATH, cache)
608
+
609
+ parts = [
610
+ f"{updated} uploaded file{'s' if updated != 1 else ''} processed",
611
+ f"{translated} translation{'s' if translated != 1 else ''}",
612
+ f"{toned} tone pass{'es' if toned != 1 else ''}",
613
+ ]
614
+ if skipped:
615
+ parts.append(f"{skipped} skipped")
616
+ return "Refresh complete: " + ", ".join(parts) + "."
617
+
618
+
619
+ def refresh(uploaded_review_files):
620
+ status = _translate_uploaded_records(uploaded_review_files)
621
+ return gr.update(choices=file_choices()), overall_stats(), status
622
+
623
+
624
+ # --- UI ----------------------------------------------------------------------
625
+
626
+ with gr.Blocks(title="Vernacular - Riverstone Translation Review") as demo:
627
+ gr.Markdown(f"# 🌍 Vernacular — {config.TARGET_LANG_NAME} translation review")
628
+ stats_md = gr.Markdown(overall_stats())
629
+
630
+ with gr.Accordion("Character wiki builder", open=True):
631
+ wiki_source_files = gr.File(
632
+ label="Upload character source files",
633
+ file_count="multiple",
634
+ file_types=[".doc", ".docx", ".xls", ".xlsx", ".json", ".txt", ".csv"],
635
+ )
636
+ existing_wiki_json = gr.File(
637
+ label="Upload existing wiki JSON",
638
+ file_count="single",
639
+ file_types=[".json"],
640
+ )
641
+ with gr.Row():
642
+ build_wiki_btn = gr.Button("Build / update wiki JSON", variant="primary")
643
+ wiki_json_file = gr.File(
644
+ label="Download wiki JSON",
645
+ value=str(WIKI_BUNDLE_PATH) if WIKI_BUNDLE_PATH.exists() else None,
646
+ interactive=False,
647
+ )
648
+ wiki_build_status = gr.Markdown("")
649
+
650
+ with gr.Row():
651
+ file_dd = gr.Dropdown(
652
+ choices=file_choices(), label="File", interactive=True, scale=5
653
+ )
654
+ refresh_btn = gr.Button("🔄 Refresh", scale=1)
655
+
656
+ current_file = gr.State(None)
657
+ current_index = gr.State(0)
658
+ uploaded_review_files = gr.State([])
659
+
660
+ with gr.Row():
661
+ with gr.Column(scale=3):
662
+ context_md = gr.Markdown("*Select a file to start reviewing.*")
663
+ source_tb = gr.Textbox(
664
+ label=f"{config.SOURCE_LANG_NAME} original",
665
+ lines=2,
666
+ interactive=False,
667
+ )
668
+ translation_tb = gr.Textbox(
669
+ label=f"{config.TARGET_LANG_NAME} translation (edit if needed)",
670
+ lines=3,
671
+ interactive=True,
672
+ )
673
+ with gr.Row():
674
+ prev_btn = gr.Button("← Prev")
675
+ item_label = gr.Markdown("0 / 0")
676
+ next_btn = gr.Button("Next →")
677
+ with gr.Row():
678
+ approve_btn = gr.Button("✅ Approve / Save edit", variant="primary")
679
+ reject_btn = gr.Button("❌ Reject", variant="stop")
680
+ status_md = gr.Markdown("")
681
+ with gr.Column(scale=2):
682
+ with gr.Accordion("🎭 Character wiki", open=False):
683
+ wiki_md = gr.Markdown("")
684
+
685
+ with gr.Accordion("📋 All strings in this file (click a row to jump)", open=False):
686
+ items_df = gr.Dataframe(
687
+ headers=["#", "", "speaker", f"{config.SOURCE_LANG_NAME} original",
688
+ f"{config.TARGET_LANG_NAME} translation"],
689
+ interactive=False,
690
+ wrap=True,
691
+ )
692
+
693
+ render_outputs = [context_md, source_tb, translation_tb, status_md, item_label,
694
+ wiki_md, stats_md, items_df]
695
+
696
+ file_dd.change(select_file, [file_dd], [current_file, current_index, *render_outputs])
697
+ items_df.select(jump_to, [current_file], [current_index, *render_outputs])
698
+ prev_btn.click(
699
+ lambda rel, i: navigate(rel, i, -1),
700
+ [current_file, current_index],
701
+ [current_index, *render_outputs],
702
+ )
703
+ next_btn.click(
704
+ lambda rel, i: navigate(rel, i, +1),
705
+ [current_file, current_index],
706
+ [current_index, *render_outputs],
707
+ )
708
+ approve_btn.click(
709
+ lambda rel, i, t: decide(rel, i, t, "approve"),
710
+ [current_file, current_index, translation_tb],
711
+ [current_index, *render_outputs, file_dd],
712
+ )
713
+ reject_btn.click(
714
+ lambda rel, i, t: decide(rel, i, t, "reject"),
715
+ [current_file, current_index, translation_tb],
716
+ [current_index, *render_outputs, file_dd],
717
+ )
718
+ refresh_btn.click(refresh, [uploaded_review_files], [file_dd, stats_md, wiki_build_status])
719
+ build_wiki_btn.click(
720
+ build_uploaded_wikis,
721
+ [wiki_source_files, existing_wiki_json],
722
+ [wiki_build_status, wiki_json_file, uploaded_review_files],
723
+ )
724
+
725
+ if __name__ == "__main__":
726
+ demo.launch()
config.py ADDED
@@ -0,0 +1,70 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Central configuration for the translation pipeline and review tool.
2
+
3
+ To target a new language, change the three TARGET_* values (and PACK_NAME).
4
+ Everything else derives from them.
5
+ """
6
+
7
+ from pathlib import Path
8
+
9
+ ROOT = Path(__file__).parent
10
+
11
+ # --- Target language -------------------------------------------------------
12
+ TARGET_LANG_CODE = "de" # ISO 639-1 code understood by TranslateGemma
13
+ TARGET_LANG_NAME = "German"
14
+ PACK_NAME = "German_JSON"
15
+
16
+ SOURCE_LANG_CODE = "en"
17
+ SOURCE_LANG_NAME = "English"
18
+
19
+ # --- Paths ------------------------------------------------------------------
20
+ SOURCE_DIR = ROOT / "English_JSON"
21
+ PACK_DIR = ROOT / PACK_NAME
22
+ TRANSLATIONS_DIR = ROOT / "translations" / TARGET_LANG_CODE
23
+ CACHE_PATH = TRANSLATIONS_DIR / ".cache.json"
24
+ FILE_CONTEXT_PATH = ROOT / "file_context.json"
25
+ # Public fallback used on HF Spaces: context for the demo sample files only.
26
+ FILE_CONTEXT_SAMPLE_PATH = ROOT / "file_context.sample.json"
27
+ FILE_MAPPING_PATH = ROOT / "FILE_MAPPING.md"
28
+ CHAR_DATA_DIR = ROOT / "character_data"
29
+ CHAR_WIKI_DIR = ROOT / "character_wikis"
30
+
31
+ # --- Hugging Face in-process model inference --------------------------------
32
+ # Both translation stages now run inside Python with Transformers. The model
33
+ # repos below are the canonical Hugging Face checkpoints, not local HTTP
34
+ # servers. TranslateGemma is gated, so the runtime needs an HF token whose
35
+ # account has accepted Google's Gemma terms.
36
+ TRANSLATE_MODEL_ID = "google/translategemma-12b-it"
37
+ TONE_MODEL_ID = "google/gemma-4-12B-it"
38
+
39
+ # "auto" follows the model card examples and works locally and on GPU Spaces.
40
+ # Set HF_DEVICE_MAP = None to load on CPU first and explicitly move to cuda.
41
+ HF_DEVICE_MAP = "auto"
42
+ HF_DTYPE = "auto"
43
+ HF_ATTN_IMPLEMENTATION = None
44
+
45
+ TRANSLATE_MAX_NEW_TOKENS = 1024
46
+ TONE_MAX_NEW_TOKENS = 512
47
+
48
+ # Gemma's officially recommended sampling parameters for creative rewriting.
49
+ TONE_GENERATION_KWARGS = {
50
+ "do_sample": True,
51
+ "temperature": 1.0,
52
+ "top_k": 64,
53
+ "top_p": 0.95,
54
+ }
55
+
56
+ # ZeroGPU only exposes the real GPU inside @spaces.GPU functions. The decorator
57
+ # is a no-op outside ZeroGPU, so keeping these values here makes deployment a
58
+ # config change instead of a code fork.
59
+ ZERO_GPU_DURATION_S = 600
60
+ ZERO_GPU_SIZE = "xlarge"
61
+
62
+ # Keeping both 12B checkpoints resident can exceed the default 48GB ZeroGPU
63
+ # slice. The pipeline unloads the other model before loading the requested one
64
+ # unless this is set to True on a larger GPU.
65
+ HF_KEEP_BOTH_MODELS = False
66
+ HF_PRELOAD_MODELS = ()
67
+
68
+ REQUEST_TIMEOUT_S = 600
69
+ MAX_ATTEMPTS = 3
70
+ RETRY_BACKOFF_S = 5
requirements.txt ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ gradio==6.16.0
2
+ spaces
3
+ torch
4
+ torchvision
5
+ accelerate
6
+ transformers
7
+ sentencepiece
8
+ protobuf
9
+ huggingface_hub