| """Batch translation driver: English_JSON -> translations/<lang>/ review records. |
| |
| Stage 1 (translate): every translatable string through TranslateGemma. |
| Stage 2 (tone): dialogue lines of mapped characters through the Ollama tone |
| model using the character's voice wiki. |
| |
| One review record file is written per source file, mirroring the pack layout. |
| Runs are resumable: existing records are merged by item key and only missing |
| work is done. Repeated strings hit a shared translation cache. |
| |
| Usage: |
| python -m pipeline.translate_pack # full run, both stages |
| python -m pipeline.translate_pack --filter "Filler Chats/Brad" |
| python -m pipeline.translate_pack --stage translate # stage 1 only |
| python -m pipeline.translate_pack --limit 5 # first 5 files |
| python -m pipeline.translate_pack --dry-run # plan only, no LLM calls |
| """ |
|
|
| import argparse |
| import json |
| import sys |
| import time |
| from pathlib import Path |
|
|
| sys.path.insert(0, str(Path(__file__).parent.parent)) |
| import config |
| from pipeline import clients |
| from pipeline.dialogue_map import build_dialogue_map, display_name, load_wiki |
| from pipeline.rules import iter_translatable, key_to_str |
|
|
| SAVE_EVERY = 25 |
| CONTEXT_LINES = 2 |
|
|
|
|
| def load_record(record_path: Path) -> dict: |
| if record_path.exists(): |
| return json.loads(record_path.read_text(encoding="utf-8")) |
| return {} |
|
|
|
|
| def save_json(path: Path, data: dict) -> None: |
| path.parent.mkdir(parents=True, exist_ok=True) |
| path.write_text(json.dumps(data, indent=2, ensure_ascii=False), encoding="utf-8") |
|
|
|
|
| def build_items(data, rel: str, char_id: str | None, existing: dict) -> list[dict]: |
| """Fresh item list from the source file, merged with prior record state.""" |
| prior = {item["key"]: item for item in existing.get("items", [])} |
| items = [] |
| for found in iter_translatable(data, rel): |
| key = key_to_str(found.path) |
| old = prior.get(key) |
| if old and old["source"] == found.source: |
| items.append(old) |
| continue |
| items.append( |
| { |
| "key": key, |
| "path": list(found.path), |
| "source": found.source, |
| "kind": found.kind, |
| "character": char_id if found.kind == "dialogue" else None, |
| "mt": None, |
| "toned": None, |
| "status": "pending", |
| "final": None, |
| } |
| ) |
| return items |
|
|
|
|
| def run_translate_stage(record: dict, record_path: Path, cache: dict) -> int: |
| done = 0 |
| for item in record["items"]: |
| if item["mt"] is not None: |
| continue |
| cached = cache.get(item["source"]) |
| if cached is not None: |
| item["mt"] = cached |
| else: |
| item["mt"] = clients.translate(item["source"]) |
| cache[item["source"]] = item["mt"] |
| done += 1 |
| if done % SAVE_EVERY == 0: |
| save_json(record_path, record) |
| save_json(config.CACHE_PATH, cache) |
| return done |
|
|
|
|
| def run_tone_stage(record: dict, record_path: Path) -> int: |
| char_id = record.get("character") |
| if not char_id: |
| return 0 |
| wiki = load_wiki(char_id) |
| name = display_name(char_id) |
| sources = [item["source"] for item in record["items"]] |
|
|
| done = 0 |
| for i, item in enumerate(record["items"]): |
| if item["kind"] != "dialogue" or item["toned"] is not None or item["mt"] is None: |
| continue |
| context = sources[max(0, i - CONTEXT_LINES) : i] |
| item["toned"] = clients.adjust_tone(item["source"], item["mt"], wiki, name, context) |
| done += 1 |
| if done % SAVE_EVERY == 0: |
| save_json(record_path, record) |
| return done |
|
|
|
|
| def needs_work(record: dict, stage: str) -> bool: |
| for item in record.get("items", []): |
| if stage in ("translate", "all") and item["mt"] is None: |
| return True |
| if ( |
| stage in ("tone", "all") |
| and item["kind"] == "dialogue" |
| and record.get("character") |
| and item["mt"] is not None |
| and item["toned"] is None |
| ): |
| return True |
| return False |
|
|
|
|
| def main() -> None: |
| parser = argparse.ArgumentParser(description=__doc__.split("\n")[0]) |
| parser.add_argument("--filter", default="", help="only files whose path contains this substring") |
| parser.add_argument("--limit", type=int, default=0, help="max number of files to process") |
| parser.add_argument("--stage", choices=["translate", "tone", "all"], default="all") |
| parser.add_argument("--dry-run", action="store_true", help="report planned work, no LLM calls") |
| args = parser.parse_args() |
| sys.stdout.reconfigure(line_buffering=True) |
|
|
| dmap = build_dialogue_map() |
| cache = json.loads(config.CACHE_PATH.read_text(encoding="utf-8")) if config.CACHE_PATH.exists() else {} |
|
|
| source_files = sorted(config.SOURCE_DIR.rglob("*.json")) |
| processed = 0 |
| totals = {"translated": 0, "toned": 0, "files": 0} |
|
|
| try: |
| for source_file in source_files: |
| rel = str(source_file.relative_to(config.SOURCE_DIR)) |
| if args.filter and args.filter not in rel: |
| continue |
| if args.limit and processed >= args.limit: |
| break |
|
|
| data = json.loads(source_file.read_text(encoding="utf-8")) |
| char_id = dmap.get(rel) |
| record_path = config.TRANSLATIONS_DIR / rel |
| record = load_record(record_path) |
| record = { |
| "file": rel, |
| "character": char_id, |
| "items": build_items(data, rel, char_id, record), |
| } |
| if not record["items"]: |
| continue |
| processed += 1 |
| if not needs_work(record, args.stage): |
| continue |
|
|
| pending_mt = sum(1 for i in record["items"] if i["mt"] is None) |
| pending_tone = sum( |
| 1 |
| for i in record["items"] |
| if i["kind"] == "dialogue" and char_id and i["toned"] is None |
| ) |
| print(f"[{rel}] strings={len(record['items'])} mt-pending={pending_mt} " |
| f"tone-pending={pending_tone if char_id else 0}" |
| + (f" character={char_id}" if char_id else "")) |
| if args.dry_run: |
| continue |
|
|
| started = time.monotonic() |
| if args.stage in ("translate", "all"): |
| totals["translated"] += run_translate_stage(record, record_path, cache) |
| if args.stage in ("tone", "all"): |
| totals["toned"] += run_tone_stage(record, record_path) |
| save_json(record_path, record) |
| save_json(config.CACHE_PATH, cache) |
| totals["files"] += 1 |
| print(f" done in {time.monotonic() - started:.0f}s") |
| except KeyboardInterrupt: |
| print("\nInterrupted - progress saved; rerun to resume.") |
| sys.exit(130) |
|
|
| print( |
| f"Finished: {totals['files']} file(s) updated, " |
| f"{totals['translated']} new translations, {totals['toned']} tone passes." |
| ) |
|
|
|
|
| if __name__ == "__main__": |
| main() |
|
|