"""Batch translation driver: English_JSON -> translations// review records. Stage 1 (translate): every translatable string through TranslateGemma. Stage 2 (tone): dialogue lines of mapped characters through the Ollama tone model using the character's voice wiki. One review record file is written per source file, mirroring the pack layout. Runs are resumable: existing records are merged by item key and only missing work is done. Repeated strings hit a shared translation cache. Usage: python -m pipeline.translate_pack # full run, both stages python -m pipeline.translate_pack --filter "Filler Chats/Brad" python -m pipeline.translate_pack --stage translate # stage 1 only python -m pipeline.translate_pack --limit 5 # first 5 files python -m pipeline.translate_pack --dry-run # plan only, no LLM calls """ import argparse import json import sys import time from pathlib import Path sys.path.insert(0, str(Path(__file__).parent.parent)) import config from pipeline import clients from pipeline.dialogue_map import build_dialogue_map, display_name, load_wiki from pipeline.rules import iter_translatable, key_to_str SAVE_EVERY = 25 # persist record/cache after this many new translations CONTEXT_LINES = 2 # preceding source lines passed to the tone model def load_record(record_path: Path) -> dict: if record_path.exists(): return json.loads(record_path.read_text(encoding="utf-8")) return {} def save_json(path: Path, data: dict) -> None: path.parent.mkdir(parents=True, exist_ok=True) path.write_text(json.dumps(data, indent=2, ensure_ascii=False), encoding="utf-8") def build_items(data, rel: str, char_id: str | None, existing: dict) -> list[dict]: """Fresh item list from the source file, merged with prior record state.""" prior = {item["key"]: item for item in existing.get("items", [])} items = [] for found in iter_translatable(data, rel): key = key_to_str(found.path) old = prior.get(key) if old and old["source"] == found.source: items.append(old) continue items.append( { "key": key, "path": list(found.path), "source": found.source, "kind": found.kind, "character": char_id if found.kind == "dialogue" else None, "mt": None, "toned": None, "status": "pending", "final": None, } ) return items def run_translate_stage(record: dict, record_path: Path, cache: dict) -> int: done = 0 for item in record["items"]: if item["mt"] is not None: continue cached = cache.get(item["source"]) if cached is not None: item["mt"] = cached else: item["mt"] = clients.translate(item["source"]) cache[item["source"]] = item["mt"] done += 1 if done % SAVE_EVERY == 0: save_json(record_path, record) save_json(config.CACHE_PATH, cache) return done def run_tone_stage(record: dict, record_path: Path) -> int: char_id = record.get("character") if not char_id: return 0 wiki = load_wiki(char_id) name = display_name(char_id) sources = [item["source"] for item in record["items"]] done = 0 for i, item in enumerate(record["items"]): if item["kind"] != "dialogue" or item["toned"] is not None or item["mt"] is None: continue context = sources[max(0, i - CONTEXT_LINES) : i] item["toned"] = clients.adjust_tone(item["source"], item["mt"], wiki, name, context) done += 1 if done % SAVE_EVERY == 0: save_json(record_path, record) return done def needs_work(record: dict, stage: str) -> bool: for item in record.get("items", []): if stage in ("translate", "all") and item["mt"] is None: return True if ( stage in ("tone", "all") and item["kind"] == "dialogue" and record.get("character") and item["mt"] is not None and item["toned"] is None ): return True return False def main() -> None: parser = argparse.ArgumentParser(description=__doc__.split("\n")[0]) parser.add_argument("--filter", default="", help="only files whose path contains this substring") parser.add_argument("--limit", type=int, default=0, help="max number of files to process") parser.add_argument("--stage", choices=["translate", "tone", "all"], default="all") parser.add_argument("--dry-run", action="store_true", help="report planned work, no LLM calls") args = parser.parse_args() sys.stdout.reconfigure(line_buffering=True) dmap = build_dialogue_map() cache = json.loads(config.CACHE_PATH.read_text(encoding="utf-8")) if config.CACHE_PATH.exists() else {} source_files = sorted(config.SOURCE_DIR.rglob("*.json")) processed = 0 totals = {"translated": 0, "toned": 0, "files": 0} try: for source_file in source_files: rel = str(source_file.relative_to(config.SOURCE_DIR)) if args.filter and args.filter not in rel: continue if args.limit and processed >= args.limit: break data = json.loads(source_file.read_text(encoding="utf-8")) char_id = dmap.get(rel) record_path = config.TRANSLATIONS_DIR / rel record = load_record(record_path) record = { "file": rel, "character": char_id, "items": build_items(data, rel, char_id, record), } if not record["items"]: continue processed += 1 if not needs_work(record, args.stage): continue pending_mt = sum(1 for i in record["items"] if i["mt"] is None) pending_tone = sum( 1 for i in record["items"] if i["kind"] == "dialogue" and char_id and i["toned"] is None ) print(f"[{rel}] strings={len(record['items'])} mt-pending={pending_mt} " f"tone-pending={pending_tone if char_id else 0}" + (f" character={char_id}" if char_id else "")) if args.dry_run: continue started = time.monotonic() if args.stage in ("translate", "all"): totals["translated"] += run_translate_stage(record, record_path, cache) if args.stage in ("tone", "all"): totals["toned"] += run_tone_stage(record, record_path) save_json(record_path, record) save_json(config.CACHE_PATH, cache) totals["files"] += 1 print(f" done in {time.monotonic() - started:.0f}s") except KeyboardInterrupt: print("\nInterrupted - progress saved; rerun to resume.") sys.exit(130) print( f"Finished: {totals['files']} file(s) updated, " f"{totals['translated']} new translations, {totals['toned']} tone passes." ) if __name__ == "__main__": main()