Vernacular / pipeline /translate_pack.py
bhardwaj08sarthak's picture
Upload folder using huggingface_hub
2619b20 verified
Raw
History Blame Contribute Delete
7.28 kB
"""Batch translation driver: English_JSON -> translations/<lang>/ review records.
Stage 1 (translate): every translatable string through TranslateGemma.
Stage 2 (tone): dialogue lines of mapped characters through the Ollama tone
model using the character's voice wiki.
One review record file is written per source file, mirroring the pack layout.
Runs are resumable: existing records are merged by item key and only missing
work is done. Repeated strings hit a shared translation cache.
Usage:
python -m pipeline.translate_pack # full run, both stages
python -m pipeline.translate_pack --filter "Filler Chats/Brad"
python -m pipeline.translate_pack --stage translate # stage 1 only
python -m pipeline.translate_pack --limit 5 # first 5 files
python -m pipeline.translate_pack --dry-run # plan only, no LLM calls
"""
import argparse
import json
import sys
import time
from pathlib import Path
sys.path.insert(0, str(Path(__file__).parent.parent))
import config
from pipeline import clients
from pipeline.dialogue_map import build_dialogue_map, display_name, load_wiki
from pipeline.rules import iter_translatable, key_to_str
SAVE_EVERY = 25 # persist record/cache after this many new translations
CONTEXT_LINES = 2 # preceding source lines passed to the tone model
def load_record(record_path: Path) -> dict:
if record_path.exists():
return json.loads(record_path.read_text(encoding="utf-8"))
return {}
def save_json(path: Path, data: dict) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(data, indent=2, ensure_ascii=False), encoding="utf-8")
def build_items(data, rel: str, char_id: str | None, existing: dict) -> list[dict]:
"""Fresh item list from the source file, merged with prior record state."""
prior = {item["key"]: item for item in existing.get("items", [])}
items = []
for found in iter_translatable(data, rel):
key = key_to_str(found.path)
old = prior.get(key)
if old and old["source"] == found.source:
items.append(old)
continue
items.append(
{
"key": key,
"path": list(found.path),
"source": found.source,
"kind": found.kind,
"character": char_id if found.kind == "dialogue" else None,
"mt": None,
"toned": None,
"status": "pending",
"final": None,
}
)
return items
def run_translate_stage(record: dict, record_path: Path, cache: dict) -> int:
done = 0
for item in record["items"]:
if item["mt"] is not None:
continue
cached = cache.get(item["source"])
if cached is not None:
item["mt"] = cached
else:
item["mt"] = clients.translate(item["source"])
cache[item["source"]] = item["mt"]
done += 1
if done % SAVE_EVERY == 0:
save_json(record_path, record)
save_json(config.CACHE_PATH, cache)
return done
def run_tone_stage(record: dict, record_path: Path) -> int:
char_id = record.get("character")
if not char_id:
return 0
wiki = load_wiki(char_id)
name = display_name(char_id)
sources = [item["source"] for item in record["items"]]
done = 0
for i, item in enumerate(record["items"]):
if item["kind"] != "dialogue" or item["toned"] is not None or item["mt"] is None:
continue
context = sources[max(0, i - CONTEXT_LINES) : i]
item["toned"] = clients.adjust_tone(item["source"], item["mt"], wiki, name, context)
done += 1
if done % SAVE_EVERY == 0:
save_json(record_path, record)
return done
def needs_work(record: dict, stage: str) -> bool:
for item in record.get("items", []):
if stage in ("translate", "all") and item["mt"] is None:
return True
if (
stage in ("tone", "all")
and item["kind"] == "dialogue"
and record.get("character")
and item["mt"] is not None
and item["toned"] is None
):
return True
return False
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
parser.add_argument("--filter", default="", help="only files whose path contains this substring")
parser.add_argument("--limit", type=int, default=0, help="max number of files to process")
parser.add_argument("--stage", choices=["translate", "tone", "all"], default="all")
parser.add_argument("--dry-run", action="store_true", help="report planned work, no LLM calls")
args = parser.parse_args()
sys.stdout.reconfigure(line_buffering=True)
dmap = build_dialogue_map()
cache = json.loads(config.CACHE_PATH.read_text(encoding="utf-8")) if config.CACHE_PATH.exists() else {}
source_files = sorted(config.SOURCE_DIR.rglob("*.json"))
processed = 0
totals = {"translated": 0, "toned": 0, "files": 0}
try:
for source_file in source_files:
rel = str(source_file.relative_to(config.SOURCE_DIR))
if args.filter and args.filter not in rel:
continue
if args.limit and processed >= args.limit:
break
data = json.loads(source_file.read_text(encoding="utf-8"))
char_id = dmap.get(rel)
record_path = config.TRANSLATIONS_DIR / rel
record = load_record(record_path)
record = {
"file": rel,
"character": char_id,
"items": build_items(data, rel, char_id, record),
}
if not record["items"]:
continue
processed += 1
if not needs_work(record, args.stage):
continue
pending_mt = sum(1 for i in record["items"] if i["mt"] is None)
pending_tone = sum(
1
for i in record["items"]
if i["kind"] == "dialogue" and char_id and i["toned"] is None
)
print(f"[{rel}] strings={len(record['items'])} mt-pending={pending_mt} "
f"tone-pending={pending_tone if char_id else 0}"
+ (f" character={char_id}" if char_id else ""))
if args.dry_run:
continue
started = time.monotonic()
if args.stage in ("translate", "all"):
totals["translated"] += run_translate_stage(record, record_path, cache)
if args.stage in ("tone", "all"):
totals["toned"] += run_tone_stage(record, record_path)
save_json(record_path, record)
save_json(config.CACHE_PATH, cache)
totals["files"] += 1
print(f" done in {time.monotonic() - started:.0f}s")
except KeyboardInterrupt:
print("\nInterrupted - progress saved; rerun to resume.")
sys.exit(130)
print(
f"Finished: {totals['files']} file(s) updated, "
f"{totals['translated']} new translations, {totals['toned']} tone passes."
)
if __name__ == "__main__":
main()