File size: 7,283 Bytes
478fb0c | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 | """Batch translation driver: English_JSON -> translations/<lang>/ review records.
Stage 1 (translate): every translatable string through TranslateGemma.
Stage 2 (tone): dialogue lines of mapped characters through the Ollama tone
model using the character's voice wiki.
One review record file is written per source file, mirroring the pack layout.
Runs are resumable: existing records are merged by item key and only missing
work is done. Repeated strings hit a shared translation cache.
Usage:
python -m pipeline.translate_pack # full run, both stages
python -m pipeline.translate_pack --filter "Filler Chats/Brad"
python -m pipeline.translate_pack --stage translate # stage 1 only
python -m pipeline.translate_pack --limit 5 # first 5 files
python -m pipeline.translate_pack --dry-run # plan only, no LLM calls
"""
import argparse
import json
import sys
import time
from pathlib import Path
sys.path.insert(0, str(Path(__file__).parent.parent))
import config
from pipeline import clients
from pipeline.dialogue_map import build_dialogue_map, display_name, load_wiki
from pipeline.rules import iter_translatable, key_to_str
SAVE_EVERY = 25 # persist record/cache after this many new translations
CONTEXT_LINES = 2 # preceding source lines passed to the tone model
def load_record(record_path: Path) -> dict:
if record_path.exists():
return json.loads(record_path.read_text(encoding="utf-8"))
return {}
def save_json(path: Path, data: dict) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(data, indent=2, ensure_ascii=False), encoding="utf-8")
def build_items(data, rel: str, char_id: str | None, existing: dict) -> list[dict]:
"""Fresh item list from the source file, merged with prior record state."""
prior = {item["key"]: item for item in existing.get("items", [])}
items = []
for found in iter_translatable(data, rel):
key = key_to_str(found.path)
old = prior.get(key)
if old and old["source"] == found.source:
items.append(old)
continue
items.append(
{
"key": key,
"path": list(found.path),
"source": found.source,
"kind": found.kind,
"character": char_id if found.kind == "dialogue" else None,
"mt": None,
"toned": None,
"status": "pending",
"final": None,
}
)
return items
def run_translate_stage(record: dict, record_path: Path, cache: dict) -> int:
done = 0
for item in record["items"]:
if item["mt"] is not None:
continue
cached = cache.get(item["source"])
if cached is not None:
item["mt"] = cached
else:
item["mt"] = clients.translate(item["source"])
cache[item["source"]] = item["mt"]
done += 1
if done % SAVE_EVERY == 0:
save_json(record_path, record)
save_json(config.CACHE_PATH, cache)
return done
def run_tone_stage(record: dict, record_path: Path) -> int:
char_id = record.get("character")
if not char_id:
return 0
wiki = load_wiki(char_id)
name = display_name(char_id)
sources = [item["source"] for item in record["items"]]
done = 0
for i, item in enumerate(record["items"]):
if item["kind"] != "dialogue" or item["toned"] is not None or item["mt"] is None:
continue
context = sources[max(0, i - CONTEXT_LINES) : i]
item["toned"] = clients.adjust_tone(item["source"], item["mt"], wiki, name, context)
done += 1
if done % SAVE_EVERY == 0:
save_json(record_path, record)
return done
def needs_work(record: dict, stage: str) -> bool:
for item in record.get("items", []):
if stage in ("translate", "all") and item["mt"] is None:
return True
if (
stage in ("tone", "all")
and item["kind"] == "dialogue"
and record.get("character")
and item["mt"] is not None
and item["toned"] is None
):
return True
return False
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
parser.add_argument("--filter", default="", help="only files whose path contains this substring")
parser.add_argument("--limit", type=int, default=0, help="max number of files to process")
parser.add_argument("--stage", choices=["translate", "tone", "all"], default="all")
parser.add_argument("--dry-run", action="store_true", help="report planned work, no LLM calls")
args = parser.parse_args()
sys.stdout.reconfigure(line_buffering=True)
dmap = build_dialogue_map()
cache = json.loads(config.CACHE_PATH.read_text(encoding="utf-8")) if config.CACHE_PATH.exists() else {}
source_files = sorted(config.SOURCE_DIR.rglob("*.json"))
processed = 0
totals = {"translated": 0, "toned": 0, "files": 0}
try:
for source_file in source_files:
rel = str(source_file.relative_to(config.SOURCE_DIR))
if args.filter and args.filter not in rel:
continue
if args.limit and processed >= args.limit:
break
data = json.loads(source_file.read_text(encoding="utf-8"))
char_id = dmap.get(rel)
record_path = config.TRANSLATIONS_DIR / rel
record = load_record(record_path)
record = {
"file": rel,
"character": char_id,
"items": build_items(data, rel, char_id, record),
}
if not record["items"]:
continue
processed += 1
if not needs_work(record, args.stage):
continue
pending_mt = sum(1 for i in record["items"] if i["mt"] is None)
pending_tone = sum(
1
for i in record["items"]
if i["kind"] == "dialogue" and char_id and i["toned"] is None
)
print(f"[{rel}] strings={len(record['items'])} mt-pending={pending_mt} "
f"tone-pending={pending_tone if char_id else 0}"
+ (f" character={char_id}" if char_id else ""))
if args.dry_run:
continue
started = time.monotonic()
if args.stage in ("translate", "all"):
totals["translated"] += run_translate_stage(record, record_path, cache)
if args.stage in ("tone", "all"):
totals["toned"] += run_tone_stage(record, record_path)
save_json(record_path, record)
save_json(config.CACHE_PATH, cache)
totals["files"] += 1
print(f" done in {time.monotonic() - started:.0f}s")
except KeyboardInterrupt:
print("\nInterrupted - progress saved; rerun to resume.")
sys.exit(130)
print(
f"Finished: {totals['files']} file(s) updated, "
f"{totals['translated']} new translations, {totals['toned']} tone passes."
)
if __name__ == "__main__":
main()
|