File size: 7,283 Bytes
478fb0c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
"""Batch translation driver: English_JSON -> translations/<lang>/ review records.

Stage 1 (translate): every translatable string through TranslateGemma.
Stage 2 (tone): dialogue lines of mapped characters through the Ollama tone
model using the character's voice wiki.

One review record file is written per source file, mirroring the pack layout.
Runs are resumable: existing records are merged by item key and only missing
work is done. Repeated strings hit a shared translation cache.

Usage:
    python -m pipeline.translate_pack                       # full run, both stages
    python -m pipeline.translate_pack --filter "Filler Chats/Brad"
    python -m pipeline.translate_pack --stage translate     # stage 1 only
    python -m pipeline.translate_pack --limit 5             # first 5 files
    python -m pipeline.translate_pack --dry-run             # plan only, no LLM calls
"""

import argparse
import json
import sys
import time
from pathlib import Path

sys.path.insert(0, str(Path(__file__).parent.parent))
import config
from pipeline import clients
from pipeline.dialogue_map import build_dialogue_map, display_name, load_wiki
from pipeline.rules import iter_translatable, key_to_str

SAVE_EVERY = 25  # persist record/cache after this many new translations
CONTEXT_LINES = 2  # preceding source lines passed to the tone model


def load_record(record_path: Path) -> dict:
    if record_path.exists():
        return json.loads(record_path.read_text(encoding="utf-8"))
    return {}


def save_json(path: Path, data: dict) -> None:
    path.parent.mkdir(parents=True, exist_ok=True)
    path.write_text(json.dumps(data, indent=2, ensure_ascii=False), encoding="utf-8")


def build_items(data, rel: str, char_id: str | None, existing: dict) -> list[dict]:
    """Fresh item list from the source file, merged with prior record state."""
    prior = {item["key"]: item for item in existing.get("items", [])}
    items = []
    for found in iter_translatable(data, rel):
        key = key_to_str(found.path)
        old = prior.get(key)
        if old and old["source"] == found.source:
            items.append(old)
            continue
        items.append(
            {
                "key": key,
                "path": list(found.path),
                "source": found.source,
                "kind": found.kind,
                "character": char_id if found.kind == "dialogue" else None,
                "mt": None,
                "toned": None,
                "status": "pending",
                "final": None,
            }
        )
    return items


def run_translate_stage(record: dict, record_path: Path, cache: dict) -> int:
    done = 0
    for item in record["items"]:
        if item["mt"] is not None:
            continue
        cached = cache.get(item["source"])
        if cached is not None:
            item["mt"] = cached
        else:
            item["mt"] = clients.translate(item["source"])
            cache[item["source"]] = item["mt"]
            done += 1
            if done % SAVE_EVERY == 0:
                save_json(record_path, record)
                save_json(config.CACHE_PATH, cache)
    return done


def run_tone_stage(record: dict, record_path: Path) -> int:
    char_id = record.get("character")
    if not char_id:
        return 0
    wiki = load_wiki(char_id)
    name = display_name(char_id)
    sources = [item["source"] for item in record["items"]]

    done = 0
    for i, item in enumerate(record["items"]):
        if item["kind"] != "dialogue" or item["toned"] is not None or item["mt"] is None:
            continue
        context = sources[max(0, i - CONTEXT_LINES) : i]
        item["toned"] = clients.adjust_tone(item["source"], item["mt"], wiki, name, context)
        done += 1
        if done % SAVE_EVERY == 0:
            save_json(record_path, record)
    return done


def needs_work(record: dict, stage: str) -> bool:
    for item in record.get("items", []):
        if stage in ("translate", "all") and item["mt"] is None:
            return True
        if (
            stage in ("tone", "all")
            and item["kind"] == "dialogue"
            and record.get("character")
            and item["mt"] is not None
            and item["toned"] is None
        ):
            return True
    return False


def main() -> None:
    parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
    parser.add_argument("--filter", default="", help="only files whose path contains this substring")
    parser.add_argument("--limit", type=int, default=0, help="max number of files to process")
    parser.add_argument("--stage", choices=["translate", "tone", "all"], default="all")
    parser.add_argument("--dry-run", action="store_true", help="report planned work, no LLM calls")
    args = parser.parse_args()
    sys.stdout.reconfigure(line_buffering=True)

    dmap = build_dialogue_map()
    cache = json.loads(config.CACHE_PATH.read_text(encoding="utf-8")) if config.CACHE_PATH.exists() else {}

    source_files = sorted(config.SOURCE_DIR.rglob("*.json"))
    processed = 0
    totals = {"translated": 0, "toned": 0, "files": 0}

    try:
        for source_file in source_files:
            rel = str(source_file.relative_to(config.SOURCE_DIR))
            if args.filter and args.filter not in rel:
                continue
            if args.limit and processed >= args.limit:
                break

            data = json.loads(source_file.read_text(encoding="utf-8"))
            char_id = dmap.get(rel)
            record_path = config.TRANSLATIONS_DIR / rel
            record = load_record(record_path)
            record = {
                "file": rel,
                "character": char_id,
                "items": build_items(data, rel, char_id, record),
            }
            if not record["items"]:
                continue
            processed += 1
            if not needs_work(record, args.stage):
                continue

            pending_mt = sum(1 for i in record["items"] if i["mt"] is None)
            pending_tone = sum(
                1
                for i in record["items"]
                if i["kind"] == "dialogue" and char_id and i["toned"] is None
            )
            print(f"[{rel}] strings={len(record['items'])} mt-pending={pending_mt} "
                  f"tone-pending={pending_tone if char_id else 0}"
                  + (f" character={char_id}" if char_id else ""))
            if args.dry_run:
                continue

            started = time.monotonic()
            if args.stage in ("translate", "all"):
                totals["translated"] += run_translate_stage(record, record_path, cache)
            if args.stage in ("tone", "all"):
                totals["toned"] += run_tone_stage(record, record_path)
            save_json(record_path, record)
            save_json(config.CACHE_PATH, cache)
            totals["files"] += 1
            print(f"  done in {time.monotonic() - started:.0f}s")
    except KeyboardInterrupt:
        print("\nInterrupted - progress saved; rerun to resume.")
        sys.exit(130)

    print(
        f"Finished: {totals['files']} file(s) updated, "
        f"{totals['translated']} new translations, {totals['toned']} tone passes."
    )


if __name__ == "__main__":
    main()