File size: 2,507 Bytes
478fb0c | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 | """Map chat files to the character (or group) whose lines they contain.
Built from character_data/{id}.json "mentions" (produced by
build_character_data.py), keeping only characters that have a voice wiki in
character_wikis/. When several characters claim the same file (group chats,
flashback variants), the character whose id matches the filename stem wins,
then the longest id as tie-break.
Speaker attribution within a mapped file (see build_character_wikis.py):
- Conversations: "message" = the mapped character, "options" = player/owner.
- Filler Chats: type "1" = player/owner, type "-1" = system, anything else =
the mapped character/group.
"""
import json
import re
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).parent.parent))
import config
def _stem_key(rel_path: str) -> str:
"""Normalised filename stem: 'Flashback 1 Ralph.json' -> 'flashback1ralph'."""
return re.sub(r"[^a-z0-9]", "", Path(rel_path).stem.lower())
def build_dialogue_map() -> dict[str, str]:
"""Return {rel_path inside the source pack: character_id}."""
claims: dict[str, list[str]] = {}
for char_file in sorted(config.CHAR_DATA_DIR.glob("*.json")):
char_id = char_file.stem
if not (config.CHAR_WIKI_DIR / f"{char_id}.md").exists():
continue
data = json.loads(char_file.read_text(encoding="utf-8"))
for mention in data.get("mentions", []):
rel = mention.removeprefix("English_JSON/")
claims.setdefault(rel, []).append(char_id)
mapping: dict[str, str] = {}
for rel, char_ids in claims.items():
stem = _stem_key(rel)
exact = [c for c in char_ids if stem.endswith(c)]
candidates = exact or char_ids
mapping[rel] = max(candidates, key=len)
return mapping
def load_wiki(char_id: str) -> str:
return (config.CHAR_WIKI_DIR / f"{char_id}.md").read_text(encoding="utf-8").strip()
def display_name(char_id: str) -> str:
"""Character display name from character_data (falls back to the id)."""
char_file = config.CHAR_DATA_DIR / f"{char_id}.json"
if char_file.exists():
names = json.loads(char_file.read_text(encoding="utf-8")).get("name") or []
if names:
return names[0]
return char_id
if __name__ == "__main__":
dmap = build_dialogue_map()
print(f"{len(dmap)} chat files mapped to {len(set(dmap.values()))} characters")
for rel, cid in sorted(dmap.items())[:10]:
print(f" {cid:15s} {rel}")
|