File size: 2,306 Bytes
478fb0c | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 | """Parse FILE_MAPPING.md into file_context.json.
FILE_MAPPING.md (teammate-generated) has one "### N. filename" section per
source file with bullet fields. We extract per file: content label, inferred
role, and detailed summary, keyed by the file's path inside the JSON pack
(e.g. "Initial/Gameplay/Adam_s Phone/Messages/General.json").
Usage:
python -m pipeline.build_file_context
"""
import json
import re
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).parent.parent))
import config
_SECTION_RE = re.compile(r"^### \d+\. `", re.MULTILINE)
_FIELD_RES = {
"rel_path": re.compile(r"^- Relative path: `(.+?)`", re.MULTILINE),
"label": re.compile(r"^- Content label: `(.+?)`", re.MULTILINE),
"role": re.compile(r"^- Inferred role: (.+?)$", re.MULTILINE),
"summary": re.compile(r"^- Detailed summary: (.+?)$", re.MULTILINE),
}
def normalize_key(rel_path: str) -> str:
"""Canonical lookup key: strip stray spaces around each path component."""
return "/".join(part.strip() for part in rel_path.replace("\\", "/").split("/"))
def _pack_key(rel_path: str) -> str:
"""'English/Initial/.../General.docx' -> 'Initial/.../General.json'."""
path = rel_path.replace("\\", "/").removeprefix("English/")
return normalize_key(str(Path(path).with_suffix(".json")))
def parse_file_mapping(markdown: str) -> dict[str, dict[str, str]]:
context: dict[str, dict[str, str]] = {}
for section in _SECTION_RE.split(markdown)[1:]:
fields = {}
for name, pattern in _FIELD_RES.items():
match = pattern.search(section)
fields[name] = match.group(1).strip() if match else ""
if not fields["rel_path"]:
continue
context[_pack_key(fields["rel_path"])] = {
"label": fields["label"],
"role": fields["role"],
"summary": fields["summary"],
}
return context
def main() -> None:
markdown = config.FILE_MAPPING_PATH.read_text(encoding="utf-8")
context = parse_file_mapping(markdown)
config.FILE_CONTEXT_PATH.write_text(
json.dumps(context, indent=2, ensure_ascii=False), encoding="utf-8"
)
print(f"Wrote context for {len(context)} files to {config.FILE_CONTEXT_PATH.name}")
if __name__ == "__main__":
main()
|