File size: 2,544 Bytes
956e59c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
"""Central configuration for the translation pipeline and review tool.

To target a new language, change the three TARGET_* values (and PACK_NAME).
Everything else derives from them.
"""

from pathlib import Path

ROOT = Path(__file__).parent

# --- Target language -------------------------------------------------------
TARGET_LANG_CODE = "de"  # ISO 639-1 code understood by TranslateGemma
TARGET_LANG_NAME = "German"
PACK_NAME = "German_JSON"

SOURCE_LANG_CODE = "en"
SOURCE_LANG_NAME = "English"

# --- Paths ------------------------------------------------------------------
SOURCE_DIR = ROOT / "English_JSON"
PACK_DIR = ROOT / PACK_NAME
TRANSLATIONS_DIR = ROOT / "translations" / TARGET_LANG_CODE
CACHE_PATH = TRANSLATIONS_DIR / ".cache.json"
FILE_CONTEXT_PATH = ROOT / "file_context.json"
# Public fallback used on HF Spaces: context for the demo sample files only.
FILE_CONTEXT_SAMPLE_PATH = ROOT / "file_context.sample.json"
FILE_MAPPING_PATH = ROOT / "FILE_MAPPING.md"
CHAR_DATA_DIR = ROOT / "character_data"
CHAR_WIKI_DIR = ROOT / "character_wikis"

# --- Hugging Face in-process model inference --------------------------------
# Both translation stages now run inside Python with Transformers. The model
# repos below are the canonical Hugging Face checkpoints, not local HTTP
# servers. TranslateGemma is gated, so the runtime needs an HF token whose
# account has accepted Google's Gemma terms.
TRANSLATE_MODEL_ID = "google/translategemma-12b-it"
TONE_MODEL_ID = "google/gemma-4-12B-it"

# "auto" follows the model card examples and works locally and on GPU Spaces.
# Set HF_DEVICE_MAP = None to load on CPU first and explicitly move to cuda.
HF_DEVICE_MAP = "auto"
HF_DTYPE = "auto"
HF_ATTN_IMPLEMENTATION = None

TRANSLATE_MAX_NEW_TOKENS = 1024
TONE_MAX_NEW_TOKENS = 512

# Gemma's officially recommended sampling parameters for creative rewriting.
TONE_GENERATION_KWARGS = {
    "do_sample": True,
    "temperature": 1.0,
    "top_k": 64,
    "top_p": 0.95,
}

# ZeroGPU only exposes the real GPU inside @spaces.GPU functions. The decorator
# is a no-op outside ZeroGPU, so keeping these values here makes deployment a
# config change instead of a code fork.
ZERO_GPU_DURATION_S = 600
ZERO_GPU_SIZE = "xlarge"

# Keeping both 12B checkpoints resident can exceed the default 48GB ZeroGPU
# slice. The pipeline unloads the other model before loading the requested one
# unless this is set to True on a larger GPU.
HF_KEEP_BOTH_MODELS = False
HF_PRELOAD_MODELS = ()

REQUEST_TIMEOUT_S = 600
MAX_ATTEMPTS = 3
RETRY_BACKOFF_S = 5