# DZAIR tokenizer rules. Version lives inside this file, never in the filename. # Rule changes are version bumps with entries in tokenizer_rules.history.md. # This file freezes task 2.5's selection; the sweep that produced it is # recorded in data/processed/tokenizer/sweep.stats.json. version: 1 algorithm: unigram implementation: sentencepiece normalization_rule_name: identity byte_fallback: true split_digits: true character_coverage: 0.9995 max_sentencepiece_length: 16 specials: pad: {piece: "[PAD]", id: 0} unk: {piece: "[UNK]", id: 1} cls: {piece: "[CLS]", id: 2} sep: {piece: "[SEP]", id: 3} mask: {piece: "[MASK]", user_defined: true} required_chars: arabic_block: "U+0600-U+06FF alpha, minus unified alefs U+0622/U+0623/U+0625, tatweel U+0640, Arabic-Indic digits, combining marks" latin: "a-z only (normalisation v1 lowercases Latin; uppercase slots would never train)" french_accents: "U+00E0 U+00E2 U+00E6 U+00E7 U+00E9 U+00E8 U+00EA U+00EB U+00EE U+00EF U+00F4 U+0153 U+00F9 U+00FB U+00FC U+00FF" digits: "0-9 ASCII" apostrophes: "U+0027 U+2019 (guaranteed, never pre-split)" hyphen: "U+002D (splits Arabizi/French compounds)" training: cut: full input_lines: 4000000 seed: 42 input_sentence_size: 4000000 shuffle_input_sentence: true sweep: [16000, 24000, 32000, 48000, 64000] nfkc_arm: dead selection: name: dzair-tok-48k vocab_size: 48000 normalization: identity criterion: min fertility at same-or-smaller vocab vs DziriBERT 1.4370 with zero UNK-class failures fertility_overall: 1.4503 reopen: Phase 3 downstream rank overturns, or flat downstream selects 64k on compression; fertility alone never reopens inference_preprocessing: - NFC (training text is normalisation-v1 output, already NFC) - lowercase Latin (training text is lowercased; raw uppercase fragments into bytes at +10% fertility, measured 2.6) - no transliteration, no script unification (lossy by Guellil evidence) robustness: report: data/processed/tokenizer/robustness.stats.json unk_hits: 0 roundtrip_failures: 0