DZAIR / tokenizer_config.json
ainouche-abderahmane's picture
Fix tokenizer class, fp16 numerics and repo layout
1c795a3 verified
Raw History Blame Contribute Delete
563 Bytes
{
"tokenizer_class": "DebertaV2Tokenizer",
"model_input_names": [
"input_ids",
"attention_mask"
],
"unk_token": "[UNK]",
"pad_token": "[PAD]",
"cls_token": "[CLS]",
"sep_token": "[SEP]",
"mask_token": "[MASK]",
"do_lower_case": false,
"keep_accents": true,
"split_by_punct": false,
"padding_side": "right",
"truncation_side": "right",
"model_max_length": 512,
"latin_lowercase_before_encode": true,
"note": "Lowercase Latin spans before encoding; the tokenizer then wraps input as [CLS] chunk [SEP]. See the model card."
}