| served_model_name: voxtral-realtime | |
| quantization: compressed-tensors | |
| kv_cache_dtype: fp8_e4m3 | |
| attention_backend: TRITON_ATTN | |
| tokenizer_mode: mistral | |
| max_model_len: 16384 | |
| max_num_seqs: 1 | |
| max_num_batched_tokens: 16384 | |
| compilation_config: | |
| cudagraph_mode: PIECEWISE | |
| speculative_config: | |
| method: ngram | |
| num_speculative_tokens: 1 | |
| prompt_lookup_max: 2 | |
| prompt_lookup_min: 1 | |