served_model_name: voxtral-realtime quantization: compressed-tensors kv_cache_dtype: fp8_e4m3 attention_backend: TRITON_ATTN tokenizer_mode: mistral max_model_len: 16384 max_num_seqs: 1 max_num_batched_tokens: 16384 compilation_config: cudagraph_mode: PIECEWISE speculative_config: method: ngram num_speculative_tokens: 1 prompt_lookup_max: 2 prompt_lookup_min: 1