#!/usr/bin/env bash # run_vllm.sh — reference vLLM launch for MiniMax-M2.7-NVFP4-GB10-AC on dual-Spark TP=2. # # Ships configured for the "Agentic" deployment profile (Marlin NVFP4 MoE + ngram speculative # decoding). Comment out the --speculative-config line below to switch to the # "Throughput-stable" profile for novel-text / batch workloads. # # See DEPLOYMENT.md in this repo for: # - Profile tradeoffs and when to pick which # - Measured numbers on 2× DGX Spark (GB10, SM 12.1) # - Observations, caveats, and links to the community threads / PRs that informed this recipe. # # Assumes: # - Ray head + worker already running (one per Spark) # - Model mounted/available at $MODEL_PATH on both hosts # - vLLM >= 0.19.x with the Marlin NVFP4 backend built in (eugr/spark-vllm-docker nightly is the reference image) set -euo pipefail MODEL_PATH="${MODEL_PATH:-/models/MiniMax-M2.7-NVFP4-GB10-AC}" SERVED_NAME="${SERVED_NAME:-minimax-m2.7-ac}" HOST="${HOST:-0.0.0.0}" PORT="${PORT:-30000}" GPU_MEM_UTIL="${GPU_MEM_UTIL:-0.88}" MAX_MODEL_LEN="${MAX_MODEL_LEN:-196608}" MAX_NUM_SEQS="${MAX_NUM_SEQS:-12}" MAX_NUM_BATCHED_TOKENS="${MAX_NUM_BATCHED_TOKENS:-32768}" TP_SIZE="${TP_SIZE:-2}" # --- Tuned environment variables ---------------------------------------------- # Forum + vendor-recipe validated for MiniMax-M2.7 NVFP4 on GB10 (SM 12.1). # On SM 12.1, the Marlin NVFP4 MoE backend is currently the fastest path — the # FlashInfer CUTLASS NVFP4 MoE path has maturity issues on this specific compute # capability (see DEPLOYMENT.md § "Why Marlin MoE on GB10"). export SAFETENSORS_FAST_GPU=1 export OMP_NUM_THREADS=8 export TORCHINDUCTOR_MAX_AUTOTUNE=0 export VLLM_ALLOW_LONG_MAX_MODEL_LEN=1 export VLLM_FLOAT32_MATMUL_PRECISION=high export VLLM_FLASHINFER_MOE_BACKEND=throughput export VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=1 # Marlin NVFP4 MoE path export VLLM_NVFP4_GEMM_BACKEND=marlin export VLLM_USE_FLASHINFER_MOE_FP4=0 export VLLM_TEST_FORCE_FP8_MARLIN=1 export VLLM_MARLIN_USE_ATOMIC_ADD=1 # --- Compilation config ------------------------------------------------------- # cudagraph_mode=none is INTENTIONAL on dual-Spark Ray TP. # PIECEWISE captures cleanly in current vLLM builds (historical deadlock is fixed) # but measurably regresses decode 12–20% on multi-node Ray TP because each piece # boundary forces a cross-node sync over QSFP56 whose cost exceeds launch-overhead # savings. Retest only if you change away from Ray or run on a single Spark. COMPILATION_CONFIG='{"cudagraph_mode":"none","inductor_compile_config":{"combo_kernels":false,"benchmark_combo_kernel":false,"max_autotune":false,"max_autotune_gemm":false}}' # --- Speculative decoding (Agentic profile) ----------------------------------- # ngram speculation wins on agentic / code traffic (repeated tool names, file paths, # JSON keys) — peak 48.34 tok/s, avg 36.44 tok/s across our 12-prompt agent set. # On synthetic benchmarks with low token repetition it slightly regresses decode. # To switch to the "Throughput-stable" profile, comment out the SPECULATIVE_CONFIG # line below and remove --speculative-config from the vllm serve invocation. SPECULATIVE_CONFIG='{"method":"ngram","num_speculative_tokens":5,"prompt_lookup_max":4,"prompt_lookup_min":2}' # --- vLLM serve --------------------------------------------------------------- exec vllm serve "$MODEL_PATH" \ --host "$HOST" --port "$PORT" \ --served-model-name "$SERVED_NAME" \ --tensor-parallel-size "$TP_SIZE" \ --distributed-executor-backend ray \ --gpu-memory-utilization "$GPU_MEM_UTIL" \ --max-model-len "$MAX_MODEL_LEN" \ --max-num-seqs "$MAX_NUM_SEQS" \ --max-num-batched-tokens "$MAX_NUM_BATCHED_TOKENS" \ --kv-cache-dtype fp8_e4m3 \ --attention-backend flashinfer \ --attention-config.use_trtllm_attention=0 \ --enable-prefix-caching \ --enable-chunked-prefill \ --trust-remote-code \ --enable-auto-tool-choice --tool-call-parser minimax_m2 \ --reasoning-parser minimax_m2_append_think \ --compilation-config "$COMPILATION_CONFIG" \ --speculative-config "$SPECULATIVE_CONFIG"