File size: 4,101 Bytes
8d3c73d
 
 
0ba0662
 
 
 
 
 
 
 
8d3c73d
 
0ba0662
 
 
8d3c73d
 
 
 
 
 
 
 
 
 
 
 
 
 
0ba0662
 
 
 
 
 
 
 
8d3c73d
 
 
0ba0662
8d3c73d
0ba0662
 
 
 
 
 
8d3c73d
 
0ba0662
8d3c73d
0ba0662
 
 
8d3c73d
 
 
0ba0662
 
 
 
 
 
 
 
 
8d3c73d
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
0ba0662
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
#!/usr/bin/env bash
# run_vllm.sh — reference vLLM launch for MiniMax-M2.7-NVFP4-GB10-AC on dual-Spark TP=2.
#
# Ships configured for the "Agentic" deployment profile (Marlin NVFP4 MoE + ngram speculative
# decoding). Comment out the --speculative-config line below to switch to the
# "Throughput-stable" profile for novel-text / batch workloads.
#
# See DEPLOYMENT.md in this repo for:
#   - Profile tradeoffs and when to pick which
#   - Measured numbers on 2× DGX Spark (GB10, SM 12.1)
#   - Observations, caveats, and links to the community threads / PRs that informed this recipe.
#
# Assumes:
#   - Ray head + worker already running (one per Spark)
#   - Model mounted/available at $MODEL_PATH on both hosts
#   - vLLM >= 0.19.x with the Marlin NVFP4 backend built in (eugr/spark-vllm-docker nightly is the reference image)

set -euo pipefail

MODEL_PATH="${MODEL_PATH:-/models/MiniMax-M2.7-NVFP4-GB10-AC}"
SERVED_NAME="${SERVED_NAME:-minimax-m2.7-ac}"
HOST="${HOST:-0.0.0.0}"
PORT="${PORT:-30000}"
GPU_MEM_UTIL="${GPU_MEM_UTIL:-0.88}"
MAX_MODEL_LEN="${MAX_MODEL_LEN:-196608}"
MAX_NUM_SEQS="${MAX_NUM_SEQS:-12}"
MAX_NUM_BATCHED_TOKENS="${MAX_NUM_BATCHED_TOKENS:-32768}"
TP_SIZE="${TP_SIZE:-2}"

# --- Tuned environment variables ----------------------------------------------
# Forum + vendor-recipe validated for MiniMax-M2.7 NVFP4 on GB10 (SM 12.1).
# On SM 12.1, the Marlin NVFP4 MoE backend is currently the fastest path — the
# FlashInfer CUTLASS NVFP4 MoE path has maturity issues on this specific compute
# capability (see DEPLOYMENT.md § "Why Marlin MoE on GB10").

export SAFETENSORS_FAST_GPU=1
export OMP_NUM_THREADS=8
export TORCHINDUCTOR_MAX_AUTOTUNE=0

export VLLM_ALLOW_LONG_MAX_MODEL_LEN=1
export VLLM_FLOAT32_MATMUL_PRECISION=high
export VLLM_FLASHINFER_MOE_BACKEND=throughput
export VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=1

# Marlin NVFP4 MoE path
export VLLM_NVFP4_GEMM_BACKEND=marlin
export VLLM_USE_FLASHINFER_MOE_FP4=0
export VLLM_TEST_FORCE_FP8_MARLIN=1
export VLLM_MARLIN_USE_ATOMIC_ADD=1

# --- Compilation config -------------------------------------------------------
# cudagraph_mode=none is INTENTIONAL on dual-Spark Ray TP.
# PIECEWISE captures cleanly in current vLLM builds (historical deadlock is fixed)
# but measurably regresses decode 12–20% on multi-node Ray TP because each piece
# boundary forces a cross-node sync over QSFP56 whose cost exceeds launch-overhead
# savings. Retest only if you change away from Ray or run on a single Spark.

COMPILATION_CONFIG='{"cudagraph_mode":"none","inductor_compile_config":{"combo_kernels":false,"benchmark_combo_kernel":false,"max_autotune":false,"max_autotune_gemm":false}}'

# --- Speculative decoding (Agentic profile) -----------------------------------
# ngram speculation wins on agentic / code traffic (repeated tool names, file paths,
# JSON keys) — peak 48.34 tok/s, avg 36.44 tok/s across our 12-prompt agent set.
# On synthetic benchmarks with low token repetition it slightly regresses decode.
# To switch to the "Throughput-stable" profile, comment out the SPECULATIVE_CONFIG
# line below and remove --speculative-config from the vllm serve invocation.

SPECULATIVE_CONFIG='{"method":"ngram","num_speculative_tokens":5,"prompt_lookup_max":4,"prompt_lookup_min":2}'

# --- vLLM serve ---------------------------------------------------------------
exec vllm serve "$MODEL_PATH" \
  --host "$HOST" --port "$PORT" \
  --served-model-name "$SERVED_NAME" \
  --tensor-parallel-size "$TP_SIZE" \
  --distributed-executor-backend ray \
  --gpu-memory-utilization "$GPU_MEM_UTIL" \
  --max-model-len "$MAX_MODEL_LEN" \
  --max-num-seqs "$MAX_NUM_SEQS" \
  --max-num-batched-tokens "$MAX_NUM_BATCHED_TOKENS" \
  --kv-cache-dtype fp8_e4m3 \
  --attention-backend flashinfer \
  --attention-config.use_trtllm_attention=0 \
  --enable-prefix-caching \
  --enable-chunked-prefill \
  --trust-remote-code \
  --enable-auto-tool-choice --tool-call-parser minimax_m2 \
  --reasoning-parser minimax_m2_append_think \
  --compilation-config "$COMPILATION_CONFIG" \
  --speculative-config "$SPECULATIVE_CONFIG"