services: glm52: image: ${VERDICTAI_IMAGE:?set VERDICTAI_IMAGE to the accepted immutable image} restart: unless-stopped gpus: all ipc: host shm_size: "32g" ulimits: memlock: -1 nofile: 1048576 ports: - "${BIND_ADDRESS:-127.0.0.1}:${PORT:-8000}:8000" environment: CUDA_VISIBLE_DEVICES: "${CUDA_VISIBLE_DEVICES:-0,1,2,3,4,5,6,7}" VLLM_PP_LAYER_PARTITION: "9,10,10,10,10,10,10,9" VLLM_WORKER_MULTIPROC_METHOD: spawn VLLM_DEEP_GEMM_WARMUP: skip volumes: - "${MODEL_DIR:?set MODEL_DIR to this repository}:/model:ro" - "${CACHE_DIR:-./.runtime-cache}:/cache:rw" command: - vllm - serve - /model - --served-model-name - ${SERVED_MODEL_NAME:-GLM-5.2-SQG-W4A8} - --host - 0.0.0.0 - --port - "8000" - --trust-remote-code - --tensor-parallel-size - "1" - --pipeline-parallel-size - "8" - --distributed-executor-backend - mp - --quantization - exl3 - --dtype - bfloat16 - --kv-cache-dtype - bfloat16 - --load-format - safetensors - --attention-backend - FLASHMLA_SPARSE - --enforce-eager - --disable-custom-all-reduce - --gpu-memory-utilization - ${GPU_MEMORY_UTILIZATION:-0.90} - --max-model-len - ${MAX_MODEL_LEN:-262144} - --speculative-config - '{"method":"mtp","num_speculative_tokens":1}' healthcheck: test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8000/v1/models >/dev/null"] interval: 30s timeout: 10s retries: 60 start_period: 20m