{ "metadata": { "version": "0.4.29", "engine": "vllm", "model": "GLM-5.2-EXL3-TR3-3.0bpw", "server": "127.0.0.1:8000", "timestamp": "2026-07-22T05:36:14.936220", "decode_mode": "duration", "primary_decode_layer": "sustained_decode", "duration_per_test": 30.0, "request_count": 0, "warmup_request_count": 0, "run_burst": false, "prefill_mode": "standalone_cold", "standalone_prefill": true, "prefill_only": true, "skip_prefill": false, "burst_e2e_status": "not_run_use_--run-burst", "burst_request_count": 0, "burst_warmup_request_count": 0, "burst_requests_per_concurrency": 5, "decode_warmup_seconds": 3.0, "decode_warmup_context": 0, "decode_warmup_concurrency": 1, "cell_warmup_timeout_seconds": 0.0, "cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0", "show_capacity_limited_values": false, "max_tokens": 8192, "temperature": null, "ignore_eos": true, "max_total_tokens": 524288, "dcp_size": 0, "metrics_available": true, "metrics_warning": "", "concurrency_levels": [ 1, 2, 4, 8, 16, 32, 64, 128 ], "context_lengths": [ 0, 16384, 32768, 65536, 131072 ], "startup_diagnostics_available": true, "nvidia_p2p_override_effective": true, "p2pmark_status": "not_run", "amd_fabric_status": "not_run" }, "startup_diagnostics": { "version": "0.4.29", "server_url": "http://127.0.0.1:8000", "hostname": "pop-os", "uname": "Linux pop-os 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux", "env": {}, "args": { "concurrency": "1,2,4,8,16,32,64,128", "contexts": "0,16k,32k,64k,128k", "max_tokens": 8192, "duration": 30.0, "request_count": 0, "run_burst": false, "standalone_prefill": true, "prefill_only": true, "skip_prefill": false, "prefill_contexts": "8k,64k,128k", "prefill_metric": "auto", "dcp_size": 0, "kv_budget": 0 }, "nvidia_p2p_override": { "effective": true, "configured": true, "params_path": "/proc/driver/nvidia/params", "params_available": true, "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf", "modprobe_available": true, "runtime": { "ForceP2P": "0x11", "RMForceP2PType": "1", "RMPcieP2PType": "2", "GrdmaPciTopoCheckOverride": "1", "EnableResizableBar": "1", "DmaRemapPeerMmio": "1" }, "expected": { "ForceP2P": "0x11", "RMForceP2PType": "1", "RMPcieP2PType": "2", "GrdmaPciTopoCheckOverride": "1", "EnableResizableBar": "1" }, "missing": [], "mismatched": {}, "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1", "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"", "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded" }, "p2pmark": { "status": "not_run" }, "amd_fabric": { "status": "not_run" }, "nvidia_smi_query": { "cmd": [ "nvidia-smi", "--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit", "--format=csv,noheader,nounits" ], "returncode": 0, "stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 595.58.03, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 595.58.03, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 595.58.03, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 595.58.03, 00000000:C1:00.0, 5, 16, 300.00", "stderr": "" }, "nvidia_smi_topo": { "cmd": [ "nvidia-smi", "topo", "-m" ], "returncode": 0, "stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks", "stderr": "" } }, "nvidia_p2p_override": { "effective": true, "configured": true, "params_path": "/proc/driver/nvidia/params", "params_available": true, "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf", "modprobe_available": true, "runtime": { "ForceP2P": "0x11", "RMForceP2PType": "1", "RMPcieP2PType": "2", "GrdmaPciTopoCheckOverride": "1", "EnableResizableBar": "1", "DmaRemapPeerMmio": "1" }, "expected": { "ForceP2P": "0x11", "RMForceP2PType": "1", "RMPcieP2PType": "2", "GrdmaPciTopoCheckOverride": "1", "EnableResizableBar": "1" }, "missing": [], "mismatched": {}, "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1", "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"", "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded" }, "p2pmark": { "status": "not_run" }, "amd_fabric": { "status": "not_run" }, "hardware_run_summary": { "samples": 79, "duration_seconds": 188.739, "gpu_count": 4, "cpu_util_avg_pct": 13.14, "cpu_temp_max_c": 81.88, "gpu_util_avg_pct": 91.6, "gpu_util_max_pct": 100.0, "mem_util_avg_pct": 20.04, "mem_util_max_pct": 36.0, "temp_avg_c": 70.94, "temp_max_c": 90.0, "power_total_avg_w": 1068.71, "power_total_max_w": 1162.61, "power_limit_total_w": 1200.0, "vram_used_avg_mb": 380759.81, "vram_used_max_mb": 381263.0, "vram_total_mb": 391548.0, "vram_used_avg_pct": 97.24, "vram_used_max_pct": 97.37, "pcie_rx_avg_mb_s": 50640.37, "pcie_rx_max_mb_s": 77266.0, "pcie_tx_avg_mb_s": 49778.35, "pcie_tx_max_mb_s": 98654.0 }, "event_log": [], "prefill": { "8192": { "ttft_seconds": 5.46, "prefill_seconds": 5.46, "tok_per_sec": 1502.0, "client_ttft_seconds": 5.46, "client_tok_per_sec": 1502.0, "prompt_tokens": 8201, "samples": 2, "method": "client", "server_validation": { "method": "prometheus", "tok_per_sec": 1507.0, "prefill_seconds": 5.443, "prompt_tokens": 8201, "request_prompt_tokens": 8201, "cached_tokens": 0, "token_source": "kv_computed", "samples": 2, "invalid_reason": "" }, "hardware_summary": { "samples": 7, "duration_seconds": 14.434, "gpu_count": 4, "cpu_util_avg_pct": 12.59, "cpu_temp_max_c": 79.38, "gpu_util_avg_pct": 64.18, "gpu_util_max_pct": 100.0, "mem_util_avg_pct": 12.86, "mem_util_max_pct": 24.0, "temp_avg_c": 61.64, "temp_max_c": 79.0, "power_total_avg_w": 927.54, "power_total_max_w": 1142.27, "power_limit_total_w": 1200.0, "vram_used_avg_mb": 379532.0, "vram_used_max_mb": 379659.0, "vram_total_mb": 391548.0, "vram_used_avg_pct": 96.93, "vram_used_max_pct": 96.96, "pcie_rx_avg_mb_s": 15854.29, "pcie_rx_max_mb_s": 37233.0, "pcie_tx_avg_mb_s": 14198.71, "pcie_tx_max_mb_s": 40655.0 } }, "65536": { "ttft_seconds": 51.64, "prefill_seconds": 51.64, "tok_per_sec": 1249.0, "client_ttft_seconds": 51.64, "client_tok_per_sec": 1249.0, "prompt_tokens": 64512, "samples": 1, "method": "client", "server_validation": { "method": "prometheus", "tok_per_sec": 1252.0, "prefill_seconds": 51.523, "prompt_tokens": 64512, "request_prompt_tokens": 64512, "cached_tokens": 0, "token_source": "kv_computed", "samples": 1, "invalid_reason": "" }, "hardware_summary": { "samples": 22, "duration_seconds": 50.859, "gpu_count": 4, "cpu_util_avg_pct": 13.5, "cpu_temp_max_c": 80.38, "gpu_util_avg_pct": 95.41, "gpu_util_max_pct": 100.0, "mem_util_avg_pct": 21.09, "mem_util_max_pct": 27.0, "temp_avg_c": 68.94, "temp_max_c": 90.0, "power_total_avg_w": 1086.17, "power_total_max_w": 1144.61, "power_limit_total_w": 1200.0, "vram_used_avg_mb": 380781.5, "vram_used_max_mb": 381109.0, "vram_total_mb": 391548.0, "vram_used_avg_pct": 97.25, "vram_used_max_pct": 97.33, "pcie_rx_avg_mb_s": 55074.05, "pcie_rx_max_mb_s": 77266.0, "pcie_tx_avg_mb_s": 54968.68, "pcie_tx_max_mb_s": 90554.0 } }, "131072": { "ttft_seconds": 109.002, "prefill_seconds": 109.002, "tok_per_sec": 1182.0, "client_ttft_seconds": 109.002, "client_tok_per_sec": 1182.0, "prompt_tokens": 128881, "samples": 1, "method": "client", "server_validation": { "method": "prometheus", "tok_per_sec": 1185.0, "prefill_seconds": 108.797, "prompt_tokens": 128881, "request_prompt_tokens": 128881, "cached_tokens": 0, "token_source": "kv_computed", "samples": 1, "invalid_reason": "" }, "hardware_summary": { "samples": 46, "duration_seconds": 108.985, "gpu_count": 4, "cpu_util_avg_pct": 13.36, "cpu_temp_max_c": 81.88, "gpu_util_avg_pct": 97.01, "gpu_util_max_pct": 100.0, "mem_util_avg_pct": 21.29, "mem_util_max_pct": 27.0, "temp_avg_c": 74.17, "temp_max_c": 90.0, "power_total_avg_w": 1105.58, "power_total_max_w": 1151.83, "power_limit_total_w": 1200.0, "vram_used_avg_mb": 381232.07, "vram_used_max_mb": 381263.0, "vram_total_mb": 391548.0, "vram_used_avg_pct": 97.37, "vram_used_max_pct": 97.37, "pcie_rx_avg_mb_s": 57352.15, "pcie_rx_max_mb_s": 76770.0, "pcie_tx_avg_mb_s": 55909.09, "pcie_tx_max_mb_s": 98654.0 } } }, "results": [], "summary_table": {}, "burst_results": [], "burst_summary_table": {}, "methodology": { "prefill": { "name": "Prefill", "present": true, "mode": "standalone_cold", "formula": "prompt_tokens / TTFT", "notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation." }, "sustained_decode": { "name": "Sustained Decode", "present": false, "formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable", "notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline." }, "burst_e2e_decode": { "name": "Burst / E2E Decode", "present": false, "status": "not run; use --run-burst", "formula": "sum(completion_tokens) / profiling_wall_time", "notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion." } } }