GLM-5.2-EXL3-TR3-3.0bpw / benchmarks /2026-07-22 /prefill-8k-64k-128k.json
brandonmusic's picture
Publish validated v2 GG/Sparkinfer runtime
c7e592f verified
Raw History Blame Contribute Delete
12.8 kB
{
"metadata": {
"version": "0.4.29",
"engine": "vllm",
"model": "GLM-5.2-EXL3-TR3-3.0bpw",
"server": "127.0.0.1:8000",
"timestamp": "2026-07-22T05:36:14.936220",
"decode_mode": "duration",
"primary_decode_layer": "sustained_decode",
"duration_per_test": 30.0,
"request_count": 0,
"warmup_request_count": 0,
"run_burst": false,
"prefill_mode": "standalone_cold",
"standalone_prefill": true,
"prefill_only": true,
"skip_prefill": false,
"burst_e2e_status": "not_run_use_--run-burst",
"burst_request_count": 0,
"burst_warmup_request_count": 0,
"burst_requests_per_concurrency": 5,
"decode_warmup_seconds": 3.0,
"decode_warmup_context": 0,
"decode_warmup_concurrency": 1,
"cell_warmup_timeout_seconds": 0.0,
"cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
"show_capacity_limited_values": false,
"max_tokens": 8192,
"temperature": null,
"ignore_eos": true,
"max_total_tokens": 524288,
"dcp_size": 0,
"metrics_available": true,
"metrics_warning": "",
"concurrency_levels": [
1,
2,
4,
8,
16,
32,
64,
128
],
"context_lengths": [
0,
16384,
32768,
65536,
131072
],
"startup_diagnostics_available": true,
"nvidia_p2p_override_effective": true,
"p2pmark_status": "not_run",
"amd_fabric_status": "not_run"
},
"startup_diagnostics": {
"version": "0.4.29",
"server_url": "http://127.0.0.1:8000",
"hostname": "pop-os",
"uname": "Linux pop-os 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
"env": {},
"args": {
"concurrency": "1,2,4,8,16,32,64,128",
"contexts": "0,16k,32k,64k,128k",
"max_tokens": 8192,
"duration": 30.0,
"request_count": 0,
"run_burst": false,
"standalone_prefill": true,
"prefill_only": true,
"skip_prefill": false,
"prefill_contexts": "8k,64k,128k",
"prefill_metric": "auto",
"dcp_size": 0,
"kv_budget": 0
},
"nvidia_p2p_override": {
"effective": true,
"configured": true,
"params_path": "/proc/driver/nvidia/params",
"params_available": true,
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
"modprobe_available": true,
"runtime": {
"ForceP2P": "0x11",
"RMForceP2PType": "1",
"RMPcieP2PType": "2",
"GrdmaPciTopoCheckOverride": "1",
"EnableResizableBar": "1",
"DmaRemapPeerMmio": "1"
},
"expected": {
"ForceP2P": "0x11",
"RMForceP2PType": "1",
"RMPcieP2PType": "2",
"GrdmaPciTopoCheckOverride": "1",
"EnableResizableBar": "1"
},
"missing": [],
"mismatched": {},
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
},
"p2pmark": {
"status": "not_run"
},
"amd_fabric": {
"status": "not_run"
},
"nvidia_smi_query": {
"cmd": [
"nvidia-smi",
"--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
"--format=csv,noheader,nounits"
],
"returncode": 0,
"stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 595.58.03, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 595.58.03, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 595.58.03, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 595.58.03, 00000000:C1:00.0, 5, 16, 300.00",
"stderr": ""
},
"nvidia_smi_topo": {
"cmd": [
"nvidia-smi",
"topo",
"-m"
],
"returncode": 0,
"stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
"stderr": ""
}
},
"nvidia_p2p_override": {
"effective": true,
"configured": true,
"params_path": "/proc/driver/nvidia/params",
"params_available": true,
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
"modprobe_available": true,
"runtime": {
"ForceP2P": "0x11",
"RMForceP2PType": "1",
"RMPcieP2PType": "2",
"GrdmaPciTopoCheckOverride": "1",
"EnableResizableBar": "1",
"DmaRemapPeerMmio": "1"
},
"expected": {
"ForceP2P": "0x11",
"RMForceP2PType": "1",
"RMPcieP2PType": "2",
"GrdmaPciTopoCheckOverride": "1",
"EnableResizableBar": "1"
},
"missing": [],
"mismatched": {},
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
},
"p2pmark": {
"status": "not_run"
},
"amd_fabric": {
"status": "not_run"
},
"hardware_run_summary": {
"samples": 79,
"duration_seconds": 188.739,
"gpu_count": 4,
"cpu_util_avg_pct": 13.14,
"cpu_temp_max_c": 81.88,
"gpu_util_avg_pct": 91.6,
"gpu_util_max_pct": 100.0,
"mem_util_avg_pct": 20.04,
"mem_util_max_pct": 36.0,
"temp_avg_c": 70.94,
"temp_max_c": 90.0,
"power_total_avg_w": 1068.71,
"power_total_max_w": 1162.61,
"power_limit_total_w": 1200.0,
"vram_used_avg_mb": 380759.81,
"vram_used_max_mb": 381263.0,
"vram_total_mb": 391548.0,
"vram_used_avg_pct": 97.24,
"vram_used_max_pct": 97.37,
"pcie_rx_avg_mb_s": 50640.37,
"pcie_rx_max_mb_s": 77266.0,
"pcie_tx_avg_mb_s": 49778.35,
"pcie_tx_max_mb_s": 98654.0
},
"event_log": [],
"prefill": {
"8192": {
"ttft_seconds": 5.46,
"prefill_seconds": 5.46,
"tok_per_sec": 1502.0,
"client_ttft_seconds": 5.46,
"client_tok_per_sec": 1502.0,
"prompt_tokens": 8201,
"samples": 2,
"method": "client",
"server_validation": {
"method": "prometheus",
"tok_per_sec": 1507.0,
"prefill_seconds": 5.443,
"prompt_tokens": 8201,
"request_prompt_tokens": 8201,
"cached_tokens": 0,
"token_source": "kv_computed",
"samples": 2,
"invalid_reason": ""
},
"hardware_summary": {
"samples": 7,
"duration_seconds": 14.434,
"gpu_count": 4,
"cpu_util_avg_pct": 12.59,
"cpu_temp_max_c": 79.38,
"gpu_util_avg_pct": 64.18,
"gpu_util_max_pct": 100.0,
"mem_util_avg_pct": 12.86,
"mem_util_max_pct": 24.0,
"temp_avg_c": 61.64,
"temp_max_c": 79.0,
"power_total_avg_w": 927.54,
"power_total_max_w": 1142.27,
"power_limit_total_w": 1200.0,
"vram_used_avg_mb": 379532.0,
"vram_used_max_mb": 379659.0,
"vram_total_mb": 391548.0,
"vram_used_avg_pct": 96.93,
"vram_used_max_pct": 96.96,
"pcie_rx_avg_mb_s": 15854.29,
"pcie_rx_max_mb_s": 37233.0,
"pcie_tx_avg_mb_s": 14198.71,
"pcie_tx_max_mb_s": 40655.0
}
},
"65536": {
"ttft_seconds": 51.64,
"prefill_seconds": 51.64,
"tok_per_sec": 1249.0,
"client_ttft_seconds": 51.64,
"client_tok_per_sec": 1249.0,
"prompt_tokens": 64512,
"samples": 1,
"method": "client",
"server_validation": {
"method": "prometheus",
"tok_per_sec": 1252.0,
"prefill_seconds": 51.523,
"prompt_tokens": 64512,
"request_prompt_tokens": 64512,
"cached_tokens": 0,
"token_source": "kv_computed",
"samples": 1,
"invalid_reason": ""
},
"hardware_summary": {
"samples": 22,
"duration_seconds": 50.859,
"gpu_count": 4,
"cpu_util_avg_pct": 13.5,
"cpu_temp_max_c": 80.38,
"gpu_util_avg_pct": 95.41,
"gpu_util_max_pct": 100.0,
"mem_util_avg_pct": 21.09,
"mem_util_max_pct": 27.0,
"temp_avg_c": 68.94,
"temp_max_c": 90.0,
"power_total_avg_w": 1086.17,
"power_total_max_w": 1144.61,
"power_limit_total_w": 1200.0,
"vram_used_avg_mb": 380781.5,
"vram_used_max_mb": 381109.0,
"vram_total_mb": 391548.0,
"vram_used_avg_pct": 97.25,
"vram_used_max_pct": 97.33,
"pcie_rx_avg_mb_s": 55074.05,
"pcie_rx_max_mb_s": 77266.0,
"pcie_tx_avg_mb_s": 54968.68,
"pcie_tx_max_mb_s": 90554.0
}
},
"131072": {
"ttft_seconds": 109.002,
"prefill_seconds": 109.002,
"tok_per_sec": 1182.0,
"client_ttft_seconds": 109.002,
"client_tok_per_sec": 1182.0,
"prompt_tokens": 128881,
"samples": 1,
"method": "client",
"server_validation": {
"method": "prometheus",
"tok_per_sec": 1185.0,
"prefill_seconds": 108.797,
"prompt_tokens": 128881,
"request_prompt_tokens": 128881,
"cached_tokens": 0,
"token_source": "kv_computed",
"samples": 1,
"invalid_reason": ""
},
"hardware_summary": {
"samples": 46,
"duration_seconds": 108.985,
"gpu_count": 4,
"cpu_util_avg_pct": 13.36,
"cpu_temp_max_c": 81.88,
"gpu_util_avg_pct": 97.01,
"gpu_util_max_pct": 100.0,
"mem_util_avg_pct": 21.29,
"mem_util_max_pct": 27.0,
"temp_avg_c": 74.17,
"temp_max_c": 90.0,
"power_total_avg_w": 1105.58,
"power_total_max_w": 1151.83,
"power_limit_total_w": 1200.0,
"vram_used_avg_mb": 381232.07,
"vram_used_max_mb": 381263.0,
"vram_total_mb": 391548.0,
"vram_used_avg_pct": 97.37,
"vram_used_max_pct": 97.37,
"pcie_rx_avg_mb_s": 57352.15,
"pcie_rx_max_mb_s": 76770.0,
"pcie_tx_avg_mb_s": 55909.09,
"pcie_tx_max_mb_s": 98654.0
}
}
},
"results": [],
"summary_table": {},
"burst_results": [],
"burst_summary_table": {},
"methodology": {
"prefill": {
"name": "Prefill",
"present": true,
"mode": "standalone_cold",
"formula": "prompt_tokens / TTFT",
"notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
},
"sustained_decode": {
"name": "Sustained Decode",
"present": false,
"formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
"notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
},
"burst_e2e_decode": {
"name": "Burst / E2E Decode",
"present": false,
"status": "not run; use --run-burst",
"formula": "sum(completion_tokens) / profiling_wall_time",
"notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
}
}
}