File size: 1,273 Bytes
0805035
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
{
  "model": "Step-3.7-Flash-ROCmFPX-Q3-QualityPlus",
  "benchmark": "Tool-Eval full",
  "harness_version": "2.0.7",
  "run_id": "2026-07-05T06-07-28.330900Z_405f5a32",
  "scenario_count": 69,
  "final_score": 88,
  "deployability": 70,
  "responsiveness": 27,
  "total_points": 122,
  "max_points": 138,
  "status_counts": {
    "pass": 57,
    "partial": 8,
    "fail": 4
  },
  "runtime": {
    "backend": "llama.cpp ROCmFPX / Vulkan",
    "context": 65536,
    "mtp": true,
    "speculative_n_max": 2,
    "speculative_p_min": 0.75,
    "target_kv": "q8_0/q8_0",
    "draft_kv": "q8_0/q8_0",
    "batch": 8192,
    "ubatch": 2048
  },
  "speed_and_memory": {
    "pp_tps": 145.65656394243823,
    "tg_tps": 32.338072639262656,
    "prompt_tokens": 34333,
    "predicted_tokens": 33430,
    "peak_pooled_gpu_gib": 96.34914779663086,
    "peak_ram_used_gib": 107.00336074829102,
    "wall_time": "21:21.23"
  },
  "notes": [
    "Local AMD Ryzen AI Max+ 395 / Strix Halo measurement.",
    "Uses the Step native tool_response chat template with protocol-boundary escaping.",
    "The public llm.ciru.ai StepFun tool-eval page currently documents the Step tool-calling methodology and matching FP4 score row; this JSON records the exact Q3 QualityPlus run summary."
  ]
}