| { | |
| "model": "glm-5.3-flash", | |
| "ctx": 300000, | |
| "gen": 2200, | |
| "rounds": [ | |
| { | |
| "round": 1, | |
| "prompt_tokens": 259632, | |
| "completion_tokens": 2712, | |
| "wall_s": 69.2, | |
| "finish_reason": "length", | |
| "four_gram_max": 2, | |
| "four_gram_top": "as a function of", | |
| "eight_gram_max": 1, | |
| "eight_gram_top": "The user wants a detailed, well-structured essay of", | |
| "max_consec_line_repeat": 1, | |
| "degenerate_4gram_ge12": false, | |
| "tail": "nce length, output length, and the specific attention implementation used. Standardized benchmarks like MLPerf provide reference implementations, but long-context evaluation often requires custom setups.\n\nKey metrics include: prefill throughput (tokens/second for processing the input), decode throughput (tokens/second for generation, measured at various context lengths), time-to-first-token (TTFT," | |
| } | |
| ], | |
| "verdict": "PASS" | |
| } |