| { |
| "model": "glm-5.3-flash-recal", |
| "ctx": 260000, |
| "gen": 2200, |
| "rounds": [ |
| { |
| "round": 1, |
| "prompt_tokens": 225032, |
| "completion_tokens": 2712, |
| "wall_s": 64.1, |
| "finish_reason": "length", |
| "four_gram_max": 2, |
| "four_gram_top": "at least 2,000 words", |
| "eight_gram_max": 1, |
| "eight_gram_top": "The user wants a detailed essay of at", |
| "max_consec_line_repeat": 1, |
| "degenerate_4gram_ge12": false, |
| "tail": "hmarks but degenerate during extended generation. The most informative evaluations combine these approaches, testing models under conditions that approximate real-world usage while maintaining the controlled variables needed for scientific comparison. As context windows continue to expand and models are deployed in increasingly demanding applications, the sophistication of evaluation methodologies" |
| }, |
| { |
| "round": 2, |
| "prompt_tokens": 224997, |
| "completion_tokens": 2712, |
| "wall_s": 58.4, |
| "finish_reason": "length", |
| "four_gram_max": 3, |
| "four_gram_top": "- Why long context", |
| "eight_gram_max": 1, |
| "eight_gram_top": "The user has provided a long filler document", |
| "max_consec_line_repeat": 1, |
| "degenerate_4gram_ge12": false, |
| "tail": "asymmetry; per-channel/per-token grouping; outliers\n- Eviction/compression alternatives (sinks, heavy hitters) as complementary\n- Evaluation protocol: sweep precision, measure NIAH/perplexity deltas, latency, memory\n- Sensitivity of retrieval to key perturbations; softmax sensitivity\n- Practical guidance: keep K higher precision or per-channel scales; quantize V aggressively; validate at multiple\n" |
| } |
| ], |
| "verdict": "PASS" |
| } |