Download benchmark-summary-20260927.json from matteiuspi/Qwen3.8-Flash-Next-W4A16-G128-300i: direct link, hf CLI and curl.
- Browser
- Download file 3.94 kB
-
https://huggingface.co/matteiuspi/Qwen3.8-Flash-Next-W4A16-G128-300i/resolve/main/benchmark-summary-20260927.json
- Command line
-
hf download hf://matteiuspi/Qwen3.8-Flash-Next-W4A16-G128-300i/benchmark-summary-20260927.json
-
curl -L -o benchmark-summary-20260927.json https://huggingface.co/matteiuspi/Qwen3.8-Flash-Next-W4A16-G128-300i/resolve/main/benchmark-summary-20260927.json
3.94 kB
| { | |
| "schema_version": 1, | |
| "date_utc": "2026-09-27", | |
| "model_id": "matteiuspi/Qwen3.8-Flash-Next-W4A16-G128-300i", | |
| "hardware": "2 x Atlas 300I Duo / 4 x Ascend 310P3", | |
| "checkpoint_format": "W4A16 G128, packed INT4 routed experts, FP16 activations", | |
| "runtime": "experimental W4 tile-reuse candidate, MTP draft depth 2, FULL_DECODE_ONLY graphs, TP4/EP", | |
| "sampling": { | |
| "temperature": 0, | |
| "seed": 42, | |
| "thinking": false, | |
| "ignore_eos": true, | |
| "completion_tokens": 512 | |
| }, | |
| "metric": "(completion_tokens - 1) / (elapsed_seconds - time_to_first_token_seconds), based on client streaming; excludes prefill", | |
| "workloads": [ | |
| { | |
| "label": "short_prompt", | |
| "source_run": "tile-http-candidate-short.jsonl", | |
| "measured_requests": [ | |
| { | |
| "prompt_tokens": 32, | |
| "cached_prompt_tokens": 0, | |
| "completion_tokens": 512, | |
| "time_to_first_token_seconds": 0.5066101450065617, | |
| "decode_seconds": 31.007059219999064, | |
| "client_decode_tokens_per_second": 16.480118168394796, | |
| "mtp_acceptance": 0.8643617021276596, | |
| "output_sha256": "05ab57b2bc71b7bd6efb1f43d5ed54885de933122a4120e11373ed6258f95ef8" | |
| }, | |
| { | |
| "prompt_tokens": 43, | |
| "cached_prompt_tokens": 0, | |
| "completion_tokens": 512, | |
| "time_to_first_token_seconds": 0.6068624640029157, | |
| "decode_seconds": 36.076975575997494, | |
| "client_decode_tokens_per_second": 14.164158492819318, | |
| "mtp_acceptance": 0.6820276497695853, | |
| "output_sha256": "d6f9cf24f10f2b6fcf8c9664648ce6d499ff2e226bbb78a9f6e4791d64d26d17" | |
| }, | |
| { | |
| "prompt_tokens": 49, | |
| "cached_prompt_tokens": 0, | |
| "completion_tokens": 512, | |
| "time_to_first_token_seconds": 0.6193453000014415, | |
| "decode_seconds": 34.727616406999005, | |
| "client_decode_tokens_per_second": 14.714514063136594, | |
| "mtp_acceptance": 0.7166666666666667, | |
| "output_sha256": "6f72caf6e5abf1b1dc3e370968ad7cb683667a6597b86d5bf55989ffce2d6af0" | |
| } | |
| ], | |
| "median_client_decode_tokens_per_second": 14.714514063136594 | |
| }, | |
| { | |
| "label": "approximately_23k_prompt", | |
| "source_run": "tile-http-candidate-long.jsonl", | |
| "measured_requests": [ | |
| { | |
| "prompt_tokens": 23407, | |
| "cached_prompt_tokens": 23168, | |
| "completion_tokens": 512, | |
| "time_to_first_token_seconds": 1.6225465909956256, | |
| "decode_seconds": 31.515383977006422, | |
| "client_decode_tokens_per_second": 16.2143034770836, | |
| "mtp_acceptance": 0.8864864864864865, | |
| "output_sha256": "cef47d1a306d8ad208bcf04501012707e5c2389d130dbc4a8e9d404662b23ef9" | |
| }, | |
| { | |
| "prompt_tokens": 23418, | |
| "cached_prompt_tokens": 23168, | |
| "completion_tokens": 512, | |
| "time_to_first_token_seconds": 1.6728898490036954, | |
| "decode_seconds": 34.99341846400057, | |
| "client_decode_tokens_per_second": 14.602745957091633, | |
| "mtp_acceptance": 0.7524509803921569, | |
| "output_sha256": "9b1f608c46be1e37542ec02ded7c803cc061326094ad6ad3c59a5c84cbcb53ad" | |
| }, | |
| { | |
| "prompt_tokens": 23424, | |
| "cached_prompt_tokens": 23168, | |
| "completion_tokens": 512, | |
| "time_to_first_token_seconds": 2.3089601309984573, | |
| "decode_seconds": 36.66200596400449, | |
| "client_decode_tokens_per_second": 13.938135313755343, | |
| "mtp_acceptance": 0.6985981308411215, | |
| "output_sha256": "6866fdd0b999ca2fc1a513eb69363ec95bd9beb8c20da37d8f232c769a70976c" | |
| } | |
| ], | |
| "median_client_decode_tokens_per_second": 14.602745957091633 | |
| } | |
| ], | |
| "scope": "Three requests per workload in one W4 runtime study. W8 was not rerun on this configuration. Fixed-length outputs and prompt checks do not establish full-model quality or 262K context correctness." | |
| } | |