{ "name": "0731-affine-awq-diag-imatrix-nr8-keeper-layout-gptq-base", "source_repository": "deepseek-ai/DeepSeek-V4-Flash-0731", "source_revision": "9e165c30e2704aec5d9d593cce3eebd58bbef1cb", "runtime": "vMLX Python", "format": "JANG MLX affine with folded AWQ and diagonal imatrix", "profile_bits": 2, "bookend_bits": 8, "bookend_group_size": 64, "attention_bits": 8, "attention_group_size": 64, "token_bookend_bits": 8, "token_bookend_group_size": 64, "routed_group_size": 64, "routed_projection_group_sizes": { "w1": 64, "w2": 32, "w3": 64 }, "routed_projection_layer_bits": { "w1": { "5": 3, "14": 3, "30": 3, "34": 3, "37": 3, "42": 3 } }, "routed_projection_layer_group_sizes": {}, "awq": { "capture": "actual post_attention_layernorm output after hc_pre", "alpha": 0.25, "clip": [ 0.5, 2.0 ], "fold": "inverse ffn_norm; router and routed/shared w1+w3 input columns", "folded_control_dtype": "F32", "folded_control_extra_payload_bytes": 90529792 }, "imatrix": { "capture": "actual routed down_proj input after DSV4 SwiGLU", "statistic": "per-channel second moment", "alpha": 0.25, "clip": [ 0.5, 2.0 ], "fold": "routed w2 columns and inverse routed w3 rows", "codec": "stock MLX affine" }, "critical_controls": "source precision", "drop_mtp": true, "target_min_bytes": 101500000000, "target_max_bytes": 102500000000, "target_display_gib": [ 94.5, 95.5 ], "header_and_index_reserve_bytes": 100000000, "projected_payload_bytes": 101971512412, "projected_payload_gib": 94.96837147697806, "generation_and_prompt_contract": "official 0731 Python encoder; default thinking/low; temperature=0.6 (coding default; vendor card 1.0) top_p=0.95 top_k=0; no repetition penalty", "description": "Fast-regime shipping base: proven keeper NR8 layout (w1 2/g64 with six 3-bit lifts, w2 2/g32, w3 2/g64 all layers) rebuilt with pool-quant and temp-0.6 stamps; receives GPTQ error-compensated routed codes." }