{ "weight_format": "affine", "profile": "JANG_2L_GS64_ProjLayerBits_Ggs64-Dgs32-Ugs64_Attn8g64_Tok8g64_NoMTP_AWQ_DiagImatrix_QAT_GPTQ", "source_model": { "repository": "deepseek-ai/DeepSeek-V4-Flash-0731", "release_identity": "DeepSeek-V4-Flash-0731", "revision": "9e165c30e2704aec5d9d593cce3eebd58bbef1cb", "source_path": "/Users/eric/sources/DeepSeek-V4-Flash-0731-9e165c30e270", "release_lock_sha256": "d8f218816eb9917c50f1773948002c670c8ebb76b862abc04790d15ca0e20d53", "config_sha256": "6c8f3d2d3b48707541b88f32f22ef3f0f8a6b57d8523281e2b8d3cdb0ae9a023", "weight_index_sha256": "98efab455cf08dfbbbaaba6f570e1bf10bf927d2b4c3c453a59c2f6f0e3be92b" }, "source_revision": "9e165c30e2704aec5d9d593cce3eebd58bbef1cb", "critical_f32_preserved": true, "awq": { "enabled": true, "method": "normalized_ffn_input_fold", "scope": "routed/shared w1+w3 plus router, inverse ffn_norm", "alpha": 0.25, "scale_clip": [ 0.5, 2.0 ], "calibrated_layers": [ 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42 ], "calibration_sha256": "22efb070d5357ee937bbe67140b710dbc6d6dceaf99d550b82ebbae94e602c11", "folded_tensor_counts": { "ffn_norm": 43, "router": 43, "routed_w1_w3": 22016, "shared_w1_w3": 86 }, "runtime_sidecar_required": false }, "imatrix": { "enabled": true, "method": "diagonal_activation_second_moment_fold", "scope": "routed w2 columns plus inverse routed w3 rows", "alpha": 0.25, "scale_clip": [ 0.5, 2.0 ], "calibrated_layers": [ 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42 ], "calibration_sha256": "22efb070d5357ee937bbe67140b710dbc6d6dceaf99d550b82ebbae94e602c11", "folded_tensor_counts": { "routed_w2": 11008, "routed_w3_inverse": 11008 }, "codec": "mlx_affine", "runtime_sidecar_required": false }, "dsv4_runtime_requirements": { "limited_swiglu_tq_patch": false, "generic_mlx_sinks": false, "native_cache_schema": "deepseek_v4_v9", "generic_turboquant_kv": false, "long_ctx_default": true, "pool_quant_default": true, "max_num_seqs": 1 }, "affine_bits": { "routed_expert": 2, "attention": 8, "shared_expert": 8, "embed_tokens": 8, "lm_head": 8, "mtp_routed_expert": 2, "mtp_non_routed_matmul": 8, "norms_router_hc": 16 }, "affine_group_size": { "routed_expert": 64, "attention": 64, "shared_expert": 64, "embed_tokens": 64, "lm_head": 64, "mtp_routed_expert": 64, "mtp_non_routed_matmul": 64 }, "quantization": { "method": "affine", "top_level_default": { "bits": 2, "group_size": 64, "mode": "affine" }, "routed_experts": { "bits": 2, "codec": "affine", "group_size": 64, "bit_plan": { "default_bits": 2, "codec": "affine", "group_size": 64, "routed_layer_bits": {}, "routed_projection_bits": {}, "mtp_routed_bits": 2, "mtp_routed_projection_bits": {}, "routed_down_layer_bits": {}, "routed_projection_layer_bits": { "w1": { "5": 3, "14": 3, "30": 3, "34": 3, "37": 3, "42": 3 } }, "routed_projection_group_sizes": { "w1": 64, "w2": 32, "w3": 64 }, "routed_projection_layer_group_sizes": {} } }, "non_routed": { "bits": 8, "codec": "affine", "group_size": 64 }, "token_bookends": { "bits": 8, "codec": "affine", "group_size": 64, "override": true }, "attention": { "bits": 8, "codec": "affine", "group_size": 64, "override": true }, "critical_control_tensors": "source-f32", "override_count": 13056, "group_totals": { "8b_g64": 512, "2b_g64": 20480, "2b_g32": 11008, "3b_g64": 1536 } }, "routed_layer_bits": {}, "routed_projection_bits": {}, "routed_down_layer_bits": {}, "routed_projection_layer_bits": { "w1": { "5": 3, "14": 3, "30": 3, "34": 3, "37": 3, "42": 3 } }, "routed_projection_group_sizes": { "w1": 64, "w2": 32, "w3": 64 }, "routed_projection_layer_group_sizes": {}, "cache": { "schema": "deepseek_v4_v9", "components": [ "swa", "csa", "hca", "compressor", "indexer" ], "sliding_window": 128, "compress_ratios": [ 0, 0, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 0, 0, 0 ], "generic_turboquant_kv": false, "pool_quant_default": true, "mtp_activation_requires_draft_cache": true }, "mtp": { "preserved": false, "runtime_self_spec_enabled": false, "mode": "dropped", "num_nextn_predict_layers": 0, "activation_requires": "separate MTP drafter, draft cache, accept/reject verifier, and DSV4 SWA+CSA/HSA composite-cache-safe rollback" }, "dspark": { "preserved": false, "runtime_enabled": false, "mode": "dropped", "stage_count": 3, "stage_ids": [ 0, 1, 2 ], "inference_n_mtp_layers": 3, "block_size": 5, "noise_token_id": 128799, "target_main_layers": [ 40, 41, 42 ], "markov_rank": 256, "activation_requires": [ "main hidden-state capture at target layers", "five-token DSpark draft block", "Markov and confidence heads", "SWA+CSA+HCA composite-cache atomic rollback" ] }, "runtime": { "bundle_has_mtp": false, "mtp_layers": 0, "mtp_mode": "dropped" }, "source_config": { "n_routed_experts": 256, "num_experts_per_tok": 6, "num_hidden_layers": 43, "num_nextn_predict_layers": 1, "inference_n_mtp_layers": 3, "sliding_window": 128, "compress_ratios": [ 0, 0, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 128, 4, 0, 0, 0 ], "hc_mult": 4, "hc_sinkhorn_iters": 20, "swiglu_limit": 10.0, "routed_scaling_factor": 1.5 }, "model_family": "deepseek_v4", "chat": { "encoder": "encoding_dsv4", "encoder_fn": "encode_messages", "chat_template_source": "official_python_encoder", "has_tokenizer_chat_template": false, "bos_token": "<\uff5cbegin\u2581of\u2581sentence\uff5c>", "eos_token": "<\uff5cend\u2581of\u2581sentence\uff5c>", "bos_token_id": 0, "eos_token_id": 1, "role_tokens": { "user": "<\uff5cUser\uff5c>", "assistant": "<\uff5cAssistant\uff5c>", "latest_reminder": "<\uff5clatest_reminder\uff5c>" }, "reasoning": { "supported": true, "modes": [ "chat", "thinking" ], "default_mode": "thinking", "default_effort": "low", "thinking_start": "", "thinking_end": "", "reasoning_effort_levels": [ "low", "high", "max" ], "drop_earlier_reasoning": true }, "tool_calling": { "supported": true, "parser": "dsml", "dsml_token": "\uff5cDSML\uff5c", "tool_calls_block": "tool_calls", "invoke_block": "invoke", "parameter_block": "parameter", "tool_output_tag": "tool_result" }, "sampling_defaults": { "_from_model_config": true, "bos_token_id": 0, "eos_token_id": 1, "do_sample": true, "temperature": 0.6, "top_p": 0.95, "transformers_version": "4.46.3", "top_k": 0, "temperature_note": "coding default override 2026-08-03; vendor card documents 1.0" } }, "qat": { "enabled": true, "method": "gptq_error_compensated_rounding", "scope": "all routed expert w1/w2/w3 codes; scales/biases min-max f16 (byte-identical grid to the non-QAT sibling); non-routed and critical controls unchanged", "objective": "per-expert routing-weighted output reconstruction; Hessian H = sum_k w_k x x^T per expert (f64 factorization, escalating damping from 0.01), blocked GPTQ column sweep on the fixed min-max grid; BRECQ sequencing (w1/w3 first, w2 on inputs derived through the quantized w1/w3 and DSV4 limited SwiGLU)", "guard": "per-expert best-of vs min-max codes; experts with <24 routed calibration rows always keep min-max codes", "calibration": "11113-row agentic-coding X1 capture through 94.995GiB NR8 keeper on vMLX Python 1.6.22, pool quant on, three prompts crossing the 2052-token indexer boundary", "codec": "unchanged stock MLX affine gather_qmm", "per_layer_recon_improvement": { "0": { "w1": 0.987, "w2": 0.945, "w3": 0.987 }, "1": { "w1": 0.982, "w2": 0.846, "w3": 0.982 }, "2": { "w1": 0.983, "w2": 0.943, "w3": 0.983 }, "3": { "w1": 0.977, "w2": 0.939, "w3": 0.977 }, "4": { "w1": 0.977, "w2": 0.877, "w3": 0.977 }, "5": { "w1": 0.979, "w2": 0.939, "w3": 0.973 }, "6": { "w1": 0.974, "w2": 0.907, "w3": 0.974 }, "7": { "w1": 0.969, "w2": 0.931, "w3": 0.969 }, "8": { "w1": 0.97, "w2": 0.931, "w3": 0.97 }, "9": { "w1": 0.973, "w2": 0.844, "w3": 0.973 }, "10": { "w1": 0.971, "w2": 0.919, "w3": 0.971 }, "11": { "w1": 0.969, "w2": 0.926, "w3": 0.969 }, "12": { "w1": 0.969, "w2": 0.92, "w3": 0.969 }, "13": { "w1": 0.963, "w2": 0.806, "w3": 0.963 }, "14": { "w1": 0.968, "w2": 0.926, "w3": 0.961 }, "15": { "w1": 0.969, "w2": 0.926, "w3": 0.969 }, "16": { "w1": 0.969, "w2": 0.926, "w3": 0.969 }, "17": { "w1": 0.966, "w2": 0.928, "w3": 0.965 }, "18": { "w1": 0.97, "w2": 0.924, "w3": 0.97 }, "19": { "w1": 0.976, "w2": 0.935, "w3": 0.976 }, "20": { "w1": 0.971, "w2": 0.944, "w3": 0.97 }, "21": { "w1": 0.971, "w2": 0.936, "w3": 0.971 }, "22": { "w1": 0.972, "w2": 0.935, "w3": 0.971 }, "23": { "w1": 0.972, "w2": 0.923, "w3": 0.971 }, "24": { "w1": 0.969, "w2": 0.944, "w3": 0.968 }, "25": { "w1": 0.969, "w2": 0.927, "w3": 0.969 }, "26": { "w1": 0.969, "w2": 0.983, "w3": 0.969 }, "27": { "w1": 0.97, "w2": 0.958, "w3": 0.97 }, "28": { "w1": 0.97, "w2": 0.927, "w3": 0.97 }, "29": { "w1": 0.972, "w2": 0.999, "w3": 0.972 }, "30": { "w1": 0.974, "w2": 0.025, "w3": 0.966 }, "31": { "w1": 0.973, "w2": 0.996, "w3": 0.973 }, "32": { "w1": 0.973, "w2": 0.993, "w3": 0.973 }, "33": { "w1": 0.974, "w2": 0.994, "w3": 0.973 }, "34": { "w1": 0.978, "w2": 0.998, "w3": 0.971 }, "35": { "w1": 0.974, "w2": 0.998, "w3": 0.974 }, "36": { "w1": 0.969, "w2": 0.998, "w3": 0.969 }, "37": { "w1": 0.979, "w2": 0.997, "w3": 0.972 }, "38": { "w1": 0.969, "w2": 0.998, "w3": 0.969 }, "39": { "w1": 0.973, "w2": 0.99, "w3": 0.973 }, "40": { "w1": 0.975, "w2": 0.998, "w3": 0.974 }, "41": { "w1": 0.972, "w2": 0.206, "w3": 0.972 }, "42": { "w1": 0.982, "w2": 0.997, "w3": 0.978 } }, "date": "2026-08-03" } }