{ "method": "Swift-specific GSQ refinement with reused ISTA GSQ-RCO per-tensor allocation profiles", "passes": [ "Joint attention and expert refinement", "Second expert-refinement pass" ], "source_models": [ { "tier": "IQ2_XS", "unsplit_sha256": "eb0dcf3c752d4b282eacc902e09200c1ac1fd974b4466e43f8a07699311f6580", "unsplit_bytes": 68152166976, "development_kld": 0.341275, "shards": [ "Swift-Qwen3.8-Flash-Next-GSQ-RCO-IQ2_XS-00001-of-00002.gguf", "Swift-Qwen3.8-Flash-Next-GSQ-RCO-IQ2_XS-00002-of-00002.gguf" ], "tensor_count": 1224 }, { "tier": "Q2_0", "unsplit_sha256": "a14046f27675e53fadccaf99e8f7583efda6239d9932001ee927e3e33e710015", "unsplit_bytes": 66549952576, "development_kld": 0.42435, "shards": [ "Swift-Qwen3.8-Flash-Next-GSQ-RCO-Q2_0-00001-of-00002.gguf", "Swift-Qwen3.8-Flash-Next-GSQ-RCO-Q2_0-00002-of-00002.gguf" ], "tensor_count": 1224 }, { "tier": "IQ3_XXS", "unsplit_sha256": "12e72a2fcb388b192c7352a1392770bfb60365a49b37875f146adf578c60e358", "unsplit_bytes": 75966072896, "development_kld": 0.240139, "shards": [ "Swift-Qwen3.8-Flash-Next-GSQ-RCO-IQ3_XXS-00001-of-00002.gguf", "Swift-Qwen3.8-Flash-Next-GSQ-RCO-IQ3_XXS-00002-of-00002.gguf" ], "tensor_count": 1224 } ], "importance_matrix": { "file": "imatrix-swiftfn-v1mix.gguf", "sha256": "769098d84e38baa285832e6335898cc5e04d578c8a65abbe4e164ce9795ac0fc" }, "notes": [ "Allocation reuse does not constitute a new RCO search on Swift.", "Native quantization-format support and validation applied for each tier.", "IQ2_XS and Q2_0 second passes were segmented, with prefix replay and a global torch RNG reset at the second segment; not bit-identical to an uninterrupted run.", "Q2_0 denotes a mixed per-tensor allocation profile, not uniform Q2_0 storage.", "This manifest identifies the exact evaluated and packaged candidates." ] }