File size: 2,060 Bytes
b22d729
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
{
  "method": "Swift-specific GSQ refinement with reused ISTA GSQ-RCO per-tensor allocation profiles",
  "passes": [
    "Joint attention and expert refinement",
    "Second expert-refinement pass"
  ],
  "source_models": [
    {
      "tier": "IQ2_XS",
      "unsplit_sha256": "eb0dcf3c752d4b282eacc902e09200c1ac1fd974b4466e43f8a07699311f6580",
      "unsplit_bytes": 68152166976,
      "development_kld": 0.341275,
      "shards": [
        "Swift-Qwen3.8-Flash-Next-GSQ-RCO-IQ2_XS-00001-of-00002.gguf",
        "Swift-Qwen3.8-Flash-Next-GSQ-RCO-IQ2_XS-00002-of-00002.gguf"
      ],
      "tensor_count": 1224
    },
    {
      "tier": "Q2_0",
      "unsplit_sha256": "a14046f27675e53fadccaf99e8f7583efda6239d9932001ee927e3e33e710015",
      "unsplit_bytes": 66549952576,
      "development_kld": 0.42435,
      "shards": [
        "Swift-Qwen3.8-Flash-Next-GSQ-RCO-Q2_0-00001-of-00002.gguf",
        "Swift-Qwen3.8-Flash-Next-GSQ-RCO-Q2_0-00002-of-00002.gguf"
      ],
      "tensor_count": 1224
    },
    {
      "tier": "IQ3_XXS",
      "unsplit_sha256": "12e72a2fcb388b192c7352a1392770bfb60365a49b37875f146adf578c60e358",
      "unsplit_bytes": 75966072896,
      "development_kld": 0.240139,
      "shards": [
        "Swift-Qwen3.8-Flash-Next-GSQ-RCO-IQ3_XXS-00001-of-00002.gguf",
        "Swift-Qwen3.8-Flash-Next-GSQ-RCO-IQ3_XXS-00002-of-00002.gguf"
      ],
      "tensor_count": 1224
    }
  ],
  "importance_matrix": {
    "file": "imatrix-swiftfn-v1mix.gguf",
    "sha256": "769098d84e38baa285832e6335898cc5e04d578c8a65abbe4e164ce9795ac0fc"
  },
  "notes": [
    "Allocation reuse does not constitute a new RCO search on Swift.",
    "Native quantization-format support and validation applied for each tier.",
    "IQ2_XS and Q2_0 second passes were segmented, with prefix replay and a global torch RNG reset at the second segment; not bit-identical to an uninterrupted run.",
    "Q2_0 denotes a mixed per-tensor allocation profile, not uniform Q2_0 storage.",
    "This manifest identifies the exact evaluated and packaged candidates."
  ]
}