brandonmusic commited on
Commit
5488407
·
verified ·
1 Parent(s): 1ad19f0

Add files using upload-large-folder tool

Browse files
MTP-VERIFY-REPORT.json ADDED
@@ -0,0 +1,17 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "status": "PASS",
3
+ "layer": 45,
4
+ "experts": 288,
5
+ "sidecars": 4,
6
+ "expert_stream_mismatches": 0,
7
+ "codec_closure": "PASS: all 864 projection encodes",
8
+ "coupled_component_validation": "PASS",
9
+ "routed_sidecar_bytes": 3853990880,
10
+ "mean_tw_nmse": {
11
+ "gate_tw_nmse": 0.005752041559620415,
12
+ "up_tw_nmse": 0.005751395422463146,
13
+ "down_tw_nmse": 0.0061286417262067724
14
+ },
15
+ "scale_policy": "signed-unit coupled vectors; CPU torch seed 530045; no prior r27 MTP scales",
16
+ "native_execution_tested": false
17
+ }
VERIFY-REPORT.json CHANGED
@@ -1,56 +1,74 @@
1
  {
2
- "checkpoint": "/root/cv2data/out/ckpt-B",
3
- "runtime_overlay_and_kernel_header_checks": "PASS (168 files)",
4
- "expert_stream_mismatches": {
5
- "3": 0,
6
- "4": 0,
7
- "5": 0,
8
- "6": 0,
9
- "7": 0,
10
- "8": 0,
11
- "9": 0,
12
- "10": 0,
13
- "11": 0,
14
- "12": 0,
15
- "13": 0,
16
- "14": 0,
17
- "15": 0,
18
- "16": 0,
19
- "17": 0,
20
- "18": 0,
21
- "19": 0,
22
- "20": 0,
23
- "21": 0,
24
- "22": 0,
25
- "23": 0,
26
- "24": 0,
27
- "25": 0,
28
- "26": 0,
29
- "27": 0,
30
- "28": 0,
31
- "29": 0,
32
- "30": 0,
33
- "31": 0,
34
- "32": 0,
35
- "33": 0,
36
- "34": 0,
37
- "35": 0,
38
- "36": 0,
39
- "37": 0,
40
- "38": 0,
41
- "39": 0,
42
- "40": 0,
43
- "41": 0,
44
- "42": 0,
45
- "43": 0,
46
- "44": 0
47
- },
48
- "routed_sidecar_bytes": 161867586048,
49
- "tp2_routed_bytes_per_gpu": 80933793024.0,
50
- "allocation_counts": {
51
- "3": 0,
52
- "4": 42,
53
- "5": 0
54
- },
55
- "status": "PASS"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
56
  }
 
1
  {
2
+ "checkpoint": "/root/cv2data/out/ckpt-B",
3
+ "runtime_overlay_and_kernel_header_checks": "PASS: 168 main files plus 4 MTP files",
4
+ "expert_stream_mismatches": {
5
+ "3": 0,
6
+ "4": 0,
7
+ "5": 0,
8
+ "6": 0,
9
+ "7": 0,
10
+ "8": 0,
11
+ "9": 0,
12
+ "10": 0,
13
+ "11": 0,
14
+ "12": 0,
15
+ "13": 0,
16
+ "14": 0,
17
+ "15": 0,
18
+ "16": 0,
19
+ "17": 0,
20
+ "18": 0,
21
+ "19": 0,
22
+ "20": 0,
23
+ "21": 0,
24
+ "22": 0,
25
+ "23": 0,
26
+ "24": 0,
27
+ "25": 0,
28
+ "26": 0,
29
+ "27": 0,
30
+ "28": 0,
31
+ "29": 0,
32
+ "30": 0,
33
+ "31": 0,
34
+ "32": 0,
35
+ "33": 0,
36
+ "34": 0,
37
+ "35": 0,
38
+ "36": 0,
39
+ "37": 0,
40
+ "38": 0,
41
+ "39": 0,
42
+ "40": 0,
43
+ "41": 0,
44
+ "42": 0,
45
+ "43": 0,
46
+ "44": 0,
47
+ "45": 0
48
+ },
49
+ "routed_sidecar_bytes": 165721576928,
50
+ "tp2_routed_bytes_per_gpu": 82860788464.0,
51
+ "allocation_counts": {
52
+ "3": 0,
53
+ "4": 43,
54
+ "5": 0
55
+ },
56
+ "status": "PASS",
57
+ "mtp_verification": {
58
+ "status": "PASS",
59
+ "layer": 45,
60
+ "experts": 288,
61
+ "sidecars": 4,
62
+ "expert_stream_mismatches": 0,
63
+ "codec_closure": "PASS: all 864 projection encodes",
64
+ "coupled_component_validation": "PASS",
65
+ "routed_sidecar_bytes": 3853990880,
66
+ "mean_tw_nmse": {
67
+ "gate_tw_nmse": 0.005752041559620415,
68
+ "up_tw_nmse": 0.005751395422463146,
69
+ "down_tw_nmse": 0.0061286417262067724
70
+ },
71
+ "scale_policy": "signed-unit coupled vectors; CPU torch seed 530045; no prior r27 MTP scales",
72
+ "native_execution_tested": false
73
+ }
74
  }
carrier/model-mtp-nonexpert.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:31fc8804fce4cb1e0207d9fb13c19a892d85b50a7f769e763cb8d9e5b6a2e6ba
3
+ size 369674104
carrier/model.safetensors.index.json CHANGED
The diff for this file is too large to render. See raw diff
 
design/design-2.json CHANGED
The diff for this file is too large to render. See raw diff
 
evidence/build-mtp-checkpoint.py ADDED
@@ -0,0 +1,144 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Assemble and verify MTP45 sidecars and retain only its unchanged carrier tensors."""
2
+ import hashlib
3
+ import json
4
+ import math
5
+ import os
6
+ import shutil
7
+ import time
8
+ from pathlib import Path
9
+
10
+ import torch
11
+ from safetensors import safe_open
12
+ from safetensors.torch import load_file, save_file
13
+
14
+ from build_checkpoint import build_rank, sha256_file, tensor_sha256, CODEC
15
+ from p8_coupled_scales import SCALE_NAMES, validate_coupled_component
16
+
17
+ out = Path(os.environ["CV2_OUT"])
18
+ root = Path(os.environ["CV2_ROOT"])
19
+ while not (out / "status/mtp-encoded.ready").exists():
20
+ time.sleep(10)
21
+ inputs = out / "mtp-inputs"
22
+ input_info = json.loads((inputs / "inputs.json").read_text())
23
+ core_sha = sha256_file(root / "code/codecv2/encode_layer_v2.py")
24
+ wrapper_sha = sha256_file(root / "ops/encode-mtp-v2.py")
25
+ chunks, metrics = {}, []
26
+ for path in sorted((out / "mtp-chunks").glob("*.safetensors")):
27
+ with safe_open(str(path), framework="pt", device="cpu") as src:
28
+ meta = src.metadata()
29
+ assert meta["schema"] == "trellismx-codec-v2-mtp-chunk.v1"
30
+ assert meta["layer"] == "45" and meta["bits"] == "4"
31
+ assert meta["encoder_sha256"] == core_sha and meta["mtp_wrapper_sha256"] == wrapper_sha
32
+ assert meta["mtp_inputs_sha256"] == sha256_file(inputs / "inputs.json")
33
+ assert float(meta["beta"]) == 0.5 and float(meta["percdamp"]) == 0.3
34
+ start, end = map(int, meta["expert_range"].split(":"))
35
+ for expert in range(start, end):
36
+ assert expert not in chunks
37
+ chunks[expert] = {p: (src.get_tensor(f"{expert}.{p}.trellis"), src.get_tensor(f"{expert}.{p}.scale_ue8m0"))
38
+ for p in ("gate", "up", "down")}
39
+ metrics.extend(json.loads(path.with_suffix(".metrics.json").read_text())["rows"])
40
+ assert set(chunks) == set(range(288))
41
+ assert {r["expert"] for r in metrics} == set(range(288)) and len(metrics) == 288
42
+ assert all(math.isfinite(v) for row in metrics for v in row.values())
43
+ dest = out / "mtp-sidecars"
44
+ (dest / "sidecars").mkdir(parents=True, exist_ok=True)
45
+ (dest / "design").mkdir(exist_ok=True)
46
+ design = {"schema": "trellismx-codecv2-mtp-design.v1", "layer": 45, "bits": 4,
47
+ "encoder_core_sha256": core_sha, "mtp_wrapper_sha256": wrapper_sha,
48
+ "beta": 0.5, "percdamp": 0.3, "coupled_sign_draw": 0,
49
+ "scale_policy": input_info["scale_policy"], "scale_sha256": input_info["scale_sha256"],
50
+ "capture_sha256": input_info["source_capture_sha256"],
51
+ "calibration": "64 fit windows; 2047 valid MTP rows each; all routed plus every fourth all-token row",
52
+ "main_model_scales_unchanged": True}
53
+ design_path = dest / "design/design-2.json"
54
+ design_path.write_text(json.dumps(design, indent=2) + "\n")
55
+ design_sha = sha256_file(design_path)
56
+ scale = load_file(str(inputs / "coupled-scales.safetensors"))
57
+ transform_sha = sha256_file(out / "ckpt-B/design/transform.json")
58
+ records, ranks = [], []
59
+ for rank in range(4):
60
+ template = out / "ckpt-B/sidecars" / f"p8-layer-044-tp4-rank-{rank}.safetensors"
61
+ tensors, meta = build_rank(template, chunks, rank, 4, design_sha)
62
+ sl = slice(rank * 512, (rank + 1) * 512)
63
+ tensors["gate_up_suh_fp16"] = scale["gate_up_suh"]
64
+ tensors["down_svh_fp16"] = scale["down_svh"]
65
+ tensors["intermediate_scales_fp16"] = torch.cat([scale[key][:, sl] for key in ("gate_svh", "up_svh", "down_suh")], dim=1).contiguous()
66
+ meta.pop("exl3_scale_source_sha256", None)
67
+ meta.update(layer="45", mtp_scale_source_sha256=input_info["scale_sha256"],
68
+ mtp_scale_policy=input_info["scale_policy"], mtp_capture_sha256=input_info["source_capture_sha256"])
69
+ for name in SCALE_NAMES:
70
+ meta["sha256_" + name] = tensor_sha256(tensors[name])
71
+ validate_coupled_component(meta, {name: tensors[name] for name in SCALE_NAMES},
72
+ layer=45, rank=rank, experts=288, hidden=4096, intermediate=512,
73
+ expected_transform_sha256=transform_sha)
74
+ name = f"sidecars/p8-layer-045-tp4-rank-{rank}.safetensors"
75
+ path = dest / name
76
+ save_file(tensors, str(path), metadata=meta)
77
+ records.append({"layer": 45, "rank": rank, "bits": 4, "path": name,
78
+ "bytes": path.stat().st_size, "sha256": sha256_file(path), "source_design_sha256": design_sha})
79
+ # Read the actual saved artifact for the full stream reassembly check.
80
+ with safe_open(str(path), framework="pt", device="cpu") as src:
81
+ ranks.append({key: src.get_tensor(key) for key in CODEC})
82
+ print(json.dumps({"saved": name, "bytes": path.stat().st_size}), flush=True)
83
+ mismatches = 0
84
+ for expert in range(288):
85
+ actual = {
86
+ "gate": (torch.cat([t["w13_trellis"][0, expert] for t in ranks], 1),
87
+ torch.cat([t["w13_scale_ue8m0"][expert, 512:] for t in ranks], 0)),
88
+ "up": (torch.cat([t["w13_trellis"][1, expert] for t in ranks], 1),
89
+ torch.cat([t["w13_scale_ue8m0"][expert, :512] for t in ranks], 0)),
90
+ "down": (torch.cat([t["w2_trellis"][expert] for t in ranks], 0),
91
+ torch.cat([t["w2_scale_ue8m0"][expert] for t in ranks], 1))}
92
+ mismatches += not all(torch.equal(a, b) for key in actual for a, b in zip(actual[key], chunks[expert][key]))
93
+ assert mismatches == 0
94
+ report = {"status": "PASS", "layer": 45, "experts": 288, "sidecars": 4,
95
+ "expert_stream_mismatches": mismatches, "codec_closure": "PASS: all 864 projection encodes",
96
+ "coupled_component_validation": "PASS", "routed_sidecar_bytes": sum(r["bytes"] for r in records),
97
+ "mean_tw_nmse": {key: sum(r[key] for r in metrics) / 288 for key in ("gate_tw_nmse", "up_tw_nmse", "down_tw_nmse")},
98
+ "scale_policy": input_info["scale_policy"], "native_execution_tested": False}
99
+ (dest / "MTP-VERIFY-REPORT.json").write_text(json.dumps(report, indent=2) + "\n")
100
+ (dest / "mtp-manifest.json").write_text(json.dumps({"files": records, "design_sha256": design_sha}, indent=2) + "\n")
101
+
102
+ # Remove only replaced MTP expert tensors; retain all 25 other MTP tensors bit for bit.
103
+ old_carrier = out / "native-carrier"
104
+ carrier = out / "native-carrier-mtp"
105
+ carrier.mkdir(exist_ok=True)
106
+ base_receipt = json.loads((out / "native-carrier-receipt.json").read_text())
107
+ kept_tensors = {}
108
+ for record in base_receipt["files"]:
109
+ name = record["path"]
110
+ if name.startswith("model-mtp-"):
111
+ with safe_open(str(old_carrier / name), framework="pt", device="cpu") as src:
112
+ for key in src.keys():
113
+ if ".layers.45.mlp.experts." not in key:
114
+ assert key not in kept_tensors
115
+ kept_tensors[key] = src.get_tensor(key)
116
+ elif name != "model.safetensors.index.json":
117
+ if not (carrier / name).exists():
118
+ os.link(old_carrier / name, carrier / name)
119
+ assert len(kept_tensors) == 25
120
+ mtp_file = "model-mtp-nonexpert.safetensors"
121
+ save_file(kept_tensors, str(carrier / mtp_file))
122
+ with safe_open(str(carrier / mtp_file), framework="pt", device="cpu") as check:
123
+ assert set(check.keys()) == set(kept_tensors)
124
+ assert all(torch.equal(check.get_tensor(key), value) for key, value in kept_tensors.items())
125
+ index = json.loads((old_carrier / "model.safetensors.index.json").read_text())
126
+ weight_map = {k: v for k, v in index["weight_map"].items() if ".layers.45.mlp.experts." not in k}
127
+ for key in kept_tensors:
128
+ weight_map[key] = mtp_file
129
+ index["weight_map"] = weight_map
130
+ index["metadata"]["total_size"] -= 7247757312 + 226492416
131
+ (carrier / "model.safetensors.index.json").write_text(json.dumps(index, indent=2) + "\n")
132
+ known = {r["path"]: r for r in base_receipt["files"]}
133
+ new_records = []
134
+ for path in sorted(carrier.iterdir()):
135
+ assert path.is_file()
136
+ digest = known[path.name]["sha256"] if path.name in known and path.name != "model.safetensors.index.json" else sha256_file(path)
137
+ new_records.append({"path": path.name, "bytes": path.stat().st_size, "sha256": digest})
138
+ receipt = {**base_receipt, "files": new_records, "bytes": sum(r["bytes"] for r in new_records),
139
+ "indexed_tensors": len(weight_map), "mtp_expert_tensors_replaced": 1728,
140
+ "mtp_nonexpert_tensors_preserved": 25, "mtp_nonexpert_tensor_equality": "PASS",
141
+ "derivation": "MTP45 FP8 experts and scales replaced by Trellis sidecars; other tensors unchanged"}
142
+ (out / "native-carrier-mtp-receipt.json").write_text(json.dumps(receipt, indent=2) + "\n")
143
+ (out / "status/mtp.ready").write_text(design_sha + "\n")
144
+ print(json.dumps(report), flush=True)
evidence/encode-mtp-v2.py ADDED
@@ -0,0 +1,91 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Encode MTP45 with the supplied codec-v2 math and its 2047-row fit captures."""
2
+ import argparse
3
+ import hashlib
4
+ import json
5
+ import os
6
+ import time
7
+ from pathlib import Path
8
+
9
+ import numpy as np
10
+ import torch
11
+ from safetensors.torch import load_file, save_file
12
+
13
+ import encode_layer_v2 as cv2
14
+ from glm53_nvfp4.p8_coupled_scale import CoupledScaleSet, COUPLED_SIGN_DRAW, encode_coupled_scale_weights, coupled_input_carrier, coupled_middle_carrier
15
+ from glm53_nvfp4.shard_index import IndexedCheckpoint
16
+
17
+ ap = argparse.ArgumentParser()
18
+ ap.add_argument("--expert-start", type=int, required=True)
19
+ ap.add_argument("--expert-end", type=int, required=True)
20
+ ap.add_argument("--out", type=Path, required=True)
21
+ args = ap.parse_args()
22
+ assert 0 <= args.expert_start < args.expert_end <= 288
23
+ assert not args.out.exists()
24
+ inputs = Path(os.environ["CV2_OUT"]) / "mtp-inputs"
25
+ info = json.loads((inputs / "inputs.json").read_text())
26
+ assert info["rows"] == 64 * 2047
27
+ hidden = np.memmap(inputs / "hidden.bf16.bin", mode="r", dtype="<u2", shape=(info["rows"], 4096))
28
+ ids = np.memmap(inputs / "topk_ids.u16le.bin", mode="r", dtype="<u2", shape=(info["rows"], 8))
29
+ routes = np.memmap(inputs / "topk_weights.f32le.bin", mode="r", dtype="<f4", shape=(info["rows"], 8))
30
+ scales = load_file(str(inputs / "coupled-scales.safetensors"))
31
+ scales_all = CoupledScaleSet(**scales, source_path=inputs, source_sha256=info["scale_sha256"], source_metadata={}, tensor_hashes={})
32
+ scales_all.validate()
33
+ device = torch.device("cuda:0")
34
+ torch.cuda.set_device(device)
35
+ every = np.asarray([w * 2047 + t for w in range(64) for t in range(0, 2047, 4)])
36
+ hid_every = torch.from_numpy(np.array(hidden[every], copy=True)).view(torch.bfloat16)
37
+ ones_every = torch.ones(len(every))
38
+ source_path = Path(os.environ["CV2_BF16"])
39
+ source = IndexedCheckpoint(source_path, source_path / "model.safetensors.index.json")
40
+ payload, metrics, h_all_in = {}, [], None
41
+ started = time.time()
42
+ for expert in range(args.expert_start, args.expert_end):
43
+ t0 = time.time()
44
+ base = source.expert_prefix(45, expert)
45
+ orig = {p: source.get(f"{base}{p}.weight").to(device).float() for p in cv2.PROJ}
46
+ sc = CoupledScaleSet(gate_up_suh=scales_all.gate_up_suh.to(device),
47
+ down_svh=scales_all.down_svh.to(device),
48
+ gate_svh=scales_all.gate_svh[expert:expert + 1].to(device),
49
+ up_svh=scales_all.up_svh[expert:expert + 1].to(device),
50
+ down_suh=scales_all.down_suh[expert:expert + 1].to(device),
51
+ source_path=inputs, source_sha256=info["scale_sha256"], source_metadata={}, tensor_hashes={})
52
+ tw = encode_coupled_scale_weights(orig["gate_proj"], orig["up_proj"], orig["down_proj"], sc,
53
+ expert=0, intermediate_draw=COUPLED_SIGN_DRAW)
54
+ carrier = lambda h: coupled_input_carrier(h, sc.gate_up_suh, quantize=True)
55
+ row_ids, slots = np.nonzero(ids == expert)
56
+ assert row_ids.size, f"No MTP fit routes for expert {expert}"
57
+ hid_r = torch.from_numpy(np.array(hidden[row_ids], copy=True)).view(torch.bfloat16)
58
+ rt_r = torch.from_numpy(np.array(routes[row_ids, slots], copy=True))
59
+ if h_all_in is None:
60
+ h_all_in = cv2.hessian_chunks(hid_every, ones_every, carrier, device)
61
+ h_in = cv2.mix(cv2.hessian_chunks(hid_r, rt_r, carrier, device), h_all_in, 0.5)
62
+ g_rec, g_tr, g_sc = cv2.encode_v2(tw[0], h_in, 4, 0.3)
63
+ u_rec, u_tr, u_sc = cv2.encode_v2(tw[1], h_in, 4, 0.3)
64
+ mid = lambda h: coupled_middle_carrier(carrier(h), g_rec, u_rec, sc, expert=0,
65
+ intermediate_draw=COUPLED_SIGN_DRAW, quantize=True)
66
+ h_dn = cv2.mix(cv2.hessian_chunks(hid_r, rt_r, mid, device),
67
+ cv2.hessian_chunks(hid_every, ones_every, mid, device), 0.5)
68
+ d_rec, d_tr, d_sc = cv2.encode_v2(tw[2], h_dn, 4, 0.3)
69
+ row = {"expert": expert, "routed_tokens": len(row_ids), "seconds": round(time.time() - t0, 2)}
70
+ for name, rec, tr, codes, target in (("gate", g_rec, g_tr, g_sc, tw[0]),
71
+ ("up", u_rec, u_tr, u_sc, tw[1]), ("down", d_rec, d_tr, d_sc, tw[2])):
72
+ payload[f"{expert}.{name}.trellis"] = tr
73
+ payload[f"{expert}.{name}.scale_ue8m0"] = codes
74
+ row[f"{name}_tw_nmse"] = float((rec - target).double().square().sum() / target.double().square().sum())
75
+ assert all(np.isfinite(v) for v in row.values())
76
+ metrics.append(row)
77
+ print(json.dumps(row), flush=True)
78
+ del orig, tw, h_in, h_dn, g_rec, u_rec, d_rec, hid_r, rt_r
79
+ torch.cuda.empty_cache()
80
+ meta = {"schema": "trellismx-codec-v2-mtp-chunk.v1", "layer": "45", "bits": "4",
81
+ "expert_range": f"{args.expert_start}:{args.expert_end}", "beta": "0.5", "percdamp": "0.3",
82
+ "encoder_sha256": hashlib.sha256(Path(cv2.__file__).read_bytes()).hexdigest(),
83
+ "mtp_wrapper_sha256": hashlib.sha256(Path(__file__).read_bytes()).hexdigest(),
84
+ "mtp_inputs_sha256": hashlib.sha256((inputs / "inputs.json").read_bytes()).hexdigest(),
85
+ "scale_policy": info["scale_policy"]}
86
+ args.out.parent.mkdir(parents=True, exist_ok=True)
87
+ tmp = args.out.with_suffix(".partial")
88
+ save_file(payload, str(tmp), metadata=meta)
89
+ tmp.rename(args.out)
90
+ args.out.with_suffix(".metrics.json").write_text(json.dumps({"meta": meta, "rows": metrics, "seconds": time.time() - started}, indent=2) + "\n")
91
+ print(json.dumps({"done": str(args.out), "seconds": time.time() - started}), flush=True)
evidence/mtp-inputs.json ADDED
@@ -0,0 +1,717 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "schema": "trellismx-codecv2-mtp-inputs.v1",
3
+ "layer": 45,
4
+ "capture_repo": "brandonmusic/GLM-5.3-Flash-BF16-Teacher-Logits",
5
+ "capture_revision": "95f4fdd94bf29989db2e0d1054e4931f55edb6aa",
6
+ "source_capture_sha256": "50afb51ecb3519599bf8722e7f2fe92e3e0edc96d3e3ba605ed59246055f8ee8",
7
+ "windows": [
8
+ {
9
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
10
+ "document_id": "reap-recall-packed-axis4_reasoning_termination-1",
11
+ "domain": "axis4_reasoning_termination",
12
+ "main_terminal_offset": 6144,
13
+ "role": "fit",
14
+ "rows": 2047,
15
+ "token_ids_sha256": "208eadbaf0f326e410557519bd28398a6cb44027a143fa7d5eb4e104e492c59f",
16
+ "window_id": "fit-0003",
17
+ "window_index": 3
18
+ },
19
+ {
20
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
21
+ "document_id": "reap-recall-packed-axis4_reasoning_termination-1",
22
+ "domain": "axis4_reasoning_termination",
23
+ "main_terminal_offset": 14336,
24
+ "role": "fit",
25
+ "rows": 2047,
26
+ "token_ids_sha256": "b4b12b709b0643763ea88f28a7159fb0d6f67ec403a769d82888097330c588fe",
27
+ "window_id": "fit-0007",
28
+ "window_index": 7
29
+ },
30
+ {
31
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
32
+ "document_id": "reap-recall-packed-axis4_reasoning_termination-1",
33
+ "domain": "axis4_reasoning_termination",
34
+ "main_terminal_offset": 22528,
35
+ "role": "fit",
36
+ "rows": 2047,
37
+ "token_ids_sha256": "03e62368844238497ffcf7a12bd9baa5bd36480ec68d2f8563c9180fd275560f",
38
+ "window_id": "fit-0011",
39
+ "window_index": 11
40
+ },
41
+ {
42
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
43
+ "document_id": "reap-recall-packed-axis4_reasoning_termination-1",
44
+ "domain": "axis4_reasoning_termination",
45
+ "main_terminal_offset": 30720,
46
+ "role": "fit",
47
+ "rows": 2047,
48
+ "token_ids_sha256": "d35fae888ad057534d85e66c7eec53d189eb31f4312ed9f84009501f20584619",
49
+ "window_id": "fit-0015",
50
+ "window_index": 15
51
+ },
52
+ {
53
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
54
+ "document_id": "reap-recall-packed-axis4_reasoning_termination-1",
55
+ "domain": "axis4_reasoning_termination",
56
+ "main_terminal_offset": 38912,
57
+ "role": "fit",
58
+ "rows": 2047,
59
+ "token_ids_sha256": "ff92c88a67993287542987983aa46d9e9a0ab636ff3ff453f0db487cd3654181",
60
+ "window_id": "fit-0019",
61
+ "window_index": 19
62
+ },
63
+ {
64
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
65
+ "document_id": "reap-recall-packed-axis1_general-1",
66
+ "domain": "axis1_general",
67
+ "main_terminal_offset": 40960,
68
+ "role": "fit",
69
+ "rows": 2047,
70
+ "token_ids_sha256": "c0f29007388fc4a03964260f8d9d5bc5e24239db854979e215866eeb4c8db3e9",
71
+ "window_id": "fit-0020",
72
+ "window_index": 20
73
+ },
74
+ {
75
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
76
+ "document_id": "reap-recall-packed-axis3_code_agentic-4",
77
+ "domain": "axis3_code_agentic",
78
+ "main_terminal_offset": 45056,
79
+ "role": "fit",
80
+ "rows": 2047,
81
+ "token_ids_sha256": "0ed0081f3ee1f0117a8b716fb530cbfa6b014d2a550e6a4a32f0e4c403559edd",
82
+ "window_id": "fit-0022",
83
+ "window_index": 22
84
+ },
85
+ {
86
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
87
+ "document_id": "reap-recall-packed-axis4_reasoning_termination-1",
88
+ "domain": "axis4_reasoning_termination",
89
+ "main_terminal_offset": 47104,
90
+ "role": "fit",
91
+ "rows": 2047,
92
+ "token_ids_sha256": "cb81685f194c02bd698d50ed0f5ec197a28dcfdca6d9a37c1af44111d9b5e438",
93
+ "window_id": "fit-0023",
94
+ "window_index": 23
95
+ },
96
+ {
97
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
98
+ "document_id": "reap-recall-packed-axis4_reasoning_termination-1",
99
+ "domain": "axis4_reasoning_termination",
100
+ "main_terminal_offset": 55296,
101
+ "role": "fit",
102
+ "rows": 2047,
103
+ "token_ids_sha256": "2a5e68ea816bcd1ae71eb65a2c2963ad693139fb9b18853cc395772310584346",
104
+ "window_id": "fit-0027",
105
+ "window_index": 27
106
+ },
107
+ {
108
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
109
+ "document_id": "reap-recall-packed-axis4_reasoning_termination-1",
110
+ "domain": "axis4_reasoning_termination",
111
+ "main_terminal_offset": 63488,
112
+ "role": "fit",
113
+ "rows": 2047,
114
+ "token_ids_sha256": "6bcd57c39f9f206255f13684ce2c00367879867f68b405589edd41b5678ac249",
115
+ "window_id": "fit-0031",
116
+ "window_index": 31
117
+ },
118
+ {
119
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
120
+ "document_id": "reap-recall-packed-axis4_reasoning_termination-1",
121
+ "domain": "axis4_reasoning_termination",
122
+ "main_terminal_offset": 71680,
123
+ "role": "fit",
124
+ "rows": 2047,
125
+ "token_ids_sha256": "e8866385831f1d5bfccba77298e19ac414b8b90dd468ef138d979f704675ce5f",
126
+ "window_id": "fit-0035",
127
+ "window_index": 35
128
+ },
129
+ {
130
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
131
+ "document_id": "reap-recall-packed-axis4_reasoning_termination-1",
132
+ "domain": "axis4_reasoning_termination",
133
+ "main_terminal_offset": 79872,
134
+ "role": "fit",
135
+ "rows": 2047,
136
+ "token_ids_sha256": "d2a9535df825df10239ecd7212ec5e0cca0e216e5303c6d7ec77cef63680cc06",
137
+ "window_id": "fit-0039",
138
+ "window_index": 39
139
+ },
140
+ {
141
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
142
+ "document_id": "reap-recall-packed-axis4_reasoning_termination-1",
143
+ "domain": "axis4_reasoning_termination",
144
+ "main_terminal_offset": 88064,
145
+ "role": "fit",
146
+ "rows": 2047,
147
+ "token_ids_sha256": "4388a3c18f9cdc932620967f6fda46146d90f722b8263482354eccdd16177c85",
148
+ "window_id": "fit-0043",
149
+ "window_index": 43
150
+ },
151
+ {
152
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
153
+ "document_id": "reap-recall-packed-axis4_reasoning_termination-1",
154
+ "domain": "axis4_reasoning_termination",
155
+ "main_terminal_offset": 96256,
156
+ "role": "fit",
157
+ "rows": 2047,
158
+ "token_ids_sha256": "0432ef02690e147e5b6d03c38f0207010e830c1344823223141b5ba49e771826",
159
+ "window_id": "fit-0047",
160
+ "window_index": 47
161
+ },
162
+ {
163
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
164
+ "document_id": "reap-recall-packed-axis4_reasoning_termination-1",
165
+ "domain": "axis4_reasoning_termination",
166
+ "main_terminal_offset": 104448,
167
+ "role": "fit",
168
+ "rows": 2047,
169
+ "token_ids_sha256": "5dd6cf9058c192340985bdc02d379b102c194c7e4da43ff209cf5de019b25b8c",
170
+ "window_id": "fit-0051",
171
+ "window_index": 51
172
+ },
173
+ {
174
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
175
+ "document_id": "reap-recall-packed-axis1_general-1",
176
+ "domain": "axis1_general",
177
+ "main_terminal_offset": 106496,
178
+ "role": "fit",
179
+ "rows": 2047,
180
+ "token_ids_sha256": "83b96172a361a5f1a8eac603c4ef5b1e96023cdd07faf388479d0147a8c289cb",
181
+ "window_id": "fit-0052",
182
+ "window_index": 52
183
+ },
184
+ {
185
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
186
+ "document_id": "reap-recall-packed-axis4_reasoning_termination-1",
187
+ "domain": "axis4_reasoning_termination",
188
+ "main_terminal_offset": 112640,
189
+ "role": "fit",
190
+ "rows": 2047,
191
+ "token_ids_sha256": "7472e1d9f1fdeb708fb933d8bf51682e4d498f4cd96b001c86e166936da4c122",
192
+ "window_id": "fit-0055",
193
+ "window_index": 55
194
+ },
195
+ {
196
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
197
+ "document_id": "reap-recall-packed-axis4_reasoning_termination-1",
198
+ "domain": "axis4_reasoning_termination",
199
+ "main_terminal_offset": 120832,
200
+ "role": "fit",
201
+ "rows": 2047,
202
+ "token_ids_sha256": "6455be9da5b596f57b219d3a01f25cc7e38d83f09738b4a041f6ddb72d5a909c",
203
+ "window_id": "fit-0059",
204
+ "window_index": 59
205
+ },
206
+ {
207
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
208
+ "document_id": "reap-recall-packed-axis4_reasoning_termination-1",
209
+ "domain": "axis4_reasoning_termination",
210
+ "main_terminal_offset": 129024,
211
+ "role": "fit",
212
+ "rows": 2047,
213
+ "token_ids_sha256": "f838036e2eb0726a528ea1e5252e58d3340fd57f0a8f655bfb16eba68b74dc3b",
214
+ "window_id": "fit-0063",
215
+ "window_index": 63
216
+ },
217
+ {
218
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
219
+ "document_id": "reap-recall-packed-axis2_legal-1",
220
+ "domain": "axis2_legal",
221
+ "main_terminal_offset": 145408,
222
+ "role": "fit",
223
+ "rows": 2047,
224
+ "token_ids_sha256": "2b9498a4c264be4b3336e94d572e22fedb24fd745537ab519fe52616c7fa8369",
225
+ "window_id": "fit-0071",
226
+ "window_index": 71
227
+ },
228
+ {
229
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
230
+ "document_id": "reap-recall-packed-axis1_general-1",
231
+ "domain": "axis1_general",
232
+ "main_terminal_offset": 161792,
233
+ "role": "fit",
234
+ "rows": 2047,
235
+ "token_ids_sha256": "51ba8b2845d334157b51261bd6f7f8144bda32f0b53ec082356f9f3bffbca0f9",
236
+ "window_id": "fit-0079",
237
+ "window_index": 79
238
+ },
239
+ {
240
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
241
+ "document_id": "reap-recall-packed-axis3_code_agentic-4",
242
+ "domain": "axis3_code_agentic",
243
+ "main_terminal_offset": 190464,
244
+ "role": "fit",
245
+ "rows": 2047,
246
+ "token_ids_sha256": "7675e123eef3e202d31be6af29c815254ae601a85119eab63f124b476af9cba2",
247
+ "window_id": "fit-0093",
248
+ "window_index": 93
249
+ },
250
+ {
251
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
252
+ "document_id": "reap-recall-packed-axis1_general-1",
253
+ "domain": "axis1_general",
254
+ "main_terminal_offset": 235520,
255
+ "role": "fit",
256
+ "rows": 2047,
257
+ "token_ids_sha256": "52c57e4c50a2f724e6b0bb02ebeff3c850a27f0b3eb95db9b33dfd2ecc36fdc6",
258
+ "window_id": "fit-0115",
259
+ "window_index": 115
260
+ },
261
+ {
262
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
263
+ "document_id": "reap-recall-packed-axis2_legal-1",
264
+ "domain": "axis2_legal",
265
+ "main_terminal_offset": 237568,
266
+ "role": "fit",
267
+ "rows": 2047,
268
+ "token_ids_sha256": "fa0c3db2b83b1236825127432a325fd888cf8f5233f1b0f9aa88e3d6c46dd816",
269
+ "window_id": "fit-0116",
270
+ "window_index": 116
271
+ },
272
+ {
273
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
274
+ "document_id": "reap-recall-packed-axis2_legal-1",
275
+ "domain": "axis2_legal",
276
+ "main_terminal_offset": 249856,
277
+ "role": "fit",
278
+ "rows": 2047,
279
+ "token_ids_sha256": "1767d9eceb5d6d223d1a6222dd7e71fec1dd775d8eab9b8b56b39064e7e41957",
280
+ "window_id": "fit-0122",
281
+ "window_index": 122
282
+ },
283
+ {
284
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
285
+ "document_id": "reap-recall-packed-axis2_legal-1",
286
+ "domain": "axis2_legal",
287
+ "main_terminal_offset": 262144,
288
+ "role": "fit",
289
+ "rows": 2047,
290
+ "token_ids_sha256": "5af6ec5c7649361f07f5217411a5ac85b9c5b4b9740967c0b8d99aba570866fc",
291
+ "window_id": "fit-0128",
292
+ "window_index": 128
293
+ },
294
+ {
295
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
296
+ "document_id": "reap-recall-packed-axis2_legal-1",
297
+ "domain": "axis2_legal",
298
+ "main_terminal_offset": 268288,
299
+ "role": "fit",
300
+ "rows": 2047,
301
+ "token_ids_sha256": "5fb1a5379d4fe994a1e86a2e1cf44aa56c52b7e02ad43a05df7c40bf43418e4f",
302
+ "window_id": "fit-0131",
303
+ "window_index": 131
304
+ },
305
+ {
306
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
307
+ "document_id": "reap-recall-packed-axis2_legal-1",
308
+ "domain": "axis2_legal",
309
+ "main_terminal_offset": 274432,
310
+ "role": "fit",
311
+ "rows": 2047,
312
+ "token_ids_sha256": "284ff8b2bdae3627c57a9da5b3e97bf54068681d41c65b9e27659aef73763581",
313
+ "window_id": "fit-0134",
314
+ "window_index": 134
315
+ },
316
+ {
317
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
318
+ "document_id": "reap-recall-packed-axis3_code_agentic-4",
319
+ "domain": "axis3_code_agentic",
320
+ "main_terminal_offset": 282624,
321
+ "role": "fit",
322
+ "rows": 2047,
323
+ "token_ids_sha256": "f8c769d67636da9dfca5d73e57292a34830cfeb60da4b67cba5f2df4b083cbe9",
324
+ "window_id": "fit-0138",
325
+ "window_index": 138
326
+ },
327
+ {
328
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
329
+ "document_id": "reap-recall-packed-axis1_general-1",
330
+ "domain": "axis1_general",
331
+ "main_terminal_offset": 284672,
332
+ "role": "fit",
333
+ "rows": 2047,
334
+ "token_ids_sha256": "639984e8d3351390d01d2f934cbc7d4635a2b281692326080ca0bee7fa82e365",
335
+ "window_id": "fit-0139",
336
+ "window_index": 139
337
+ },
338
+ {
339
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
340
+ "document_id": "reap-recall-packed-axis2_legal-1",
341
+ "domain": "axis2_legal",
342
+ "main_terminal_offset": 292864,
343
+ "role": "fit",
344
+ "rows": 2047,
345
+ "token_ids_sha256": "d98dc54bc9af44ae5f06bffa644c90d68ded7f231ee86daef18228b9104fd018",
346
+ "window_id": "fit-0143",
347
+ "window_index": 143
348
+ },
349
+ {
350
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
351
+ "document_id": "reap-recall-packed-axis3_code_agentic-4",
352
+ "domain": "axis3_code_agentic",
353
+ "main_terminal_offset": 301056,
354
+ "role": "fit",
355
+ "rows": 2047,
356
+ "token_ids_sha256": "a14a85dbd0f53c19dba5c9f952fcecc0bb9ab2ee419b3138278878fbbc46bede",
357
+ "window_id": "fit-0147",
358
+ "window_index": 147
359
+ },
360
+ {
361
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
362
+ "document_id": "reap-recall-packed-axis2_legal-1",
363
+ "domain": "axis2_legal",
364
+ "main_terminal_offset": 317440,
365
+ "role": "fit",
366
+ "rows": 2047,
367
+ "token_ids_sha256": "f18685915b8f8b2c8dbc7c411471d5c6e56bfdd9c3158241ad9cc7adf9bf1a85",
368
+ "window_id": "fit-0155",
369
+ "window_index": 155
370
+ },
371
+ {
372
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
373
+ "document_id": "reap-recall-packed-axis3_code_agentic-4",
374
+ "domain": "axis3_code_agentic",
375
+ "main_terminal_offset": 319488,
376
+ "role": "fit",
377
+ "rows": 2047,
378
+ "token_ids_sha256": "5e9bf81a13d603aeee09856b17cd8085475c8252af6f4376565886cf7c8c824c",
379
+ "window_id": "fit-0156",
380
+ "window_index": 156
381
+ },
382
+ {
383
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
384
+ "document_id": "reap-recall-packed-axis1_general-1",
385
+ "domain": "axis1_general",
386
+ "main_terminal_offset": 333824,
387
+ "role": "fit",
388
+ "rows": 2047,
389
+ "token_ids_sha256": "d3274fddf9feaff8a8460e7cb8dcec6de21d7ee34d8df54c8a14784995bc93ff",
390
+ "window_id": "fit-0163",
391
+ "window_index": 163
392
+ },
393
+ {
394
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
395
+ "document_id": "reap-recall-packed-axis1_general-1",
396
+ "domain": "axis1_general",
397
+ "main_terminal_offset": 352256,
398
+ "role": "fit",
399
+ "rows": 2047,
400
+ "token_ids_sha256": "2f02a7a1854c920b37b555302659fb206af3227ad18c97f98f9d71487cf140f3",
401
+ "window_id": "fit-0172",
402
+ "window_index": 172
403
+ },
404
+ {
405
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
406
+ "document_id": "reap-recall-packed-axis2_legal-1",
407
+ "domain": "axis2_legal",
408
+ "main_terminal_offset": 354304,
409
+ "role": "fit",
410
+ "rows": 2047,
411
+ "token_ids_sha256": "b90385699a4e5ad8511be2f979d8b1eefb4f51f15009d0a9cfea504823f7e804",
412
+ "window_id": "fit-0173",
413
+ "window_index": 173
414
+ },
415
+ {
416
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
417
+ "document_id": "reap-recall-packed-axis3_code_agentic-4",
418
+ "domain": "axis3_code_agentic",
419
+ "main_terminal_offset": 399360,
420
+ "role": "fit",
421
+ "rows": 2047,
422
+ "token_ids_sha256": "e94b7adcc6c9b3ea370d72b3a93e0e587f3f15d5efa7a51dc5dbb851ef008d0b",
423
+ "window_id": "fit-0195",
424
+ "window_index": 195
425
+ },
426
+ {
427
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
428
+ "document_id": "reap-recall-packed-axis2_legal-1",
429
+ "domain": "axis2_legal",
430
+ "main_terminal_offset": 403456,
431
+ "role": "fit",
432
+ "rows": 2047,
433
+ "token_ids_sha256": "34364d64711a7325d1c600a90cb67e75cab6e57a5659cee1f186bdad4f92e9d3",
434
+ "window_id": "fit-0197",
435
+ "window_index": 197
436
+ },
437
+ {
438
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
439
+ "document_id": "reap-recall-packed-axis1_general-1",
440
+ "domain": "axis1_general",
441
+ "main_terminal_offset": 413696,
442
+ "role": "fit",
443
+ "rows": 2047,
444
+ "token_ids_sha256": "11bc2a5757f9ef787e1f21ef06fc6d13c6a174a12486dc8fbb43194775120123",
445
+ "window_id": "fit-0202",
446
+ "window_index": 202
447
+ },
448
+ {
449
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
450
+ "document_id": "reap-recall-packed-axis2_legal-1",
451
+ "domain": "axis2_legal",
452
+ "main_terminal_offset": 421888,
453
+ "role": "fit",
454
+ "rows": 2047,
455
+ "token_ids_sha256": "dd9e2596cf1f732177b22cc3abbc8d14df675f7478cb7273b8b02522516d8f59",
456
+ "window_id": "fit-0206",
457
+ "window_index": 206
458
+ },
459
+ {
460
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
461
+ "document_id": "reap-recall-packed-axis3_code_agentic-4",
462
+ "domain": "axis3_code_agentic",
463
+ "main_terminal_offset": 430080,
464
+ "role": "fit",
465
+ "rows": 2047,
466
+ "token_ids_sha256": "d7508373c2c195b1dff5dc0b35fb171e00348cd735296383336fa461cd678bdc",
467
+ "window_id": "fit-0210",
468
+ "window_index": 210
469
+ },
470
+ {
471
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
472
+ "document_id": "reap-recall-packed-axis2_legal-1",
473
+ "domain": "axis2_legal",
474
+ "main_terminal_offset": 440320,
475
+ "role": "fit",
476
+ "rows": 2047,
477
+ "token_ids_sha256": "81fc07989fe68186d98c90b7807d07520d83c8070821307ad2d69c9d1d90b120",
478
+ "window_id": "fit-0215",
479
+ "window_index": 215
480
+ },
481
+ {
482
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
483
+ "document_id": "reap-recall-packed-axis2_legal-1",
484
+ "domain": "axis2_legal",
485
+ "main_terminal_offset": 458752,
486
+ "role": "fit",
487
+ "rows": 2047,
488
+ "token_ids_sha256": "95c807d7a5aad5ea11c2e22591930669aaca061412d41754cdde15a6a4546900",
489
+ "window_id": "fit-0224",
490
+ "window_index": 224
491
+ },
492
+ {
493
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
494
+ "document_id": "reap-recall-packed-axis2_legal-1",
495
+ "domain": "axis2_legal",
496
+ "main_terminal_offset": 464896,
497
+ "role": "fit",
498
+ "rows": 2047,
499
+ "token_ids_sha256": "4a2158f357933bf460001d22ee4b6fcdd3609f03e0b417d8e04c2ce049dadf51",
500
+ "window_id": "fit-0227",
501
+ "window_index": 227
502
+ },
503
+ {
504
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
505
+ "document_id": "reap-recall-packed-axis3_code_agentic-4",
506
+ "domain": "axis3_code_agentic",
507
+ "main_terminal_offset": 466944,
508
+ "role": "fit",
509
+ "rows": 2047,
510
+ "token_ids_sha256": "1ef5b044392253389dd12a22c1978e706ab6584dbf301d6253d1099a742bbefe",
511
+ "window_id": "fit-0228",
512
+ "window_index": 228
513
+ },
514
+ {
515
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
516
+ "document_id": "reap-recall-packed-axis3_code_agentic-4",
517
+ "domain": "axis3_code_agentic",
518
+ "main_terminal_offset": 485376,
519
+ "role": "fit",
520
+ "rows": 2047,
521
+ "token_ids_sha256": "cb88cfc15311ce056f24e24a627de2fbc91f1eea2ced97785029d5ef29d1e02e",
522
+ "window_id": "fit-0237",
523
+ "window_index": 237
524
+ },
525
+ {
526
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
527
+ "document_id": "reap-recall-packed-axis1_general-1",
528
+ "domain": "axis1_general",
529
+ "main_terminal_offset": 505856,
530
+ "role": "fit",
531
+ "rows": 2047,
532
+ "token_ids_sha256": "a2a532c83873a63fbd1ecd3dc072059389886f2eb24fbe14ecf2f36f7e016f87",
533
+ "window_id": "fit-0247",
534
+ "window_index": 247
535
+ },
536
+ {
537
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
538
+ "document_id": "reap-recall-packed-axis3_code_agentic-4",
539
+ "domain": "axis3_code_agentic",
540
+ "main_terminal_offset": 534528,
541
+ "role": "fit",
542
+ "rows": 2047,
543
+ "token_ids_sha256": "17037b1512bacfd160cff6d58a099ae398cbeed5bc171ab1c8e0f0af8933b19e",
544
+ "window_id": "fit-0261",
545
+ "window_index": 261
546
+ },
547
+ {
548
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
549
+ "document_id": "reap-recall-packed-axis1_general-1",
550
+ "domain": "axis1_general",
551
+ "main_terminal_offset": 542720,
552
+ "role": "fit",
553
+ "rows": 2047,
554
+ "token_ids_sha256": "c64237273b209a945c7c818ebd1d422cb2ab50e4ebd7ef4b8401b242a9029e7b",
555
+ "window_id": "fit-0265",
556
+ "window_index": 265
557
+ },
558
+ {
559
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
560
+ "document_id": "reap-recall-packed-axis2_legal-1",
561
+ "domain": "axis2_legal",
562
+ "main_terminal_offset": 563200,
563
+ "role": "fit",
564
+ "rows": 2047,
565
+ "token_ids_sha256": "1e994c189fbd3d369abf496db5ff48750f71b252f645b09462230218985abdc2",
566
+ "window_id": "fit-0275",
567
+ "window_index": 275
568
+ },
569
+ {
570
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
571
+ "document_id": "reap-recall-packed-axis3_code_agentic-4",
572
+ "domain": "axis3_code_agentic",
573
+ "main_terminal_offset": 577536,
574
+ "role": "fit",
575
+ "rows": 2047,
576
+ "token_ids_sha256": "a3bcd38c2780d32246f3b8b630bad422b4dd444572d648ca21e720cbd4c28128",
577
+ "window_id": "fit-0282",
578
+ "window_index": 282
579
+ },
580
+ {
581
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
582
+ "document_id": "reap-recall-packed-axis1_general-1",
583
+ "domain": "axis1_general",
584
+ "main_terminal_offset": 616448,
585
+ "role": "fit",
586
+ "rows": 2047,
587
+ "token_ids_sha256": "a8fe54b5fc7f4fa5936b330d93d76da5dfdacd9712ef7a6f477600e910a2125b",
588
+ "window_id": "fit-0301",
589
+ "window_index": 301
590
+ },
591
+ {
592
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
593
+ "document_id": "reap-recall-packed-axis1_general-1",
594
+ "domain": "axis1_general",
595
+ "main_terminal_offset": 647168,
596
+ "role": "fit",
597
+ "rows": 2047,
598
+ "token_ids_sha256": "0060f819a176ec0abfa02ad323c9015afd7f5d6d6e06c99d819f16cb8f7a8a25",
599
+ "window_id": "fit-0316",
600
+ "window_index": 316
601
+ },
602
+ {
603
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
604
+ "document_id": "reap-recall-packed-axis2_legal-1",
605
+ "domain": "axis2_legal",
606
+ "main_terminal_offset": 649216,
607
+ "role": "fit",
608
+ "rows": 2047,
609
+ "token_ids_sha256": "f728ace6743429f911cdec7051a6dff7db0b0f5f8f915b42767fac4c95f8d849",
610
+ "window_id": "fit-0317",
611
+ "window_index": 317
612
+ },
613
+ {
614
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
615
+ "document_id": "reap-recall-packed-axis3_code_agentic-4",
616
+ "domain": "axis3_code_agentic",
617
+ "main_terminal_offset": 677888,
618
+ "role": "fit",
619
+ "rows": 2047,
620
+ "token_ids_sha256": "429eb9411d4ec92a26abf72cf302ef72d94f2df9c9011be2e9a91fcaaec7bece",
621
+ "window_id": "fit-0331",
622
+ "window_index": 331
623
+ },
624
+ {
625
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
626
+ "document_id": "reap-recall-packed-axis1_general-1",
627
+ "domain": "axis1_general",
628
+ "main_terminal_offset": 708608,
629
+ "role": "fit",
630
+ "rows": 2047,
631
+ "token_ids_sha256": "5655a824cde5dbc1600eee1f27b4b7cd0494dd1826139e93966984e62b46d789",
632
+ "window_id": "fit-0346",
633
+ "window_index": 346
634
+ },
635
+ {
636
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
637
+ "document_id": "reap-recall-packed-axis1_general-1",
638
+ "domain": "axis1_general",
639
+ "main_terminal_offset": 712704,
640
+ "role": "fit",
641
+ "rows": 2047,
642
+ "token_ids_sha256": "49baabdd793585f57ed18f17f025b09e16bd8eb26a1ab826e2072bb140b80026",
643
+ "window_id": "fit-0348",
644
+ "window_index": 348
645
+ },
646
+ {
647
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
648
+ "document_id": "reap-recall-packed-axis1_general-1",
649
+ "domain": "axis1_general",
650
+ "main_terminal_offset": 737280,
651
+ "role": "fit",
652
+ "rows": 2047,
653
+ "token_ids_sha256": "023afe71006b811629b43bb3d6d3044d120726ed525723961d5ff083ff4c32bc",
654
+ "window_id": "fit-0360",
655
+ "window_index": 360
656
+ },
657
+ {
658
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
659
+ "document_id": "reap-recall-packed-axis3_code_agentic-4",
660
+ "domain": "axis3_code_agentic",
661
+ "main_terminal_offset": 739328,
662
+ "role": "fit",
663
+ "rows": 2047,
664
+ "token_ids_sha256": "44e340f3fcb5a830cbccc7683a48b4691f3709bbaa9bff7fab2c34fdf03c3cf1",
665
+ "window_id": "fit-0361",
666
+ "window_index": 361
667
+ },
668
+ {
669
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
670
+ "document_id": "reap-recall-packed-axis3_code_agentic-4",
671
+ "domain": "axis3_code_agentic",
672
+ "main_terminal_offset": 747520,
673
+ "role": "fit",
674
+ "rows": 2047,
675
+ "token_ids_sha256": "eb638a132cf39bd4e4d14d6a976a1e0d47646ca1da63218c95cdcd66f24ddd9b",
676
+ "window_id": "fit-0365",
677
+ "window_index": 365
678
+ },
679
+ {
680
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
681
+ "document_id": "reap-recall-packed-axis3_code_agentic-4",
682
+ "domain": "axis3_code_agentic",
683
+ "main_terminal_offset": 751616,
684
+ "role": "fit",
685
+ "rows": 2047,
686
+ "token_ids_sha256": "59f18d04f36791b4dbb1a158692a0c6b2b53bf2c7b9ae63a51c3183270688225",
687
+ "window_id": "fit-0367",
688
+ "window_index": 367
689
+ },
690
+ {
691
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
692
+ "document_id": "reap-recall-packed-axis1_general-1",
693
+ "domain": "axis1_general",
694
+ "main_terminal_offset": 765952,
695
+ "role": "fit",
696
+ "rows": 2047,
697
+ "token_ids_sha256": "ea0b1e6680826d54e456803cae763837712748fc73f5477bb414ca708fea55aa",
698
+ "window_id": "fit-0374",
699
+ "window_index": 374
700
+ },
701
+ {
702
+ "attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
703
+ "document_id": "reap-recall-packed-axis3_code_agentic-4",
704
+ "domain": "axis3_code_agentic",
705
+ "main_terminal_offset": 772096,
706
+ "role": "fit",
707
+ "rows": 2047,
708
+ "token_ids_sha256": "75ff1700624bc9c4e5251f3e8afe5b3fc6293199739cceaaca79de0c1ac6b175",
709
+ "window_id": "fit-0377",
710
+ "window_index": 377
711
+ }
712
+ ],
713
+ "rows_per_window": 2047,
714
+ "rows": 131008,
715
+ "scale_policy": "signed-unit coupled vectors; CPU torch seed 530045; no prior r27 MTP scales",
716
+ "scale_sha256": "ad170baa8259a6af3f234f154c0b9cca188b8d1f7cf6de0c89788bccbf66827b"
717
+ }
evidence/native-carrier-mtp-receipt.json ADDED
@@ -0,0 +1,112 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "repo": "local-inference-lab/GLM-5.3-Flash-NVFP4",
3
+ "revision": "520de24eabf507659eaef7c70f14fd584527facc",
4
+ "source_verification": "PASS",
5
+ "index_header_coverage": "PASS",
6
+ "indexed_tensors": 37906,
7
+ "omitted_replaced_routed_tensors": 108864,
8
+ "bytes": 19388363261,
9
+ "files": [
10
+ {
11
+ "path": "LICENSE",
12
+ "bytes": 1070,
13
+ "sha256": "30b85b6b9659f2e78aa259f8faf5d920a68dee7c9ced3fa6dba1f19f2bc4fca1"
14
+ },
15
+ {
16
+ "path": "README.md",
17
+ "bytes": 7243,
18
+ "sha256": "3f0894b80aefb75c2afbaa0d4fb0f0f0993f14cb9700c6a5077ea6c201637b7b"
19
+ },
20
+ {
21
+ "path": "amax.safetensors",
22
+ "bytes": 9779904,
23
+ "sha256": "db629a0e7cc2b77ea8a4ec560d326ae94c2af561f4fdbd39d8f611818da43d2b"
24
+ },
25
+ {
26
+ "path": "amax_checkpoint.json",
27
+ "bytes": 1137,
28
+ "sha256": "e2ab014bc3354ad679c4057c5890e786a45dd6517b61d049a652ac76ec9cba6c"
29
+ },
30
+ {
31
+ "path": "amax_checkpoint.safetensors",
32
+ "bytes": 9780048,
33
+ "sha256": "d3ef0aa55826c7f2a29c65a1d8b32368a26a527cf36f27c1665bd0bec5d34fa1"
34
+ },
35
+ {
36
+ "path": "chat_template.jinja",
37
+ "bytes": 10644,
38
+ "sha256": "34d5ee66b12fa6446cdae131c352b8f68cd85369e0e6fda115583805fada3891"
39
+ },
40
+ {
41
+ "path": "config.json",
42
+ "bytes": 15761,
43
+ "sha256": "676382abd1e90a6c85f0c8f33d45441ecd45fd514fd7b63ce5610e732d8e4996"
44
+ },
45
+ {
46
+ "path": "generation_config.json",
47
+ "bytes": 2233,
48
+ "sha256": "80475d7d0e56cf2729f1e0e60bfec01bda743f3829bc5c42e14f54fb6d52e612"
49
+ },
50
+ {
51
+ "path": "hf_quant_config.json",
52
+ "bytes": 8052,
53
+ "sha256": "9c084477c9fe496929a15dde8fb795d13b143e32911e613fe7209121b5606ee9"
54
+ },
55
+ {
56
+ "path": "model-hf-nonexpert-00001-of-00004.safetensors",
57
+ "bytes": 5356357736,
58
+ "sha256": "9f7ede71b2213a6962ea3801015e84edd185ddf00e1b8a0aa9ea6288c28014bf"
59
+ },
60
+ {
61
+ "path": "model-hf-nonexpert-00002-of-00004.safetensors",
62
+ "bytes": 5318341144,
63
+ "sha256": "b26a6aeebb246e95fcb5d54f0b6f8dd66fdfeb17fe54652ae835af1b7d64f6ef"
64
+ },
65
+ {
66
+ "path": "model-hf-nonexpert-00003-of-00004.safetensors",
67
+ "bytes": 5343414728,
68
+ "sha256": "0d56d129e4bd3d732968e7c5dbf3fc11e3e23566619842832325c44306cc68a5"
69
+ },
70
+ {
71
+ "path": "model-hf-nonexpert-00004-of-00004.safetensors",
72
+ "bytes": 2951938800,
73
+ "sha256": "43f38b4fe13a7002a5cf2841083caaac5baa65ebf72e7fb4d9c58bf9bff0280a"
74
+ },
75
+ {
76
+ "path": "model-inputscales.safetensors",
77
+ "bytes": 4726664,
78
+ "sha256": "4255779f031450572af8548c610fd9abfe7df89704985d18595c21788593cd05"
79
+ },
80
+ {
81
+ "path": "model-mtp-nonexpert.safetensors",
82
+ "bytes": 369674104,
83
+ "sha256": "31fc8804fce4cb1e0207d9fb13c19a892d85b50a7f769e763cb8d9e5b6a2e6ba"
84
+ },
85
+ {
86
+ "path": "model.safetensors.index.json",
87
+ "bytes": 4084881,
88
+ "sha256": "7edc92ec0cf197039a40235bd4f4906c6ecad85db7295d51479f51d859df310d"
89
+ },
90
+ {
91
+ "path": "processor_config.json",
92
+ "bytes": 909,
93
+ "sha256": "aae38374c94b08cc9b0547c6e64f05b951bd9735cea571c6988f5ed552bed3ed"
94
+ },
95
+ {
96
+ "path": "tokenizer.json",
97
+ "bytes": 20217442,
98
+ "sha256": "19e773648cb4e65de8660ea6365e10acca112d42a854923df93db4a6f333a82d"
99
+ },
100
+ {
101
+ "path": "tokenizer_config.json",
102
+ "bytes": 761,
103
+ "sha256": "98b1271574f41abf89427ae2dda030d94dc9478f0edc5a8bd240db213c6fd5fc"
104
+ }
105
+ ],
106
+ "layout": "carrier/ under checkpoint root; MODEL_ROOT points to carrier/",
107
+ "native_execution_tested": false,
108
+ "mtp_expert_tensors_replaced": 1728,
109
+ "mtp_nonexpert_tensors_preserved": 25,
110
+ "mtp_nonexpert_tensor_equality": "PASS",
111
+ "derivation": "MTP45 FP8 experts and scales replaced by Trellis sidecars; other tensors unchanged"
112
+ }
evidence/prepare-mtp-inputs.py ADDED
@@ -0,0 +1,73 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Prepare the existing BF16 MTP captures for the owner's new Trellis MTP arm."""
2
+ import hashlib
3
+ import json
4
+ import os
5
+ from pathlib import Path
6
+
7
+ import numpy as np
8
+ import torch
9
+ from huggingface_hub import snapshot_download
10
+ from safetensors.torch import save_file
11
+
12
+ root = Path(os.environ["CV2_ROOT"])
13
+ out = Path(os.environ["CV2_OUT"])
14
+ dest = out / "mtp-inputs"
15
+ dest.mkdir(exist_ok=True)
16
+ repo = "brandonmusic/GLM-5.3-Flash-BF16-Teacher-Logits"
17
+ rev = "95f4fdd94bf29989db2e0d1054e4931f55edb6aa"
18
+ snapshot_download(repo, repo_type="dataset", revision=rev, local_dir=dest / "source",
19
+ allow_patterns=["calibration/mtp45-ep4-full/*"], max_workers=4)
20
+ source = dest / "source/calibration/mtp45-ep4-full"
21
+ manifest = json.loads((source / "capture-manifest.json").read_text())
22
+ receipt = json.loads((source / "capture-receipt.json").read_text())
23
+ assert receipt["complete"] is True
24
+ assert hashlib.sha256((source / "capture-manifest.json").read_bytes()).hexdigest() == receipt["capture_manifest_file_sha256"]
25
+ assert manifest["layer"] == 45 and manifest["geometry"]["experts"] == 288
26
+ assert manifest["model_revision"] == "a6c167b62691b2bac901344b65cb651a70f53e43"
27
+ for info in manifest["files"].values():
28
+ path = source / info["path"]
29
+ assert path.stat().st_size == info["bytes"]
30
+ with path.open("rb") as stream:
31
+ assert hashlib.file_digest(stream, "sha256").hexdigest() == info["sha256"]
32
+ roles = json.loads((root / "data/roles-v5.json").read_text())["roles"]["fit"]
33
+ assert len(roles) == 64
34
+ role_map = {r["id"]: r for r in roles}
35
+ windows = [w for w in manifest["windows"] if w["window_id"] in role_map]
36
+ assert len(windows) == 64
37
+ offset = 0
38
+ offsets = {}
39
+ for window in manifest["windows"]:
40
+ offsets[window["window_id"]] = offset
41
+ offset += window["rows"]
42
+ assert offset == manifest["rows"]
43
+ for window in windows:
44
+ assert window["role"] == "fit" and window["rows"] == 2047
45
+ assert window["token_ids_sha256"] == role_map[window["window_id"]]["input_sha256"]
46
+ streams = [("hidden_bf16", "<u2", 4096, "hidden.bf16.bin"),
47
+ ("topk_ids_u16le", "<u2", 8, "topk_ids.u16le.bin"),
48
+ ("topk_weights_f32le", "<f4", 8, "topk_weights.f32le.bin")]
49
+ for key, dtype, width, name in streams:
50
+ array = np.memmap(source / manifest["files"][key]["path"], mode="r", dtype=dtype,
51
+ shape=(manifest["rows"], width))
52
+ with (dest / name).open("wb") as stream:
53
+ for window in windows:
54
+ start = offsets[window["window_id"]]
55
+ stream.write(np.asarray(array[start:start + window["rows"]]).tobytes())
56
+ # No prior Trellis/EXL3 MTP scale vectors exist in the supplied r27 overlay.
57
+ # New signed-unit vectors provide an orthogonal coupled H512/H128 transform;
58
+ # the unchanged codec-v2 block-scale refit handles weight magnitudes.
59
+ generator = torch.Generator(device="cpu").manual_seed(530045)
60
+ def signs(shape):
61
+ return torch.randint(0, 2, shape, generator=generator).mul_(2).sub_(1).to(torch.float16).contiguous()
62
+ scales = {"gate_up_suh": signs((4096,)), "down_svh": signs((4096,)),
63
+ "gate_svh": signs((288, 2048)), "up_svh": signs((288, 2048)),
64
+ "down_suh": signs((288, 2048))}
65
+ save_file(scales, str(dest / "coupled-scales.safetensors"))
66
+ info = {"schema": "trellismx-codecv2-mtp-inputs.v1", "layer": 45,
67
+ "capture_repo": repo, "capture_revision": rev, "source_capture_sha256": manifest["capture_sha256"],
68
+ "windows": windows, "rows_per_window": 2047, "rows": 64 * 2047,
69
+ "scale_policy": "signed-unit coupled vectors; CPU torch seed 530045; no prior r27 MTP scales",
70
+ "scale_sha256": hashlib.sha256((dest / "coupled-scales.safetensors").read_bytes()).hexdigest()}
71
+ (dest / "inputs.json").write_text(json.dumps(info, indent=2) + "\n")
72
+ (out / "status/mtp-inputs.ready").write_text(info["scale_sha256"] + "\n")
73
+ print(json.dumps({k: v for k, v in info.items() if k != "windows"}), flush=True)
runtime/Dockerfile ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ FROM verdictai/trellismx@sha256:a0392e1c370eb933d87511928ea3aba6d5d63965cfe6eeaf9cc026a667e98ddc
2
+ COPY patch-manifest.json install-mtp-runtime.py /opt/codecv2-mtp/
3
+ COPY patches /opt/codecv2-mtp/patches
4
+ RUN python3 /opt/codecv2-mtp/install-mtp-runtime.py
5
+ ENV MODEL_ROOT=/model/carrier VLLM_TRELLISMX_CHECKPOINT=/model
6
+ ENTRYPOINT ["/bin/bash", "/release/serve-rp2.sh"]
runtime/LICENSE-APACHE-2.0 ADDED
@@ -0,0 +1,201 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Apache License
2
+ Version 2.0, January 2004
3
+ http://www.apache.org/licenses/
4
+
5
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
6
+
7
+ 1. Definitions.
8
+
9
+ "License" shall mean the terms and conditions for use, reproduction,
10
+ and distribution as defined by Sections 1 through 9 of this document.
11
+
12
+ "Licensor" shall mean the copyright owner or entity authorized by
13
+ the copyright owner that is granting the License.
14
+
15
+ "Legal Entity" shall mean the union of the acting entity and all
16
+ other entities that control, are controlled by, or are under common
17
+ control with that entity. For the purposes of this definition,
18
+ "control" means (i) the power, direct or indirect, to cause the
19
+ direction or management of such entity, whether by contract or
20
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
21
+ outstanding shares, or (iii) beneficial ownership of such entity.
22
+
23
+ "You" (or "Your") shall mean an individual or Legal Entity
24
+ exercising permissions granted by this License.
25
+
26
+ "Source" form shall mean the preferred form for making modifications,
27
+ including but not limited to software source code, documentation
28
+ source, and configuration files.
29
+
30
+ "Object" form shall mean any form resulting from mechanical
31
+ transformation or translation of a Source form, including but
32
+ not limited to compiled object code, generated documentation,
33
+ and conversions to other media types.
34
+
35
+ "Work" shall mean the work of authorship, whether in Source or
36
+ Object form, made available under the License, as indicated by a
37
+ copyright notice that is included in or attached to the work
38
+ (an example is provided in the Appendix below).
39
+
40
+ "Derivative Works" shall mean any work, whether in Source or Object
41
+ form, that is based on (or derived from) the Work and for which the
42
+ editorial revisions, annotations, elaborations, or other modifications
43
+ represent, as a whole, an original work of authorship. For the purposes
44
+ of this License, Derivative Works shall not include works that remain
45
+ separable from, or merely link (or bind by name) to the interfaces of,
46
+ the Work and Derivative Works thereof.
47
+
48
+ "Contribution" shall mean any work of authorship, including
49
+ the original version of the Work and any modifications or additions
50
+ to that Work or Derivative Works thereof, that is intentionally
51
+ submitted to Licensor for inclusion in the Work by the copyright owner
52
+ or by an individual or Legal Entity authorized to submit on behalf of
53
+ the copyright owner. For the purposes of this definition, "submitted"
54
+ means any form of electronic, verbal, or written communication sent
55
+ to the Licensor or its representatives, including but not limited to
56
+ communication on electronic mailing lists, source code control systems,
57
+ and issue tracking systems that are managed by, or on behalf of, the
58
+ Licensor for the purpose of discussing and improving the Work, but
59
+ excluding communication that is conspicuously marked or otherwise
60
+ designated in writing by the copyright owner as "Not a Contribution."
61
+
62
+ "Contributor" shall mean Licensor and any individual or Legal Entity
63
+ on behalf of whom a Contribution has been received by Licensor and
64
+ subsequently incorporated within the Work.
65
+
66
+ 2. Grant of Copyright License. Subject to the terms and conditions of
67
+ this License, each Contributor hereby grants to You a perpetual,
68
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
69
+ copyright license to reproduce, prepare Derivative Works of,
70
+ publicly display, publicly perform, sublicense, and distribute the
71
+ Work and such Derivative Works in Source or Object form.
72
+
73
+ 3. Grant of Patent License. Subject to the terms and conditions of
74
+ this License, each Contributor hereby grants to You a perpetual,
75
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
76
+ (except as stated in this section) patent license to make, have made,
77
+ use, offer to sell, sell, import, and otherwise transfer the Work,
78
+ where such license applies only to those patent claims licensable
79
+ by such Contributor that are necessarily infringed by their
80
+ Contribution(s) alone or by combination of their Contribution(s)
81
+ with the Work to which such Contribution(s) was submitted. If You
82
+ institute patent litigation against any entity (including a
83
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
84
+ or a Contribution incorporated within the Work constitutes direct
85
+ or contributory patent infringement, then any patent licenses
86
+ granted to You under this License for that Work shall terminate
87
+ as of the date such litigation is filed.
88
+
89
+ 4. Redistribution. You may reproduce and distribute copies of the
90
+ Work or Derivative Works thereof in any medium, with or without
91
+ modifications, and in Source or Object form, provided that You
92
+ meet the following conditions:
93
+
94
+ (a) You must give any other recipients of the Work or
95
+ Derivative Works a copy of this License; and
96
+
97
+ (b) You must cause any modified files to carry prominent notices
98
+ stating that You changed the files; and
99
+
100
+ (c) You must retain, in the Source form of any Derivative Works
101
+ that You distribute, all copyright, patent, trademark, and
102
+ attribution notices from the Source form of the Work,
103
+ excluding those notices that do not pertain to any part of
104
+ the Derivative Works; and
105
+
106
+ (d) If the Work includes a "NOTICE" text file as part of its
107
+ distribution, then any Derivative Works that You distribute must
108
+ include a readable copy of the attribution notices contained
109
+ within such NOTICE file, excluding those notices that do not
110
+ pertain to any part of the Derivative Works, in at least one
111
+ of the following places: within a NOTICE text file distributed
112
+ as part of the Derivative Works; within the Source form or
113
+ documentation, if provided along with the Derivative Works; or,
114
+ within a display generated by the Derivative Works, if and
115
+ wherever such third-party notices normally appear. The contents
116
+ of the NOTICE file are for informational purposes only and
117
+ do not modify the License. You may add Your own attribution
118
+ notices within Derivative Works that You distribute, alongside
119
+ or as an addendum to the NOTICE text from the Work, provided
120
+ that such additional attribution notices cannot be construed
121
+ as modifying the License.
122
+
123
+ You may add Your own copyright statement to Your modifications and
124
+ may provide additional or different license terms and conditions
125
+ for use, reproduction, or distribution of Your modifications, or
126
+ for any such Derivative Works as a whole, provided Your use,
127
+ reproduction, and distribution of the Work otherwise complies with
128
+ the conditions stated in this License.
129
+
130
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
131
+ any Contribution intentionally submitted for inclusion in the Work
132
+ by You to the Licensor shall be under the terms and conditions of
133
+ this License, without any additional terms or conditions.
134
+ Notwithstanding the above, nothing herein shall supersede or modify
135
+ the terms of any separate license agreement you may have executed
136
+ with Licensor regarding such Contributions.
137
+
138
+ 6. Trademarks. This License does not grant permission to use the trade
139
+ names, trademarks, service marks, or product names of the Licensor,
140
+ except as required for reasonable and customary use in describing the
141
+ origin of the Work and reproducing the content of the NOTICE file.
142
+
143
+ 7. Disclaimer of Warranty. Unless required by applicable law or
144
+ agreed to in writing, Licensor provides the Work (and each
145
+ Contributor provides its Contributions) on an "AS IS" BASIS,
146
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
147
+ implied, including, without limitation, any warranties or conditions
148
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
149
+ PARTICULAR PURPOSE. You are solely responsible for determining the
150
+ appropriateness of using or redistributing the Work and assume any
151
+ risks associated with Your exercise of permissions under this License.
152
+
153
+ 8. Limitation of Liability. In no event and under no legal theory,
154
+ whether in tort (including negligence), contract, or otherwise,
155
+ unless required by applicable law (such as deliberate and grossly
156
+ negligent acts) or agreed to in writing, shall any Contributor be
157
+ liable to You for damages, including any direct, indirect, special,
158
+ incidental, or consequential damages of any character arising as a
159
+ result of this License or out of the use or inability to use the
160
+ Work (including but not limited to damages for loss of goodwill,
161
+ work stoppage, computer failure or malfunction, or any and all
162
+ other commercial damages or losses), even if such Contributor
163
+ has been advised of the possibility of such damages.
164
+
165
+ 9. Accepting Warranty or Additional Liability. While redistributing
166
+ the Work or Derivative Works thereof, You may choose to offer,
167
+ and charge a fee for, acceptance of support, warranty, indemnity,
168
+ or other liability obligations and/or rights consistent with this
169
+ License. However, in accepting such obligations, You may act only
170
+ on Your own behalf and on Your sole responsibility, not on behalf
171
+ of any other Contributor, and only if You agree to indemnify,
172
+ defend, and hold each Contributor harmless for any liability
173
+ incurred by, or claims asserted against, such Contributor by reason
174
+ of your accepting any such warranty or additional liability.
175
+
176
+ END OF TERMS AND CONDITIONS
177
+
178
+ APPENDIX: How to apply the Apache License to your work.
179
+
180
+ To apply the Apache License to your work, attach the following
181
+ boilerplate notice, with the fields enclosed by brackets "[]"
182
+ replaced with your own identifying information. (Don't include
183
+ the brackets!) The text should be enclosed in the appropriate
184
+ comment syntax for the file format. We also recommend that a
185
+ file or class name and description of purpose be included on the
186
+ same "printed page" as the copyright notice for easier
187
+ identification within third-party archives.
188
+
189
+ Copyright [yyyy] [name of copyright owner]
190
+
191
+ Licensed under the Apache License, Version 2.0 (the "License");
192
+ you may not use this file except in compliance with the License.
193
+ You may obtain a copy of the License at
194
+
195
+ http://www.apache.org/licenses/LICENSE-2.0
196
+
197
+ Unless required by applicable law or agreed to in writing, software
198
+ distributed under the License is distributed on an "AS IS" BASIS,
199
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
200
+ See the License for the specific language governing permissions and
201
+ limitations under the License.
runtime/install-mtp-runtime.py ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Install the three pinned MTP loader changes inside the existing RP2 image."""
2
+ import hashlib
3
+ import json
4
+ import os
5
+ from pathlib import Path
6
+
7
+ root = Path(__file__).resolve().parent
8
+ manifest = json.loads((root / "patch-manifest.json").read_text())
9
+ pending = []
10
+ for record in manifest["files"]:
11
+ target = Path(record["target"])
12
+ data = (root / "patches" / record["source"]).read_bytes()
13
+ assert hashlib.sha256(data).hexdigest() == record["patched_sha256"]
14
+ actual = hashlib.sha256(target.read_bytes()).hexdigest()
15
+ if actual == record["patched_sha256"]:
16
+ continue
17
+ if actual != record["original_sha256"]:
18
+ raise RuntimeError(f"Unsupported runtime source at {target}; use {manifest['base_image']}")
19
+ compile(data, str(target), "exec")
20
+ pending.append((target, data))
21
+ for target, data in pending:
22
+ temporary = target.with_suffix(".codecv2-mtp.tmp")
23
+ temporary.write_bytes(data)
24
+ temporary.chmod(target.stat().st_mode & 0o777)
25
+ os.replace(temporary, target)
26
+ print("TrellisMX MTP45 runtime installed; existing main-model kernels preserved", flush=True)
runtime/patch-manifest.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "base_image": "verdictai/trellismx@sha256:a0392e1c370eb933d87511928ea3aba6d5d63965cfe6eeaf9cc026a667e98ddc",
3
+ "files": [
4
+ {
5
+ "source": "vllm_utils_trellismx.py",
6
+ "target": "/opt/glm53-flash/vllm/vllm/utils/trellismx.py",
7
+ "original_sha256": "27be1af286b3fe6eee1fa11602f9fd87142ffa0e13bb9895a9a96346beb91fa3",
8
+ "patched_sha256": "c228f6c52c8f34cb63d37809cea03a944c1fcdbf013459434f29c95eb43d821f"
9
+ },
10
+ {
11
+ "source": "vllm_quant_trellismx.py",
12
+ "target": "/opt/glm53-flash/vllm/vllm/model_executor/layers/quantization/trellismx.py",
13
+ "original_sha256": "26512cae767be9b0f03a0dfbe2280b8f8778b7889eca146e03fbc6e740a3568e",
14
+ "patched_sha256": "6d849966ed3ea4b5f32d08702d58d20f6b24985de5a65f100ef76ee679ea267c"
15
+ },
16
+ {
17
+ "source": "p8_native_kernel.py",
18
+ "target": "/opt/glm53-flash/b12x/b12x/moe/_shared/trellismx/p8_native_kernel.py",
19
+ "original_sha256": "fc88f5ed2a466bbca8459b83dadfd14d5063a4e19f45626a31ca6f7c52c02078",
20
+ "patched_sha256": "6f3bb3c2c61f1881b2e0420a6606fb0cc3d8c37b7939ad64c15a043bfdb18167"
21
+ }
22
+ ]
23
+ }
runtime/patches/p8_native_kernel.py ADDED
@@ -0,0 +1,887 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Experimental TP-local P8 procedural-MCG MoE runtime for GLM-5.3.
2
+
3
+ This is a direct device path: a K3, K4 or K5 trellis stream is decoded to E4M3 inside
4
+ the MMA kernel and physical UE8M0/32 scales are consumed by the tensor core.
5
+ It implements the frozen TP4 identity-boundary P8 contract and an opt-in M1
6
+ H128 suh/svh scale component for a GLM routed layer whose sidecar carries the
7
+ matching immutable layer identity.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ from dataclasses import dataclass
12
+ from pathlib import Path
13
+ import os
14
+
15
+ import cutlass
16
+ import cutlass.cute as cute
17
+ import torch
18
+ from cutlass.base_dsl.compiler import OptLevel
19
+ from cutlass.cute.runtime import make_ptr
20
+ from safetensors import safe_open
21
+ from .p8_coupled_scales import (
22
+ COUPLED_SCHEMA as P8_COUPLED_SCHEMA,
23
+ SCALE_NAMES,
24
+ SCHEMA as P8_SCALE_COMPONENT_SCHEMA,
25
+ validate_coupled_component,
26
+ validate_scale_component,
27
+ )
28
+ from .policy_smallm_schedule import P8SmallMGeometry, p8_small_m_scratch_layout, use_small_m
29
+ from .tile_policy import select_tile
30
+
31
+ from b12x._lib.compiler import KernelCompileSpec, compile as b12x_compile
32
+ from b12x._lib.utils import get_max_active_clusters
33
+ from b12x.moe._shared.kernels.route_hoist_dynamic import MoEDynamicKernelBackend
34
+ from b12x.moe.fused_moe._impl import (
35
+ _DynamicMoEW4A8Launch,
36
+ _e8m0_scale_to_w4a8_sfb_inplace,
37
+ _launch_dynamic_topk_sum,
38
+ current_cuda_stream,
39
+ )
40
+
41
+
42
+ def _gptr(dtype, tensor: torch.Tensor, align: int = 16):
43
+ return make_ptr(
44
+ dtype, tensor.data_ptr(), cute.AddressSpace.gmem, assumed_align=align
45
+ )
46
+
47
+
48
+ def _fake_i32(shape: tuple[int, ...]):
49
+ return cute.runtime.make_fake_compact_tensor(
50
+ cutlass.Int32, shape, assumed_align=4
51
+ )
52
+
53
+
54
+ def _fake_f32(shape: tuple[int, ...]):
55
+ return cute.runtime.make_fake_compact_tensor(
56
+ cutlass.Float32, shape, assumed_align=16
57
+ )
58
+
59
+
60
+ @dataclass
61
+ class _CompiledArm:
62
+ compiled: object
63
+ tile_m: int
64
+ materialized: bool
65
+ mac: int
66
+
67
+
68
+ class P8NativeTPMoE:
69
+ """Own one TP rank's physical P8 payload and launch compiled MoE kernels."""
70
+
71
+ def __init__(
72
+ self,
73
+ sidecar: Path | tuple[Path, Path],
74
+ *,
75
+ device: torch.device,
76
+ tp_rank: int,
77
+ layer: int = 3,
78
+ expected_design_sha256: str | None = None,
79
+ expected_transform_sha256: str | None = None,
80
+ topk: int = 8,
81
+ hidden: int = 4096,
82
+ intermediate: int = 512,
83
+ swiglu_limit: float = 10.0,
84
+ force_materialized: bool | None = None,
85
+ mac_override: int | None = None,
86
+ deterministic_output: bool = True,
87
+ small_m_scheduler: bool = False,
88
+ fc1_tile_n: int = 128,
89
+ debug_capture: bool = False,
90
+ diagnostic_raw_fc1: bool = False,
91
+ fuse_scratch_zero: bool = False,
92
+ compact_scale_storage: bool = False,
93
+ compact_input_storage: bool = False,
94
+ shared_workspace: bool = False,
95
+ world_size: int = 4,
96
+ tp4_parent_sha256: tuple[str, str] | None = None,
97
+ prefill_chunk_tokens: int = 0,
98
+ grid_policy: bool | None = None,
99
+ grouped_m16: bool = False,
100
+ fuse_grouped_scratch: bool = False,
101
+ tile_major_tasks: bool = False,
102
+ fc1_pipeline_stages: int = 2,
103
+ fc1_warps: int = 4,
104
+ fc1_a_swizzle_rotate: bool = False,
105
+ fc1_warp_quant: bool | None = None,
106
+ fc1_exact_staging: bool = False,
107
+ fc1_broadcast_a: bool | None = None,
108
+ ) -> None:
109
+ self.device = torch.device(device)
110
+ # D-x2 (PREREG amendment 2): captured once; joins the compile spec below.
111
+ self.p8_down_remainder = os.environ.get("B12X_P8_DOWN_REMAINDER") == "1"
112
+ self._dx2_planes = 2 if self.p8_down_remainder else 1
113
+ self._dx2_phases = os.environ.get("B12X_P8_DOWN_REMAINDER_PHASES", "both")
114
+ self.p8_dx2_rowpack = os.environ.get("B12X_P8_DOWN_REMAINDER") in ("rp", "rp2")
115
+ self.p8_input_rowpack = os.environ.get("B12X_P8_DOWN_REMAINDER") == "rp2"
116
+ # EPI-PAR: bit-identical parallel small-M FC1 epilogue; joins the compile spec below.
117
+ self.p8_epi_par = os.environ.get("B12X_P8_EPI_PAR") == "1"
118
+ if self.p8_epi_par:
119
+ print(f"P8_EPI_PAR_ACTIVE layer={layer} rank={tp_rank}", flush=True)
120
+ if self.p8_dx2_rowpack:
121
+ print(f"P8_DX2_ROWPACK_ACTIVE layer={layer} rank={tp_rank} input_hop={int(self.p8_input_rowpack)}", flush=True)
122
+ self.grouped_m16 = bool(grouped_m16)
123
+ self.fuse_grouped_scratch = bool(fuse_grouped_scratch)
124
+ self.tile_major_tasks = bool(tile_major_tasks)
125
+ if fc1_pipeline_stages not in (2, 3):
126
+ raise ValueError('FC1 pipeline supports two or three stages')
127
+ self.fc1_pipeline_stages = fc1_pipeline_stages
128
+ if fc1_warps not in (4, 8):
129
+ raise ValueError('FC1 supports four or eight warps')
130
+ self.fc1_warps = fc1_warps
131
+ self.fc1_exact_staging = bool(fc1_exact_staging)
132
+ self.fc1_broadcast_a = (os.environ.get('GLM53_P8_FC1_BROADCAST_A') == '1'
133
+ if fc1_broadcast_a is None else bool(fc1_broadcast_a))
134
+ self.fc1_a_swizzle_rotate = bool(fc1_a_swizzle_rotate)
135
+ self.fc1_warp_quant = (os.environ.get('GLM53_P8_FC1_WARP_QUANT') == '1'
136
+ if fc1_warp_quant is None else bool(fc1_warp_quant))
137
+ self.grid_policy = (os.environ.get('GLM53_P8_GRID_POLICY') == '1'
138
+ if grid_policy is None else bool(grid_policy))
139
+ self.prefill_chunk_tokens = int(prefill_chunk_tokens)
140
+ if self.prefill_chunk_tokens < 0 or self.prefill_chunk_tokens % 64:
141
+ raise ValueError("P8 prefill chunk must be zero or a positive multiple of 64")
142
+ self.compact_scale_storage = bool(compact_scale_storage)
143
+ self.compact_input_storage = bool(compact_input_storage)
144
+ self.shared_workspace = bool(shared_workspace)
145
+ self.tp_rank = int(tp_rank)
146
+ self.world_size = int(world_size)
147
+ if self.world_size != 4 or self.tp_rank not in range(4):
148
+ raise ValueError("P8 native supports TP4 only; TP2 validators are unsupported")
149
+ self.layer = int(layer)
150
+ if not 3 <= self.layer <= 45:
151
+ raise ValueError("P8 native layer must be in GLM routed layers 3..44 or MTP45")
152
+ self.topk = int(topk)
153
+ self.hidden = int(hidden)
154
+ self.intermediate = int(intermediate)
155
+ self.swiglu_limit = float(swiglu_limit)
156
+ self.force_materialized = force_materialized
157
+ self.mac_override = None if mac_override is None else int(mac_override)
158
+ self.deterministic_output = bool(deterministic_output)
159
+ # Explicit developmental opt-in. M2/M3 and prefill retain baseline
160
+ # selection; this is not enabled through a serving environment flag.
161
+ self.small_m_scheduler = bool(small_m_scheduler)
162
+ self.fc1_tile_n = int(fc1_tile_n)
163
+ self.debug_capture = bool(debug_capture)
164
+ self.diagnostic_raw_fc1 = bool(diagnostic_raw_fc1)
165
+ self.fuse_scratch_zero = bool(fuse_scratch_zero)
166
+ self._scratch_layout = (
167
+ p8_small_m_scratch_layout(intermediate=self.intermediate) if self.fuse_scratch_zero else None
168
+ )
169
+ self.debug_tensors = {}
170
+ if self.fc1_tile_n not in (32, 64, 128):
171
+ raise ValueError("FC1 tile N must be 32, 64, or 128")
172
+ if self.fc1_tile_n != 128 and (
173
+ not self.small_m_scheduler or self.swiglu_limit != 10.0
174
+ ):
175
+ raise ValueError("Narrow FC1 requires small-M and SwiGLU limit 10")
176
+ if self.small_m_scheduler and (
177
+ not self.deterministic_output or force_materialized is not None
178
+ or (topk, hidden, intermediate) != (8, 4096, 2048 // self.world_size)
179
+ ):
180
+ raise ValueError("P8 small-M requires deterministic GLM TP4 and automatic fallback")
181
+ if self.diagnostic_raw_fc1 and (
182
+ not self.debug_capture
183
+ or not self.small_m_scheduler
184
+ or self.fc1_tile_n != 128
185
+ or self.fuse_scratch_zero
186
+ ):
187
+ raise ValueError(
188
+ "raw FC1 diagnostic requires debug M1 small-M N128 capture"
189
+ )
190
+ if self.mac_override is not None and self.mac_override <= 0:
191
+ raise ValueError("mac_override must be positive")
192
+ if tp4_parent_sha256 is not None:
193
+ if self.world_size != 2:
194
+ raise ValueError("Parent-pair adapter requires TP2")
195
+ from glm53_nvfp4.p8_tp2_repack import open_tp2_pair
196
+ source = open_tp2_pair(sidecar, tp4_parent_sha256, layer=self.layer, rank=self.tp_rank)
197
+ else:
198
+ source = safe_open(sidecar, framework="pt", device="cpu")
199
+ with source as src:
200
+ metadata = src.metadata() or {}
201
+ schema = metadata.get("schema")
202
+ source_design_sha256 = metadata.get("source_design_sha256")
203
+ bits_text = metadata.get("bits", "")
204
+ if bits_text not in {"3", "4", "5"}:
205
+ raise RuntimeError(f"invalid P8 trellis rate: {bits_text!r}")
206
+ self.trellis_bits = int(bits_text)
207
+ base_required = {
208
+ "layer": str(self.layer),
209
+ "rank": str(self.tp_rank),
210
+ "world_size": str(self.world_size),
211
+ "bits": bits_text,
212
+ "alphabet": "e4m3",
213
+ "scale": "ue8m0-k32",
214
+ "law": "procedural-mcg-alpha2",
215
+ "ldlq": "false",
216
+ }
217
+ identity_schema = schema in {
218
+ "glm53-p8-identity-mcg-tp4-rank.v1",
219
+ "glm53-p8-mcg-tp4-rank.v2",
220
+ }
221
+ scale_component_schema = schema == P8_SCALE_COMPONENT_SCHEMA
222
+ full_coupled_schema = schema == P8_COUPLED_SCHEMA.replace("tp4", f"tp{self.world_size}")
223
+ if self.world_size == 2 and not full_coupled_schema:
224
+ raise RuntimeError("TP2 port only supports the full-coupled schema")
225
+ if (
226
+ not (identity_schema or scale_component_schema or full_coupled_schema)
227
+ or any(metadata.get(key) != value for key, value in base_required.items())
228
+ or (identity_schema and metadata.get("boundary") != "identity")
229
+ ):
230
+ raise RuntimeError(f"invalid P8 native sidecar metadata: {metadata}")
231
+ if full_coupled_schema or schema in {
232
+ "glm53-p8-mcg-tp4-rank.v2",
233
+ P8_SCALE_COMPONENT_SCHEMA,
234
+ P8_COUPLED_SCHEMA,
235
+ }:
236
+ if (
237
+ not isinstance(source_design_sha256, str)
238
+ or len(source_design_sha256) != 64
239
+ or any(char not in "0123456789abcdef" for char in source_design_sha256)
240
+ ):
241
+ raise RuntimeError("v2 P8 sidecar lacks a valid source design hash")
242
+ if (
243
+ expected_design_sha256 is not None
244
+ and source_design_sha256 != expected_design_sha256
245
+ ):
246
+ raise RuntimeError("P8 sidecar does not match the expected design")
247
+ elif expected_design_sha256 is not None:
248
+ raise RuntimeError("historical P8 sidecars cannot satisfy a v2 design pin")
249
+ w13 = src.get_tensor("w13_trellis")
250
+ w2 = src.get_tensor("w2_trellis")
251
+ w13_scale = src.get_tensor("w13_scale_ue8m0")
252
+ w2_scale = src.get_tensor("w2_scale_ue8m0")
253
+ scale_tensors = (
254
+ {name: src.get_tensor(name) for name in SCALE_NAMES}
255
+ if scale_component_schema or full_coupled_schema
256
+ else None
257
+ )
258
+ self.source_design_sha256 = source_design_sha256
259
+ experts = int(w13.shape[1])
260
+ stream_words = 16 * self.trellis_bits
261
+ if tuple(w13.shape) != (
262
+ 2, experts, hidden // 16, intermediate // 16, stream_words
263
+ ):
264
+ raise RuntimeError(f"unexpected W13 trellis shape {tuple(w13.shape)}")
265
+ if tuple(w2.shape) != (
266
+ experts, intermediate // 16, hidden // 16, stream_words
267
+ ):
268
+ raise RuntimeError(f"unexpected W2 trellis shape {tuple(w2.shape)}")
269
+ if tuple(w13_scale.shape) != (experts, 2 * intermediate, hidden // 32):
270
+ raise RuntimeError(f"unexpected W13 scale shape {tuple(w13_scale.shape)}")
271
+ if tuple(w2_scale.shape) != (experts, hidden, intermediate // 32):
272
+ raise RuntimeError(f"unexpected W2 scale shape {tuple(w2_scale.shape)}")
273
+ self.experts = experts
274
+ self.scale_component = None
275
+ self.full_coupled = bool(full_coupled_schema)
276
+ if (self.grouped_m16 or self.fuse_grouped_scratch) and not self.full_coupled:
277
+ raise ValueError("grouped scratch requires the full-coupled kernel owner")
278
+ if self.fuse_grouped_scratch and not self.grouped_m16:
279
+ raise ValueError("fused grouped scratch requires grouped_m16")
280
+ if self.shared_workspace and (not self.full_coupled or self.debug_capture or self.fuse_scratch_zero):
281
+ raise ValueError('shared workspace requires full coupling without retained debug tensors or fused arena')
282
+ if self.compact_scale_storage and not self.full_coupled:
283
+ raise ValueError("compact scales require the external full-coupled owners")
284
+ if self.full_coupled and expected_transform_sha256 is None:
285
+ raise RuntimeError(
286
+ "full-coupled P8 requires an externally pinned encoder transform"
287
+ )
288
+ if scale_tensors is not None:
289
+ validator = (
290
+ validate_coupled_component
291
+ if self.full_coupled
292
+ else validate_scale_component
293
+ )
294
+ # Preserve compatibility with the pinned TP4 validator during
295
+ # isolated storage diagnostics. TP2 requires the ported validator.
296
+ validator_kwargs = {} if self.world_size == 4 else {"world_size": self.world_size}
297
+ if self.full_coupled:
298
+ validator_kwargs["expected_transform_sha256"] = (
299
+ expected_transform_sha256
300
+ )
301
+ self.scale_component = validator(
302
+ metadata, scale_tensors, layer=self.layer, rank=self.tp_rank,
303
+ experts=experts, hidden=hidden, intermediate=intermediate,
304
+ **validator_kwargs,
305
+ )
306
+ if not self.small_m_scheduler or self.fc1_tile_n != 128:
307
+ raise RuntimeError(
308
+ "P8 scale component requires the M1 N128 owner path"
309
+ )
310
+ # The trellis storage is byte-for-byte the same size as the packed
311
+ # E2M1 descriptor carrier expected by the inherited W4A8 launch ABI.
312
+ # Alias it for the descriptor-only arguments instead of allocating a
313
+ # second ~0.9 GiB of unread dummy weights per layer and TP rank. The
314
+ # kernel reads the procedural stream through the uint32 pointers below;
315
+ # it never dereferences the descriptor carrier values.
316
+ w13_stream_storage = w13.to(device=self.device).contiguous()
317
+ w2_stream_storage = w2.to(device=self.device).contiguous()
318
+ self.w13_stream = w13_stream_storage.view(torch.int32).reshape(-1)
319
+ if self.fc1_exact_staging:
320
+ if ((self.world_size, self.experts, self.hidden, self.intermediate, self.fc1_tile_n)
321
+ != (4, 288, 4096, 512, 128) or not self.full_coupled):
322
+ raise ValueError('exact staging requires full-coupled TP4 E288/H4096/I512 N128')
323
+ expected_words = 2 * 288 * 4096 * 512 * self.trellis_bits // 32
324
+ if self.w13_stream.numel() != expected_words:
325
+ raise ValueError('exact staging FC1 stream extent mismatch')
326
+ self.w2_stream = w2_stream_storage.view(torch.int32).reshape(-1)
327
+ w13_scale = w13_scale.to(device=self.device).contiguous()
328
+ w2_scale = w2_scale.to(device=self.device).contiguous()
329
+ # The monolithic kernel consumes the logical [E, N, K/32] UE8M0
330
+ # plane through the sfb_*_mx ABI slots. The split materialized
331
+ # kernels consume a separately repacked copy through *_sfb_rp.
332
+ # Keep both representations: a one-byte sentinel in the logical slots
333
+ # is an out-of-bounds scale read, not an identity scale.
334
+ self.w13_scale_mx = w13_scale.reshape(-1)
335
+ self.w2_scale_mx = w2_scale.reshape(-1)
336
+ self.w13_sfb = _e8m0_scale_to_w4a8_sfb_inplace(
337
+ w13_scale.clone(),
338
+ weight_E=experts,
339
+ rows=2 * intermediate,
340
+ k_dim=hidden,
341
+ gated_half_rows=intermediate,
342
+ ).reshape(-1)
343
+ self.w2_sfb = _e8m0_scale_to_w4a8_sfb_inplace(
344
+ w2_scale.clone(),
345
+ weight_E=experts,
346
+ rows=hidden,
347
+ k_dim=intermediate,
348
+ ).reshape(-1)
349
+ if self.compact_scale_storage:
350
+ # Full coupling forces external materialized FC1/FC2 for every M.
351
+ # Logical-scale ABI arguments remain non-null aliases, but are
352
+ # not read by those owners. Never enable for a monolithic arm.
353
+ # This is opt-in pending device closure against separate storage.
354
+ self.w13_scale_mx = self.w13_sfb
355
+ self.w2_scale_mx = self.w2_sfb
356
+ # These are descriptor carriers only; they alias the trellis storage
357
+ # above and therefore add zero payload bytes. Their trailing extent is
358
+ # the PACKED row length, which is bits/8 bytes per weight: hidden // 2
359
+ # only at K4. Deriving it from the stored rate keeps K4 byte-identical
360
+ # while letting K3 and K5 describe their own shorter or longer rows.
361
+ w13_row_bytes = hidden * self.trellis_bits // 8
362
+ w2_row_bytes = intermediate * self.trellis_bits // 8
363
+ w13_dummy_bytes = experts * 2 * intermediate * w13_row_bytes
364
+ w2_dummy_bytes = experts * hidden * w2_row_bytes
365
+ self.w13_dummy = w13_stream_storage.view(torch.uint8).reshape(-1)[
366
+ :w13_dummy_bytes
367
+ ].reshape(
368
+ experts, 2 * intermediate, w13_row_bytes
369
+ )
370
+ self.w2_dummy = w2_stream_storage.view(torch.uint8).reshape(-1)[
371
+ :w2_dummy_bytes
372
+ ].reshape(
373
+ experts, hidden, w2_row_bytes
374
+ )
375
+ self.sentinel = torch.zeros(1, dtype=torch.uint8, device=self.device)
376
+ self.zero_lut = torch.zeros(1, dtype=torch.uint8, device=self.device)
377
+ # MCG never dereferences the LUT pointer. The diagnostic-only arm
378
+ # reuses that dead ABI slot for exactly 128 FP32 trace values (512 B),
379
+ # initialized to an all-ones NaN sentinel so partial writes fail closed.
380
+ self.input_prequant_trace = (
381
+ torch.full((512,), 0xFF, dtype=torch.uint8, device=self.device)
382
+ if self.diagnostic_raw_fc1
383
+ else self.zero_lut
384
+ )
385
+ self.zero_rotation = torch.zeros(1, dtype=torch.float16, device=self.device)
386
+ self.scale_component_packed = (
387
+ self.scale_component.packed.to(device=self.device)
388
+ if self.scale_component is not None
389
+ else self.zero_rotation
390
+ )
391
+ self.ones = torch.ones(experts, dtype=torch.float32, device=self.device)
392
+ if self.small_m_scheduler and self.experts != 288:
393
+ raise ValueError("P8 small-M requires 288 experts")
394
+ # v11: the small-M owner path is compiled per stored rate (K3/K4/K5);
395
+ # M>1 on a non-K4 layer is served row by row through that same exact
396
+ # kernel because the grouped M64 prefill kernels remain K4-only.
397
+ self._compiled: dict[tuple[bool, bool], _CompiledArm] = {}
398
+ self._coupled_reducer = None
399
+
400
+ def _compile(self, materialized: bool, small_m: bool = False, expected_m: int | None = None) -> _CompiledArm:
401
+ if self.compact_scale_storage and not (self.full_coupled and materialized):
402
+ raise RuntimeError("compact scales cannot enter a monolithic path")
403
+ selected_m = select_tile(expected_m if expected_m is not None else (1 if small_m else 4096))[0]
404
+ cache_key = (materialized, small_m, selected_m)
405
+ cached = self._compiled.get(cache_key)
406
+ if cached is not None:
407
+ return cached
408
+ tile_m = selected_m
409
+ if self.grouped_m16:
410
+ raise RuntimeError("fixed M16 override conflicts with requested tile policy")
411
+ mac = (
412
+ self.mac_override
413
+ if self.mac_override is not None
414
+ else (64 if materialized else int(get_max_active_clusters(1)))
415
+ )
416
+ kernel = MoEDynamicKernelBackend(
417
+ 16,
418
+ (tile_m, 128),
419
+ activation="silu",
420
+ quant_recipe="w4a8_trellis",
421
+ w4a8_repacked=True,
422
+ num_topk=self.topk,
423
+ trellis_bits=self.trellis_bits,
424
+ trellis_codebook="mcg",
425
+ trellis_scaled=True,
426
+ trellis_identity_boundary=not self.full_coupled,
427
+ direct_routing=small_m,
428
+ materialize_intermediate=materialized,
429
+ p8_small_m=small_m,
430
+ p8_fc1_tile_n=self.fc1_tile_n if small_m else 128,
431
+ p8_scale_sandwich=self.scale_component is not None,
432
+ p8_full_coupled=self.full_coupled,
433
+ share_input_across_experts=materialized,
434
+ deterministic_output=self.deterministic_output,
435
+ swiglu_limit=self.swiglu_limit,
436
+ )
437
+ if self.p8_down_remainder:
438
+ if not self.full_coupled:
439
+ raise RuntimeError("D-x2 is implemented for the full-coupled P8 paths only")
440
+ if self.fc1_warp_quant:
441
+ raise RuntimeError("D-x2 needs fc1_warp_quant=False: the warp-quant FC1 epilogue has no remainder writer")
442
+ from b12x.moe._shared.kernels.p8_down_remainder import enable_down_remainder
443
+ enable_down_remainder(kernel, self._dx2_phases)
444
+ if self.p8_dx2_rowpack and small_m:
445
+ if not self.full_coupled:
446
+ raise RuntimeError("D-x2-RP is implemented for the full-coupled P8 paths only")
447
+ if self.fc1_warp_quant:
448
+ raise RuntimeError("D-x2-RP/RP2 need fc1_warp_quant=False: the warp-quant FC1 epilogue has no remainder writer")
449
+ if self.tile_major_tasks or self.grouped_m16:
450
+ raise RuntimeError("D-x2-RP/RP2 need one route per direct tile (tile_major_tasks=False, grouped_m16=False)")
451
+ if self.p8_input_rowpack and not self.fc1_broadcast_a:
452
+ raise RuntimeError("D-x2-RP2 needs fc1_broadcast_a=True: otherwise ordinary FC1 staging also writes A row 8")
453
+ from b12x.moe._shared.kernels.p8_down_remainder import enable_rowpack
454
+ enable_rowpack(kernel, input_hop=self.p8_input_rowpack)
455
+ if self.diagnostic_raw_fc1:
456
+ if not (small_m and self.full_coupled):
457
+ raise RuntimeError(
458
+ "raw FC1 diagnostic dispatched outside full-coupled M1"
459
+ )
460
+ from b12x.moe._shared.kernels.p8_h128_fc1 import (
461
+ P8H128FC1RawCaptureKernel,
462
+ )
463
+
464
+ kernel.materialized_phase1_kernel = P8H128FC1RawCaptureKernel()
465
+ kernel.p8_input_prequant_diagnostic = True
466
+ if self.full_coupled:
467
+ # Both M1 and grouped owners share the scale/sign plane geometry.
468
+ kernel.materialized_phase1_kernel.p8_intermediate = self.intermediate
469
+ kernel.materialized_phase2_kernel.p8_intermediate = self.intermediate
470
+ if small_m:
471
+ kernel.materialized_phase1_kernel.p8_tile_major = self.tile_major_tasks
472
+ kernel.materialized_phase2_kernel.p8_tile_major = self.tile_major_tasks
473
+ fc1 = kernel.materialized_phase1_kernel
474
+ fc1.p8_a_swizzle_rotate = self.fc1_a_swizzle_rotate
475
+ fc1.p8_warp_quant = self.fc1_warp_quant
476
+ fc1.p8_epi_par = self.p8_epi_par
477
+ fc1.p8_exact_staging = self.fc1_exact_staging
478
+ # Direct owner invokes _run_task with valid_rows=1. Never
479
+ # apply this specialization to grouped multi-row owners.
480
+ fc1.p8_broadcast_a = self.fc1_broadcast_a
481
+ fc1.num_warps = self.fc1_warps
482
+ fc1.threads_per_cta = 32 * self.fc1_warps
483
+ fc1.p8_n8_per_warp = 16 // self.fc1_warps
484
+ fc1.owned_row_groups = fc1.tile_m // self.fc1_warps
485
+ fc1.p8_pipeline_stages = self.fc1_pipeline_stages
486
+ fc1.shared_bytes = max(fc1.shared_bytes, self.fc1_pipeline_stages * fc1.stage_bytes)
487
+ fc1.shared_words = (fc1.shared_bytes + 3) // 4
488
+ fc1.trellis_lut_offset = fc1.shared_bytes
489
+ launch = _DynamicMoEW4A8Launch(
490
+ kernel,
491
+ k=self.hidden,
492
+ n=self.intermediate,
493
+ w1_n=2 * self.intermediate,
494
+ num_topk=self.topk,
495
+ )
496
+
497
+ def ptr(dtype, address: int, align: int = 16):
498
+ return make_ptr(dtype, address, cute.AddressSpace.gmem, assumed_align=align)
499
+
500
+ def fake_ptr_u8():
501
+ return ptr(cutlass.Uint8, 16)
502
+
503
+ def fake_ptr_i32():
504
+ return ptr(cutlass.Int32, 4, 4)
505
+
506
+ def fake_ptr_u32():
507
+ return ptr(cutlass.Uint32, 16)
508
+
509
+ b_w13_fake = cute.runtime.make_fake_compact_tensor(
510
+ cutlass.Float4E2M1FN,
511
+ (2 * self.intermediate, self.hidden, self.experts),
512
+ stride_order=(1, 0, 2),
513
+ assumed_align=16,
514
+ )
515
+ b_w2_fake = cute.runtime.make_fake_compact_tensor(
516
+ cutlass.Float4E2M1FN,
517
+ (self.hidden, self.intermediate, self.experts),
518
+ stride_order=(1, 0, 2),
519
+ assumed_align=16,
520
+ )
521
+ compiled = b12x_compile(
522
+ launch,
523
+ ptr(cutlass.BFloat16, 16),
524
+ fake_ptr_i32(),
525
+ ptr(cutlass.Float32, 4, 4),
526
+ ptr(cutlass.Float4E2M1FN, 16),
527
+ ptr(cutlass.Float8E4M3FN, 16),
528
+ fake_ptr_u8(),
529
+ fake_ptr_u8(),
530
+ fake_ptr_u32(),
531
+ _fake_i32((1,)), _fake_i32((1,)), _fake_i32((1,)),
532
+ _fake_i32((1,)), _fake_i32((1,)), _fake_i32((1,)), _fake_i32((1,)),
533
+ fake_ptr_i32(), fake_ptr_i32(), fake_ptr_i32(),
534
+ fake_ptr_i32(), fake_ptr_i32(), fake_ptr_i32(), fake_ptr_i32(),
535
+ b_w13_fake,
536
+ ptr(cutlass.Float8E4M3FN, 16),
537
+ b_w2_fake,
538
+ ptr(cutlass.Float8E4M3FN, 16),
539
+ fake_ptr_u8(), fake_ptr_u8(), fake_ptr_u8(), fake_ptr_u8(),
540
+ fake_ptr_u32(), fake_ptr_u32(), fake_ptr_u32(), fake_ptr_u32(),
541
+ _fake_i32((self.experts,)),
542
+ _fake_i32((self.experts,)),
543
+ _fake_i32((self.experts + 1,)),
544
+ _fake_f32((self.experts,)), _fake_f32((self.experts,)),
545
+ _fake_f32((self.experts,)), _fake_f32((self.experts,)),
546
+ ptr(
547
+ cutlass.Float32 if self.full_coupled else cutlass.BFloat16,
548
+ 16,
549
+ ),
550
+ fake_ptr_i32(),
551
+ ptr(cutlass.Float32, 16),
552
+ 1, 1, 1, 1, 1, 1, 1,
553
+ current_cuda_stream(),
554
+ fake_ptr_u8(),
555
+ ptr(cutlass.Float16, 16),
556
+ # The compile spec is the JIT cache key and includes every specialized
557
+ # field, especially stored rate, topology and epilogue dimensions.
558
+ compile_spec=KernelCompileSpec.from_fields(
559
+ "glm53.p8.native.tp",
560
+ 4,
561
+ ("fc1_row_alias376", 1),
562
+ ("requested_m_regime_direct_policy", 1),
563
+ ("fc1_route_hoist", 1),
564
+ ("tile_m", tile_m),
565
+ ("trellis_bits", self.trellis_bits),
566
+ ("mcg_k5_funnel", int(small_m and self.trellis_bits == 5)),
567
+ ("fc2_carveout100_grid564", int(small_m)),
568
+ ("fc2_k5_funnel", int(small_m and self.trellis_bits == 5)),
569
+ ("grouped_fc2_grid376", int(not small_m)),
570
+ ("tile_major_tasks", int(self.tile_major_tasks and small_m)),
571
+ ("fc1_pipeline_stages", self.fc1_pipeline_stages if small_m else 2),
572
+ ("fc1_warps", self.fc1_warps if small_m else 4),
573
+ ("fc1_a_swizzle_rotate", int(self.fc1_a_swizzle_rotate and small_m)),
574
+ ("fc1_warp_quant", int(self.fc1_warp_quant and small_m)),
575
+ ("fc1_exact_staging", int(self.fc1_exact_staging and small_m)),
576
+ ("fc1_broadcast_a", int(self.fc1_broadcast_a and small_m)),
577
+ ("materialized", int(materialized)),
578
+ ("small_m_scheduler", int(small_m)),
579
+ ("fc1_tile_n", self.fc1_tile_n if small_m else 128),
580
+ ("experts", self.experts),
581
+ ("hidden", self.hidden),
582
+ ("intermediate", self.intermediate),
583
+ ("topk", self.topk),
584
+ ("rank", self.tp_rank),
585
+ ("scaled", 1),
586
+ ("identity", int(not self.full_coupled)),
587
+ ("scale_sandwich", int(self.scale_component is not None)),
588
+ ("full_coupled", int(self.full_coupled)),
589
+ ("raw_fc1_diagnostic", int(self.diagnostic_raw_fc1)),
590
+ ("input_prequant_diagnostic", int(self.diagnostic_raw_fc1)),
591
+ ("codebook", "mcg"),
592
+ ("deterministic_output", int(self.deterministic_output)),
593
+ ("down_remainder", int(self.p8_down_remainder)),
594
+ ("down_remainder_phases", self._dx2_phases if self.p8_down_remainder else "none"),
595
+ ("dx2_rowpack", int(self.p8_dx2_rowpack and small_m)),
596
+ ("dx2_input_rowpack", int(self.p8_input_rowpack and small_m)),
597
+ ("epi_par", int(self.p8_epi_par and small_m)),
598
+ ),
599
+ dsl_compile_options=OptLevel(2),
600
+ )
601
+ arm = _CompiledArm(compiled=compiled, tile_m=tile_m, materialized=materialized, mac=mac)
602
+ self._compiled[cache_key] = arm
603
+ return arm
604
+
605
+ def _compile_full_coupled_reducer(self):
606
+ if not self.full_coupled:
607
+ raise RuntimeError("coupled reducer requested for non-coupled P8")
608
+ if self._coupled_reducer is not None:
609
+ return self._coupled_reducer
610
+ from b12x.moe._shared.kernels.p8_coupled_topk import (
611
+ P8CoupledTopKSumKernel,
612
+ )
613
+
614
+ reducer = P8CoupledTopKSumKernel(topk=self.topk, hidden=self.hidden)
615
+ self._coupled_reducer = b12x_compile(
616
+ reducer,
617
+ make_ptr(cutlass.Float32, 16, cute.AddressSpace.gmem, assumed_align=16),
618
+ make_ptr(cutlass.Float32, 4, cute.AddressSpace.gmem, assumed_align=4),
619
+ make_ptr(cutlass.BFloat16, 16, cute.AddressSpace.gmem, assumed_align=16),
620
+ 1,
621
+ current_cuda_stream(),
622
+ compile_spec=KernelCompileSpec.from_fields(
623
+ "glm53.p8.coupled_topk_h512",
624
+ 2,
625
+ ("topk", self.topk),
626
+ ("hidden", self.hidden),
627
+ ("rank", self.tp_rank),
628
+ ("route_dtype", "fp32"),
629
+ ("output_dtype", "bf16"),
630
+ ),
631
+ dsl_compile_options=OptLevel(2),
632
+ )
633
+ return self._coupled_reducer
634
+
635
+ @torch.inference_mode()
636
+ def __call__(
637
+ self,
638
+ x: torch.Tensor,
639
+ topk_weights: torch.Tensor,
640
+ topk_ids: torch.Tensor,
641
+ ) -> torch.Tensor:
642
+ if x.dtype != torch.bfloat16 or x.ndim != 2 or x.shape[1] != self.hidden:
643
+ raise RuntimeError(f"P8 native input contract mismatch: {x.dtype} {tuple(x.shape)}")
644
+ m = int(x.shape[0])
645
+ if tuple(topk_ids.shape) != (m, self.topk) or tuple(topk_weights.shape) != (m, self.topk):
646
+ raise RuntimeError("P8 native routing shape mismatch")
647
+ if self.prefill_chunk_tokens and m > self.prefill_chunk_tokens:
648
+ if not self.full_coupled or self.debug_capture:
649
+ raise RuntimeError("bounded prefill requires full coupling without debug capture")
650
+ # MoE is token-local. Bound route-output and materialized carrier
651
+ # storage without changing the scheduler batch or attention work.
652
+ # Reuse allocations on the same stream; no host readback or sync.
653
+ output = torch.empty_like(x)
654
+ for start in range(0, m, self.prefill_chunk_tokens):
655
+ stop = min(start + self.prefill_chunk_tokens, m)
656
+ output[start:stop].copy_(self(x[start:stop], topk_weights[start:stop], topk_ids[start:stop]))
657
+ return output
658
+ # Every stored rate now has a fused grouped M64/N128 owner, so a non-K4 layer at M>1
659
+ # runs the same native path as K4 rather than looping the M1 kernel row by row. The
660
+ # row-by-row fallback is deliberately gone: a rate without a grouped specialization
661
+ # must fail closed instead of silently serving at a fraction of the speed.
662
+ if self.scale_component is not None and not self.full_coupled and m != 1:
663
+ raise RuntimeError("P8 scale sandwich currently supports M=1 only")
664
+ # Match the W4A8 planner's measured M16-to-M64 transition: sparse
665
+ # decode and ordinary prefill stay monolithic; only dense routed
666
+ # batches pay for the split materialized phase kernels.
667
+ materialized = (
668
+ m * self.topk >= 36 * self.experts
669
+ if self.force_materialized is None
670
+ else self.force_materialized
671
+ )
672
+ small_m = use_small_m(self.small_m_scheduler, m)
673
+ if self.full_coupled:
674
+ # Experimental direct-route batches through M16; requires matching
675
+ # multirow FC1 and input-prologue patches. Not serving-qualified.
676
+ small_m = m <= 16 and not self.grouped_m16
677
+ materialized = True
678
+ materialized = materialized or small_m
679
+ arm = self._compile(materialized, small_m=small_m, expected_m=m)
680
+ tile_m = arm.tile_m
681
+ x = x.contiguous()
682
+ flat_ids = topk_ids.to(dtype=torch.int32).contiguous().reshape(-1)
683
+ flat_weights = topk_weights.to(dtype=torch.float32).contiguous().reshape(-1)
684
+ physical_tiles = (
685
+ m * self.topk if small_m
686
+ else self.experts + (m * self.topk + tile_m - 1) // tile_m
687
+ )
688
+ rows_padded = physical_tiles * tile_m
689
+ gate_tile_count = ((2 * self.intermediate) // 128) // 2
690
+ max_tasks = physical_tiles * max(gate_tile_count, 1)
691
+ fused_scratch_zero = ((self.fuse_scratch_zero and small_m) or
692
+ (self.fuse_grouped_scratch and self.grouped_m16))
693
+ shared_kernel_output = None
694
+ if fused_scratch_zero or self.shared_workspace:
695
+ if fused_scratch_zero:
696
+ from .direct_policy_scratch import direct_scratch_layout
697
+ self._scratch_layout = (direct_scratch_layout(m, self.intermediate, tile_m=tile_m, planes=self._dx2_planes) if small_m
698
+ else p8_small_m_scratch_layout(intermediate=self.intermediate, tokens=m, planes=self._dx2_planes,
699
+ shared=True, grouped=True, tile_m=tile_m, direct=False))
700
+ layout = (p8_small_m_scratch_layout(intermediate=self.intermediate, tokens=m, shared=True, planes=self._dx2_planes,
701
+ grouped=self.full_coupled and materialized and not small_m, tile_m=tile_m, direct=small_m)
702
+ if self.shared_workspace else self._scratch_layout)
703
+ assert layout is not None
704
+ # A single GPU fill initializes all original bytes plus alignment
705
+ # padding. The views add no casts, copies, or device kernels.
706
+ if self.shared_workspace:
707
+ from vllm.v1.worker.workspace import current_workspace_manager
708
+ arena, shared_kernel_output = current_workspace_manager().get_simultaneous(
709
+ ((layout.nbytes,), torch.uint8),
710
+ ((m * self.topk, self.hidden), torch.float32),
711
+ )
712
+ arena.zero_()
713
+ else:
714
+ arena = torch.zeros(layout.nbytes, dtype=torch.uint8, device=self.device)
715
+ buffers = {
716
+ region.name: arena.narrow(0, region.offset, region.nbytes)
717
+ .view(getattr(torch, region.dtype)).reshape(region.shape)
718
+ for region in layout.regions
719
+ }
720
+ packed_a = buffers["packed_a"]
721
+ scale_flat = buffers["scale_flat"]
722
+ intermediate_u32 = buffers["intermediate_u32"]
723
+ barrier_count = buffers["barrier_count"]
724
+ barrier_epoch = buffers["barrier_epoch"]
725
+ pair_head = buffers["pair_head"]
726
+ producers_done = buffers["producers_done"]
727
+ all_published = buffers["all_published"]
728
+ task_head = buffers["task_head"]
729
+ task_tail = buffers["task_tail"]
730
+ task_ready = buffers["task_ready"]
731
+ task_expert = buffers["task_expert"]
732
+ task_m_tile = buffers["task_m_tile"]
733
+ task_slice_begin = buffers["task_slice_begin"]
734
+ task_slice_count = buffers["task_slice_count"]
735
+ task_valid_rows = buffers["task_valid_rows"]
736
+ tile_write_count = buffers["tile_write_count"]
737
+ row_counts = buffers["row_counts"]
738
+ expert_write_rows = buffers["expert_write_rows"]
739
+ expert_tile_base = buffers["expert_tile_base"]
740
+ token_map = buffers["token_map"]
741
+ token_weights = buffers["token_weights"]
742
+ output = (torch.zeros(m, self.hidden, dtype=torch.bfloat16, device=self.device)
743
+ if self.shared_workspace else buffers["output"])
744
+ else:
745
+ # Coupled grouped FC1 addresses the shared carrier by token index,
746
+ # not padded expert row (p8_h128_fc1 src_word uses tok). Keep the
747
+ # original extent for every other owner, including M1.
748
+ input_rows = (m if self.compact_input_storage and self.full_coupled
749
+ and materialized and not small_m else rows_padded)
750
+ packed_a = torch.zeros(input_rows * self.hidden, dtype=torch.uint8, device=self.device)
751
+ scale_elements = (
752
+ m * (self.hidden // 32)
753
+ if self.full_coupled and materialized and not small_m
754
+ else (self.experts + m * self.topk + 1)
755
+ * tile_m
756
+ * (self.hidden // 8)
757
+ )
758
+ scale_flat = torch.zeros(
759
+ scale_elements, dtype=torch.uint8, device=self.device
760
+ )
761
+ intermediate_count = self._dx2_planes * rows_padded * (self.intermediate + self.intermediate // 32) // 4
762
+ intermediate_u32 = torch.zeros(intermediate_count, dtype=torch.int32, device=self.device)
763
+
764
+ def z1():
765
+ return torch.zeros(1, dtype=torch.int32, device=self.device)
766
+
767
+ def ztask():
768
+ return torch.zeros(max_tasks, dtype=torch.int32, device=self.device)
769
+
770
+ barrier_count, barrier_epoch = z1(), z1()
771
+ pair_head, producers_done, all_published = z1(), z1(), z1()
772
+ task_head, task_tail = z1(), z1()
773
+ task_ready, task_expert, task_m_tile = ztask(), ztask(), ztask()
774
+ task_slice_begin, task_slice_count, task_valid_rows = ztask(), ztask(), ztask()
775
+ tile_write_count = torch.zeros(physical_tiles, dtype=torch.int32, device=self.device)
776
+ row_counts = torch.zeros(self.experts, dtype=torch.int32, device=self.device)
777
+ expert_write_rows = torch.zeros(self.experts, dtype=torch.int32, device=self.device)
778
+ expert_tile_base = torch.zeros(self.experts + 1, dtype=torch.int32, device=self.device)
779
+ token_map = torch.zeros(rows_padded, dtype=torch.int32, device=self.device)
780
+ token_weights = torch.zeros(rows_padded, dtype=torch.float32, device=self.device)
781
+ output = torch.zeros(m, self.hidden, dtype=torch.bfloat16, device=self.device)
782
+ if self.diagnostic_raw_fc1:
783
+ # Keep the ordinary allocation block exactly unchanged. Only the
784
+ # diagnostic arm fills NaN payload sentinels and zeroes counters.
785
+ intermediate_u32.fill_(-1)
786
+ trace_base = rows_padded * (self.intermediate // 4)
787
+ intermediate_u32[trace_base + 32 : trace_base + 64].zero_()
788
+ kernel_output = (
789
+ shared_kernel_output if shared_kernel_output is not None else torch.empty(
790
+ m * self.topk,
791
+ self.hidden,
792
+ dtype=torch.float32 if self.full_coupled else torch.bfloat16,
793
+ device=self.device,
794
+ )
795
+ if self.deterministic_output
796
+ else output
797
+ )
798
+ launch_mac = arm.mac
799
+ if self.grid_policy and self.world_size == 4 and self.mac_override is None:
800
+ from .p8_multirow_scratch import direct_grid_capacity
801
+ launch_mac = direct_grid_capacity(m, arm.mac)
802
+ arm.compiled(
803
+ _gptr(cutlass.BFloat16, x),
804
+ _gptr(cutlass.Int32, flat_ids, 4),
805
+ _gptr(cutlass.Float32, flat_weights, 4),
806
+ _gptr(cutlass.Float4E2M1FN, packed_a),
807
+ _gptr(cutlass.Float8E4M3FN, scale_flat),
808
+ _gptr(cutlass.Uint8, packed_a),
809
+ _gptr(cutlass.Uint8, scale_flat),
810
+ _gptr(cutlass.Uint32, intermediate_u32),
811
+ barrier_count, barrier_epoch, pair_head, producers_done, all_published,
812
+ task_head, task_tail,
813
+ _gptr(cutlass.Int32, task_ready, 4),
814
+ _gptr(cutlass.Int32, task_expert, 4),
815
+ _gptr(cutlass.Int32, task_m_tile, 4),
816
+ _gptr(cutlass.Int32, task_slice_begin, 4),
817
+ _gptr(cutlass.Int32, task_slice_count, 4),
818
+ _gptr(cutlass.Int32, task_valid_rows, 4),
819
+ _gptr(cutlass.Int32, tile_write_count, 4),
820
+ self.w13_dummy,
821
+ _gptr(cutlass.Float8E4M3FN, self.sentinel),
822
+ self.w2_dummy,
823
+ _gptr(cutlass.Float8E4M3FN, self.sentinel),
824
+ _gptr(cutlass.Uint8, self.w13_scale_mx),
825
+ _gptr(cutlass.Uint8, self.w2_scale_mx),
826
+ _gptr(cutlass.Uint8, self.sentinel),
827
+ _gptr(cutlass.Uint8, self.sentinel),
828
+ _gptr(cutlass.Uint32, self.w13_stream),
829
+ _gptr(cutlass.Uint32, self.w13_sfb),
830
+ _gptr(cutlass.Uint32, self.w2_stream),
831
+ _gptr(cutlass.Uint32, self.w2_sfb),
832
+ row_counts, expert_write_rows, expert_tile_base,
833
+ self.ones, self.ones, self.ones, self.ones,
834
+ _gptr(
835
+ cutlass.Float32 if self.full_coupled else cutlass.BFloat16,
836
+ kernel_output,
837
+ ),
838
+ _gptr(cutlass.Int32, token_map, 4),
839
+ _gptr(cutlass.Float32, token_weights, 4),
840
+ m,
841
+ m * self.topk,
842
+ m * self.topk if self.deterministic_output else m,
843
+ rows_padded,
844
+ max_tasks,
845
+ physical_tiles,
846
+ launch_mac,
847
+ current_cuda_stream(),
848
+ _gptr(cutlass.Uint8, self.input_prequant_trace),
849
+ _gptr(cutlass.Float16, self.scale_component_packed),
850
+ )
851
+ if self.deterministic_output and not self.diagnostic_raw_fc1:
852
+ if self.full_coupled:
853
+ reducer = self._compile_full_coupled_reducer()
854
+ reducer(
855
+ _gptr(cutlass.Float32, kernel_output),
856
+ _gptr(cutlass.Float32, flat_weights, 4),
857
+ _gptr(cutlass.BFloat16, output),
858
+ m,
859
+ current_cuda_stream(),
860
+ )
861
+ else:
862
+ _launch_dynamic_topk_sum(
863
+ route_output=kernel_output,
864
+ output=output,
865
+ m=m,
866
+ num_topk=self.topk,
867
+ k=self.hidden,
868
+ stream=current_cuda_stream(),
869
+ )
870
+ if self.debug_capture:
871
+ self.debug_tensors = {
872
+ "packed_a": packed_a, "scale_flat": scale_flat,
873
+ "intermediate_u32": intermediate_u32,
874
+ "route_output": kernel_output,
875
+ "token_map": token_map, "row_counts": row_counts,
876
+ "expert_tile_base": expert_tile_base,
877
+ }
878
+ self.debug_dispatch = {"small_m": small_m, "materialized": materialized,
879
+ "fused_scratch_zero": fused_scratch_zero,
880
+ "fc1_tile_n": self.fc1_tile_n if small_m else 128,
881
+ "tile_m": tile_m}
882
+ if self.diagnostic_raw_fc1:
883
+ self.debug_dispatch["diagnostic_raw_fc1"] = True
884
+ self.debug_tensors["input_prequant_trace"] = (
885
+ self.input_prequant_trace
886
+ )
887
+ return output
runtime/patches/vllm_quant_trellismx.py ADDED
@@ -0,0 +1,195 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # SPDX-FileCopyrightText: Copyright contributors to the vLLM project
3
+ """Opt-in native P8 routed experts over a ModelOpt NVFP4 carrier.
4
+
5
+ The runtime is separately licensed and lazily imported. Dense, attention,
6
+ router, shared-expert and MTP tensors retain the carrier's quantization.
7
+ """
8
+
9
+ from vllm import envs
10
+
11
+ import regex as re
12
+ import torch
13
+
14
+ from vllm.config import get_current_vllm_config
15
+ from vllm.logger import init_logger
16
+ from vllm.model_executor.layers.fused_moe.activation import MoEActivation
17
+ from vllm.model_executor.layers.fused_moe.fused_moe_method_base import (
18
+ FusedMoEMethodBase,
19
+ )
20
+ from vllm.model_executor.layers.quantization.modelopt import ModelOptNvFp4FusedMoE
21
+ from vllm.utils.b12x import B12xWarmupUnit
22
+ from vllm.utils.trellismx import load_overlay, routed_layer
23
+
24
+ logger = init_logger(__name__)
25
+
26
+
27
+ def maybe_trellismx_method(config, layer, prefix):
28
+ directory = envs.VLLM_TRELLISMX_CHECKPOINT
29
+ index = routed_layer(prefix)
30
+ if not directory:
31
+ return None
32
+ text_config = get_current_vllm_config().model_config.hf_text_config
33
+ if text_config.model_type not in ("glm5_next_text", "glm5_next_mtp"):
34
+ raise ValueError(
35
+ "TrellisMX overlay currently requires the GLM5Next text adapter"
36
+ )
37
+ if index is None:
38
+ if re.fullmatch(
39
+ r"(?:model\.language_model|language_model\.model|model)"
40
+ r"\.layers\.45\.(?:mtp_block\.)?mlp\.experts",
41
+ prefix,
42
+ ):
43
+ return None
44
+ raise ValueError(f"Unrecognized TrellisMX routed-expert prefix: {prefix}")
45
+ if index == 45:
46
+ if (45, 0) not in load_overlay(directory).records:
47
+ return None
48
+ # MTP carrier metadata describes the replaced MXFP8 tensors. Use the
49
+ # established NVFP4 allocation ABI before loading native P8 sidecars.
50
+ config = getattr(config, "nvfp4_config", config)
51
+ if getattr(config, "quant_method", None) != "NVFP4":
52
+ raise ValueError("TrellisMX requires the pinned ModelOpt NVFP4 carrier")
53
+ return TrellisMXMoEMethod(config, layer.moe_config, directory, index)
54
+
55
+
56
+ class TrellisMXMoEMethod(ModelOptNvFp4FusedMoE):
57
+ """Keep the carrier's weight-loader ABI; execute routed weights using P8."""
58
+
59
+ def __init__(self, config, moe_config, directory, layer_index):
60
+ # Do not select or compile an NVFP4 expert backend we never execute.
61
+ FusedMoEMethodBase.__init__(self, moe_config)
62
+ self.quant_config = config
63
+ self.use_a16 = False
64
+ self.use_global_sf = False
65
+ parallel = moe_config.moe_parallel_config
66
+ if (
67
+ parallel.tp_size != 4
68
+ or parallel.ep_size != 1
69
+ or moe_config.hidden_dim != 4096
70
+ or moe_config.intermediate_size_per_partition != 512
71
+ or moe_config.num_experts != 288
72
+ or moe_config.experts_per_token != 8
73
+ or moe_config.has_bias
74
+ or moe_config.is_lora_enabled
75
+ or moe_config.activation != MoEActivation.SILU
76
+ or moe_config.in_dtype != torch.bfloat16
77
+ or moe_config.swiglu_limit != 10.0
78
+ ):
79
+ raise ValueError(
80
+ "TrellisMX GLM adapter requires TP4, EP1 and GLM Flash shapes"
81
+ )
82
+ self.layer_index = layer_index
83
+ self.rank = parallel.tp_rank
84
+ self.overlay = load_overlay(directory)
85
+ self.runtime = None
86
+
87
+ @property
88
+ def is_monolithic(self):
89
+ return False
90
+
91
+ @property
92
+ def supports_eplb(self):
93
+ return False
94
+
95
+ def get_fused_moe_quant_config(self, layer):
96
+ return None
97
+
98
+ def process_weights_after_loading(self, layer):
99
+ if self.runtime is not None:
100
+ raise RuntimeError("TrellisMX hot weight replacement is unsupported")
101
+ from b12x.moe._shared.trellismx.p8_native_kernel import P8NativeTPMoE
102
+
103
+ device = layer.w13_weight.device
104
+ if device.type != "cuda" or torch.cuda.get_device_capability(device) != (12, 0):
105
+ raise ValueError("This TrellisMX runtime requires SM120 CUDA")
106
+ sidecar = self.overlay.sidecar(self.layer_index, self.rank)
107
+ record = self.overlay.records[self.layer_index, self.rank]
108
+ runtime = P8NativeTPMoE(
109
+ sidecar,
110
+ device=device,
111
+ tp_rank=self.rank,
112
+ world_size=4,
113
+ layer=self.layer_index,
114
+ expected_design_sha256=record["source_design_sha256"],
115
+ expected_transform_sha256=self.overlay.transform_hash,
116
+ topk=8,
117
+ hidden=4096,
118
+ intermediate=512,
119
+ swiglu_limit=10.0,
120
+ small_m_scheduler=True,
121
+ fc1_tile_n=128,
122
+ fuse_scratch_zero=True,
123
+ prefill_chunk_tokens=0,
124
+ grid_policy=True,
125
+ fc1_warp_quant=False,
126
+ fc1_broadcast_a=True,
127
+ )
128
+ # Release only replaced routed storage, after a successful load. Never
129
+ # touch the runner's router/shared experts or the separately owned MTP.
130
+ released = 0
131
+ for name in (
132
+ "w13_weight",
133
+ "w2_weight",
134
+ "w13_weight_scale",
135
+ "w2_weight_scale",
136
+ "w13_weight_scale_2",
137
+ "w2_weight_scale_2",
138
+ "w13_input_scale",
139
+ "w2_input_scale",
140
+ ):
141
+ value = getattr(layer, name)
142
+ released += value.numel() * value.element_size()
143
+ setattr(
144
+ layer,
145
+ name,
146
+ torch.nn.Parameter(
147
+ torch.empty(0, dtype=value.dtype, device=value.device),
148
+ requires_grad=False,
149
+ ),
150
+ )
151
+ self.runtime = runtime
152
+ layer.b12x_warmup_provider = self
153
+ logger.info(
154
+ "TrellisMX layer=%d rank=%d K%d E4M3/UE8M0-32 "
155
+ "coupled-h512-h128 released_carrier_bytes=%d",
156
+ self.layer_index,
157
+ self.rank,
158
+ record["bits"],
159
+ released,
160
+ )
161
+
162
+ def apply(
163
+ self, layer, x, topk_weights, topk_ids, shared_experts, shared_experts_input
164
+ ):
165
+ if self.runtime is None:
166
+ raise RuntimeError("TrellisMX routed weights were not loaded")
167
+ if x.shape[0] == 0:
168
+ return torch.empty_like(x)
169
+ return self.runtime(x, topk_weights, topk_ids)
170
+
171
+ def apply_monolithic(self, *args, **kwargs):
172
+ raise RuntimeError("TrellisMX routing belongs to the Jovian MoE runner")
173
+
174
+ def get_b12x_warmup_unit(self, layer, token_counts, output_dtype):
175
+ runtime = self.runtime
176
+ if runtime is None:
177
+ raise RuntimeError("TrellisMX warmup before weight load")
178
+
179
+ def compile():
180
+ for tokens in token_counts:
181
+ x = torch.zeros(
182
+ (tokens, 4096), dtype=output_dtype, device=runtime.device
183
+ )
184
+ ids = torch.arange(8, dtype=torch.int32, device=runtime.device)
185
+ ids = ids.expand(tokens, 8).contiguous()
186
+ weights = torch.full((tokens, 8), 0.125, device=runtime.device)
187
+ runtime(x, weights, ids)
188
+
189
+ # Each layer owns compiled descriptors and scratch; do not deduplicate
190
+ # across layers solely because K/shape match.
191
+ return B12xWarmupUnit(
192
+ name="TrellisMX",
193
+ key=(type(self), self.layer_index, self.rank),
194
+ compile=compile,
195
+ )
runtime/patches/vllm_utils_trellismx.py ADDED
@@ -0,0 +1,130 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # SPDX-FileCopyrightText: Copyright contributors to the vLLM project
3
+ """CPU-only validation for the versioned TrellisMX routed overlay."""
4
+
5
+ import hashlib
6
+ import json
7
+ import struct
8
+ from dataclasses import dataclass
9
+ from functools import lru_cache
10
+ from pathlib import Path
11
+
12
+ import regex as re
13
+
14
+ _PREFIX = re.compile(
15
+ r"(?:model\.language_model|language_model\.model|model)"
16
+ r"\.layers\.(\d+)\.(?:mtp_block\.)?mlp\.experts"
17
+ )
18
+
19
+
20
+ def routed_layer(prefix: str) -> int | None:
21
+ match = _PREFIX.fullmatch(prefix)
22
+ if match is None:
23
+ return None
24
+ layer = int(match[1])
25
+ return layer if 3 <= layer <= 45 else None
26
+
27
+
28
+ def sha256(path: Path) -> str:
29
+ digest = hashlib.sha256()
30
+ with path.open("rb") as source:
31
+ for chunk in iter(lambda: source.read(8 * 1024 * 1024), b""):
32
+ digest.update(chunk)
33
+ return digest.hexdigest()
34
+
35
+
36
+ def _local_file(root: Path, relative: str) -> Path:
37
+ path = root / relative
38
+ # Symlinked local checkpoints are supported, but traversal in manifests is not.
39
+ if Path(relative).is_absolute() or ".." in Path(relative).parts:
40
+ raise ValueError(f"Unsafe TrellisMX path: {relative}")
41
+ if not path.is_file():
42
+ raise ValueError(f"Missing TrellisMX artifact: {path}")
43
+ return path
44
+
45
+
46
+ @dataclass(frozen=True)
47
+ class Overlay:
48
+ root: Path
49
+ records: dict
50
+ transform_hash: str
51
+
52
+ def sidecar(self, layer: int, rank: int, *, verify: bool = True) -> Path:
53
+ record = self.records[layer, rank]
54
+ path = _local_file(self.root, record["path"])
55
+ if verify and sha256(path) != record["sha256"]:
56
+ raise ValueError(f"TrellisMX weight hash mismatch: {path}")
57
+ with path.open("rb") as stream:
58
+ length_bytes = stream.read(8)
59
+ if len(length_bytes) != 8:
60
+ raise ValueError(f"Truncated safetensors header: {path}")
61
+ length = struct.unpack("<Q", length_bytes)[0]
62
+ if not 2 <= length <= 16 * 1024 * 1024:
63
+ raise ValueError(f"Invalid safetensors header size: {path}")
64
+ header = json.loads(stream.read(length))
65
+ metadata = header.get("__metadata__", {})
66
+ expected = {
67
+ "schema": "glm53-p8-coupled-h512-h128-tp4-rank.v1",
68
+ "layer": str(layer),
69
+ "rank": str(rank),
70
+ "world_size": "4",
71
+ "bits": str(record["bits"]),
72
+ "alphabet": "e4m3",
73
+ "scale": "ue8m0-k32",
74
+ "law": "procedural-mcg-alpha2",
75
+ "source_design_sha256": record["source_design_sha256"],
76
+ }
77
+ for name, value in expected.items():
78
+ if metadata.get(name) != value:
79
+ raise ValueError(f"TrellisMX {path}: incompatible {name}")
80
+ return path
81
+
82
+
83
+ @lru_cache(maxsize=4)
84
+ def load_overlay(directory: str) -> Overlay:
85
+ root = Path(directory).absolute()
86
+ manifest = json.loads(_local_file(root, "trellismx-manifest.json").read_text())
87
+ if manifest.get("schema") != "trellismx.hf-overlay-release.v1":
88
+ raise ValueError("Unsupported TrellisMX overlay schema")
89
+ if manifest.get("carrier") != {
90
+ "repo_id": "local-inference-lab/GLM-5.3-Flash-NVFP4",
91
+ "revision": "520de24eabf507659eaef7c70f14fd584527facc",
92
+ }:
93
+ raise ValueError("Unsupported TrellisMX carrier identity")
94
+ allocation = manifest.get("allocation", {})
95
+ expected_layers = {int(n) for n in allocation}
96
+ if expected_layers not in (set(range(3, 45)), set(range(3, 46))):
97
+ raise ValueError("GLM TrellisMX requires layers 3..44 and optional MTP45")
98
+ if any(type(bits) is not int or bits not in (4, 5) for bits in allocation.values()):
99
+ raise ValueError("This TrellisMX adapter supports K4/K5 only")
100
+ designs = {
101
+ sha256(_local_file(root, f"design/design-{index}.json")) for index in range(3)
102
+ }
103
+ transform = _local_file(root, "design/transform.json")
104
+ transform_config = json.loads(transform.read_text())
105
+ if (
106
+ transform_config.get("boundary") != "coupled-h512-h128-suh-svh-v1"
107
+ or transform_config.get("sign_draw") != 0
108
+ or transform_config.get("activation") != "silu-cap10"
109
+ ):
110
+ raise ValueError("Unsupported TrellisMX coupled transform")
111
+ records = {}
112
+ for record in manifest.get("files", []):
113
+ key = record["layer"], record["rank"]
114
+ if key in records or key[0] not in expected_layers or key[1] not in range(4):
115
+ raise ValueError("Duplicate or invalid TrellisMX layer/rank")
116
+ expected_path = f"sidecars/p8-layer-{key[0]:03d}-tp4-rank-{key[1]}.safetensors"
117
+ if record["path"] != expected_path:
118
+ raise ValueError("Unexpected TrellisMX sidecar path")
119
+ if record["bits"] != allocation[str(key[0])]:
120
+ raise ValueError("TrellisMX allocation and sidecar rate disagree")
121
+ if record["source_design_sha256"] not in designs:
122
+ raise ValueError("TrellisMX source design outside allowlist")
123
+ if not re.fullmatch(r"[0-9a-f]{64}", record["sha256"]):
124
+ raise ValueError("Invalid TrellisMX weight hash")
125
+ if _local_file(root, record["path"]).stat().st_size != record["bytes"]:
126
+ raise ValueError("TrellisMX sidecar size mismatch")
127
+ records[key] = record
128
+ if set(records) != {(layer, rank) for layer in expected_layers for rank in range(4)}:
129
+ raise ValueError("Incomplete TrellisMX TP4 inventory")
130
+ return Overlay(root, records, sha256(transform))
runtime/serve-codecv2-mtp.sh ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ #!/bin/bash
2
+ set -euo pipefail
3
+ runtime_dir=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)
4
+ checkpoint_dir=$(dirname -- "$runtime_dir")
5
+ python3 "$runtime_dir/install-mtp-runtime.py"
6
+ export MODEL_ROOT=${MODEL_ROOT:-$checkpoint_dir/carrier}
7
+ export VLLM_TRELLISMX_CHECKPOINT=${VLLM_TRELLISMX_CHECKPOINT:-$checkpoint_dir}
8
+ exec /release/serve-rp2.sh "$@"
sidecars/p8-layer-045-tp4-rank-0.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1c699c04c08180eadad2f40a6c371295d1c46618db7029e7b7587e73a93c6dd6
3
+ size 963497720
sidecars/p8-layer-045-tp4-rank-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4ac3d8d98b153f23d65c690cf0e24a98974ba52611642619544ad0311d3277c1
3
+ size 963497720
sidecars/p8-layer-045-tp4-rank-2.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2750ae58989848196200dfa2991f40824ff6b52066f53319afd8dcc113c9bde6
3
+ size 963497720
sidecars/p8-layer-045-tp4-rank-3.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c9071b28f75dfbbb6b16017721ddc185d2eb9ab021d81bdc1ad4d632cb846ddb
3
+ size 963497720
trellismx-manifest.json CHANGED
@@ -1,6 +1,6 @@
1
  {
2
  "schema": "trellismx.hf-overlay-release.v1",
3
- "checkpoint": "codec-v2-uniform-k4",
4
  "packaging": "routed-overlay-requires-pinned-stock-carrier",
5
  "source_manifest_sha256": "dd4b391c015f9738cf7e4ed2c43b57cebc859f703e3502ca76ec3eb98d7397d8",
6
  "carrier": {
@@ -49,20 +49,18 @@
49
  "6": 4,
50
  "7": 4,
51
  "8": 4,
52
- "9": 4
 
53
  },
54
  "payload": {
55
- "bytes_under_identity_files": -151878720,
56
- "coupled_metadata_bytes": 151436544,
57
- "file_bytes": 161867586048,
58
- "identity_k4_file_bytes": 161715707328,
59
- "logical_elements": 304405807104,
60
- "safetensors_header_bytes": 564480,
61
- "same_size_or_smaller": false,
62
- "stored_bpw_including_metadata": 4.253994694462529,
63
  "weight_payload_bpw": 4.25,
64
- "weight_payload_bytes": 161715585024,
65
- "note": "codec-v2 re-encode of every routed layer; see design/design-1.json"
66
  },
67
  "files": [
68
  {
@@ -1576,7 +1574,62 @@
1576
  "rank": 3,
1577
  "bits": 4,
1578
  "source_design_sha256": "82687b2cace99c681d01e84c5bef5cf5874091231072f03755b307f4a69b9c4d"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1579
  }
1580
  ],
1581
- "runtime_image": "verdictai/trellismx@sha256:609a5fc1cd7d994ba32d9c03626c414d315947eb9f13fab474a15bc8dfbe0129"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1582
  }
 
1
  {
2
  "schema": "trellismx.hf-overlay-release.v1",
3
+ "checkpoint": "codec-v2-uniform-k4-mtp45-k4",
4
  "packaging": "routed-overlay-requires-pinned-stock-carrier",
5
  "source_manifest_sha256": "dd4b391c015f9738cf7e4ed2c43b57cebc859f703e3502ca76ec3eb98d7397d8",
6
  "carrier": {
 
49
  "6": 4,
50
  "7": 4,
51
  "8": 4,
52
+ "9": 4,
53
+ "45": 4
54
  },
55
  "payload": {
56
+ "coupled_metadata_bytes": 155042176,
57
+ "file_bytes": 165721576928,
58
+ "logical_elements": 311653564416,
59
+ "safetensors_header_bytes": 578656,
60
+ "stored_bpw_including_metadata": 4.2539947133553015,
 
 
 
61
  "weight_payload_bpw": 4.25,
62
+ "weight_payload_bytes": 165565956096,
63
+ "note": "codec-v2 re-encode of every routed layer; see design/design-1.json; MTP45 K4 included; per-module details in MTP-VERIFY-REPORT.json"
64
  },
65
  "files": [
66
  {
 
1574
  "rank": 3,
1575
  "bits": 4,
1576
  "source_design_sha256": "82687b2cace99c681d01e84c5bef5cf5874091231072f03755b307f4a69b9c4d"
1577
+ },
1578
+ {
1579
+ "layer": 45,
1580
+ "rank": 0,
1581
+ "bits": 4,
1582
+ "path": "sidecars/p8-layer-045-tp4-rank-0.safetensors",
1583
+ "bytes": 963497720,
1584
+ "sha256": "1c699c04c08180eadad2f40a6c371295d1c46618db7029e7b7587e73a93c6dd6",
1585
+ "source_design_sha256": "dc3f7e59c570d09f2dcf93921b4ce060bf36dfdcc1634c6cd615229bbc055d10"
1586
+ },
1587
+ {
1588
+ "layer": 45,
1589
+ "rank": 1,
1590
+ "bits": 4,
1591
+ "path": "sidecars/p8-layer-045-tp4-rank-1.safetensors",
1592
+ "bytes": 963497720,
1593
+ "sha256": "4ac3d8d98b153f23d65c690cf0e24a98974ba52611642619544ad0311d3277c1",
1594
+ "source_design_sha256": "dc3f7e59c570d09f2dcf93921b4ce060bf36dfdcc1634c6cd615229bbc055d10"
1595
+ },
1596
+ {
1597
+ "layer": 45,
1598
+ "rank": 2,
1599
+ "bits": 4,
1600
+ "path": "sidecars/p8-layer-045-tp4-rank-2.safetensors",
1601
+ "bytes": 963497720,
1602
+ "sha256": "2750ae58989848196200dfa2991f40824ff6b52066f53319afd8dcc113c9bde6",
1603
+ "source_design_sha256": "dc3f7e59c570d09f2dcf93921b4ce060bf36dfdcc1634c6cd615229bbc055d10"
1604
+ },
1605
+ {
1606
+ "layer": 45,
1607
+ "rank": 3,
1608
+ "bits": 4,
1609
+ "path": "sidecars/p8-layer-045-tp4-rank-3.safetensors",
1610
+ "bytes": 963497720,
1611
+ "sha256": "c9071b28f75dfbbb6b16017721ddc185d2eb9ab021d81bdc1ad4d632cb846ddb",
1612
+ "source_design_sha256": "dc3f7e59c570d09f2dcf93921b4ce060bf36dfdcc1634c6cd615229bbc055d10"
1613
  }
1614
  ],
1615
+ "runtime_image": "verdictai/trellismx@sha256:609a5fc1cd7d994ba32d9c03626c414d315947eb9f13fab474a15bc8dfbe0129",
1616
+ "mtp": {
1617
+ "layer": 45,
1618
+ "bits": 4,
1619
+ "scale_policy": "signed-unit coupled vectors; CPU torch seed 530045; no prior r27 MTP scales",
1620
+ "native_runtime_patch": "runtime/serve-codecv2-mtp.sh"
1621
+ },
1622
+ "main_model_payload": {
1623
+ "bytes_under_identity_files": -151878720,
1624
+ "coupled_metadata_bytes": 151436544,
1625
+ "file_bytes": 161867586048,
1626
+ "identity_k4_file_bytes": 161715707328,
1627
+ "logical_elements": 304405807104,
1628
+ "safetensors_header_bytes": 564480,
1629
+ "same_size_or_smaller": false,
1630
+ "stored_bpw_including_metadata": 4.253994694462529,
1631
+ "weight_payload_bpw": 4.25,
1632
+ "weight_payload_bytes": 161715585024,
1633
+ "note": "codec-v2 re-encode of every routed layer; see design/design-1.json"
1634
+ }
1635
  }