Add files using upload-large-folder tool
Browse files- MTP-VERIFY-REPORT.json +17 -0
- VERIFY-REPORT.json +72 -54
- carrier/model-mtp-nonexpert.safetensors +3 -0
- carrier/model.safetensors.index.json +0 -0
- design/design-2.json +0 -0
- evidence/build-mtp-checkpoint.py +144 -0
- evidence/encode-mtp-v2.py +91 -0
- evidence/mtp-inputs.json +717 -0
- evidence/native-carrier-mtp-receipt.json +112 -0
- evidence/prepare-mtp-inputs.py +73 -0
- runtime/Dockerfile +6 -0
- runtime/LICENSE-APACHE-2.0 +201 -0
- runtime/install-mtp-runtime.py +26 -0
- runtime/patch-manifest.json +23 -0
- runtime/patches/p8_native_kernel.py +887 -0
- runtime/patches/vllm_quant_trellismx.py +195 -0
- runtime/patches/vllm_utils_trellismx.py +130 -0
- runtime/serve-codecv2-mtp.sh +8 -0
- sidecars/p8-layer-045-tp4-rank-0.safetensors +3 -0
- sidecars/p8-layer-045-tp4-rank-1.safetensors +3 -0
- sidecars/p8-layer-045-tp4-rank-2.safetensors +3 -0
- sidecars/p8-layer-045-tp4-rank-3.safetensors +3 -0
- trellismx-manifest.json +66 -13
MTP-VERIFY-REPORT.json
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"status": "PASS",
|
| 3 |
+
"layer": 45,
|
| 4 |
+
"experts": 288,
|
| 5 |
+
"sidecars": 4,
|
| 6 |
+
"expert_stream_mismatches": 0,
|
| 7 |
+
"codec_closure": "PASS: all 864 projection encodes",
|
| 8 |
+
"coupled_component_validation": "PASS",
|
| 9 |
+
"routed_sidecar_bytes": 3853990880,
|
| 10 |
+
"mean_tw_nmse": {
|
| 11 |
+
"gate_tw_nmse": 0.005752041559620415,
|
| 12 |
+
"up_tw_nmse": 0.005751395422463146,
|
| 13 |
+
"down_tw_nmse": 0.0061286417262067724
|
| 14 |
+
},
|
| 15 |
+
"scale_policy": "signed-unit coupled vectors; CPU torch seed 530045; no prior r27 MTP scales",
|
| 16 |
+
"native_execution_tested": false
|
| 17 |
+
}
|
VERIFY-REPORT.json
CHANGED
|
@@ -1,56 +1,74 @@
|
|
| 1 |
{
|
| 2 |
-
|
| 3 |
-
|
| 4 |
-
|
| 5 |
-
|
| 6 |
-
|
| 7 |
-
|
| 8 |
-
|
| 9 |
-
|
| 10 |
-
|
| 11 |
-
|
| 12 |
-
|
| 13 |
-
|
| 14 |
-
|
| 15 |
-
|
| 16 |
-
|
| 17 |
-
|
| 18 |
-
|
| 19 |
-
|
| 20 |
-
|
| 21 |
-
|
| 22 |
-
|
| 23 |
-
|
| 24 |
-
|
| 25 |
-
|
| 26 |
-
|
| 27 |
-
|
| 28 |
-
|
| 29 |
-
|
| 30 |
-
|
| 31 |
-
|
| 32 |
-
|
| 33 |
-
|
| 34 |
-
|
| 35 |
-
|
| 36 |
-
|
| 37 |
-
|
| 38 |
-
|
| 39 |
-
|
| 40 |
-
|
| 41 |
-
|
| 42 |
-
|
| 43 |
-
|
| 44 |
-
|
| 45 |
-
|
| 46 |
-
|
| 47 |
-
|
| 48 |
-
|
| 49 |
-
|
| 50 |
-
|
| 51 |
-
"
|
| 52 |
-
|
| 53 |
-
|
| 54 |
-
|
| 55 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 56 |
}
|
|
|
|
| 1 |
{
|
| 2 |
+
"checkpoint": "/root/cv2data/out/ckpt-B",
|
| 3 |
+
"runtime_overlay_and_kernel_header_checks": "PASS: 168 main files plus 4 MTP files",
|
| 4 |
+
"expert_stream_mismatches": {
|
| 5 |
+
"3": 0,
|
| 6 |
+
"4": 0,
|
| 7 |
+
"5": 0,
|
| 8 |
+
"6": 0,
|
| 9 |
+
"7": 0,
|
| 10 |
+
"8": 0,
|
| 11 |
+
"9": 0,
|
| 12 |
+
"10": 0,
|
| 13 |
+
"11": 0,
|
| 14 |
+
"12": 0,
|
| 15 |
+
"13": 0,
|
| 16 |
+
"14": 0,
|
| 17 |
+
"15": 0,
|
| 18 |
+
"16": 0,
|
| 19 |
+
"17": 0,
|
| 20 |
+
"18": 0,
|
| 21 |
+
"19": 0,
|
| 22 |
+
"20": 0,
|
| 23 |
+
"21": 0,
|
| 24 |
+
"22": 0,
|
| 25 |
+
"23": 0,
|
| 26 |
+
"24": 0,
|
| 27 |
+
"25": 0,
|
| 28 |
+
"26": 0,
|
| 29 |
+
"27": 0,
|
| 30 |
+
"28": 0,
|
| 31 |
+
"29": 0,
|
| 32 |
+
"30": 0,
|
| 33 |
+
"31": 0,
|
| 34 |
+
"32": 0,
|
| 35 |
+
"33": 0,
|
| 36 |
+
"34": 0,
|
| 37 |
+
"35": 0,
|
| 38 |
+
"36": 0,
|
| 39 |
+
"37": 0,
|
| 40 |
+
"38": 0,
|
| 41 |
+
"39": 0,
|
| 42 |
+
"40": 0,
|
| 43 |
+
"41": 0,
|
| 44 |
+
"42": 0,
|
| 45 |
+
"43": 0,
|
| 46 |
+
"44": 0,
|
| 47 |
+
"45": 0
|
| 48 |
+
},
|
| 49 |
+
"routed_sidecar_bytes": 165721576928,
|
| 50 |
+
"tp2_routed_bytes_per_gpu": 82860788464.0,
|
| 51 |
+
"allocation_counts": {
|
| 52 |
+
"3": 0,
|
| 53 |
+
"4": 43,
|
| 54 |
+
"5": 0
|
| 55 |
+
},
|
| 56 |
+
"status": "PASS",
|
| 57 |
+
"mtp_verification": {
|
| 58 |
+
"status": "PASS",
|
| 59 |
+
"layer": 45,
|
| 60 |
+
"experts": 288,
|
| 61 |
+
"sidecars": 4,
|
| 62 |
+
"expert_stream_mismatches": 0,
|
| 63 |
+
"codec_closure": "PASS: all 864 projection encodes",
|
| 64 |
+
"coupled_component_validation": "PASS",
|
| 65 |
+
"routed_sidecar_bytes": 3853990880,
|
| 66 |
+
"mean_tw_nmse": {
|
| 67 |
+
"gate_tw_nmse": 0.005752041559620415,
|
| 68 |
+
"up_tw_nmse": 0.005751395422463146,
|
| 69 |
+
"down_tw_nmse": 0.0061286417262067724
|
| 70 |
+
},
|
| 71 |
+
"scale_policy": "signed-unit coupled vectors; CPU torch seed 530045; no prior r27 MTP scales",
|
| 72 |
+
"native_execution_tested": false
|
| 73 |
+
}
|
| 74 |
}
|
carrier/model-mtp-nonexpert.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:31fc8804fce4cb1e0207d9fb13c19a892d85b50a7f769e763cb8d9e5b6a2e6ba
|
| 3 |
+
size 369674104
|
carrier/model.safetensors.index.json
CHANGED
|
The diff for this file is too large to render.
See raw diff
|
|
|
design/design-2.json
CHANGED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evidence/build-mtp-checkpoint.py
ADDED
|
@@ -0,0 +1,144 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Assemble and verify MTP45 sidecars and retain only its unchanged carrier tensors."""
|
| 2 |
+
import hashlib
|
| 3 |
+
import json
|
| 4 |
+
import math
|
| 5 |
+
import os
|
| 6 |
+
import shutil
|
| 7 |
+
import time
|
| 8 |
+
from pathlib import Path
|
| 9 |
+
|
| 10 |
+
import torch
|
| 11 |
+
from safetensors import safe_open
|
| 12 |
+
from safetensors.torch import load_file, save_file
|
| 13 |
+
|
| 14 |
+
from build_checkpoint import build_rank, sha256_file, tensor_sha256, CODEC
|
| 15 |
+
from p8_coupled_scales import SCALE_NAMES, validate_coupled_component
|
| 16 |
+
|
| 17 |
+
out = Path(os.environ["CV2_OUT"])
|
| 18 |
+
root = Path(os.environ["CV2_ROOT"])
|
| 19 |
+
while not (out / "status/mtp-encoded.ready").exists():
|
| 20 |
+
time.sleep(10)
|
| 21 |
+
inputs = out / "mtp-inputs"
|
| 22 |
+
input_info = json.loads((inputs / "inputs.json").read_text())
|
| 23 |
+
core_sha = sha256_file(root / "code/codecv2/encode_layer_v2.py")
|
| 24 |
+
wrapper_sha = sha256_file(root / "ops/encode-mtp-v2.py")
|
| 25 |
+
chunks, metrics = {}, []
|
| 26 |
+
for path in sorted((out / "mtp-chunks").glob("*.safetensors")):
|
| 27 |
+
with safe_open(str(path), framework="pt", device="cpu") as src:
|
| 28 |
+
meta = src.metadata()
|
| 29 |
+
assert meta["schema"] == "trellismx-codec-v2-mtp-chunk.v1"
|
| 30 |
+
assert meta["layer"] == "45" and meta["bits"] == "4"
|
| 31 |
+
assert meta["encoder_sha256"] == core_sha and meta["mtp_wrapper_sha256"] == wrapper_sha
|
| 32 |
+
assert meta["mtp_inputs_sha256"] == sha256_file(inputs / "inputs.json")
|
| 33 |
+
assert float(meta["beta"]) == 0.5 and float(meta["percdamp"]) == 0.3
|
| 34 |
+
start, end = map(int, meta["expert_range"].split(":"))
|
| 35 |
+
for expert in range(start, end):
|
| 36 |
+
assert expert not in chunks
|
| 37 |
+
chunks[expert] = {p: (src.get_tensor(f"{expert}.{p}.trellis"), src.get_tensor(f"{expert}.{p}.scale_ue8m0"))
|
| 38 |
+
for p in ("gate", "up", "down")}
|
| 39 |
+
metrics.extend(json.loads(path.with_suffix(".metrics.json").read_text())["rows"])
|
| 40 |
+
assert set(chunks) == set(range(288))
|
| 41 |
+
assert {r["expert"] for r in metrics} == set(range(288)) and len(metrics) == 288
|
| 42 |
+
assert all(math.isfinite(v) for row in metrics for v in row.values())
|
| 43 |
+
dest = out / "mtp-sidecars"
|
| 44 |
+
(dest / "sidecars").mkdir(parents=True, exist_ok=True)
|
| 45 |
+
(dest / "design").mkdir(exist_ok=True)
|
| 46 |
+
design = {"schema": "trellismx-codecv2-mtp-design.v1", "layer": 45, "bits": 4,
|
| 47 |
+
"encoder_core_sha256": core_sha, "mtp_wrapper_sha256": wrapper_sha,
|
| 48 |
+
"beta": 0.5, "percdamp": 0.3, "coupled_sign_draw": 0,
|
| 49 |
+
"scale_policy": input_info["scale_policy"], "scale_sha256": input_info["scale_sha256"],
|
| 50 |
+
"capture_sha256": input_info["source_capture_sha256"],
|
| 51 |
+
"calibration": "64 fit windows; 2047 valid MTP rows each; all routed plus every fourth all-token row",
|
| 52 |
+
"main_model_scales_unchanged": True}
|
| 53 |
+
design_path = dest / "design/design-2.json"
|
| 54 |
+
design_path.write_text(json.dumps(design, indent=2) + "\n")
|
| 55 |
+
design_sha = sha256_file(design_path)
|
| 56 |
+
scale = load_file(str(inputs / "coupled-scales.safetensors"))
|
| 57 |
+
transform_sha = sha256_file(out / "ckpt-B/design/transform.json")
|
| 58 |
+
records, ranks = [], []
|
| 59 |
+
for rank in range(4):
|
| 60 |
+
template = out / "ckpt-B/sidecars" / f"p8-layer-044-tp4-rank-{rank}.safetensors"
|
| 61 |
+
tensors, meta = build_rank(template, chunks, rank, 4, design_sha)
|
| 62 |
+
sl = slice(rank * 512, (rank + 1) * 512)
|
| 63 |
+
tensors["gate_up_suh_fp16"] = scale["gate_up_suh"]
|
| 64 |
+
tensors["down_svh_fp16"] = scale["down_svh"]
|
| 65 |
+
tensors["intermediate_scales_fp16"] = torch.cat([scale[key][:, sl] for key in ("gate_svh", "up_svh", "down_suh")], dim=1).contiguous()
|
| 66 |
+
meta.pop("exl3_scale_source_sha256", None)
|
| 67 |
+
meta.update(layer="45", mtp_scale_source_sha256=input_info["scale_sha256"],
|
| 68 |
+
mtp_scale_policy=input_info["scale_policy"], mtp_capture_sha256=input_info["source_capture_sha256"])
|
| 69 |
+
for name in SCALE_NAMES:
|
| 70 |
+
meta["sha256_" + name] = tensor_sha256(tensors[name])
|
| 71 |
+
validate_coupled_component(meta, {name: tensors[name] for name in SCALE_NAMES},
|
| 72 |
+
layer=45, rank=rank, experts=288, hidden=4096, intermediate=512,
|
| 73 |
+
expected_transform_sha256=transform_sha)
|
| 74 |
+
name = f"sidecars/p8-layer-045-tp4-rank-{rank}.safetensors"
|
| 75 |
+
path = dest / name
|
| 76 |
+
save_file(tensors, str(path), metadata=meta)
|
| 77 |
+
records.append({"layer": 45, "rank": rank, "bits": 4, "path": name,
|
| 78 |
+
"bytes": path.stat().st_size, "sha256": sha256_file(path), "source_design_sha256": design_sha})
|
| 79 |
+
# Read the actual saved artifact for the full stream reassembly check.
|
| 80 |
+
with safe_open(str(path), framework="pt", device="cpu") as src:
|
| 81 |
+
ranks.append({key: src.get_tensor(key) for key in CODEC})
|
| 82 |
+
print(json.dumps({"saved": name, "bytes": path.stat().st_size}), flush=True)
|
| 83 |
+
mismatches = 0
|
| 84 |
+
for expert in range(288):
|
| 85 |
+
actual = {
|
| 86 |
+
"gate": (torch.cat([t["w13_trellis"][0, expert] for t in ranks], 1),
|
| 87 |
+
torch.cat([t["w13_scale_ue8m0"][expert, 512:] for t in ranks], 0)),
|
| 88 |
+
"up": (torch.cat([t["w13_trellis"][1, expert] for t in ranks], 1),
|
| 89 |
+
torch.cat([t["w13_scale_ue8m0"][expert, :512] for t in ranks], 0)),
|
| 90 |
+
"down": (torch.cat([t["w2_trellis"][expert] for t in ranks], 0),
|
| 91 |
+
torch.cat([t["w2_scale_ue8m0"][expert] for t in ranks], 1))}
|
| 92 |
+
mismatches += not all(torch.equal(a, b) for key in actual for a, b in zip(actual[key], chunks[expert][key]))
|
| 93 |
+
assert mismatches == 0
|
| 94 |
+
report = {"status": "PASS", "layer": 45, "experts": 288, "sidecars": 4,
|
| 95 |
+
"expert_stream_mismatches": mismatches, "codec_closure": "PASS: all 864 projection encodes",
|
| 96 |
+
"coupled_component_validation": "PASS", "routed_sidecar_bytes": sum(r["bytes"] for r in records),
|
| 97 |
+
"mean_tw_nmse": {key: sum(r[key] for r in metrics) / 288 for key in ("gate_tw_nmse", "up_tw_nmse", "down_tw_nmse")},
|
| 98 |
+
"scale_policy": input_info["scale_policy"], "native_execution_tested": False}
|
| 99 |
+
(dest / "MTP-VERIFY-REPORT.json").write_text(json.dumps(report, indent=2) + "\n")
|
| 100 |
+
(dest / "mtp-manifest.json").write_text(json.dumps({"files": records, "design_sha256": design_sha}, indent=2) + "\n")
|
| 101 |
+
|
| 102 |
+
# Remove only replaced MTP expert tensors; retain all 25 other MTP tensors bit for bit.
|
| 103 |
+
old_carrier = out / "native-carrier"
|
| 104 |
+
carrier = out / "native-carrier-mtp"
|
| 105 |
+
carrier.mkdir(exist_ok=True)
|
| 106 |
+
base_receipt = json.loads((out / "native-carrier-receipt.json").read_text())
|
| 107 |
+
kept_tensors = {}
|
| 108 |
+
for record in base_receipt["files"]:
|
| 109 |
+
name = record["path"]
|
| 110 |
+
if name.startswith("model-mtp-"):
|
| 111 |
+
with safe_open(str(old_carrier / name), framework="pt", device="cpu") as src:
|
| 112 |
+
for key in src.keys():
|
| 113 |
+
if ".layers.45.mlp.experts." not in key:
|
| 114 |
+
assert key not in kept_tensors
|
| 115 |
+
kept_tensors[key] = src.get_tensor(key)
|
| 116 |
+
elif name != "model.safetensors.index.json":
|
| 117 |
+
if not (carrier / name).exists():
|
| 118 |
+
os.link(old_carrier / name, carrier / name)
|
| 119 |
+
assert len(kept_tensors) == 25
|
| 120 |
+
mtp_file = "model-mtp-nonexpert.safetensors"
|
| 121 |
+
save_file(kept_tensors, str(carrier / mtp_file))
|
| 122 |
+
with safe_open(str(carrier / mtp_file), framework="pt", device="cpu") as check:
|
| 123 |
+
assert set(check.keys()) == set(kept_tensors)
|
| 124 |
+
assert all(torch.equal(check.get_tensor(key), value) for key, value in kept_tensors.items())
|
| 125 |
+
index = json.loads((old_carrier / "model.safetensors.index.json").read_text())
|
| 126 |
+
weight_map = {k: v for k, v in index["weight_map"].items() if ".layers.45.mlp.experts." not in k}
|
| 127 |
+
for key in kept_tensors:
|
| 128 |
+
weight_map[key] = mtp_file
|
| 129 |
+
index["weight_map"] = weight_map
|
| 130 |
+
index["metadata"]["total_size"] -= 7247757312 + 226492416
|
| 131 |
+
(carrier / "model.safetensors.index.json").write_text(json.dumps(index, indent=2) + "\n")
|
| 132 |
+
known = {r["path"]: r for r in base_receipt["files"]}
|
| 133 |
+
new_records = []
|
| 134 |
+
for path in sorted(carrier.iterdir()):
|
| 135 |
+
assert path.is_file()
|
| 136 |
+
digest = known[path.name]["sha256"] if path.name in known and path.name != "model.safetensors.index.json" else sha256_file(path)
|
| 137 |
+
new_records.append({"path": path.name, "bytes": path.stat().st_size, "sha256": digest})
|
| 138 |
+
receipt = {**base_receipt, "files": new_records, "bytes": sum(r["bytes"] for r in new_records),
|
| 139 |
+
"indexed_tensors": len(weight_map), "mtp_expert_tensors_replaced": 1728,
|
| 140 |
+
"mtp_nonexpert_tensors_preserved": 25, "mtp_nonexpert_tensor_equality": "PASS",
|
| 141 |
+
"derivation": "MTP45 FP8 experts and scales replaced by Trellis sidecars; other tensors unchanged"}
|
| 142 |
+
(out / "native-carrier-mtp-receipt.json").write_text(json.dumps(receipt, indent=2) + "\n")
|
| 143 |
+
(out / "status/mtp.ready").write_text(design_sha + "\n")
|
| 144 |
+
print(json.dumps(report), flush=True)
|
evidence/encode-mtp-v2.py
ADDED
|
@@ -0,0 +1,91 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Encode MTP45 with the supplied codec-v2 math and its 2047-row fit captures."""
|
| 2 |
+
import argparse
|
| 3 |
+
import hashlib
|
| 4 |
+
import json
|
| 5 |
+
import os
|
| 6 |
+
import time
|
| 7 |
+
from pathlib import Path
|
| 8 |
+
|
| 9 |
+
import numpy as np
|
| 10 |
+
import torch
|
| 11 |
+
from safetensors.torch import load_file, save_file
|
| 12 |
+
|
| 13 |
+
import encode_layer_v2 as cv2
|
| 14 |
+
from glm53_nvfp4.p8_coupled_scale import CoupledScaleSet, COUPLED_SIGN_DRAW, encode_coupled_scale_weights, coupled_input_carrier, coupled_middle_carrier
|
| 15 |
+
from glm53_nvfp4.shard_index import IndexedCheckpoint
|
| 16 |
+
|
| 17 |
+
ap = argparse.ArgumentParser()
|
| 18 |
+
ap.add_argument("--expert-start", type=int, required=True)
|
| 19 |
+
ap.add_argument("--expert-end", type=int, required=True)
|
| 20 |
+
ap.add_argument("--out", type=Path, required=True)
|
| 21 |
+
args = ap.parse_args()
|
| 22 |
+
assert 0 <= args.expert_start < args.expert_end <= 288
|
| 23 |
+
assert not args.out.exists()
|
| 24 |
+
inputs = Path(os.environ["CV2_OUT"]) / "mtp-inputs"
|
| 25 |
+
info = json.loads((inputs / "inputs.json").read_text())
|
| 26 |
+
assert info["rows"] == 64 * 2047
|
| 27 |
+
hidden = np.memmap(inputs / "hidden.bf16.bin", mode="r", dtype="<u2", shape=(info["rows"], 4096))
|
| 28 |
+
ids = np.memmap(inputs / "topk_ids.u16le.bin", mode="r", dtype="<u2", shape=(info["rows"], 8))
|
| 29 |
+
routes = np.memmap(inputs / "topk_weights.f32le.bin", mode="r", dtype="<f4", shape=(info["rows"], 8))
|
| 30 |
+
scales = load_file(str(inputs / "coupled-scales.safetensors"))
|
| 31 |
+
scales_all = CoupledScaleSet(**scales, source_path=inputs, source_sha256=info["scale_sha256"], source_metadata={}, tensor_hashes={})
|
| 32 |
+
scales_all.validate()
|
| 33 |
+
device = torch.device("cuda:0")
|
| 34 |
+
torch.cuda.set_device(device)
|
| 35 |
+
every = np.asarray([w * 2047 + t for w in range(64) for t in range(0, 2047, 4)])
|
| 36 |
+
hid_every = torch.from_numpy(np.array(hidden[every], copy=True)).view(torch.bfloat16)
|
| 37 |
+
ones_every = torch.ones(len(every))
|
| 38 |
+
source_path = Path(os.environ["CV2_BF16"])
|
| 39 |
+
source = IndexedCheckpoint(source_path, source_path / "model.safetensors.index.json")
|
| 40 |
+
payload, metrics, h_all_in = {}, [], None
|
| 41 |
+
started = time.time()
|
| 42 |
+
for expert in range(args.expert_start, args.expert_end):
|
| 43 |
+
t0 = time.time()
|
| 44 |
+
base = source.expert_prefix(45, expert)
|
| 45 |
+
orig = {p: source.get(f"{base}{p}.weight").to(device).float() for p in cv2.PROJ}
|
| 46 |
+
sc = CoupledScaleSet(gate_up_suh=scales_all.gate_up_suh.to(device),
|
| 47 |
+
down_svh=scales_all.down_svh.to(device),
|
| 48 |
+
gate_svh=scales_all.gate_svh[expert:expert + 1].to(device),
|
| 49 |
+
up_svh=scales_all.up_svh[expert:expert + 1].to(device),
|
| 50 |
+
down_suh=scales_all.down_suh[expert:expert + 1].to(device),
|
| 51 |
+
source_path=inputs, source_sha256=info["scale_sha256"], source_metadata={}, tensor_hashes={})
|
| 52 |
+
tw = encode_coupled_scale_weights(orig["gate_proj"], orig["up_proj"], orig["down_proj"], sc,
|
| 53 |
+
expert=0, intermediate_draw=COUPLED_SIGN_DRAW)
|
| 54 |
+
carrier = lambda h: coupled_input_carrier(h, sc.gate_up_suh, quantize=True)
|
| 55 |
+
row_ids, slots = np.nonzero(ids == expert)
|
| 56 |
+
assert row_ids.size, f"No MTP fit routes for expert {expert}"
|
| 57 |
+
hid_r = torch.from_numpy(np.array(hidden[row_ids], copy=True)).view(torch.bfloat16)
|
| 58 |
+
rt_r = torch.from_numpy(np.array(routes[row_ids, slots], copy=True))
|
| 59 |
+
if h_all_in is None:
|
| 60 |
+
h_all_in = cv2.hessian_chunks(hid_every, ones_every, carrier, device)
|
| 61 |
+
h_in = cv2.mix(cv2.hessian_chunks(hid_r, rt_r, carrier, device), h_all_in, 0.5)
|
| 62 |
+
g_rec, g_tr, g_sc = cv2.encode_v2(tw[0], h_in, 4, 0.3)
|
| 63 |
+
u_rec, u_tr, u_sc = cv2.encode_v2(tw[1], h_in, 4, 0.3)
|
| 64 |
+
mid = lambda h: coupled_middle_carrier(carrier(h), g_rec, u_rec, sc, expert=0,
|
| 65 |
+
intermediate_draw=COUPLED_SIGN_DRAW, quantize=True)
|
| 66 |
+
h_dn = cv2.mix(cv2.hessian_chunks(hid_r, rt_r, mid, device),
|
| 67 |
+
cv2.hessian_chunks(hid_every, ones_every, mid, device), 0.5)
|
| 68 |
+
d_rec, d_tr, d_sc = cv2.encode_v2(tw[2], h_dn, 4, 0.3)
|
| 69 |
+
row = {"expert": expert, "routed_tokens": len(row_ids), "seconds": round(time.time() - t0, 2)}
|
| 70 |
+
for name, rec, tr, codes, target in (("gate", g_rec, g_tr, g_sc, tw[0]),
|
| 71 |
+
("up", u_rec, u_tr, u_sc, tw[1]), ("down", d_rec, d_tr, d_sc, tw[2])):
|
| 72 |
+
payload[f"{expert}.{name}.trellis"] = tr
|
| 73 |
+
payload[f"{expert}.{name}.scale_ue8m0"] = codes
|
| 74 |
+
row[f"{name}_tw_nmse"] = float((rec - target).double().square().sum() / target.double().square().sum())
|
| 75 |
+
assert all(np.isfinite(v) for v in row.values())
|
| 76 |
+
metrics.append(row)
|
| 77 |
+
print(json.dumps(row), flush=True)
|
| 78 |
+
del orig, tw, h_in, h_dn, g_rec, u_rec, d_rec, hid_r, rt_r
|
| 79 |
+
torch.cuda.empty_cache()
|
| 80 |
+
meta = {"schema": "trellismx-codec-v2-mtp-chunk.v1", "layer": "45", "bits": "4",
|
| 81 |
+
"expert_range": f"{args.expert_start}:{args.expert_end}", "beta": "0.5", "percdamp": "0.3",
|
| 82 |
+
"encoder_sha256": hashlib.sha256(Path(cv2.__file__).read_bytes()).hexdigest(),
|
| 83 |
+
"mtp_wrapper_sha256": hashlib.sha256(Path(__file__).read_bytes()).hexdigest(),
|
| 84 |
+
"mtp_inputs_sha256": hashlib.sha256((inputs / "inputs.json").read_bytes()).hexdigest(),
|
| 85 |
+
"scale_policy": info["scale_policy"]}
|
| 86 |
+
args.out.parent.mkdir(parents=True, exist_ok=True)
|
| 87 |
+
tmp = args.out.with_suffix(".partial")
|
| 88 |
+
save_file(payload, str(tmp), metadata=meta)
|
| 89 |
+
tmp.rename(args.out)
|
| 90 |
+
args.out.with_suffix(".metrics.json").write_text(json.dumps({"meta": meta, "rows": metrics, "seconds": time.time() - started}, indent=2) + "\n")
|
| 91 |
+
print(json.dumps({"done": str(args.out), "seconds": time.time() - started}), flush=True)
|
evidence/mtp-inputs.json
ADDED
|
@@ -0,0 +1,717 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"schema": "trellismx-codecv2-mtp-inputs.v1",
|
| 3 |
+
"layer": 45,
|
| 4 |
+
"capture_repo": "brandonmusic/GLM-5.3-Flash-BF16-Teacher-Logits",
|
| 5 |
+
"capture_revision": "95f4fdd94bf29989db2e0d1054e4931f55edb6aa",
|
| 6 |
+
"source_capture_sha256": "50afb51ecb3519599bf8722e7f2fe92e3e0edc96d3e3ba605ed59246055f8ee8",
|
| 7 |
+
"windows": [
|
| 8 |
+
{
|
| 9 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 10 |
+
"document_id": "reap-recall-packed-axis4_reasoning_termination-1",
|
| 11 |
+
"domain": "axis4_reasoning_termination",
|
| 12 |
+
"main_terminal_offset": 6144,
|
| 13 |
+
"role": "fit",
|
| 14 |
+
"rows": 2047,
|
| 15 |
+
"token_ids_sha256": "208eadbaf0f326e410557519bd28398a6cb44027a143fa7d5eb4e104e492c59f",
|
| 16 |
+
"window_id": "fit-0003",
|
| 17 |
+
"window_index": 3
|
| 18 |
+
},
|
| 19 |
+
{
|
| 20 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 21 |
+
"document_id": "reap-recall-packed-axis4_reasoning_termination-1",
|
| 22 |
+
"domain": "axis4_reasoning_termination",
|
| 23 |
+
"main_terminal_offset": 14336,
|
| 24 |
+
"role": "fit",
|
| 25 |
+
"rows": 2047,
|
| 26 |
+
"token_ids_sha256": "b4b12b709b0643763ea88f28a7159fb0d6f67ec403a769d82888097330c588fe",
|
| 27 |
+
"window_id": "fit-0007",
|
| 28 |
+
"window_index": 7
|
| 29 |
+
},
|
| 30 |
+
{
|
| 31 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 32 |
+
"document_id": "reap-recall-packed-axis4_reasoning_termination-1",
|
| 33 |
+
"domain": "axis4_reasoning_termination",
|
| 34 |
+
"main_terminal_offset": 22528,
|
| 35 |
+
"role": "fit",
|
| 36 |
+
"rows": 2047,
|
| 37 |
+
"token_ids_sha256": "03e62368844238497ffcf7a12bd9baa5bd36480ec68d2f8563c9180fd275560f",
|
| 38 |
+
"window_id": "fit-0011",
|
| 39 |
+
"window_index": 11
|
| 40 |
+
},
|
| 41 |
+
{
|
| 42 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 43 |
+
"document_id": "reap-recall-packed-axis4_reasoning_termination-1",
|
| 44 |
+
"domain": "axis4_reasoning_termination",
|
| 45 |
+
"main_terminal_offset": 30720,
|
| 46 |
+
"role": "fit",
|
| 47 |
+
"rows": 2047,
|
| 48 |
+
"token_ids_sha256": "d35fae888ad057534d85e66c7eec53d189eb31f4312ed9f84009501f20584619",
|
| 49 |
+
"window_id": "fit-0015",
|
| 50 |
+
"window_index": 15
|
| 51 |
+
},
|
| 52 |
+
{
|
| 53 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 54 |
+
"document_id": "reap-recall-packed-axis4_reasoning_termination-1",
|
| 55 |
+
"domain": "axis4_reasoning_termination",
|
| 56 |
+
"main_terminal_offset": 38912,
|
| 57 |
+
"role": "fit",
|
| 58 |
+
"rows": 2047,
|
| 59 |
+
"token_ids_sha256": "ff92c88a67993287542987983aa46d9e9a0ab636ff3ff453f0db487cd3654181",
|
| 60 |
+
"window_id": "fit-0019",
|
| 61 |
+
"window_index": 19
|
| 62 |
+
},
|
| 63 |
+
{
|
| 64 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 65 |
+
"document_id": "reap-recall-packed-axis1_general-1",
|
| 66 |
+
"domain": "axis1_general",
|
| 67 |
+
"main_terminal_offset": 40960,
|
| 68 |
+
"role": "fit",
|
| 69 |
+
"rows": 2047,
|
| 70 |
+
"token_ids_sha256": "c0f29007388fc4a03964260f8d9d5bc5e24239db854979e215866eeb4c8db3e9",
|
| 71 |
+
"window_id": "fit-0020",
|
| 72 |
+
"window_index": 20
|
| 73 |
+
},
|
| 74 |
+
{
|
| 75 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 76 |
+
"document_id": "reap-recall-packed-axis3_code_agentic-4",
|
| 77 |
+
"domain": "axis3_code_agentic",
|
| 78 |
+
"main_terminal_offset": 45056,
|
| 79 |
+
"role": "fit",
|
| 80 |
+
"rows": 2047,
|
| 81 |
+
"token_ids_sha256": "0ed0081f3ee1f0117a8b716fb530cbfa6b014d2a550e6a4a32f0e4c403559edd",
|
| 82 |
+
"window_id": "fit-0022",
|
| 83 |
+
"window_index": 22
|
| 84 |
+
},
|
| 85 |
+
{
|
| 86 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 87 |
+
"document_id": "reap-recall-packed-axis4_reasoning_termination-1",
|
| 88 |
+
"domain": "axis4_reasoning_termination",
|
| 89 |
+
"main_terminal_offset": 47104,
|
| 90 |
+
"role": "fit",
|
| 91 |
+
"rows": 2047,
|
| 92 |
+
"token_ids_sha256": "cb81685f194c02bd698d50ed0f5ec197a28dcfdca6d9a37c1af44111d9b5e438",
|
| 93 |
+
"window_id": "fit-0023",
|
| 94 |
+
"window_index": 23
|
| 95 |
+
},
|
| 96 |
+
{
|
| 97 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 98 |
+
"document_id": "reap-recall-packed-axis4_reasoning_termination-1",
|
| 99 |
+
"domain": "axis4_reasoning_termination",
|
| 100 |
+
"main_terminal_offset": 55296,
|
| 101 |
+
"role": "fit",
|
| 102 |
+
"rows": 2047,
|
| 103 |
+
"token_ids_sha256": "2a5e68ea816bcd1ae71eb65a2c2963ad693139fb9b18853cc395772310584346",
|
| 104 |
+
"window_id": "fit-0027",
|
| 105 |
+
"window_index": 27
|
| 106 |
+
},
|
| 107 |
+
{
|
| 108 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 109 |
+
"document_id": "reap-recall-packed-axis4_reasoning_termination-1",
|
| 110 |
+
"domain": "axis4_reasoning_termination",
|
| 111 |
+
"main_terminal_offset": 63488,
|
| 112 |
+
"role": "fit",
|
| 113 |
+
"rows": 2047,
|
| 114 |
+
"token_ids_sha256": "6bcd57c39f9f206255f13684ce2c00367879867f68b405589edd41b5678ac249",
|
| 115 |
+
"window_id": "fit-0031",
|
| 116 |
+
"window_index": 31
|
| 117 |
+
},
|
| 118 |
+
{
|
| 119 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 120 |
+
"document_id": "reap-recall-packed-axis4_reasoning_termination-1",
|
| 121 |
+
"domain": "axis4_reasoning_termination",
|
| 122 |
+
"main_terminal_offset": 71680,
|
| 123 |
+
"role": "fit",
|
| 124 |
+
"rows": 2047,
|
| 125 |
+
"token_ids_sha256": "e8866385831f1d5bfccba77298e19ac414b8b90dd468ef138d979f704675ce5f",
|
| 126 |
+
"window_id": "fit-0035",
|
| 127 |
+
"window_index": 35
|
| 128 |
+
},
|
| 129 |
+
{
|
| 130 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 131 |
+
"document_id": "reap-recall-packed-axis4_reasoning_termination-1",
|
| 132 |
+
"domain": "axis4_reasoning_termination",
|
| 133 |
+
"main_terminal_offset": 79872,
|
| 134 |
+
"role": "fit",
|
| 135 |
+
"rows": 2047,
|
| 136 |
+
"token_ids_sha256": "d2a9535df825df10239ecd7212ec5e0cca0e216e5303c6d7ec77cef63680cc06",
|
| 137 |
+
"window_id": "fit-0039",
|
| 138 |
+
"window_index": 39
|
| 139 |
+
},
|
| 140 |
+
{
|
| 141 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 142 |
+
"document_id": "reap-recall-packed-axis4_reasoning_termination-1",
|
| 143 |
+
"domain": "axis4_reasoning_termination",
|
| 144 |
+
"main_terminal_offset": 88064,
|
| 145 |
+
"role": "fit",
|
| 146 |
+
"rows": 2047,
|
| 147 |
+
"token_ids_sha256": "4388a3c18f9cdc932620967f6fda46146d90f722b8263482354eccdd16177c85",
|
| 148 |
+
"window_id": "fit-0043",
|
| 149 |
+
"window_index": 43
|
| 150 |
+
},
|
| 151 |
+
{
|
| 152 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 153 |
+
"document_id": "reap-recall-packed-axis4_reasoning_termination-1",
|
| 154 |
+
"domain": "axis4_reasoning_termination",
|
| 155 |
+
"main_terminal_offset": 96256,
|
| 156 |
+
"role": "fit",
|
| 157 |
+
"rows": 2047,
|
| 158 |
+
"token_ids_sha256": "0432ef02690e147e5b6d03c38f0207010e830c1344823223141b5ba49e771826",
|
| 159 |
+
"window_id": "fit-0047",
|
| 160 |
+
"window_index": 47
|
| 161 |
+
},
|
| 162 |
+
{
|
| 163 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 164 |
+
"document_id": "reap-recall-packed-axis4_reasoning_termination-1",
|
| 165 |
+
"domain": "axis4_reasoning_termination",
|
| 166 |
+
"main_terminal_offset": 104448,
|
| 167 |
+
"role": "fit",
|
| 168 |
+
"rows": 2047,
|
| 169 |
+
"token_ids_sha256": "5dd6cf9058c192340985bdc02d379b102c194c7e4da43ff209cf5de019b25b8c",
|
| 170 |
+
"window_id": "fit-0051",
|
| 171 |
+
"window_index": 51
|
| 172 |
+
},
|
| 173 |
+
{
|
| 174 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 175 |
+
"document_id": "reap-recall-packed-axis1_general-1",
|
| 176 |
+
"domain": "axis1_general",
|
| 177 |
+
"main_terminal_offset": 106496,
|
| 178 |
+
"role": "fit",
|
| 179 |
+
"rows": 2047,
|
| 180 |
+
"token_ids_sha256": "83b96172a361a5f1a8eac603c4ef5b1e96023cdd07faf388479d0147a8c289cb",
|
| 181 |
+
"window_id": "fit-0052",
|
| 182 |
+
"window_index": 52
|
| 183 |
+
},
|
| 184 |
+
{
|
| 185 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 186 |
+
"document_id": "reap-recall-packed-axis4_reasoning_termination-1",
|
| 187 |
+
"domain": "axis4_reasoning_termination",
|
| 188 |
+
"main_terminal_offset": 112640,
|
| 189 |
+
"role": "fit",
|
| 190 |
+
"rows": 2047,
|
| 191 |
+
"token_ids_sha256": "7472e1d9f1fdeb708fb933d8bf51682e4d498f4cd96b001c86e166936da4c122",
|
| 192 |
+
"window_id": "fit-0055",
|
| 193 |
+
"window_index": 55
|
| 194 |
+
},
|
| 195 |
+
{
|
| 196 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 197 |
+
"document_id": "reap-recall-packed-axis4_reasoning_termination-1",
|
| 198 |
+
"domain": "axis4_reasoning_termination",
|
| 199 |
+
"main_terminal_offset": 120832,
|
| 200 |
+
"role": "fit",
|
| 201 |
+
"rows": 2047,
|
| 202 |
+
"token_ids_sha256": "6455be9da5b596f57b219d3a01f25cc7e38d83f09738b4a041f6ddb72d5a909c",
|
| 203 |
+
"window_id": "fit-0059",
|
| 204 |
+
"window_index": 59
|
| 205 |
+
},
|
| 206 |
+
{
|
| 207 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 208 |
+
"document_id": "reap-recall-packed-axis4_reasoning_termination-1",
|
| 209 |
+
"domain": "axis4_reasoning_termination",
|
| 210 |
+
"main_terminal_offset": 129024,
|
| 211 |
+
"role": "fit",
|
| 212 |
+
"rows": 2047,
|
| 213 |
+
"token_ids_sha256": "f838036e2eb0726a528ea1e5252e58d3340fd57f0a8f655bfb16eba68b74dc3b",
|
| 214 |
+
"window_id": "fit-0063",
|
| 215 |
+
"window_index": 63
|
| 216 |
+
},
|
| 217 |
+
{
|
| 218 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 219 |
+
"document_id": "reap-recall-packed-axis2_legal-1",
|
| 220 |
+
"domain": "axis2_legal",
|
| 221 |
+
"main_terminal_offset": 145408,
|
| 222 |
+
"role": "fit",
|
| 223 |
+
"rows": 2047,
|
| 224 |
+
"token_ids_sha256": "2b9498a4c264be4b3336e94d572e22fedb24fd745537ab519fe52616c7fa8369",
|
| 225 |
+
"window_id": "fit-0071",
|
| 226 |
+
"window_index": 71
|
| 227 |
+
},
|
| 228 |
+
{
|
| 229 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 230 |
+
"document_id": "reap-recall-packed-axis1_general-1",
|
| 231 |
+
"domain": "axis1_general",
|
| 232 |
+
"main_terminal_offset": 161792,
|
| 233 |
+
"role": "fit",
|
| 234 |
+
"rows": 2047,
|
| 235 |
+
"token_ids_sha256": "51ba8b2845d334157b51261bd6f7f8144bda32f0b53ec082356f9f3bffbca0f9",
|
| 236 |
+
"window_id": "fit-0079",
|
| 237 |
+
"window_index": 79
|
| 238 |
+
},
|
| 239 |
+
{
|
| 240 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 241 |
+
"document_id": "reap-recall-packed-axis3_code_agentic-4",
|
| 242 |
+
"domain": "axis3_code_agentic",
|
| 243 |
+
"main_terminal_offset": 190464,
|
| 244 |
+
"role": "fit",
|
| 245 |
+
"rows": 2047,
|
| 246 |
+
"token_ids_sha256": "7675e123eef3e202d31be6af29c815254ae601a85119eab63f124b476af9cba2",
|
| 247 |
+
"window_id": "fit-0093",
|
| 248 |
+
"window_index": 93
|
| 249 |
+
},
|
| 250 |
+
{
|
| 251 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 252 |
+
"document_id": "reap-recall-packed-axis1_general-1",
|
| 253 |
+
"domain": "axis1_general",
|
| 254 |
+
"main_terminal_offset": 235520,
|
| 255 |
+
"role": "fit",
|
| 256 |
+
"rows": 2047,
|
| 257 |
+
"token_ids_sha256": "52c57e4c50a2f724e6b0bb02ebeff3c850a27f0b3eb95db9b33dfd2ecc36fdc6",
|
| 258 |
+
"window_id": "fit-0115",
|
| 259 |
+
"window_index": 115
|
| 260 |
+
},
|
| 261 |
+
{
|
| 262 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 263 |
+
"document_id": "reap-recall-packed-axis2_legal-1",
|
| 264 |
+
"domain": "axis2_legal",
|
| 265 |
+
"main_terminal_offset": 237568,
|
| 266 |
+
"role": "fit",
|
| 267 |
+
"rows": 2047,
|
| 268 |
+
"token_ids_sha256": "fa0c3db2b83b1236825127432a325fd888cf8f5233f1b0f9aa88e3d6c46dd816",
|
| 269 |
+
"window_id": "fit-0116",
|
| 270 |
+
"window_index": 116
|
| 271 |
+
},
|
| 272 |
+
{
|
| 273 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 274 |
+
"document_id": "reap-recall-packed-axis2_legal-1",
|
| 275 |
+
"domain": "axis2_legal",
|
| 276 |
+
"main_terminal_offset": 249856,
|
| 277 |
+
"role": "fit",
|
| 278 |
+
"rows": 2047,
|
| 279 |
+
"token_ids_sha256": "1767d9eceb5d6d223d1a6222dd7e71fec1dd775d8eab9b8b56b39064e7e41957",
|
| 280 |
+
"window_id": "fit-0122",
|
| 281 |
+
"window_index": 122
|
| 282 |
+
},
|
| 283 |
+
{
|
| 284 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 285 |
+
"document_id": "reap-recall-packed-axis2_legal-1",
|
| 286 |
+
"domain": "axis2_legal",
|
| 287 |
+
"main_terminal_offset": 262144,
|
| 288 |
+
"role": "fit",
|
| 289 |
+
"rows": 2047,
|
| 290 |
+
"token_ids_sha256": "5af6ec5c7649361f07f5217411a5ac85b9c5b4b9740967c0b8d99aba570866fc",
|
| 291 |
+
"window_id": "fit-0128",
|
| 292 |
+
"window_index": 128
|
| 293 |
+
},
|
| 294 |
+
{
|
| 295 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 296 |
+
"document_id": "reap-recall-packed-axis2_legal-1",
|
| 297 |
+
"domain": "axis2_legal",
|
| 298 |
+
"main_terminal_offset": 268288,
|
| 299 |
+
"role": "fit",
|
| 300 |
+
"rows": 2047,
|
| 301 |
+
"token_ids_sha256": "5fb1a5379d4fe994a1e86a2e1cf44aa56c52b7e02ad43a05df7c40bf43418e4f",
|
| 302 |
+
"window_id": "fit-0131",
|
| 303 |
+
"window_index": 131
|
| 304 |
+
},
|
| 305 |
+
{
|
| 306 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 307 |
+
"document_id": "reap-recall-packed-axis2_legal-1",
|
| 308 |
+
"domain": "axis2_legal",
|
| 309 |
+
"main_terminal_offset": 274432,
|
| 310 |
+
"role": "fit",
|
| 311 |
+
"rows": 2047,
|
| 312 |
+
"token_ids_sha256": "284ff8b2bdae3627c57a9da5b3e97bf54068681d41c65b9e27659aef73763581",
|
| 313 |
+
"window_id": "fit-0134",
|
| 314 |
+
"window_index": 134
|
| 315 |
+
},
|
| 316 |
+
{
|
| 317 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 318 |
+
"document_id": "reap-recall-packed-axis3_code_agentic-4",
|
| 319 |
+
"domain": "axis3_code_agentic",
|
| 320 |
+
"main_terminal_offset": 282624,
|
| 321 |
+
"role": "fit",
|
| 322 |
+
"rows": 2047,
|
| 323 |
+
"token_ids_sha256": "f8c769d67636da9dfca5d73e57292a34830cfeb60da4b67cba5f2df4b083cbe9",
|
| 324 |
+
"window_id": "fit-0138",
|
| 325 |
+
"window_index": 138
|
| 326 |
+
},
|
| 327 |
+
{
|
| 328 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 329 |
+
"document_id": "reap-recall-packed-axis1_general-1",
|
| 330 |
+
"domain": "axis1_general",
|
| 331 |
+
"main_terminal_offset": 284672,
|
| 332 |
+
"role": "fit",
|
| 333 |
+
"rows": 2047,
|
| 334 |
+
"token_ids_sha256": "639984e8d3351390d01d2f934cbc7d4635a2b281692326080ca0bee7fa82e365",
|
| 335 |
+
"window_id": "fit-0139",
|
| 336 |
+
"window_index": 139
|
| 337 |
+
},
|
| 338 |
+
{
|
| 339 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 340 |
+
"document_id": "reap-recall-packed-axis2_legal-1",
|
| 341 |
+
"domain": "axis2_legal",
|
| 342 |
+
"main_terminal_offset": 292864,
|
| 343 |
+
"role": "fit",
|
| 344 |
+
"rows": 2047,
|
| 345 |
+
"token_ids_sha256": "d98dc54bc9af44ae5f06bffa644c90d68ded7f231ee86daef18228b9104fd018",
|
| 346 |
+
"window_id": "fit-0143",
|
| 347 |
+
"window_index": 143
|
| 348 |
+
},
|
| 349 |
+
{
|
| 350 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 351 |
+
"document_id": "reap-recall-packed-axis3_code_agentic-4",
|
| 352 |
+
"domain": "axis3_code_agentic",
|
| 353 |
+
"main_terminal_offset": 301056,
|
| 354 |
+
"role": "fit",
|
| 355 |
+
"rows": 2047,
|
| 356 |
+
"token_ids_sha256": "a14a85dbd0f53c19dba5c9f952fcecc0bb9ab2ee419b3138278878fbbc46bede",
|
| 357 |
+
"window_id": "fit-0147",
|
| 358 |
+
"window_index": 147
|
| 359 |
+
},
|
| 360 |
+
{
|
| 361 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 362 |
+
"document_id": "reap-recall-packed-axis2_legal-1",
|
| 363 |
+
"domain": "axis2_legal",
|
| 364 |
+
"main_terminal_offset": 317440,
|
| 365 |
+
"role": "fit",
|
| 366 |
+
"rows": 2047,
|
| 367 |
+
"token_ids_sha256": "f18685915b8f8b2c8dbc7c411471d5c6e56bfdd9c3158241ad9cc7adf9bf1a85",
|
| 368 |
+
"window_id": "fit-0155",
|
| 369 |
+
"window_index": 155
|
| 370 |
+
},
|
| 371 |
+
{
|
| 372 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 373 |
+
"document_id": "reap-recall-packed-axis3_code_agentic-4",
|
| 374 |
+
"domain": "axis3_code_agentic",
|
| 375 |
+
"main_terminal_offset": 319488,
|
| 376 |
+
"role": "fit",
|
| 377 |
+
"rows": 2047,
|
| 378 |
+
"token_ids_sha256": "5e9bf81a13d603aeee09856b17cd8085475c8252af6f4376565886cf7c8c824c",
|
| 379 |
+
"window_id": "fit-0156",
|
| 380 |
+
"window_index": 156
|
| 381 |
+
},
|
| 382 |
+
{
|
| 383 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 384 |
+
"document_id": "reap-recall-packed-axis1_general-1",
|
| 385 |
+
"domain": "axis1_general",
|
| 386 |
+
"main_terminal_offset": 333824,
|
| 387 |
+
"role": "fit",
|
| 388 |
+
"rows": 2047,
|
| 389 |
+
"token_ids_sha256": "d3274fddf9feaff8a8460e7cb8dcec6de21d7ee34d8df54c8a14784995bc93ff",
|
| 390 |
+
"window_id": "fit-0163",
|
| 391 |
+
"window_index": 163
|
| 392 |
+
},
|
| 393 |
+
{
|
| 394 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 395 |
+
"document_id": "reap-recall-packed-axis1_general-1",
|
| 396 |
+
"domain": "axis1_general",
|
| 397 |
+
"main_terminal_offset": 352256,
|
| 398 |
+
"role": "fit",
|
| 399 |
+
"rows": 2047,
|
| 400 |
+
"token_ids_sha256": "2f02a7a1854c920b37b555302659fb206af3227ad18c97f98f9d71487cf140f3",
|
| 401 |
+
"window_id": "fit-0172",
|
| 402 |
+
"window_index": 172
|
| 403 |
+
},
|
| 404 |
+
{
|
| 405 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 406 |
+
"document_id": "reap-recall-packed-axis2_legal-1",
|
| 407 |
+
"domain": "axis2_legal",
|
| 408 |
+
"main_terminal_offset": 354304,
|
| 409 |
+
"role": "fit",
|
| 410 |
+
"rows": 2047,
|
| 411 |
+
"token_ids_sha256": "b90385699a4e5ad8511be2f979d8b1eefb4f51f15009d0a9cfea504823f7e804",
|
| 412 |
+
"window_id": "fit-0173",
|
| 413 |
+
"window_index": 173
|
| 414 |
+
},
|
| 415 |
+
{
|
| 416 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 417 |
+
"document_id": "reap-recall-packed-axis3_code_agentic-4",
|
| 418 |
+
"domain": "axis3_code_agentic",
|
| 419 |
+
"main_terminal_offset": 399360,
|
| 420 |
+
"role": "fit",
|
| 421 |
+
"rows": 2047,
|
| 422 |
+
"token_ids_sha256": "e94b7adcc6c9b3ea370d72b3a93e0e587f3f15d5efa7a51dc5dbb851ef008d0b",
|
| 423 |
+
"window_id": "fit-0195",
|
| 424 |
+
"window_index": 195
|
| 425 |
+
},
|
| 426 |
+
{
|
| 427 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 428 |
+
"document_id": "reap-recall-packed-axis2_legal-1",
|
| 429 |
+
"domain": "axis2_legal",
|
| 430 |
+
"main_terminal_offset": 403456,
|
| 431 |
+
"role": "fit",
|
| 432 |
+
"rows": 2047,
|
| 433 |
+
"token_ids_sha256": "34364d64711a7325d1c600a90cb67e75cab6e57a5659cee1f186bdad4f92e9d3",
|
| 434 |
+
"window_id": "fit-0197",
|
| 435 |
+
"window_index": 197
|
| 436 |
+
},
|
| 437 |
+
{
|
| 438 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 439 |
+
"document_id": "reap-recall-packed-axis1_general-1",
|
| 440 |
+
"domain": "axis1_general",
|
| 441 |
+
"main_terminal_offset": 413696,
|
| 442 |
+
"role": "fit",
|
| 443 |
+
"rows": 2047,
|
| 444 |
+
"token_ids_sha256": "11bc2a5757f9ef787e1f21ef06fc6d13c6a174a12486dc8fbb43194775120123",
|
| 445 |
+
"window_id": "fit-0202",
|
| 446 |
+
"window_index": 202
|
| 447 |
+
},
|
| 448 |
+
{
|
| 449 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 450 |
+
"document_id": "reap-recall-packed-axis2_legal-1",
|
| 451 |
+
"domain": "axis2_legal",
|
| 452 |
+
"main_terminal_offset": 421888,
|
| 453 |
+
"role": "fit",
|
| 454 |
+
"rows": 2047,
|
| 455 |
+
"token_ids_sha256": "dd9e2596cf1f732177b22cc3abbc8d14df675f7478cb7273b8b02522516d8f59",
|
| 456 |
+
"window_id": "fit-0206",
|
| 457 |
+
"window_index": 206
|
| 458 |
+
},
|
| 459 |
+
{
|
| 460 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 461 |
+
"document_id": "reap-recall-packed-axis3_code_agentic-4",
|
| 462 |
+
"domain": "axis3_code_agentic",
|
| 463 |
+
"main_terminal_offset": 430080,
|
| 464 |
+
"role": "fit",
|
| 465 |
+
"rows": 2047,
|
| 466 |
+
"token_ids_sha256": "d7508373c2c195b1dff5dc0b35fb171e00348cd735296383336fa461cd678bdc",
|
| 467 |
+
"window_id": "fit-0210",
|
| 468 |
+
"window_index": 210
|
| 469 |
+
},
|
| 470 |
+
{
|
| 471 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 472 |
+
"document_id": "reap-recall-packed-axis2_legal-1",
|
| 473 |
+
"domain": "axis2_legal",
|
| 474 |
+
"main_terminal_offset": 440320,
|
| 475 |
+
"role": "fit",
|
| 476 |
+
"rows": 2047,
|
| 477 |
+
"token_ids_sha256": "81fc07989fe68186d98c90b7807d07520d83c8070821307ad2d69c9d1d90b120",
|
| 478 |
+
"window_id": "fit-0215",
|
| 479 |
+
"window_index": 215
|
| 480 |
+
},
|
| 481 |
+
{
|
| 482 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 483 |
+
"document_id": "reap-recall-packed-axis2_legal-1",
|
| 484 |
+
"domain": "axis2_legal",
|
| 485 |
+
"main_terminal_offset": 458752,
|
| 486 |
+
"role": "fit",
|
| 487 |
+
"rows": 2047,
|
| 488 |
+
"token_ids_sha256": "95c807d7a5aad5ea11c2e22591930669aaca061412d41754cdde15a6a4546900",
|
| 489 |
+
"window_id": "fit-0224",
|
| 490 |
+
"window_index": 224
|
| 491 |
+
},
|
| 492 |
+
{
|
| 493 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 494 |
+
"document_id": "reap-recall-packed-axis2_legal-1",
|
| 495 |
+
"domain": "axis2_legal",
|
| 496 |
+
"main_terminal_offset": 464896,
|
| 497 |
+
"role": "fit",
|
| 498 |
+
"rows": 2047,
|
| 499 |
+
"token_ids_sha256": "4a2158f357933bf460001d22ee4b6fcdd3609f03e0b417d8e04c2ce049dadf51",
|
| 500 |
+
"window_id": "fit-0227",
|
| 501 |
+
"window_index": 227
|
| 502 |
+
},
|
| 503 |
+
{
|
| 504 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 505 |
+
"document_id": "reap-recall-packed-axis3_code_agentic-4",
|
| 506 |
+
"domain": "axis3_code_agentic",
|
| 507 |
+
"main_terminal_offset": 466944,
|
| 508 |
+
"role": "fit",
|
| 509 |
+
"rows": 2047,
|
| 510 |
+
"token_ids_sha256": "1ef5b044392253389dd12a22c1978e706ab6584dbf301d6253d1099a742bbefe",
|
| 511 |
+
"window_id": "fit-0228",
|
| 512 |
+
"window_index": 228
|
| 513 |
+
},
|
| 514 |
+
{
|
| 515 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 516 |
+
"document_id": "reap-recall-packed-axis3_code_agentic-4",
|
| 517 |
+
"domain": "axis3_code_agentic",
|
| 518 |
+
"main_terminal_offset": 485376,
|
| 519 |
+
"role": "fit",
|
| 520 |
+
"rows": 2047,
|
| 521 |
+
"token_ids_sha256": "cb88cfc15311ce056f24e24a627de2fbc91f1eea2ced97785029d5ef29d1e02e",
|
| 522 |
+
"window_id": "fit-0237",
|
| 523 |
+
"window_index": 237
|
| 524 |
+
},
|
| 525 |
+
{
|
| 526 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 527 |
+
"document_id": "reap-recall-packed-axis1_general-1",
|
| 528 |
+
"domain": "axis1_general",
|
| 529 |
+
"main_terminal_offset": 505856,
|
| 530 |
+
"role": "fit",
|
| 531 |
+
"rows": 2047,
|
| 532 |
+
"token_ids_sha256": "a2a532c83873a63fbd1ecd3dc072059389886f2eb24fbe14ecf2f36f7e016f87",
|
| 533 |
+
"window_id": "fit-0247",
|
| 534 |
+
"window_index": 247
|
| 535 |
+
},
|
| 536 |
+
{
|
| 537 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 538 |
+
"document_id": "reap-recall-packed-axis3_code_agentic-4",
|
| 539 |
+
"domain": "axis3_code_agentic",
|
| 540 |
+
"main_terminal_offset": 534528,
|
| 541 |
+
"role": "fit",
|
| 542 |
+
"rows": 2047,
|
| 543 |
+
"token_ids_sha256": "17037b1512bacfd160cff6d58a099ae398cbeed5bc171ab1c8e0f0af8933b19e",
|
| 544 |
+
"window_id": "fit-0261",
|
| 545 |
+
"window_index": 261
|
| 546 |
+
},
|
| 547 |
+
{
|
| 548 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 549 |
+
"document_id": "reap-recall-packed-axis1_general-1",
|
| 550 |
+
"domain": "axis1_general",
|
| 551 |
+
"main_terminal_offset": 542720,
|
| 552 |
+
"role": "fit",
|
| 553 |
+
"rows": 2047,
|
| 554 |
+
"token_ids_sha256": "c64237273b209a945c7c818ebd1d422cb2ab50e4ebd7ef4b8401b242a9029e7b",
|
| 555 |
+
"window_id": "fit-0265",
|
| 556 |
+
"window_index": 265
|
| 557 |
+
},
|
| 558 |
+
{
|
| 559 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 560 |
+
"document_id": "reap-recall-packed-axis2_legal-1",
|
| 561 |
+
"domain": "axis2_legal",
|
| 562 |
+
"main_terminal_offset": 563200,
|
| 563 |
+
"role": "fit",
|
| 564 |
+
"rows": 2047,
|
| 565 |
+
"token_ids_sha256": "1e994c189fbd3d369abf496db5ff48750f71b252f645b09462230218985abdc2",
|
| 566 |
+
"window_id": "fit-0275",
|
| 567 |
+
"window_index": 275
|
| 568 |
+
},
|
| 569 |
+
{
|
| 570 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 571 |
+
"document_id": "reap-recall-packed-axis3_code_agentic-4",
|
| 572 |
+
"domain": "axis3_code_agentic",
|
| 573 |
+
"main_terminal_offset": 577536,
|
| 574 |
+
"role": "fit",
|
| 575 |
+
"rows": 2047,
|
| 576 |
+
"token_ids_sha256": "a3bcd38c2780d32246f3b8b630bad422b4dd444572d648ca21e720cbd4c28128",
|
| 577 |
+
"window_id": "fit-0282",
|
| 578 |
+
"window_index": 282
|
| 579 |
+
},
|
| 580 |
+
{
|
| 581 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 582 |
+
"document_id": "reap-recall-packed-axis1_general-1",
|
| 583 |
+
"domain": "axis1_general",
|
| 584 |
+
"main_terminal_offset": 616448,
|
| 585 |
+
"role": "fit",
|
| 586 |
+
"rows": 2047,
|
| 587 |
+
"token_ids_sha256": "a8fe54b5fc7f4fa5936b330d93d76da5dfdacd9712ef7a6f477600e910a2125b",
|
| 588 |
+
"window_id": "fit-0301",
|
| 589 |
+
"window_index": 301
|
| 590 |
+
},
|
| 591 |
+
{
|
| 592 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 593 |
+
"document_id": "reap-recall-packed-axis1_general-1",
|
| 594 |
+
"domain": "axis1_general",
|
| 595 |
+
"main_terminal_offset": 647168,
|
| 596 |
+
"role": "fit",
|
| 597 |
+
"rows": 2047,
|
| 598 |
+
"token_ids_sha256": "0060f819a176ec0abfa02ad323c9015afd7f5d6d6e06c99d819f16cb8f7a8a25",
|
| 599 |
+
"window_id": "fit-0316",
|
| 600 |
+
"window_index": 316
|
| 601 |
+
},
|
| 602 |
+
{
|
| 603 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 604 |
+
"document_id": "reap-recall-packed-axis2_legal-1",
|
| 605 |
+
"domain": "axis2_legal",
|
| 606 |
+
"main_terminal_offset": 649216,
|
| 607 |
+
"role": "fit",
|
| 608 |
+
"rows": 2047,
|
| 609 |
+
"token_ids_sha256": "f728ace6743429f911cdec7051a6dff7db0b0f5f8f915b42767fac4c95f8d849",
|
| 610 |
+
"window_id": "fit-0317",
|
| 611 |
+
"window_index": 317
|
| 612 |
+
},
|
| 613 |
+
{
|
| 614 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 615 |
+
"document_id": "reap-recall-packed-axis3_code_agentic-4",
|
| 616 |
+
"domain": "axis3_code_agentic",
|
| 617 |
+
"main_terminal_offset": 677888,
|
| 618 |
+
"role": "fit",
|
| 619 |
+
"rows": 2047,
|
| 620 |
+
"token_ids_sha256": "429eb9411d4ec92a26abf72cf302ef72d94f2df9c9011be2e9a91fcaaec7bece",
|
| 621 |
+
"window_id": "fit-0331",
|
| 622 |
+
"window_index": 331
|
| 623 |
+
},
|
| 624 |
+
{
|
| 625 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 626 |
+
"document_id": "reap-recall-packed-axis1_general-1",
|
| 627 |
+
"domain": "axis1_general",
|
| 628 |
+
"main_terminal_offset": 708608,
|
| 629 |
+
"role": "fit",
|
| 630 |
+
"rows": 2047,
|
| 631 |
+
"token_ids_sha256": "5655a824cde5dbc1600eee1f27b4b7cd0494dd1826139e93966984e62b46d789",
|
| 632 |
+
"window_id": "fit-0346",
|
| 633 |
+
"window_index": 346
|
| 634 |
+
},
|
| 635 |
+
{
|
| 636 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 637 |
+
"document_id": "reap-recall-packed-axis1_general-1",
|
| 638 |
+
"domain": "axis1_general",
|
| 639 |
+
"main_terminal_offset": 712704,
|
| 640 |
+
"role": "fit",
|
| 641 |
+
"rows": 2047,
|
| 642 |
+
"token_ids_sha256": "49baabdd793585f57ed18f17f025b09e16bd8eb26a1ab826e2072bb140b80026",
|
| 643 |
+
"window_id": "fit-0348",
|
| 644 |
+
"window_index": 348
|
| 645 |
+
},
|
| 646 |
+
{
|
| 647 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 648 |
+
"document_id": "reap-recall-packed-axis1_general-1",
|
| 649 |
+
"domain": "axis1_general",
|
| 650 |
+
"main_terminal_offset": 737280,
|
| 651 |
+
"role": "fit",
|
| 652 |
+
"rows": 2047,
|
| 653 |
+
"token_ids_sha256": "023afe71006b811629b43bb3d6d3044d120726ed525723961d5ff083ff4c32bc",
|
| 654 |
+
"window_id": "fit-0360",
|
| 655 |
+
"window_index": 360
|
| 656 |
+
},
|
| 657 |
+
{
|
| 658 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 659 |
+
"document_id": "reap-recall-packed-axis3_code_agentic-4",
|
| 660 |
+
"domain": "axis3_code_agentic",
|
| 661 |
+
"main_terminal_offset": 739328,
|
| 662 |
+
"role": "fit",
|
| 663 |
+
"rows": 2047,
|
| 664 |
+
"token_ids_sha256": "44e340f3fcb5a830cbccc7683a48b4691f3709bbaa9bff7fab2c34fdf03c3cf1",
|
| 665 |
+
"window_id": "fit-0361",
|
| 666 |
+
"window_index": 361
|
| 667 |
+
},
|
| 668 |
+
{
|
| 669 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 670 |
+
"document_id": "reap-recall-packed-axis3_code_agentic-4",
|
| 671 |
+
"domain": "axis3_code_agentic",
|
| 672 |
+
"main_terminal_offset": 747520,
|
| 673 |
+
"role": "fit",
|
| 674 |
+
"rows": 2047,
|
| 675 |
+
"token_ids_sha256": "eb638a132cf39bd4e4d14d6a976a1e0d47646ca1da63218c95cdcd66f24ddd9b",
|
| 676 |
+
"window_id": "fit-0365",
|
| 677 |
+
"window_index": 365
|
| 678 |
+
},
|
| 679 |
+
{
|
| 680 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 681 |
+
"document_id": "reap-recall-packed-axis3_code_agentic-4",
|
| 682 |
+
"domain": "axis3_code_agentic",
|
| 683 |
+
"main_terminal_offset": 751616,
|
| 684 |
+
"role": "fit",
|
| 685 |
+
"rows": 2047,
|
| 686 |
+
"token_ids_sha256": "59f18d04f36791b4dbb1a158692a0c6b2b53bf2c7b9ae63a51c3183270688225",
|
| 687 |
+
"window_id": "fit-0367",
|
| 688 |
+
"window_index": 367
|
| 689 |
+
},
|
| 690 |
+
{
|
| 691 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 692 |
+
"document_id": "reap-recall-packed-axis1_general-1",
|
| 693 |
+
"domain": "axis1_general",
|
| 694 |
+
"main_terminal_offset": 765952,
|
| 695 |
+
"role": "fit",
|
| 696 |
+
"rows": 2047,
|
| 697 |
+
"token_ids_sha256": "ea0b1e6680826d54e456803cae763837712748fc73f5477bb414ca708fea55aa",
|
| 698 |
+
"window_id": "fit-0374",
|
| 699 |
+
"window_index": 374
|
| 700 |
+
},
|
| 701 |
+
{
|
| 702 |
+
"attention_mask_sha256": "3f9144983ff37e8e104c28604ca9fc99957529a5c1e1a17904cc38f64b472a86",
|
| 703 |
+
"document_id": "reap-recall-packed-axis3_code_agentic-4",
|
| 704 |
+
"domain": "axis3_code_agentic",
|
| 705 |
+
"main_terminal_offset": 772096,
|
| 706 |
+
"role": "fit",
|
| 707 |
+
"rows": 2047,
|
| 708 |
+
"token_ids_sha256": "75ff1700624bc9c4e5251f3e8afe5b3fc6293199739cceaaca79de0c1ac6b175",
|
| 709 |
+
"window_id": "fit-0377",
|
| 710 |
+
"window_index": 377
|
| 711 |
+
}
|
| 712 |
+
],
|
| 713 |
+
"rows_per_window": 2047,
|
| 714 |
+
"rows": 131008,
|
| 715 |
+
"scale_policy": "signed-unit coupled vectors; CPU torch seed 530045; no prior r27 MTP scales",
|
| 716 |
+
"scale_sha256": "ad170baa8259a6af3f234f154c0b9cca188b8d1f7cf6de0c89788bccbf66827b"
|
| 717 |
+
}
|
evidence/native-carrier-mtp-receipt.json
ADDED
|
@@ -0,0 +1,112 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"repo": "local-inference-lab/GLM-5.3-Flash-NVFP4",
|
| 3 |
+
"revision": "520de24eabf507659eaef7c70f14fd584527facc",
|
| 4 |
+
"source_verification": "PASS",
|
| 5 |
+
"index_header_coverage": "PASS",
|
| 6 |
+
"indexed_tensors": 37906,
|
| 7 |
+
"omitted_replaced_routed_tensors": 108864,
|
| 8 |
+
"bytes": 19388363261,
|
| 9 |
+
"files": [
|
| 10 |
+
{
|
| 11 |
+
"path": "LICENSE",
|
| 12 |
+
"bytes": 1070,
|
| 13 |
+
"sha256": "30b85b6b9659f2e78aa259f8faf5d920a68dee7c9ced3fa6dba1f19f2bc4fca1"
|
| 14 |
+
},
|
| 15 |
+
{
|
| 16 |
+
"path": "README.md",
|
| 17 |
+
"bytes": 7243,
|
| 18 |
+
"sha256": "3f0894b80aefb75c2afbaa0d4fb0f0f0993f14cb9700c6a5077ea6c201637b7b"
|
| 19 |
+
},
|
| 20 |
+
{
|
| 21 |
+
"path": "amax.safetensors",
|
| 22 |
+
"bytes": 9779904,
|
| 23 |
+
"sha256": "db629a0e7cc2b77ea8a4ec560d326ae94c2af561f4fdbd39d8f611818da43d2b"
|
| 24 |
+
},
|
| 25 |
+
{
|
| 26 |
+
"path": "amax_checkpoint.json",
|
| 27 |
+
"bytes": 1137,
|
| 28 |
+
"sha256": "e2ab014bc3354ad679c4057c5890e786a45dd6517b61d049a652ac76ec9cba6c"
|
| 29 |
+
},
|
| 30 |
+
{
|
| 31 |
+
"path": "amax_checkpoint.safetensors",
|
| 32 |
+
"bytes": 9780048,
|
| 33 |
+
"sha256": "d3ef0aa55826c7f2a29c65a1d8b32368a26a527cf36f27c1665bd0bec5d34fa1"
|
| 34 |
+
},
|
| 35 |
+
{
|
| 36 |
+
"path": "chat_template.jinja",
|
| 37 |
+
"bytes": 10644,
|
| 38 |
+
"sha256": "34d5ee66b12fa6446cdae131c352b8f68cd85369e0e6fda115583805fada3891"
|
| 39 |
+
},
|
| 40 |
+
{
|
| 41 |
+
"path": "config.json",
|
| 42 |
+
"bytes": 15761,
|
| 43 |
+
"sha256": "676382abd1e90a6c85f0c8f33d45441ecd45fd514fd7b63ce5610e732d8e4996"
|
| 44 |
+
},
|
| 45 |
+
{
|
| 46 |
+
"path": "generation_config.json",
|
| 47 |
+
"bytes": 2233,
|
| 48 |
+
"sha256": "80475d7d0e56cf2729f1e0e60bfec01bda743f3829bc5c42e14f54fb6d52e612"
|
| 49 |
+
},
|
| 50 |
+
{
|
| 51 |
+
"path": "hf_quant_config.json",
|
| 52 |
+
"bytes": 8052,
|
| 53 |
+
"sha256": "9c084477c9fe496929a15dde8fb795d13b143e32911e613fe7209121b5606ee9"
|
| 54 |
+
},
|
| 55 |
+
{
|
| 56 |
+
"path": "model-hf-nonexpert-00001-of-00004.safetensors",
|
| 57 |
+
"bytes": 5356357736,
|
| 58 |
+
"sha256": "9f7ede71b2213a6962ea3801015e84edd185ddf00e1b8a0aa9ea6288c28014bf"
|
| 59 |
+
},
|
| 60 |
+
{
|
| 61 |
+
"path": "model-hf-nonexpert-00002-of-00004.safetensors",
|
| 62 |
+
"bytes": 5318341144,
|
| 63 |
+
"sha256": "b26a6aeebb246e95fcb5d54f0b6f8dd66fdfeb17fe54652ae835af1b7d64f6ef"
|
| 64 |
+
},
|
| 65 |
+
{
|
| 66 |
+
"path": "model-hf-nonexpert-00003-of-00004.safetensors",
|
| 67 |
+
"bytes": 5343414728,
|
| 68 |
+
"sha256": "0d56d129e4bd3d732968e7c5dbf3fc11e3e23566619842832325c44306cc68a5"
|
| 69 |
+
},
|
| 70 |
+
{
|
| 71 |
+
"path": "model-hf-nonexpert-00004-of-00004.safetensors",
|
| 72 |
+
"bytes": 2951938800,
|
| 73 |
+
"sha256": "43f38b4fe13a7002a5cf2841083caaac5baa65ebf72e7fb4d9c58bf9bff0280a"
|
| 74 |
+
},
|
| 75 |
+
{
|
| 76 |
+
"path": "model-inputscales.safetensors",
|
| 77 |
+
"bytes": 4726664,
|
| 78 |
+
"sha256": "4255779f031450572af8548c610fd9abfe7df89704985d18595c21788593cd05"
|
| 79 |
+
},
|
| 80 |
+
{
|
| 81 |
+
"path": "model-mtp-nonexpert.safetensors",
|
| 82 |
+
"bytes": 369674104,
|
| 83 |
+
"sha256": "31fc8804fce4cb1e0207d9fb13c19a892d85b50a7f769e763cb8d9e5b6a2e6ba"
|
| 84 |
+
},
|
| 85 |
+
{
|
| 86 |
+
"path": "model.safetensors.index.json",
|
| 87 |
+
"bytes": 4084881,
|
| 88 |
+
"sha256": "7edc92ec0cf197039a40235bd4f4906c6ecad85db7295d51479f51d859df310d"
|
| 89 |
+
},
|
| 90 |
+
{
|
| 91 |
+
"path": "processor_config.json",
|
| 92 |
+
"bytes": 909,
|
| 93 |
+
"sha256": "aae38374c94b08cc9b0547c6e64f05b951bd9735cea571c6988f5ed552bed3ed"
|
| 94 |
+
},
|
| 95 |
+
{
|
| 96 |
+
"path": "tokenizer.json",
|
| 97 |
+
"bytes": 20217442,
|
| 98 |
+
"sha256": "19e773648cb4e65de8660ea6365e10acca112d42a854923df93db4a6f333a82d"
|
| 99 |
+
},
|
| 100 |
+
{
|
| 101 |
+
"path": "tokenizer_config.json",
|
| 102 |
+
"bytes": 761,
|
| 103 |
+
"sha256": "98b1271574f41abf89427ae2dda030d94dc9478f0edc5a8bd240db213c6fd5fc"
|
| 104 |
+
}
|
| 105 |
+
],
|
| 106 |
+
"layout": "carrier/ under checkpoint root; MODEL_ROOT points to carrier/",
|
| 107 |
+
"native_execution_tested": false,
|
| 108 |
+
"mtp_expert_tensors_replaced": 1728,
|
| 109 |
+
"mtp_nonexpert_tensors_preserved": 25,
|
| 110 |
+
"mtp_nonexpert_tensor_equality": "PASS",
|
| 111 |
+
"derivation": "MTP45 FP8 experts and scales replaced by Trellis sidecars; other tensors unchanged"
|
| 112 |
+
}
|
evidence/prepare-mtp-inputs.py
ADDED
|
@@ -0,0 +1,73 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Prepare the existing BF16 MTP captures for the owner's new Trellis MTP arm."""
|
| 2 |
+
import hashlib
|
| 3 |
+
import json
|
| 4 |
+
import os
|
| 5 |
+
from pathlib import Path
|
| 6 |
+
|
| 7 |
+
import numpy as np
|
| 8 |
+
import torch
|
| 9 |
+
from huggingface_hub import snapshot_download
|
| 10 |
+
from safetensors.torch import save_file
|
| 11 |
+
|
| 12 |
+
root = Path(os.environ["CV2_ROOT"])
|
| 13 |
+
out = Path(os.environ["CV2_OUT"])
|
| 14 |
+
dest = out / "mtp-inputs"
|
| 15 |
+
dest.mkdir(exist_ok=True)
|
| 16 |
+
repo = "brandonmusic/GLM-5.3-Flash-BF16-Teacher-Logits"
|
| 17 |
+
rev = "95f4fdd94bf29989db2e0d1054e4931f55edb6aa"
|
| 18 |
+
snapshot_download(repo, repo_type="dataset", revision=rev, local_dir=dest / "source",
|
| 19 |
+
allow_patterns=["calibration/mtp45-ep4-full/*"], max_workers=4)
|
| 20 |
+
source = dest / "source/calibration/mtp45-ep4-full"
|
| 21 |
+
manifest = json.loads((source / "capture-manifest.json").read_text())
|
| 22 |
+
receipt = json.loads((source / "capture-receipt.json").read_text())
|
| 23 |
+
assert receipt["complete"] is True
|
| 24 |
+
assert hashlib.sha256((source / "capture-manifest.json").read_bytes()).hexdigest() == receipt["capture_manifest_file_sha256"]
|
| 25 |
+
assert manifest["layer"] == 45 and manifest["geometry"]["experts"] == 288
|
| 26 |
+
assert manifest["model_revision"] == "a6c167b62691b2bac901344b65cb651a70f53e43"
|
| 27 |
+
for info in manifest["files"].values():
|
| 28 |
+
path = source / info["path"]
|
| 29 |
+
assert path.stat().st_size == info["bytes"]
|
| 30 |
+
with path.open("rb") as stream:
|
| 31 |
+
assert hashlib.file_digest(stream, "sha256").hexdigest() == info["sha256"]
|
| 32 |
+
roles = json.loads((root / "data/roles-v5.json").read_text())["roles"]["fit"]
|
| 33 |
+
assert len(roles) == 64
|
| 34 |
+
role_map = {r["id"]: r for r in roles}
|
| 35 |
+
windows = [w for w in manifest["windows"] if w["window_id"] in role_map]
|
| 36 |
+
assert len(windows) == 64
|
| 37 |
+
offset = 0
|
| 38 |
+
offsets = {}
|
| 39 |
+
for window in manifest["windows"]:
|
| 40 |
+
offsets[window["window_id"]] = offset
|
| 41 |
+
offset += window["rows"]
|
| 42 |
+
assert offset == manifest["rows"]
|
| 43 |
+
for window in windows:
|
| 44 |
+
assert window["role"] == "fit" and window["rows"] == 2047
|
| 45 |
+
assert window["token_ids_sha256"] == role_map[window["window_id"]]["input_sha256"]
|
| 46 |
+
streams = [("hidden_bf16", "<u2", 4096, "hidden.bf16.bin"),
|
| 47 |
+
("topk_ids_u16le", "<u2", 8, "topk_ids.u16le.bin"),
|
| 48 |
+
("topk_weights_f32le", "<f4", 8, "topk_weights.f32le.bin")]
|
| 49 |
+
for key, dtype, width, name in streams:
|
| 50 |
+
array = np.memmap(source / manifest["files"][key]["path"], mode="r", dtype=dtype,
|
| 51 |
+
shape=(manifest["rows"], width))
|
| 52 |
+
with (dest / name).open("wb") as stream:
|
| 53 |
+
for window in windows:
|
| 54 |
+
start = offsets[window["window_id"]]
|
| 55 |
+
stream.write(np.asarray(array[start:start + window["rows"]]).tobytes())
|
| 56 |
+
# No prior Trellis/EXL3 MTP scale vectors exist in the supplied r27 overlay.
|
| 57 |
+
# New signed-unit vectors provide an orthogonal coupled H512/H128 transform;
|
| 58 |
+
# the unchanged codec-v2 block-scale refit handles weight magnitudes.
|
| 59 |
+
generator = torch.Generator(device="cpu").manual_seed(530045)
|
| 60 |
+
def signs(shape):
|
| 61 |
+
return torch.randint(0, 2, shape, generator=generator).mul_(2).sub_(1).to(torch.float16).contiguous()
|
| 62 |
+
scales = {"gate_up_suh": signs((4096,)), "down_svh": signs((4096,)),
|
| 63 |
+
"gate_svh": signs((288, 2048)), "up_svh": signs((288, 2048)),
|
| 64 |
+
"down_suh": signs((288, 2048))}
|
| 65 |
+
save_file(scales, str(dest / "coupled-scales.safetensors"))
|
| 66 |
+
info = {"schema": "trellismx-codecv2-mtp-inputs.v1", "layer": 45,
|
| 67 |
+
"capture_repo": repo, "capture_revision": rev, "source_capture_sha256": manifest["capture_sha256"],
|
| 68 |
+
"windows": windows, "rows_per_window": 2047, "rows": 64 * 2047,
|
| 69 |
+
"scale_policy": "signed-unit coupled vectors; CPU torch seed 530045; no prior r27 MTP scales",
|
| 70 |
+
"scale_sha256": hashlib.sha256((dest / "coupled-scales.safetensors").read_bytes()).hexdigest()}
|
| 71 |
+
(dest / "inputs.json").write_text(json.dumps(info, indent=2) + "\n")
|
| 72 |
+
(out / "status/mtp-inputs.ready").write_text(info["scale_sha256"] + "\n")
|
| 73 |
+
print(json.dumps({k: v for k, v in info.items() if k != "windows"}), flush=True)
|
runtime/Dockerfile
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
FROM verdictai/trellismx@sha256:a0392e1c370eb933d87511928ea3aba6d5d63965cfe6eeaf9cc026a667e98ddc
|
| 2 |
+
COPY patch-manifest.json install-mtp-runtime.py /opt/codecv2-mtp/
|
| 3 |
+
COPY patches /opt/codecv2-mtp/patches
|
| 4 |
+
RUN python3 /opt/codecv2-mtp/install-mtp-runtime.py
|
| 5 |
+
ENV MODEL_ROOT=/model/carrier VLLM_TRELLISMX_CHECKPOINT=/model
|
| 6 |
+
ENTRYPOINT ["/bin/bash", "/release/serve-rp2.sh"]
|
runtime/LICENSE-APACHE-2.0
ADDED
|
@@ -0,0 +1,201 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Apache License
|
| 2 |
+
Version 2.0, January 2004
|
| 3 |
+
http://www.apache.org/licenses/
|
| 4 |
+
|
| 5 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 6 |
+
|
| 7 |
+
1. Definitions.
|
| 8 |
+
|
| 9 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 10 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 11 |
+
|
| 12 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 13 |
+
the copyright owner that is granting the License.
|
| 14 |
+
|
| 15 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 16 |
+
other entities that control, are controlled by, or are under common
|
| 17 |
+
control with that entity. For the purposes of this definition,
|
| 18 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 19 |
+
direction or management of such entity, whether by contract or
|
| 20 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 21 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 22 |
+
|
| 23 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 24 |
+
exercising permissions granted by this License.
|
| 25 |
+
|
| 26 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 27 |
+
including but not limited to software source code, documentation
|
| 28 |
+
source, and configuration files.
|
| 29 |
+
|
| 30 |
+
"Object" form shall mean any form resulting from mechanical
|
| 31 |
+
transformation or translation of a Source form, including but
|
| 32 |
+
not limited to compiled object code, generated documentation,
|
| 33 |
+
and conversions to other media types.
|
| 34 |
+
|
| 35 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 36 |
+
Object form, made available under the License, as indicated by a
|
| 37 |
+
copyright notice that is included in or attached to the work
|
| 38 |
+
(an example is provided in the Appendix below).
|
| 39 |
+
|
| 40 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 41 |
+
form, that is based on (or derived from) the Work and for which the
|
| 42 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 43 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 44 |
+
of this License, Derivative Works shall not include works that remain
|
| 45 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 46 |
+
the Work and Derivative Works thereof.
|
| 47 |
+
|
| 48 |
+
"Contribution" shall mean any work of authorship, including
|
| 49 |
+
the original version of the Work and any modifications or additions
|
| 50 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 51 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 52 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 53 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 54 |
+
means any form of electronic, verbal, or written communication sent
|
| 55 |
+
to the Licensor or its representatives, including but not limited to
|
| 56 |
+
communication on electronic mailing lists, source code control systems,
|
| 57 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 58 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 59 |
+
excluding communication that is conspicuously marked or otherwise
|
| 60 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 61 |
+
|
| 62 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 63 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 64 |
+
subsequently incorporated within the Work.
|
| 65 |
+
|
| 66 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 67 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 68 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 69 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 70 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 71 |
+
Work and such Derivative Works in Source or Object form.
|
| 72 |
+
|
| 73 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 74 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 75 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 76 |
+
(except as stated in this section) patent license to make, have made,
|
| 77 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 78 |
+
where such license applies only to those patent claims licensable
|
| 79 |
+
by such Contributor that are necessarily infringed by their
|
| 80 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 81 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 82 |
+
institute patent litigation against any entity (including a
|
| 83 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 84 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 85 |
+
or contributory patent infringement, then any patent licenses
|
| 86 |
+
granted to You under this License for that Work shall terminate
|
| 87 |
+
as of the date such litigation is filed.
|
| 88 |
+
|
| 89 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 90 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 91 |
+
modifications, and in Source or Object form, provided that You
|
| 92 |
+
meet the following conditions:
|
| 93 |
+
|
| 94 |
+
(a) You must give any other recipients of the Work or
|
| 95 |
+
Derivative Works a copy of this License; and
|
| 96 |
+
|
| 97 |
+
(b) You must cause any modified files to carry prominent notices
|
| 98 |
+
stating that You changed the files; and
|
| 99 |
+
|
| 100 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 101 |
+
that You distribute, all copyright, patent, trademark, and
|
| 102 |
+
attribution notices from the Source form of the Work,
|
| 103 |
+
excluding those notices that do not pertain to any part of
|
| 104 |
+
the Derivative Works; and
|
| 105 |
+
|
| 106 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 107 |
+
distribution, then any Derivative Works that You distribute must
|
| 108 |
+
include a readable copy of the attribution notices contained
|
| 109 |
+
within such NOTICE file, excluding those notices that do not
|
| 110 |
+
pertain to any part of the Derivative Works, in at least one
|
| 111 |
+
of the following places: within a NOTICE text file distributed
|
| 112 |
+
as part of the Derivative Works; within the Source form or
|
| 113 |
+
documentation, if provided along with the Derivative Works; or,
|
| 114 |
+
within a display generated by the Derivative Works, if and
|
| 115 |
+
wherever such third-party notices normally appear. The contents
|
| 116 |
+
of the NOTICE file are for informational purposes only and
|
| 117 |
+
do not modify the License. You may add Your own attribution
|
| 118 |
+
notices within Derivative Works that You distribute, alongside
|
| 119 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 120 |
+
that such additional attribution notices cannot be construed
|
| 121 |
+
as modifying the License.
|
| 122 |
+
|
| 123 |
+
You may add Your own copyright statement to Your modifications and
|
| 124 |
+
may provide additional or different license terms and conditions
|
| 125 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 126 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 127 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 128 |
+
the conditions stated in this License.
|
| 129 |
+
|
| 130 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 131 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 132 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 133 |
+
this License, without any additional terms or conditions.
|
| 134 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 135 |
+
the terms of any separate license agreement you may have executed
|
| 136 |
+
with Licensor regarding such Contributions.
|
| 137 |
+
|
| 138 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 139 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 140 |
+
except as required for reasonable and customary use in describing the
|
| 141 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 142 |
+
|
| 143 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 144 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 145 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 146 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 147 |
+
implied, including, without limitation, any warranties or conditions
|
| 148 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 149 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 150 |
+
appropriateness of using or redistributing the Work and assume any
|
| 151 |
+
risks associated with Your exercise of permissions under this License.
|
| 152 |
+
|
| 153 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 154 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 155 |
+
unless required by applicable law (such as deliberate and grossly
|
| 156 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 157 |
+
liable to You for damages, including any direct, indirect, special,
|
| 158 |
+
incidental, or consequential damages of any character arising as a
|
| 159 |
+
result of this License or out of the use or inability to use the
|
| 160 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 161 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 162 |
+
other commercial damages or losses), even if such Contributor
|
| 163 |
+
has been advised of the possibility of such damages.
|
| 164 |
+
|
| 165 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 166 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 167 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 168 |
+
or other liability obligations and/or rights consistent with this
|
| 169 |
+
License. However, in accepting such obligations, You may act only
|
| 170 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 171 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 172 |
+
defend, and hold each Contributor harmless for any liability
|
| 173 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 174 |
+
of your accepting any such warranty or additional liability.
|
| 175 |
+
|
| 176 |
+
END OF TERMS AND CONDITIONS
|
| 177 |
+
|
| 178 |
+
APPENDIX: How to apply the Apache License to your work.
|
| 179 |
+
|
| 180 |
+
To apply the Apache License to your work, attach the following
|
| 181 |
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 182 |
+
replaced with your own identifying information. (Don't include
|
| 183 |
+
the brackets!) The text should be enclosed in the appropriate
|
| 184 |
+
comment syntax for the file format. We also recommend that a
|
| 185 |
+
file or class name and description of purpose be included on the
|
| 186 |
+
same "printed page" as the copyright notice for easier
|
| 187 |
+
identification within third-party archives.
|
| 188 |
+
|
| 189 |
+
Copyright [yyyy] [name of copyright owner]
|
| 190 |
+
|
| 191 |
+
Licensed under the Apache License, Version 2.0 (the "License");
|
| 192 |
+
you may not use this file except in compliance with the License.
|
| 193 |
+
You may obtain a copy of the License at
|
| 194 |
+
|
| 195 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 196 |
+
|
| 197 |
+
Unless required by applicable law or agreed to in writing, software
|
| 198 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 199 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 200 |
+
See the License for the specific language governing permissions and
|
| 201 |
+
limitations under the License.
|
runtime/install-mtp-runtime.py
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Install the three pinned MTP loader changes inside the existing RP2 image."""
|
| 2 |
+
import hashlib
|
| 3 |
+
import json
|
| 4 |
+
import os
|
| 5 |
+
from pathlib import Path
|
| 6 |
+
|
| 7 |
+
root = Path(__file__).resolve().parent
|
| 8 |
+
manifest = json.loads((root / "patch-manifest.json").read_text())
|
| 9 |
+
pending = []
|
| 10 |
+
for record in manifest["files"]:
|
| 11 |
+
target = Path(record["target"])
|
| 12 |
+
data = (root / "patches" / record["source"]).read_bytes()
|
| 13 |
+
assert hashlib.sha256(data).hexdigest() == record["patched_sha256"]
|
| 14 |
+
actual = hashlib.sha256(target.read_bytes()).hexdigest()
|
| 15 |
+
if actual == record["patched_sha256"]:
|
| 16 |
+
continue
|
| 17 |
+
if actual != record["original_sha256"]:
|
| 18 |
+
raise RuntimeError(f"Unsupported runtime source at {target}; use {manifest['base_image']}")
|
| 19 |
+
compile(data, str(target), "exec")
|
| 20 |
+
pending.append((target, data))
|
| 21 |
+
for target, data in pending:
|
| 22 |
+
temporary = target.with_suffix(".codecv2-mtp.tmp")
|
| 23 |
+
temporary.write_bytes(data)
|
| 24 |
+
temporary.chmod(target.stat().st_mode & 0o777)
|
| 25 |
+
os.replace(temporary, target)
|
| 26 |
+
print("TrellisMX MTP45 runtime installed; existing main-model kernels preserved", flush=True)
|
runtime/patch-manifest.json
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"base_image": "verdictai/trellismx@sha256:a0392e1c370eb933d87511928ea3aba6d5d63965cfe6eeaf9cc026a667e98ddc",
|
| 3 |
+
"files": [
|
| 4 |
+
{
|
| 5 |
+
"source": "vllm_utils_trellismx.py",
|
| 6 |
+
"target": "/opt/glm53-flash/vllm/vllm/utils/trellismx.py",
|
| 7 |
+
"original_sha256": "27be1af286b3fe6eee1fa11602f9fd87142ffa0e13bb9895a9a96346beb91fa3",
|
| 8 |
+
"patched_sha256": "c228f6c52c8f34cb63d37809cea03a944c1fcdbf013459434f29c95eb43d821f"
|
| 9 |
+
},
|
| 10 |
+
{
|
| 11 |
+
"source": "vllm_quant_trellismx.py",
|
| 12 |
+
"target": "/opt/glm53-flash/vllm/vllm/model_executor/layers/quantization/trellismx.py",
|
| 13 |
+
"original_sha256": "26512cae767be9b0f03a0dfbe2280b8f8778b7889eca146e03fbc6e740a3568e",
|
| 14 |
+
"patched_sha256": "6d849966ed3ea4b5f32d08702d58d20f6b24985de5a65f100ef76ee679ea267c"
|
| 15 |
+
},
|
| 16 |
+
{
|
| 17 |
+
"source": "p8_native_kernel.py",
|
| 18 |
+
"target": "/opt/glm53-flash/b12x/b12x/moe/_shared/trellismx/p8_native_kernel.py",
|
| 19 |
+
"original_sha256": "fc88f5ed2a466bbca8459b83dadfd14d5063a4e19f45626a31ca6f7c52c02078",
|
| 20 |
+
"patched_sha256": "6f3bb3c2c61f1881b2e0420a6606fb0cc3d8c37b7939ad64c15a043bfdb18167"
|
| 21 |
+
}
|
| 22 |
+
]
|
| 23 |
+
}
|
runtime/patches/p8_native_kernel.py
ADDED
|
@@ -0,0 +1,887 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Experimental TP-local P8 procedural-MCG MoE runtime for GLM-5.3.
|
| 2 |
+
|
| 3 |
+
This is a direct device path: a K3, K4 or K5 trellis stream is decoded to E4M3 inside
|
| 4 |
+
the MMA kernel and physical UE8M0/32 scales are consumed by the tensor core.
|
| 5 |
+
It implements the frozen TP4 identity-boundary P8 contract and an opt-in M1
|
| 6 |
+
H128 suh/svh scale component for a GLM routed layer whose sidecar carries the
|
| 7 |
+
matching immutable layer identity.
|
| 8 |
+
"""
|
| 9 |
+
from __future__ import annotations
|
| 10 |
+
|
| 11 |
+
from dataclasses import dataclass
|
| 12 |
+
from pathlib import Path
|
| 13 |
+
import os
|
| 14 |
+
|
| 15 |
+
import cutlass
|
| 16 |
+
import cutlass.cute as cute
|
| 17 |
+
import torch
|
| 18 |
+
from cutlass.base_dsl.compiler import OptLevel
|
| 19 |
+
from cutlass.cute.runtime import make_ptr
|
| 20 |
+
from safetensors import safe_open
|
| 21 |
+
from .p8_coupled_scales import (
|
| 22 |
+
COUPLED_SCHEMA as P8_COUPLED_SCHEMA,
|
| 23 |
+
SCALE_NAMES,
|
| 24 |
+
SCHEMA as P8_SCALE_COMPONENT_SCHEMA,
|
| 25 |
+
validate_coupled_component,
|
| 26 |
+
validate_scale_component,
|
| 27 |
+
)
|
| 28 |
+
from .policy_smallm_schedule import P8SmallMGeometry, p8_small_m_scratch_layout, use_small_m
|
| 29 |
+
from .tile_policy import select_tile
|
| 30 |
+
|
| 31 |
+
from b12x._lib.compiler import KernelCompileSpec, compile as b12x_compile
|
| 32 |
+
from b12x._lib.utils import get_max_active_clusters
|
| 33 |
+
from b12x.moe._shared.kernels.route_hoist_dynamic import MoEDynamicKernelBackend
|
| 34 |
+
from b12x.moe.fused_moe._impl import (
|
| 35 |
+
_DynamicMoEW4A8Launch,
|
| 36 |
+
_e8m0_scale_to_w4a8_sfb_inplace,
|
| 37 |
+
_launch_dynamic_topk_sum,
|
| 38 |
+
current_cuda_stream,
|
| 39 |
+
)
|
| 40 |
+
|
| 41 |
+
|
| 42 |
+
def _gptr(dtype, tensor: torch.Tensor, align: int = 16):
|
| 43 |
+
return make_ptr(
|
| 44 |
+
dtype, tensor.data_ptr(), cute.AddressSpace.gmem, assumed_align=align
|
| 45 |
+
)
|
| 46 |
+
|
| 47 |
+
|
| 48 |
+
def _fake_i32(shape: tuple[int, ...]):
|
| 49 |
+
return cute.runtime.make_fake_compact_tensor(
|
| 50 |
+
cutlass.Int32, shape, assumed_align=4
|
| 51 |
+
)
|
| 52 |
+
|
| 53 |
+
|
| 54 |
+
def _fake_f32(shape: tuple[int, ...]):
|
| 55 |
+
return cute.runtime.make_fake_compact_tensor(
|
| 56 |
+
cutlass.Float32, shape, assumed_align=16
|
| 57 |
+
)
|
| 58 |
+
|
| 59 |
+
|
| 60 |
+
@dataclass
|
| 61 |
+
class _CompiledArm:
|
| 62 |
+
compiled: object
|
| 63 |
+
tile_m: int
|
| 64 |
+
materialized: bool
|
| 65 |
+
mac: int
|
| 66 |
+
|
| 67 |
+
|
| 68 |
+
class P8NativeTPMoE:
|
| 69 |
+
"""Own one TP rank's physical P8 payload and launch compiled MoE kernels."""
|
| 70 |
+
|
| 71 |
+
def __init__(
|
| 72 |
+
self,
|
| 73 |
+
sidecar: Path | tuple[Path, Path],
|
| 74 |
+
*,
|
| 75 |
+
device: torch.device,
|
| 76 |
+
tp_rank: int,
|
| 77 |
+
layer: int = 3,
|
| 78 |
+
expected_design_sha256: str | None = None,
|
| 79 |
+
expected_transform_sha256: str | None = None,
|
| 80 |
+
topk: int = 8,
|
| 81 |
+
hidden: int = 4096,
|
| 82 |
+
intermediate: int = 512,
|
| 83 |
+
swiglu_limit: float = 10.0,
|
| 84 |
+
force_materialized: bool | None = None,
|
| 85 |
+
mac_override: int | None = None,
|
| 86 |
+
deterministic_output: bool = True,
|
| 87 |
+
small_m_scheduler: bool = False,
|
| 88 |
+
fc1_tile_n: int = 128,
|
| 89 |
+
debug_capture: bool = False,
|
| 90 |
+
diagnostic_raw_fc1: bool = False,
|
| 91 |
+
fuse_scratch_zero: bool = False,
|
| 92 |
+
compact_scale_storage: bool = False,
|
| 93 |
+
compact_input_storage: bool = False,
|
| 94 |
+
shared_workspace: bool = False,
|
| 95 |
+
world_size: int = 4,
|
| 96 |
+
tp4_parent_sha256: tuple[str, str] | None = None,
|
| 97 |
+
prefill_chunk_tokens: int = 0,
|
| 98 |
+
grid_policy: bool | None = None,
|
| 99 |
+
grouped_m16: bool = False,
|
| 100 |
+
fuse_grouped_scratch: bool = False,
|
| 101 |
+
tile_major_tasks: bool = False,
|
| 102 |
+
fc1_pipeline_stages: int = 2,
|
| 103 |
+
fc1_warps: int = 4,
|
| 104 |
+
fc1_a_swizzle_rotate: bool = False,
|
| 105 |
+
fc1_warp_quant: bool | None = None,
|
| 106 |
+
fc1_exact_staging: bool = False,
|
| 107 |
+
fc1_broadcast_a: bool | None = None,
|
| 108 |
+
) -> None:
|
| 109 |
+
self.device = torch.device(device)
|
| 110 |
+
# D-x2 (PREREG amendment 2): captured once; joins the compile spec below.
|
| 111 |
+
self.p8_down_remainder = os.environ.get("B12X_P8_DOWN_REMAINDER") == "1"
|
| 112 |
+
self._dx2_planes = 2 if self.p8_down_remainder else 1
|
| 113 |
+
self._dx2_phases = os.environ.get("B12X_P8_DOWN_REMAINDER_PHASES", "both")
|
| 114 |
+
self.p8_dx2_rowpack = os.environ.get("B12X_P8_DOWN_REMAINDER") in ("rp", "rp2")
|
| 115 |
+
self.p8_input_rowpack = os.environ.get("B12X_P8_DOWN_REMAINDER") == "rp2"
|
| 116 |
+
# EPI-PAR: bit-identical parallel small-M FC1 epilogue; joins the compile spec below.
|
| 117 |
+
self.p8_epi_par = os.environ.get("B12X_P8_EPI_PAR") == "1"
|
| 118 |
+
if self.p8_epi_par:
|
| 119 |
+
print(f"P8_EPI_PAR_ACTIVE layer={layer} rank={tp_rank}", flush=True)
|
| 120 |
+
if self.p8_dx2_rowpack:
|
| 121 |
+
print(f"P8_DX2_ROWPACK_ACTIVE layer={layer} rank={tp_rank} input_hop={int(self.p8_input_rowpack)}", flush=True)
|
| 122 |
+
self.grouped_m16 = bool(grouped_m16)
|
| 123 |
+
self.fuse_grouped_scratch = bool(fuse_grouped_scratch)
|
| 124 |
+
self.tile_major_tasks = bool(tile_major_tasks)
|
| 125 |
+
if fc1_pipeline_stages not in (2, 3):
|
| 126 |
+
raise ValueError('FC1 pipeline supports two or three stages')
|
| 127 |
+
self.fc1_pipeline_stages = fc1_pipeline_stages
|
| 128 |
+
if fc1_warps not in (4, 8):
|
| 129 |
+
raise ValueError('FC1 supports four or eight warps')
|
| 130 |
+
self.fc1_warps = fc1_warps
|
| 131 |
+
self.fc1_exact_staging = bool(fc1_exact_staging)
|
| 132 |
+
self.fc1_broadcast_a = (os.environ.get('GLM53_P8_FC1_BROADCAST_A') == '1'
|
| 133 |
+
if fc1_broadcast_a is None else bool(fc1_broadcast_a))
|
| 134 |
+
self.fc1_a_swizzle_rotate = bool(fc1_a_swizzle_rotate)
|
| 135 |
+
self.fc1_warp_quant = (os.environ.get('GLM53_P8_FC1_WARP_QUANT') == '1'
|
| 136 |
+
if fc1_warp_quant is None else bool(fc1_warp_quant))
|
| 137 |
+
self.grid_policy = (os.environ.get('GLM53_P8_GRID_POLICY') == '1'
|
| 138 |
+
if grid_policy is None else bool(grid_policy))
|
| 139 |
+
self.prefill_chunk_tokens = int(prefill_chunk_tokens)
|
| 140 |
+
if self.prefill_chunk_tokens < 0 or self.prefill_chunk_tokens % 64:
|
| 141 |
+
raise ValueError("P8 prefill chunk must be zero or a positive multiple of 64")
|
| 142 |
+
self.compact_scale_storage = bool(compact_scale_storage)
|
| 143 |
+
self.compact_input_storage = bool(compact_input_storage)
|
| 144 |
+
self.shared_workspace = bool(shared_workspace)
|
| 145 |
+
self.tp_rank = int(tp_rank)
|
| 146 |
+
self.world_size = int(world_size)
|
| 147 |
+
if self.world_size != 4 or self.tp_rank not in range(4):
|
| 148 |
+
raise ValueError("P8 native supports TP4 only; TP2 validators are unsupported")
|
| 149 |
+
self.layer = int(layer)
|
| 150 |
+
if not 3 <= self.layer <= 45:
|
| 151 |
+
raise ValueError("P8 native layer must be in GLM routed layers 3..44 or MTP45")
|
| 152 |
+
self.topk = int(topk)
|
| 153 |
+
self.hidden = int(hidden)
|
| 154 |
+
self.intermediate = int(intermediate)
|
| 155 |
+
self.swiglu_limit = float(swiglu_limit)
|
| 156 |
+
self.force_materialized = force_materialized
|
| 157 |
+
self.mac_override = None if mac_override is None else int(mac_override)
|
| 158 |
+
self.deterministic_output = bool(deterministic_output)
|
| 159 |
+
# Explicit developmental opt-in. M2/M3 and prefill retain baseline
|
| 160 |
+
# selection; this is not enabled through a serving environment flag.
|
| 161 |
+
self.small_m_scheduler = bool(small_m_scheduler)
|
| 162 |
+
self.fc1_tile_n = int(fc1_tile_n)
|
| 163 |
+
self.debug_capture = bool(debug_capture)
|
| 164 |
+
self.diagnostic_raw_fc1 = bool(diagnostic_raw_fc1)
|
| 165 |
+
self.fuse_scratch_zero = bool(fuse_scratch_zero)
|
| 166 |
+
self._scratch_layout = (
|
| 167 |
+
p8_small_m_scratch_layout(intermediate=self.intermediate) if self.fuse_scratch_zero else None
|
| 168 |
+
)
|
| 169 |
+
self.debug_tensors = {}
|
| 170 |
+
if self.fc1_tile_n not in (32, 64, 128):
|
| 171 |
+
raise ValueError("FC1 tile N must be 32, 64, or 128")
|
| 172 |
+
if self.fc1_tile_n != 128 and (
|
| 173 |
+
not self.small_m_scheduler or self.swiglu_limit != 10.0
|
| 174 |
+
):
|
| 175 |
+
raise ValueError("Narrow FC1 requires small-M and SwiGLU limit 10")
|
| 176 |
+
if self.small_m_scheduler and (
|
| 177 |
+
not self.deterministic_output or force_materialized is not None
|
| 178 |
+
or (topk, hidden, intermediate) != (8, 4096, 2048 // self.world_size)
|
| 179 |
+
):
|
| 180 |
+
raise ValueError("P8 small-M requires deterministic GLM TP4 and automatic fallback")
|
| 181 |
+
if self.diagnostic_raw_fc1 and (
|
| 182 |
+
not self.debug_capture
|
| 183 |
+
or not self.small_m_scheduler
|
| 184 |
+
or self.fc1_tile_n != 128
|
| 185 |
+
or self.fuse_scratch_zero
|
| 186 |
+
):
|
| 187 |
+
raise ValueError(
|
| 188 |
+
"raw FC1 diagnostic requires debug M1 small-M N128 capture"
|
| 189 |
+
)
|
| 190 |
+
if self.mac_override is not None and self.mac_override <= 0:
|
| 191 |
+
raise ValueError("mac_override must be positive")
|
| 192 |
+
if tp4_parent_sha256 is not None:
|
| 193 |
+
if self.world_size != 2:
|
| 194 |
+
raise ValueError("Parent-pair adapter requires TP2")
|
| 195 |
+
from glm53_nvfp4.p8_tp2_repack import open_tp2_pair
|
| 196 |
+
source = open_tp2_pair(sidecar, tp4_parent_sha256, layer=self.layer, rank=self.tp_rank)
|
| 197 |
+
else:
|
| 198 |
+
source = safe_open(sidecar, framework="pt", device="cpu")
|
| 199 |
+
with source as src:
|
| 200 |
+
metadata = src.metadata() or {}
|
| 201 |
+
schema = metadata.get("schema")
|
| 202 |
+
source_design_sha256 = metadata.get("source_design_sha256")
|
| 203 |
+
bits_text = metadata.get("bits", "")
|
| 204 |
+
if bits_text not in {"3", "4", "5"}:
|
| 205 |
+
raise RuntimeError(f"invalid P8 trellis rate: {bits_text!r}")
|
| 206 |
+
self.trellis_bits = int(bits_text)
|
| 207 |
+
base_required = {
|
| 208 |
+
"layer": str(self.layer),
|
| 209 |
+
"rank": str(self.tp_rank),
|
| 210 |
+
"world_size": str(self.world_size),
|
| 211 |
+
"bits": bits_text,
|
| 212 |
+
"alphabet": "e4m3",
|
| 213 |
+
"scale": "ue8m0-k32",
|
| 214 |
+
"law": "procedural-mcg-alpha2",
|
| 215 |
+
"ldlq": "false",
|
| 216 |
+
}
|
| 217 |
+
identity_schema = schema in {
|
| 218 |
+
"glm53-p8-identity-mcg-tp4-rank.v1",
|
| 219 |
+
"glm53-p8-mcg-tp4-rank.v2",
|
| 220 |
+
}
|
| 221 |
+
scale_component_schema = schema == P8_SCALE_COMPONENT_SCHEMA
|
| 222 |
+
full_coupled_schema = schema == P8_COUPLED_SCHEMA.replace("tp4", f"tp{self.world_size}")
|
| 223 |
+
if self.world_size == 2 and not full_coupled_schema:
|
| 224 |
+
raise RuntimeError("TP2 port only supports the full-coupled schema")
|
| 225 |
+
if (
|
| 226 |
+
not (identity_schema or scale_component_schema or full_coupled_schema)
|
| 227 |
+
or any(metadata.get(key) != value for key, value in base_required.items())
|
| 228 |
+
or (identity_schema and metadata.get("boundary") != "identity")
|
| 229 |
+
):
|
| 230 |
+
raise RuntimeError(f"invalid P8 native sidecar metadata: {metadata}")
|
| 231 |
+
if full_coupled_schema or schema in {
|
| 232 |
+
"glm53-p8-mcg-tp4-rank.v2",
|
| 233 |
+
P8_SCALE_COMPONENT_SCHEMA,
|
| 234 |
+
P8_COUPLED_SCHEMA,
|
| 235 |
+
}:
|
| 236 |
+
if (
|
| 237 |
+
not isinstance(source_design_sha256, str)
|
| 238 |
+
or len(source_design_sha256) != 64
|
| 239 |
+
or any(char not in "0123456789abcdef" for char in source_design_sha256)
|
| 240 |
+
):
|
| 241 |
+
raise RuntimeError("v2 P8 sidecar lacks a valid source design hash")
|
| 242 |
+
if (
|
| 243 |
+
expected_design_sha256 is not None
|
| 244 |
+
and source_design_sha256 != expected_design_sha256
|
| 245 |
+
):
|
| 246 |
+
raise RuntimeError("P8 sidecar does not match the expected design")
|
| 247 |
+
elif expected_design_sha256 is not None:
|
| 248 |
+
raise RuntimeError("historical P8 sidecars cannot satisfy a v2 design pin")
|
| 249 |
+
w13 = src.get_tensor("w13_trellis")
|
| 250 |
+
w2 = src.get_tensor("w2_trellis")
|
| 251 |
+
w13_scale = src.get_tensor("w13_scale_ue8m0")
|
| 252 |
+
w2_scale = src.get_tensor("w2_scale_ue8m0")
|
| 253 |
+
scale_tensors = (
|
| 254 |
+
{name: src.get_tensor(name) for name in SCALE_NAMES}
|
| 255 |
+
if scale_component_schema or full_coupled_schema
|
| 256 |
+
else None
|
| 257 |
+
)
|
| 258 |
+
self.source_design_sha256 = source_design_sha256
|
| 259 |
+
experts = int(w13.shape[1])
|
| 260 |
+
stream_words = 16 * self.trellis_bits
|
| 261 |
+
if tuple(w13.shape) != (
|
| 262 |
+
2, experts, hidden // 16, intermediate // 16, stream_words
|
| 263 |
+
):
|
| 264 |
+
raise RuntimeError(f"unexpected W13 trellis shape {tuple(w13.shape)}")
|
| 265 |
+
if tuple(w2.shape) != (
|
| 266 |
+
experts, intermediate // 16, hidden // 16, stream_words
|
| 267 |
+
):
|
| 268 |
+
raise RuntimeError(f"unexpected W2 trellis shape {tuple(w2.shape)}")
|
| 269 |
+
if tuple(w13_scale.shape) != (experts, 2 * intermediate, hidden // 32):
|
| 270 |
+
raise RuntimeError(f"unexpected W13 scale shape {tuple(w13_scale.shape)}")
|
| 271 |
+
if tuple(w2_scale.shape) != (experts, hidden, intermediate // 32):
|
| 272 |
+
raise RuntimeError(f"unexpected W2 scale shape {tuple(w2_scale.shape)}")
|
| 273 |
+
self.experts = experts
|
| 274 |
+
self.scale_component = None
|
| 275 |
+
self.full_coupled = bool(full_coupled_schema)
|
| 276 |
+
if (self.grouped_m16 or self.fuse_grouped_scratch) and not self.full_coupled:
|
| 277 |
+
raise ValueError("grouped scratch requires the full-coupled kernel owner")
|
| 278 |
+
if self.fuse_grouped_scratch and not self.grouped_m16:
|
| 279 |
+
raise ValueError("fused grouped scratch requires grouped_m16")
|
| 280 |
+
if self.shared_workspace and (not self.full_coupled or self.debug_capture or self.fuse_scratch_zero):
|
| 281 |
+
raise ValueError('shared workspace requires full coupling without retained debug tensors or fused arena')
|
| 282 |
+
if self.compact_scale_storage and not self.full_coupled:
|
| 283 |
+
raise ValueError("compact scales require the external full-coupled owners")
|
| 284 |
+
if self.full_coupled and expected_transform_sha256 is None:
|
| 285 |
+
raise RuntimeError(
|
| 286 |
+
"full-coupled P8 requires an externally pinned encoder transform"
|
| 287 |
+
)
|
| 288 |
+
if scale_tensors is not None:
|
| 289 |
+
validator = (
|
| 290 |
+
validate_coupled_component
|
| 291 |
+
if self.full_coupled
|
| 292 |
+
else validate_scale_component
|
| 293 |
+
)
|
| 294 |
+
# Preserve compatibility with the pinned TP4 validator during
|
| 295 |
+
# isolated storage diagnostics. TP2 requires the ported validator.
|
| 296 |
+
validator_kwargs = {} if self.world_size == 4 else {"world_size": self.world_size}
|
| 297 |
+
if self.full_coupled:
|
| 298 |
+
validator_kwargs["expected_transform_sha256"] = (
|
| 299 |
+
expected_transform_sha256
|
| 300 |
+
)
|
| 301 |
+
self.scale_component = validator(
|
| 302 |
+
metadata, scale_tensors, layer=self.layer, rank=self.tp_rank,
|
| 303 |
+
experts=experts, hidden=hidden, intermediate=intermediate,
|
| 304 |
+
**validator_kwargs,
|
| 305 |
+
)
|
| 306 |
+
if not self.small_m_scheduler or self.fc1_tile_n != 128:
|
| 307 |
+
raise RuntimeError(
|
| 308 |
+
"P8 scale component requires the M1 N128 owner path"
|
| 309 |
+
)
|
| 310 |
+
# The trellis storage is byte-for-byte the same size as the packed
|
| 311 |
+
# E2M1 descriptor carrier expected by the inherited W4A8 launch ABI.
|
| 312 |
+
# Alias it for the descriptor-only arguments instead of allocating a
|
| 313 |
+
# second ~0.9 GiB of unread dummy weights per layer and TP rank. The
|
| 314 |
+
# kernel reads the procedural stream through the uint32 pointers below;
|
| 315 |
+
# it never dereferences the descriptor carrier values.
|
| 316 |
+
w13_stream_storage = w13.to(device=self.device).contiguous()
|
| 317 |
+
w2_stream_storage = w2.to(device=self.device).contiguous()
|
| 318 |
+
self.w13_stream = w13_stream_storage.view(torch.int32).reshape(-1)
|
| 319 |
+
if self.fc1_exact_staging:
|
| 320 |
+
if ((self.world_size, self.experts, self.hidden, self.intermediate, self.fc1_tile_n)
|
| 321 |
+
!= (4, 288, 4096, 512, 128) or not self.full_coupled):
|
| 322 |
+
raise ValueError('exact staging requires full-coupled TP4 E288/H4096/I512 N128')
|
| 323 |
+
expected_words = 2 * 288 * 4096 * 512 * self.trellis_bits // 32
|
| 324 |
+
if self.w13_stream.numel() != expected_words:
|
| 325 |
+
raise ValueError('exact staging FC1 stream extent mismatch')
|
| 326 |
+
self.w2_stream = w2_stream_storage.view(torch.int32).reshape(-1)
|
| 327 |
+
w13_scale = w13_scale.to(device=self.device).contiguous()
|
| 328 |
+
w2_scale = w2_scale.to(device=self.device).contiguous()
|
| 329 |
+
# The monolithic kernel consumes the logical [E, N, K/32] UE8M0
|
| 330 |
+
# plane through the sfb_*_mx ABI slots. The split materialized
|
| 331 |
+
# kernels consume a separately repacked copy through *_sfb_rp.
|
| 332 |
+
# Keep both representations: a one-byte sentinel in the logical slots
|
| 333 |
+
# is an out-of-bounds scale read, not an identity scale.
|
| 334 |
+
self.w13_scale_mx = w13_scale.reshape(-1)
|
| 335 |
+
self.w2_scale_mx = w2_scale.reshape(-1)
|
| 336 |
+
self.w13_sfb = _e8m0_scale_to_w4a8_sfb_inplace(
|
| 337 |
+
w13_scale.clone(),
|
| 338 |
+
weight_E=experts,
|
| 339 |
+
rows=2 * intermediate,
|
| 340 |
+
k_dim=hidden,
|
| 341 |
+
gated_half_rows=intermediate,
|
| 342 |
+
).reshape(-1)
|
| 343 |
+
self.w2_sfb = _e8m0_scale_to_w4a8_sfb_inplace(
|
| 344 |
+
w2_scale.clone(),
|
| 345 |
+
weight_E=experts,
|
| 346 |
+
rows=hidden,
|
| 347 |
+
k_dim=intermediate,
|
| 348 |
+
).reshape(-1)
|
| 349 |
+
if self.compact_scale_storage:
|
| 350 |
+
# Full coupling forces external materialized FC1/FC2 for every M.
|
| 351 |
+
# Logical-scale ABI arguments remain non-null aliases, but are
|
| 352 |
+
# not read by those owners. Never enable for a monolithic arm.
|
| 353 |
+
# This is opt-in pending device closure against separate storage.
|
| 354 |
+
self.w13_scale_mx = self.w13_sfb
|
| 355 |
+
self.w2_scale_mx = self.w2_sfb
|
| 356 |
+
# These are descriptor carriers only; they alias the trellis storage
|
| 357 |
+
# above and therefore add zero payload bytes. Their trailing extent is
|
| 358 |
+
# the PACKED row length, which is bits/8 bytes per weight: hidden // 2
|
| 359 |
+
# only at K4. Deriving it from the stored rate keeps K4 byte-identical
|
| 360 |
+
# while letting K3 and K5 describe their own shorter or longer rows.
|
| 361 |
+
w13_row_bytes = hidden * self.trellis_bits // 8
|
| 362 |
+
w2_row_bytes = intermediate * self.trellis_bits // 8
|
| 363 |
+
w13_dummy_bytes = experts * 2 * intermediate * w13_row_bytes
|
| 364 |
+
w2_dummy_bytes = experts * hidden * w2_row_bytes
|
| 365 |
+
self.w13_dummy = w13_stream_storage.view(torch.uint8).reshape(-1)[
|
| 366 |
+
:w13_dummy_bytes
|
| 367 |
+
].reshape(
|
| 368 |
+
experts, 2 * intermediate, w13_row_bytes
|
| 369 |
+
)
|
| 370 |
+
self.w2_dummy = w2_stream_storage.view(torch.uint8).reshape(-1)[
|
| 371 |
+
:w2_dummy_bytes
|
| 372 |
+
].reshape(
|
| 373 |
+
experts, hidden, w2_row_bytes
|
| 374 |
+
)
|
| 375 |
+
self.sentinel = torch.zeros(1, dtype=torch.uint8, device=self.device)
|
| 376 |
+
self.zero_lut = torch.zeros(1, dtype=torch.uint8, device=self.device)
|
| 377 |
+
# MCG never dereferences the LUT pointer. The diagnostic-only arm
|
| 378 |
+
# reuses that dead ABI slot for exactly 128 FP32 trace values (512 B),
|
| 379 |
+
# initialized to an all-ones NaN sentinel so partial writes fail closed.
|
| 380 |
+
self.input_prequant_trace = (
|
| 381 |
+
torch.full((512,), 0xFF, dtype=torch.uint8, device=self.device)
|
| 382 |
+
if self.diagnostic_raw_fc1
|
| 383 |
+
else self.zero_lut
|
| 384 |
+
)
|
| 385 |
+
self.zero_rotation = torch.zeros(1, dtype=torch.float16, device=self.device)
|
| 386 |
+
self.scale_component_packed = (
|
| 387 |
+
self.scale_component.packed.to(device=self.device)
|
| 388 |
+
if self.scale_component is not None
|
| 389 |
+
else self.zero_rotation
|
| 390 |
+
)
|
| 391 |
+
self.ones = torch.ones(experts, dtype=torch.float32, device=self.device)
|
| 392 |
+
if self.small_m_scheduler and self.experts != 288:
|
| 393 |
+
raise ValueError("P8 small-M requires 288 experts")
|
| 394 |
+
# v11: the small-M owner path is compiled per stored rate (K3/K4/K5);
|
| 395 |
+
# M>1 on a non-K4 layer is served row by row through that same exact
|
| 396 |
+
# kernel because the grouped M64 prefill kernels remain K4-only.
|
| 397 |
+
self._compiled: dict[tuple[bool, bool], _CompiledArm] = {}
|
| 398 |
+
self._coupled_reducer = None
|
| 399 |
+
|
| 400 |
+
def _compile(self, materialized: bool, small_m: bool = False, expected_m: int | None = None) -> _CompiledArm:
|
| 401 |
+
if self.compact_scale_storage and not (self.full_coupled and materialized):
|
| 402 |
+
raise RuntimeError("compact scales cannot enter a monolithic path")
|
| 403 |
+
selected_m = select_tile(expected_m if expected_m is not None else (1 if small_m else 4096))[0]
|
| 404 |
+
cache_key = (materialized, small_m, selected_m)
|
| 405 |
+
cached = self._compiled.get(cache_key)
|
| 406 |
+
if cached is not None:
|
| 407 |
+
return cached
|
| 408 |
+
tile_m = selected_m
|
| 409 |
+
if self.grouped_m16:
|
| 410 |
+
raise RuntimeError("fixed M16 override conflicts with requested tile policy")
|
| 411 |
+
mac = (
|
| 412 |
+
self.mac_override
|
| 413 |
+
if self.mac_override is not None
|
| 414 |
+
else (64 if materialized else int(get_max_active_clusters(1)))
|
| 415 |
+
)
|
| 416 |
+
kernel = MoEDynamicKernelBackend(
|
| 417 |
+
16,
|
| 418 |
+
(tile_m, 128),
|
| 419 |
+
activation="silu",
|
| 420 |
+
quant_recipe="w4a8_trellis",
|
| 421 |
+
w4a8_repacked=True,
|
| 422 |
+
num_topk=self.topk,
|
| 423 |
+
trellis_bits=self.trellis_bits,
|
| 424 |
+
trellis_codebook="mcg",
|
| 425 |
+
trellis_scaled=True,
|
| 426 |
+
trellis_identity_boundary=not self.full_coupled,
|
| 427 |
+
direct_routing=small_m,
|
| 428 |
+
materialize_intermediate=materialized,
|
| 429 |
+
p8_small_m=small_m,
|
| 430 |
+
p8_fc1_tile_n=self.fc1_tile_n if small_m else 128,
|
| 431 |
+
p8_scale_sandwich=self.scale_component is not None,
|
| 432 |
+
p8_full_coupled=self.full_coupled,
|
| 433 |
+
share_input_across_experts=materialized,
|
| 434 |
+
deterministic_output=self.deterministic_output,
|
| 435 |
+
swiglu_limit=self.swiglu_limit,
|
| 436 |
+
)
|
| 437 |
+
if self.p8_down_remainder:
|
| 438 |
+
if not self.full_coupled:
|
| 439 |
+
raise RuntimeError("D-x2 is implemented for the full-coupled P8 paths only")
|
| 440 |
+
if self.fc1_warp_quant:
|
| 441 |
+
raise RuntimeError("D-x2 needs fc1_warp_quant=False: the warp-quant FC1 epilogue has no remainder writer")
|
| 442 |
+
from b12x.moe._shared.kernels.p8_down_remainder import enable_down_remainder
|
| 443 |
+
enable_down_remainder(kernel, self._dx2_phases)
|
| 444 |
+
if self.p8_dx2_rowpack and small_m:
|
| 445 |
+
if not self.full_coupled:
|
| 446 |
+
raise RuntimeError("D-x2-RP is implemented for the full-coupled P8 paths only")
|
| 447 |
+
if self.fc1_warp_quant:
|
| 448 |
+
raise RuntimeError("D-x2-RP/RP2 need fc1_warp_quant=False: the warp-quant FC1 epilogue has no remainder writer")
|
| 449 |
+
if self.tile_major_tasks or self.grouped_m16:
|
| 450 |
+
raise RuntimeError("D-x2-RP/RP2 need one route per direct tile (tile_major_tasks=False, grouped_m16=False)")
|
| 451 |
+
if self.p8_input_rowpack and not self.fc1_broadcast_a:
|
| 452 |
+
raise RuntimeError("D-x2-RP2 needs fc1_broadcast_a=True: otherwise ordinary FC1 staging also writes A row 8")
|
| 453 |
+
from b12x.moe._shared.kernels.p8_down_remainder import enable_rowpack
|
| 454 |
+
enable_rowpack(kernel, input_hop=self.p8_input_rowpack)
|
| 455 |
+
if self.diagnostic_raw_fc1:
|
| 456 |
+
if not (small_m and self.full_coupled):
|
| 457 |
+
raise RuntimeError(
|
| 458 |
+
"raw FC1 diagnostic dispatched outside full-coupled M1"
|
| 459 |
+
)
|
| 460 |
+
from b12x.moe._shared.kernels.p8_h128_fc1 import (
|
| 461 |
+
P8H128FC1RawCaptureKernel,
|
| 462 |
+
)
|
| 463 |
+
|
| 464 |
+
kernel.materialized_phase1_kernel = P8H128FC1RawCaptureKernel()
|
| 465 |
+
kernel.p8_input_prequant_diagnostic = True
|
| 466 |
+
if self.full_coupled:
|
| 467 |
+
# Both M1 and grouped owners share the scale/sign plane geometry.
|
| 468 |
+
kernel.materialized_phase1_kernel.p8_intermediate = self.intermediate
|
| 469 |
+
kernel.materialized_phase2_kernel.p8_intermediate = self.intermediate
|
| 470 |
+
if small_m:
|
| 471 |
+
kernel.materialized_phase1_kernel.p8_tile_major = self.tile_major_tasks
|
| 472 |
+
kernel.materialized_phase2_kernel.p8_tile_major = self.tile_major_tasks
|
| 473 |
+
fc1 = kernel.materialized_phase1_kernel
|
| 474 |
+
fc1.p8_a_swizzle_rotate = self.fc1_a_swizzle_rotate
|
| 475 |
+
fc1.p8_warp_quant = self.fc1_warp_quant
|
| 476 |
+
fc1.p8_epi_par = self.p8_epi_par
|
| 477 |
+
fc1.p8_exact_staging = self.fc1_exact_staging
|
| 478 |
+
# Direct owner invokes _run_task with valid_rows=1. Never
|
| 479 |
+
# apply this specialization to grouped multi-row owners.
|
| 480 |
+
fc1.p8_broadcast_a = self.fc1_broadcast_a
|
| 481 |
+
fc1.num_warps = self.fc1_warps
|
| 482 |
+
fc1.threads_per_cta = 32 * self.fc1_warps
|
| 483 |
+
fc1.p8_n8_per_warp = 16 // self.fc1_warps
|
| 484 |
+
fc1.owned_row_groups = fc1.tile_m // self.fc1_warps
|
| 485 |
+
fc1.p8_pipeline_stages = self.fc1_pipeline_stages
|
| 486 |
+
fc1.shared_bytes = max(fc1.shared_bytes, self.fc1_pipeline_stages * fc1.stage_bytes)
|
| 487 |
+
fc1.shared_words = (fc1.shared_bytes + 3) // 4
|
| 488 |
+
fc1.trellis_lut_offset = fc1.shared_bytes
|
| 489 |
+
launch = _DynamicMoEW4A8Launch(
|
| 490 |
+
kernel,
|
| 491 |
+
k=self.hidden,
|
| 492 |
+
n=self.intermediate,
|
| 493 |
+
w1_n=2 * self.intermediate,
|
| 494 |
+
num_topk=self.topk,
|
| 495 |
+
)
|
| 496 |
+
|
| 497 |
+
def ptr(dtype, address: int, align: int = 16):
|
| 498 |
+
return make_ptr(dtype, address, cute.AddressSpace.gmem, assumed_align=align)
|
| 499 |
+
|
| 500 |
+
def fake_ptr_u8():
|
| 501 |
+
return ptr(cutlass.Uint8, 16)
|
| 502 |
+
|
| 503 |
+
def fake_ptr_i32():
|
| 504 |
+
return ptr(cutlass.Int32, 4, 4)
|
| 505 |
+
|
| 506 |
+
def fake_ptr_u32():
|
| 507 |
+
return ptr(cutlass.Uint32, 16)
|
| 508 |
+
|
| 509 |
+
b_w13_fake = cute.runtime.make_fake_compact_tensor(
|
| 510 |
+
cutlass.Float4E2M1FN,
|
| 511 |
+
(2 * self.intermediate, self.hidden, self.experts),
|
| 512 |
+
stride_order=(1, 0, 2),
|
| 513 |
+
assumed_align=16,
|
| 514 |
+
)
|
| 515 |
+
b_w2_fake = cute.runtime.make_fake_compact_tensor(
|
| 516 |
+
cutlass.Float4E2M1FN,
|
| 517 |
+
(self.hidden, self.intermediate, self.experts),
|
| 518 |
+
stride_order=(1, 0, 2),
|
| 519 |
+
assumed_align=16,
|
| 520 |
+
)
|
| 521 |
+
compiled = b12x_compile(
|
| 522 |
+
launch,
|
| 523 |
+
ptr(cutlass.BFloat16, 16),
|
| 524 |
+
fake_ptr_i32(),
|
| 525 |
+
ptr(cutlass.Float32, 4, 4),
|
| 526 |
+
ptr(cutlass.Float4E2M1FN, 16),
|
| 527 |
+
ptr(cutlass.Float8E4M3FN, 16),
|
| 528 |
+
fake_ptr_u8(),
|
| 529 |
+
fake_ptr_u8(),
|
| 530 |
+
fake_ptr_u32(),
|
| 531 |
+
_fake_i32((1,)), _fake_i32((1,)), _fake_i32((1,)),
|
| 532 |
+
_fake_i32((1,)), _fake_i32((1,)), _fake_i32((1,)), _fake_i32((1,)),
|
| 533 |
+
fake_ptr_i32(), fake_ptr_i32(), fake_ptr_i32(),
|
| 534 |
+
fake_ptr_i32(), fake_ptr_i32(), fake_ptr_i32(), fake_ptr_i32(),
|
| 535 |
+
b_w13_fake,
|
| 536 |
+
ptr(cutlass.Float8E4M3FN, 16),
|
| 537 |
+
b_w2_fake,
|
| 538 |
+
ptr(cutlass.Float8E4M3FN, 16),
|
| 539 |
+
fake_ptr_u8(), fake_ptr_u8(), fake_ptr_u8(), fake_ptr_u8(),
|
| 540 |
+
fake_ptr_u32(), fake_ptr_u32(), fake_ptr_u32(), fake_ptr_u32(),
|
| 541 |
+
_fake_i32((self.experts,)),
|
| 542 |
+
_fake_i32((self.experts,)),
|
| 543 |
+
_fake_i32((self.experts + 1,)),
|
| 544 |
+
_fake_f32((self.experts,)), _fake_f32((self.experts,)),
|
| 545 |
+
_fake_f32((self.experts,)), _fake_f32((self.experts,)),
|
| 546 |
+
ptr(
|
| 547 |
+
cutlass.Float32 if self.full_coupled else cutlass.BFloat16,
|
| 548 |
+
16,
|
| 549 |
+
),
|
| 550 |
+
fake_ptr_i32(),
|
| 551 |
+
ptr(cutlass.Float32, 16),
|
| 552 |
+
1, 1, 1, 1, 1, 1, 1,
|
| 553 |
+
current_cuda_stream(),
|
| 554 |
+
fake_ptr_u8(),
|
| 555 |
+
ptr(cutlass.Float16, 16),
|
| 556 |
+
# The compile spec is the JIT cache key and includes every specialized
|
| 557 |
+
# field, especially stored rate, topology and epilogue dimensions.
|
| 558 |
+
compile_spec=KernelCompileSpec.from_fields(
|
| 559 |
+
"glm53.p8.native.tp",
|
| 560 |
+
4,
|
| 561 |
+
("fc1_row_alias376", 1),
|
| 562 |
+
("requested_m_regime_direct_policy", 1),
|
| 563 |
+
("fc1_route_hoist", 1),
|
| 564 |
+
("tile_m", tile_m),
|
| 565 |
+
("trellis_bits", self.trellis_bits),
|
| 566 |
+
("mcg_k5_funnel", int(small_m and self.trellis_bits == 5)),
|
| 567 |
+
("fc2_carveout100_grid564", int(small_m)),
|
| 568 |
+
("fc2_k5_funnel", int(small_m and self.trellis_bits == 5)),
|
| 569 |
+
("grouped_fc2_grid376", int(not small_m)),
|
| 570 |
+
("tile_major_tasks", int(self.tile_major_tasks and small_m)),
|
| 571 |
+
("fc1_pipeline_stages", self.fc1_pipeline_stages if small_m else 2),
|
| 572 |
+
("fc1_warps", self.fc1_warps if small_m else 4),
|
| 573 |
+
("fc1_a_swizzle_rotate", int(self.fc1_a_swizzle_rotate and small_m)),
|
| 574 |
+
("fc1_warp_quant", int(self.fc1_warp_quant and small_m)),
|
| 575 |
+
("fc1_exact_staging", int(self.fc1_exact_staging and small_m)),
|
| 576 |
+
("fc1_broadcast_a", int(self.fc1_broadcast_a and small_m)),
|
| 577 |
+
("materialized", int(materialized)),
|
| 578 |
+
("small_m_scheduler", int(small_m)),
|
| 579 |
+
("fc1_tile_n", self.fc1_tile_n if small_m else 128),
|
| 580 |
+
("experts", self.experts),
|
| 581 |
+
("hidden", self.hidden),
|
| 582 |
+
("intermediate", self.intermediate),
|
| 583 |
+
("topk", self.topk),
|
| 584 |
+
("rank", self.tp_rank),
|
| 585 |
+
("scaled", 1),
|
| 586 |
+
("identity", int(not self.full_coupled)),
|
| 587 |
+
("scale_sandwich", int(self.scale_component is not None)),
|
| 588 |
+
("full_coupled", int(self.full_coupled)),
|
| 589 |
+
("raw_fc1_diagnostic", int(self.diagnostic_raw_fc1)),
|
| 590 |
+
("input_prequant_diagnostic", int(self.diagnostic_raw_fc1)),
|
| 591 |
+
("codebook", "mcg"),
|
| 592 |
+
("deterministic_output", int(self.deterministic_output)),
|
| 593 |
+
("down_remainder", int(self.p8_down_remainder)),
|
| 594 |
+
("down_remainder_phases", self._dx2_phases if self.p8_down_remainder else "none"),
|
| 595 |
+
("dx2_rowpack", int(self.p8_dx2_rowpack and small_m)),
|
| 596 |
+
("dx2_input_rowpack", int(self.p8_input_rowpack and small_m)),
|
| 597 |
+
("epi_par", int(self.p8_epi_par and small_m)),
|
| 598 |
+
),
|
| 599 |
+
dsl_compile_options=OptLevel(2),
|
| 600 |
+
)
|
| 601 |
+
arm = _CompiledArm(compiled=compiled, tile_m=tile_m, materialized=materialized, mac=mac)
|
| 602 |
+
self._compiled[cache_key] = arm
|
| 603 |
+
return arm
|
| 604 |
+
|
| 605 |
+
def _compile_full_coupled_reducer(self):
|
| 606 |
+
if not self.full_coupled:
|
| 607 |
+
raise RuntimeError("coupled reducer requested for non-coupled P8")
|
| 608 |
+
if self._coupled_reducer is not None:
|
| 609 |
+
return self._coupled_reducer
|
| 610 |
+
from b12x.moe._shared.kernels.p8_coupled_topk import (
|
| 611 |
+
P8CoupledTopKSumKernel,
|
| 612 |
+
)
|
| 613 |
+
|
| 614 |
+
reducer = P8CoupledTopKSumKernel(topk=self.topk, hidden=self.hidden)
|
| 615 |
+
self._coupled_reducer = b12x_compile(
|
| 616 |
+
reducer,
|
| 617 |
+
make_ptr(cutlass.Float32, 16, cute.AddressSpace.gmem, assumed_align=16),
|
| 618 |
+
make_ptr(cutlass.Float32, 4, cute.AddressSpace.gmem, assumed_align=4),
|
| 619 |
+
make_ptr(cutlass.BFloat16, 16, cute.AddressSpace.gmem, assumed_align=16),
|
| 620 |
+
1,
|
| 621 |
+
current_cuda_stream(),
|
| 622 |
+
compile_spec=KernelCompileSpec.from_fields(
|
| 623 |
+
"glm53.p8.coupled_topk_h512",
|
| 624 |
+
2,
|
| 625 |
+
("topk", self.topk),
|
| 626 |
+
("hidden", self.hidden),
|
| 627 |
+
("rank", self.tp_rank),
|
| 628 |
+
("route_dtype", "fp32"),
|
| 629 |
+
("output_dtype", "bf16"),
|
| 630 |
+
),
|
| 631 |
+
dsl_compile_options=OptLevel(2),
|
| 632 |
+
)
|
| 633 |
+
return self._coupled_reducer
|
| 634 |
+
|
| 635 |
+
@torch.inference_mode()
|
| 636 |
+
def __call__(
|
| 637 |
+
self,
|
| 638 |
+
x: torch.Tensor,
|
| 639 |
+
topk_weights: torch.Tensor,
|
| 640 |
+
topk_ids: torch.Tensor,
|
| 641 |
+
) -> torch.Tensor:
|
| 642 |
+
if x.dtype != torch.bfloat16 or x.ndim != 2 or x.shape[1] != self.hidden:
|
| 643 |
+
raise RuntimeError(f"P8 native input contract mismatch: {x.dtype} {tuple(x.shape)}")
|
| 644 |
+
m = int(x.shape[0])
|
| 645 |
+
if tuple(topk_ids.shape) != (m, self.topk) or tuple(topk_weights.shape) != (m, self.topk):
|
| 646 |
+
raise RuntimeError("P8 native routing shape mismatch")
|
| 647 |
+
if self.prefill_chunk_tokens and m > self.prefill_chunk_tokens:
|
| 648 |
+
if not self.full_coupled or self.debug_capture:
|
| 649 |
+
raise RuntimeError("bounded prefill requires full coupling without debug capture")
|
| 650 |
+
# MoE is token-local. Bound route-output and materialized carrier
|
| 651 |
+
# storage without changing the scheduler batch or attention work.
|
| 652 |
+
# Reuse allocations on the same stream; no host readback or sync.
|
| 653 |
+
output = torch.empty_like(x)
|
| 654 |
+
for start in range(0, m, self.prefill_chunk_tokens):
|
| 655 |
+
stop = min(start + self.prefill_chunk_tokens, m)
|
| 656 |
+
output[start:stop].copy_(self(x[start:stop], topk_weights[start:stop], topk_ids[start:stop]))
|
| 657 |
+
return output
|
| 658 |
+
# Every stored rate now has a fused grouped M64/N128 owner, so a non-K4 layer at M>1
|
| 659 |
+
# runs the same native path as K4 rather than looping the M1 kernel row by row. The
|
| 660 |
+
# row-by-row fallback is deliberately gone: a rate without a grouped specialization
|
| 661 |
+
# must fail closed instead of silently serving at a fraction of the speed.
|
| 662 |
+
if self.scale_component is not None and not self.full_coupled and m != 1:
|
| 663 |
+
raise RuntimeError("P8 scale sandwich currently supports M=1 only")
|
| 664 |
+
# Match the W4A8 planner's measured M16-to-M64 transition: sparse
|
| 665 |
+
# decode and ordinary prefill stay monolithic; only dense routed
|
| 666 |
+
# batches pay for the split materialized phase kernels.
|
| 667 |
+
materialized = (
|
| 668 |
+
m * self.topk >= 36 * self.experts
|
| 669 |
+
if self.force_materialized is None
|
| 670 |
+
else self.force_materialized
|
| 671 |
+
)
|
| 672 |
+
small_m = use_small_m(self.small_m_scheduler, m)
|
| 673 |
+
if self.full_coupled:
|
| 674 |
+
# Experimental direct-route batches through M16; requires matching
|
| 675 |
+
# multirow FC1 and input-prologue patches. Not serving-qualified.
|
| 676 |
+
small_m = m <= 16 and not self.grouped_m16
|
| 677 |
+
materialized = True
|
| 678 |
+
materialized = materialized or small_m
|
| 679 |
+
arm = self._compile(materialized, small_m=small_m, expected_m=m)
|
| 680 |
+
tile_m = arm.tile_m
|
| 681 |
+
x = x.contiguous()
|
| 682 |
+
flat_ids = topk_ids.to(dtype=torch.int32).contiguous().reshape(-1)
|
| 683 |
+
flat_weights = topk_weights.to(dtype=torch.float32).contiguous().reshape(-1)
|
| 684 |
+
physical_tiles = (
|
| 685 |
+
m * self.topk if small_m
|
| 686 |
+
else self.experts + (m * self.topk + tile_m - 1) // tile_m
|
| 687 |
+
)
|
| 688 |
+
rows_padded = physical_tiles * tile_m
|
| 689 |
+
gate_tile_count = ((2 * self.intermediate) // 128) // 2
|
| 690 |
+
max_tasks = physical_tiles * max(gate_tile_count, 1)
|
| 691 |
+
fused_scratch_zero = ((self.fuse_scratch_zero and small_m) or
|
| 692 |
+
(self.fuse_grouped_scratch and self.grouped_m16))
|
| 693 |
+
shared_kernel_output = None
|
| 694 |
+
if fused_scratch_zero or self.shared_workspace:
|
| 695 |
+
if fused_scratch_zero:
|
| 696 |
+
from .direct_policy_scratch import direct_scratch_layout
|
| 697 |
+
self._scratch_layout = (direct_scratch_layout(m, self.intermediate, tile_m=tile_m, planes=self._dx2_planes) if small_m
|
| 698 |
+
else p8_small_m_scratch_layout(intermediate=self.intermediate, tokens=m, planes=self._dx2_planes,
|
| 699 |
+
shared=True, grouped=True, tile_m=tile_m, direct=False))
|
| 700 |
+
layout = (p8_small_m_scratch_layout(intermediate=self.intermediate, tokens=m, shared=True, planes=self._dx2_planes,
|
| 701 |
+
grouped=self.full_coupled and materialized and not small_m, tile_m=tile_m, direct=small_m)
|
| 702 |
+
if self.shared_workspace else self._scratch_layout)
|
| 703 |
+
assert layout is not None
|
| 704 |
+
# A single GPU fill initializes all original bytes plus alignment
|
| 705 |
+
# padding. The views add no casts, copies, or device kernels.
|
| 706 |
+
if self.shared_workspace:
|
| 707 |
+
from vllm.v1.worker.workspace import current_workspace_manager
|
| 708 |
+
arena, shared_kernel_output = current_workspace_manager().get_simultaneous(
|
| 709 |
+
((layout.nbytes,), torch.uint8),
|
| 710 |
+
((m * self.topk, self.hidden), torch.float32),
|
| 711 |
+
)
|
| 712 |
+
arena.zero_()
|
| 713 |
+
else:
|
| 714 |
+
arena = torch.zeros(layout.nbytes, dtype=torch.uint8, device=self.device)
|
| 715 |
+
buffers = {
|
| 716 |
+
region.name: arena.narrow(0, region.offset, region.nbytes)
|
| 717 |
+
.view(getattr(torch, region.dtype)).reshape(region.shape)
|
| 718 |
+
for region in layout.regions
|
| 719 |
+
}
|
| 720 |
+
packed_a = buffers["packed_a"]
|
| 721 |
+
scale_flat = buffers["scale_flat"]
|
| 722 |
+
intermediate_u32 = buffers["intermediate_u32"]
|
| 723 |
+
barrier_count = buffers["barrier_count"]
|
| 724 |
+
barrier_epoch = buffers["barrier_epoch"]
|
| 725 |
+
pair_head = buffers["pair_head"]
|
| 726 |
+
producers_done = buffers["producers_done"]
|
| 727 |
+
all_published = buffers["all_published"]
|
| 728 |
+
task_head = buffers["task_head"]
|
| 729 |
+
task_tail = buffers["task_tail"]
|
| 730 |
+
task_ready = buffers["task_ready"]
|
| 731 |
+
task_expert = buffers["task_expert"]
|
| 732 |
+
task_m_tile = buffers["task_m_tile"]
|
| 733 |
+
task_slice_begin = buffers["task_slice_begin"]
|
| 734 |
+
task_slice_count = buffers["task_slice_count"]
|
| 735 |
+
task_valid_rows = buffers["task_valid_rows"]
|
| 736 |
+
tile_write_count = buffers["tile_write_count"]
|
| 737 |
+
row_counts = buffers["row_counts"]
|
| 738 |
+
expert_write_rows = buffers["expert_write_rows"]
|
| 739 |
+
expert_tile_base = buffers["expert_tile_base"]
|
| 740 |
+
token_map = buffers["token_map"]
|
| 741 |
+
token_weights = buffers["token_weights"]
|
| 742 |
+
output = (torch.zeros(m, self.hidden, dtype=torch.bfloat16, device=self.device)
|
| 743 |
+
if self.shared_workspace else buffers["output"])
|
| 744 |
+
else:
|
| 745 |
+
# Coupled grouped FC1 addresses the shared carrier by token index,
|
| 746 |
+
# not padded expert row (p8_h128_fc1 src_word uses tok). Keep the
|
| 747 |
+
# original extent for every other owner, including M1.
|
| 748 |
+
input_rows = (m if self.compact_input_storage and self.full_coupled
|
| 749 |
+
and materialized and not small_m else rows_padded)
|
| 750 |
+
packed_a = torch.zeros(input_rows * self.hidden, dtype=torch.uint8, device=self.device)
|
| 751 |
+
scale_elements = (
|
| 752 |
+
m * (self.hidden // 32)
|
| 753 |
+
if self.full_coupled and materialized and not small_m
|
| 754 |
+
else (self.experts + m * self.topk + 1)
|
| 755 |
+
* tile_m
|
| 756 |
+
* (self.hidden // 8)
|
| 757 |
+
)
|
| 758 |
+
scale_flat = torch.zeros(
|
| 759 |
+
scale_elements, dtype=torch.uint8, device=self.device
|
| 760 |
+
)
|
| 761 |
+
intermediate_count = self._dx2_planes * rows_padded * (self.intermediate + self.intermediate // 32) // 4
|
| 762 |
+
intermediate_u32 = torch.zeros(intermediate_count, dtype=torch.int32, device=self.device)
|
| 763 |
+
|
| 764 |
+
def z1():
|
| 765 |
+
return torch.zeros(1, dtype=torch.int32, device=self.device)
|
| 766 |
+
|
| 767 |
+
def ztask():
|
| 768 |
+
return torch.zeros(max_tasks, dtype=torch.int32, device=self.device)
|
| 769 |
+
|
| 770 |
+
barrier_count, barrier_epoch = z1(), z1()
|
| 771 |
+
pair_head, producers_done, all_published = z1(), z1(), z1()
|
| 772 |
+
task_head, task_tail = z1(), z1()
|
| 773 |
+
task_ready, task_expert, task_m_tile = ztask(), ztask(), ztask()
|
| 774 |
+
task_slice_begin, task_slice_count, task_valid_rows = ztask(), ztask(), ztask()
|
| 775 |
+
tile_write_count = torch.zeros(physical_tiles, dtype=torch.int32, device=self.device)
|
| 776 |
+
row_counts = torch.zeros(self.experts, dtype=torch.int32, device=self.device)
|
| 777 |
+
expert_write_rows = torch.zeros(self.experts, dtype=torch.int32, device=self.device)
|
| 778 |
+
expert_tile_base = torch.zeros(self.experts + 1, dtype=torch.int32, device=self.device)
|
| 779 |
+
token_map = torch.zeros(rows_padded, dtype=torch.int32, device=self.device)
|
| 780 |
+
token_weights = torch.zeros(rows_padded, dtype=torch.float32, device=self.device)
|
| 781 |
+
output = torch.zeros(m, self.hidden, dtype=torch.bfloat16, device=self.device)
|
| 782 |
+
if self.diagnostic_raw_fc1:
|
| 783 |
+
# Keep the ordinary allocation block exactly unchanged. Only the
|
| 784 |
+
# diagnostic arm fills NaN payload sentinels and zeroes counters.
|
| 785 |
+
intermediate_u32.fill_(-1)
|
| 786 |
+
trace_base = rows_padded * (self.intermediate // 4)
|
| 787 |
+
intermediate_u32[trace_base + 32 : trace_base + 64].zero_()
|
| 788 |
+
kernel_output = (
|
| 789 |
+
shared_kernel_output if shared_kernel_output is not None else torch.empty(
|
| 790 |
+
m * self.topk,
|
| 791 |
+
self.hidden,
|
| 792 |
+
dtype=torch.float32 if self.full_coupled else torch.bfloat16,
|
| 793 |
+
device=self.device,
|
| 794 |
+
)
|
| 795 |
+
if self.deterministic_output
|
| 796 |
+
else output
|
| 797 |
+
)
|
| 798 |
+
launch_mac = arm.mac
|
| 799 |
+
if self.grid_policy and self.world_size == 4 and self.mac_override is None:
|
| 800 |
+
from .p8_multirow_scratch import direct_grid_capacity
|
| 801 |
+
launch_mac = direct_grid_capacity(m, arm.mac)
|
| 802 |
+
arm.compiled(
|
| 803 |
+
_gptr(cutlass.BFloat16, x),
|
| 804 |
+
_gptr(cutlass.Int32, flat_ids, 4),
|
| 805 |
+
_gptr(cutlass.Float32, flat_weights, 4),
|
| 806 |
+
_gptr(cutlass.Float4E2M1FN, packed_a),
|
| 807 |
+
_gptr(cutlass.Float8E4M3FN, scale_flat),
|
| 808 |
+
_gptr(cutlass.Uint8, packed_a),
|
| 809 |
+
_gptr(cutlass.Uint8, scale_flat),
|
| 810 |
+
_gptr(cutlass.Uint32, intermediate_u32),
|
| 811 |
+
barrier_count, barrier_epoch, pair_head, producers_done, all_published,
|
| 812 |
+
task_head, task_tail,
|
| 813 |
+
_gptr(cutlass.Int32, task_ready, 4),
|
| 814 |
+
_gptr(cutlass.Int32, task_expert, 4),
|
| 815 |
+
_gptr(cutlass.Int32, task_m_tile, 4),
|
| 816 |
+
_gptr(cutlass.Int32, task_slice_begin, 4),
|
| 817 |
+
_gptr(cutlass.Int32, task_slice_count, 4),
|
| 818 |
+
_gptr(cutlass.Int32, task_valid_rows, 4),
|
| 819 |
+
_gptr(cutlass.Int32, tile_write_count, 4),
|
| 820 |
+
self.w13_dummy,
|
| 821 |
+
_gptr(cutlass.Float8E4M3FN, self.sentinel),
|
| 822 |
+
self.w2_dummy,
|
| 823 |
+
_gptr(cutlass.Float8E4M3FN, self.sentinel),
|
| 824 |
+
_gptr(cutlass.Uint8, self.w13_scale_mx),
|
| 825 |
+
_gptr(cutlass.Uint8, self.w2_scale_mx),
|
| 826 |
+
_gptr(cutlass.Uint8, self.sentinel),
|
| 827 |
+
_gptr(cutlass.Uint8, self.sentinel),
|
| 828 |
+
_gptr(cutlass.Uint32, self.w13_stream),
|
| 829 |
+
_gptr(cutlass.Uint32, self.w13_sfb),
|
| 830 |
+
_gptr(cutlass.Uint32, self.w2_stream),
|
| 831 |
+
_gptr(cutlass.Uint32, self.w2_sfb),
|
| 832 |
+
row_counts, expert_write_rows, expert_tile_base,
|
| 833 |
+
self.ones, self.ones, self.ones, self.ones,
|
| 834 |
+
_gptr(
|
| 835 |
+
cutlass.Float32 if self.full_coupled else cutlass.BFloat16,
|
| 836 |
+
kernel_output,
|
| 837 |
+
),
|
| 838 |
+
_gptr(cutlass.Int32, token_map, 4),
|
| 839 |
+
_gptr(cutlass.Float32, token_weights, 4),
|
| 840 |
+
m,
|
| 841 |
+
m * self.topk,
|
| 842 |
+
m * self.topk if self.deterministic_output else m,
|
| 843 |
+
rows_padded,
|
| 844 |
+
max_tasks,
|
| 845 |
+
physical_tiles,
|
| 846 |
+
launch_mac,
|
| 847 |
+
current_cuda_stream(),
|
| 848 |
+
_gptr(cutlass.Uint8, self.input_prequant_trace),
|
| 849 |
+
_gptr(cutlass.Float16, self.scale_component_packed),
|
| 850 |
+
)
|
| 851 |
+
if self.deterministic_output and not self.diagnostic_raw_fc1:
|
| 852 |
+
if self.full_coupled:
|
| 853 |
+
reducer = self._compile_full_coupled_reducer()
|
| 854 |
+
reducer(
|
| 855 |
+
_gptr(cutlass.Float32, kernel_output),
|
| 856 |
+
_gptr(cutlass.Float32, flat_weights, 4),
|
| 857 |
+
_gptr(cutlass.BFloat16, output),
|
| 858 |
+
m,
|
| 859 |
+
current_cuda_stream(),
|
| 860 |
+
)
|
| 861 |
+
else:
|
| 862 |
+
_launch_dynamic_topk_sum(
|
| 863 |
+
route_output=kernel_output,
|
| 864 |
+
output=output,
|
| 865 |
+
m=m,
|
| 866 |
+
num_topk=self.topk,
|
| 867 |
+
k=self.hidden,
|
| 868 |
+
stream=current_cuda_stream(),
|
| 869 |
+
)
|
| 870 |
+
if self.debug_capture:
|
| 871 |
+
self.debug_tensors = {
|
| 872 |
+
"packed_a": packed_a, "scale_flat": scale_flat,
|
| 873 |
+
"intermediate_u32": intermediate_u32,
|
| 874 |
+
"route_output": kernel_output,
|
| 875 |
+
"token_map": token_map, "row_counts": row_counts,
|
| 876 |
+
"expert_tile_base": expert_tile_base,
|
| 877 |
+
}
|
| 878 |
+
self.debug_dispatch = {"small_m": small_m, "materialized": materialized,
|
| 879 |
+
"fused_scratch_zero": fused_scratch_zero,
|
| 880 |
+
"fc1_tile_n": self.fc1_tile_n if small_m else 128,
|
| 881 |
+
"tile_m": tile_m}
|
| 882 |
+
if self.diagnostic_raw_fc1:
|
| 883 |
+
self.debug_dispatch["diagnostic_raw_fc1"] = True
|
| 884 |
+
self.debug_tensors["input_prequant_trace"] = (
|
| 885 |
+
self.input_prequant_trace
|
| 886 |
+
)
|
| 887 |
+
return output
|
runtime/patches/vllm_quant_trellismx.py
ADDED
|
@@ -0,0 +1,195 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# SPDX-License-Identifier: Apache-2.0
|
| 2 |
+
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
| 3 |
+
"""Opt-in native P8 routed experts over a ModelOpt NVFP4 carrier.
|
| 4 |
+
|
| 5 |
+
The runtime is separately licensed and lazily imported. Dense, attention,
|
| 6 |
+
router, shared-expert and MTP tensors retain the carrier's quantization.
|
| 7 |
+
"""
|
| 8 |
+
|
| 9 |
+
from vllm import envs
|
| 10 |
+
|
| 11 |
+
import regex as re
|
| 12 |
+
import torch
|
| 13 |
+
|
| 14 |
+
from vllm.config import get_current_vllm_config
|
| 15 |
+
from vllm.logger import init_logger
|
| 16 |
+
from vllm.model_executor.layers.fused_moe.activation import MoEActivation
|
| 17 |
+
from vllm.model_executor.layers.fused_moe.fused_moe_method_base import (
|
| 18 |
+
FusedMoEMethodBase,
|
| 19 |
+
)
|
| 20 |
+
from vllm.model_executor.layers.quantization.modelopt import ModelOptNvFp4FusedMoE
|
| 21 |
+
from vllm.utils.b12x import B12xWarmupUnit
|
| 22 |
+
from vllm.utils.trellismx import load_overlay, routed_layer
|
| 23 |
+
|
| 24 |
+
logger = init_logger(__name__)
|
| 25 |
+
|
| 26 |
+
|
| 27 |
+
def maybe_trellismx_method(config, layer, prefix):
|
| 28 |
+
directory = envs.VLLM_TRELLISMX_CHECKPOINT
|
| 29 |
+
index = routed_layer(prefix)
|
| 30 |
+
if not directory:
|
| 31 |
+
return None
|
| 32 |
+
text_config = get_current_vllm_config().model_config.hf_text_config
|
| 33 |
+
if text_config.model_type not in ("glm5_next_text", "glm5_next_mtp"):
|
| 34 |
+
raise ValueError(
|
| 35 |
+
"TrellisMX overlay currently requires the GLM5Next text adapter"
|
| 36 |
+
)
|
| 37 |
+
if index is None:
|
| 38 |
+
if re.fullmatch(
|
| 39 |
+
r"(?:model\.language_model|language_model\.model|model)"
|
| 40 |
+
r"\.layers\.45\.(?:mtp_block\.)?mlp\.experts",
|
| 41 |
+
prefix,
|
| 42 |
+
):
|
| 43 |
+
return None
|
| 44 |
+
raise ValueError(f"Unrecognized TrellisMX routed-expert prefix: {prefix}")
|
| 45 |
+
if index == 45:
|
| 46 |
+
if (45, 0) not in load_overlay(directory).records:
|
| 47 |
+
return None
|
| 48 |
+
# MTP carrier metadata describes the replaced MXFP8 tensors. Use the
|
| 49 |
+
# established NVFP4 allocation ABI before loading native P8 sidecars.
|
| 50 |
+
config = getattr(config, "nvfp4_config", config)
|
| 51 |
+
if getattr(config, "quant_method", None) != "NVFP4":
|
| 52 |
+
raise ValueError("TrellisMX requires the pinned ModelOpt NVFP4 carrier")
|
| 53 |
+
return TrellisMXMoEMethod(config, layer.moe_config, directory, index)
|
| 54 |
+
|
| 55 |
+
|
| 56 |
+
class TrellisMXMoEMethod(ModelOptNvFp4FusedMoE):
|
| 57 |
+
"""Keep the carrier's weight-loader ABI; execute routed weights using P8."""
|
| 58 |
+
|
| 59 |
+
def __init__(self, config, moe_config, directory, layer_index):
|
| 60 |
+
# Do not select or compile an NVFP4 expert backend we never execute.
|
| 61 |
+
FusedMoEMethodBase.__init__(self, moe_config)
|
| 62 |
+
self.quant_config = config
|
| 63 |
+
self.use_a16 = False
|
| 64 |
+
self.use_global_sf = False
|
| 65 |
+
parallel = moe_config.moe_parallel_config
|
| 66 |
+
if (
|
| 67 |
+
parallel.tp_size != 4
|
| 68 |
+
or parallel.ep_size != 1
|
| 69 |
+
or moe_config.hidden_dim != 4096
|
| 70 |
+
or moe_config.intermediate_size_per_partition != 512
|
| 71 |
+
or moe_config.num_experts != 288
|
| 72 |
+
or moe_config.experts_per_token != 8
|
| 73 |
+
or moe_config.has_bias
|
| 74 |
+
or moe_config.is_lora_enabled
|
| 75 |
+
or moe_config.activation != MoEActivation.SILU
|
| 76 |
+
or moe_config.in_dtype != torch.bfloat16
|
| 77 |
+
or moe_config.swiglu_limit != 10.0
|
| 78 |
+
):
|
| 79 |
+
raise ValueError(
|
| 80 |
+
"TrellisMX GLM adapter requires TP4, EP1 and GLM Flash shapes"
|
| 81 |
+
)
|
| 82 |
+
self.layer_index = layer_index
|
| 83 |
+
self.rank = parallel.tp_rank
|
| 84 |
+
self.overlay = load_overlay(directory)
|
| 85 |
+
self.runtime = None
|
| 86 |
+
|
| 87 |
+
@property
|
| 88 |
+
def is_monolithic(self):
|
| 89 |
+
return False
|
| 90 |
+
|
| 91 |
+
@property
|
| 92 |
+
def supports_eplb(self):
|
| 93 |
+
return False
|
| 94 |
+
|
| 95 |
+
def get_fused_moe_quant_config(self, layer):
|
| 96 |
+
return None
|
| 97 |
+
|
| 98 |
+
def process_weights_after_loading(self, layer):
|
| 99 |
+
if self.runtime is not None:
|
| 100 |
+
raise RuntimeError("TrellisMX hot weight replacement is unsupported")
|
| 101 |
+
from b12x.moe._shared.trellismx.p8_native_kernel import P8NativeTPMoE
|
| 102 |
+
|
| 103 |
+
device = layer.w13_weight.device
|
| 104 |
+
if device.type != "cuda" or torch.cuda.get_device_capability(device) != (12, 0):
|
| 105 |
+
raise ValueError("This TrellisMX runtime requires SM120 CUDA")
|
| 106 |
+
sidecar = self.overlay.sidecar(self.layer_index, self.rank)
|
| 107 |
+
record = self.overlay.records[self.layer_index, self.rank]
|
| 108 |
+
runtime = P8NativeTPMoE(
|
| 109 |
+
sidecar,
|
| 110 |
+
device=device,
|
| 111 |
+
tp_rank=self.rank,
|
| 112 |
+
world_size=4,
|
| 113 |
+
layer=self.layer_index,
|
| 114 |
+
expected_design_sha256=record["source_design_sha256"],
|
| 115 |
+
expected_transform_sha256=self.overlay.transform_hash,
|
| 116 |
+
topk=8,
|
| 117 |
+
hidden=4096,
|
| 118 |
+
intermediate=512,
|
| 119 |
+
swiglu_limit=10.0,
|
| 120 |
+
small_m_scheduler=True,
|
| 121 |
+
fc1_tile_n=128,
|
| 122 |
+
fuse_scratch_zero=True,
|
| 123 |
+
prefill_chunk_tokens=0,
|
| 124 |
+
grid_policy=True,
|
| 125 |
+
fc1_warp_quant=False,
|
| 126 |
+
fc1_broadcast_a=True,
|
| 127 |
+
)
|
| 128 |
+
# Release only replaced routed storage, after a successful load. Never
|
| 129 |
+
# touch the runner's router/shared experts or the separately owned MTP.
|
| 130 |
+
released = 0
|
| 131 |
+
for name in (
|
| 132 |
+
"w13_weight",
|
| 133 |
+
"w2_weight",
|
| 134 |
+
"w13_weight_scale",
|
| 135 |
+
"w2_weight_scale",
|
| 136 |
+
"w13_weight_scale_2",
|
| 137 |
+
"w2_weight_scale_2",
|
| 138 |
+
"w13_input_scale",
|
| 139 |
+
"w2_input_scale",
|
| 140 |
+
):
|
| 141 |
+
value = getattr(layer, name)
|
| 142 |
+
released += value.numel() * value.element_size()
|
| 143 |
+
setattr(
|
| 144 |
+
layer,
|
| 145 |
+
name,
|
| 146 |
+
torch.nn.Parameter(
|
| 147 |
+
torch.empty(0, dtype=value.dtype, device=value.device),
|
| 148 |
+
requires_grad=False,
|
| 149 |
+
),
|
| 150 |
+
)
|
| 151 |
+
self.runtime = runtime
|
| 152 |
+
layer.b12x_warmup_provider = self
|
| 153 |
+
logger.info(
|
| 154 |
+
"TrellisMX layer=%d rank=%d K%d E4M3/UE8M0-32 "
|
| 155 |
+
"coupled-h512-h128 released_carrier_bytes=%d",
|
| 156 |
+
self.layer_index,
|
| 157 |
+
self.rank,
|
| 158 |
+
record["bits"],
|
| 159 |
+
released,
|
| 160 |
+
)
|
| 161 |
+
|
| 162 |
+
def apply(
|
| 163 |
+
self, layer, x, topk_weights, topk_ids, shared_experts, shared_experts_input
|
| 164 |
+
):
|
| 165 |
+
if self.runtime is None:
|
| 166 |
+
raise RuntimeError("TrellisMX routed weights were not loaded")
|
| 167 |
+
if x.shape[0] == 0:
|
| 168 |
+
return torch.empty_like(x)
|
| 169 |
+
return self.runtime(x, topk_weights, topk_ids)
|
| 170 |
+
|
| 171 |
+
def apply_monolithic(self, *args, **kwargs):
|
| 172 |
+
raise RuntimeError("TrellisMX routing belongs to the Jovian MoE runner")
|
| 173 |
+
|
| 174 |
+
def get_b12x_warmup_unit(self, layer, token_counts, output_dtype):
|
| 175 |
+
runtime = self.runtime
|
| 176 |
+
if runtime is None:
|
| 177 |
+
raise RuntimeError("TrellisMX warmup before weight load")
|
| 178 |
+
|
| 179 |
+
def compile():
|
| 180 |
+
for tokens in token_counts:
|
| 181 |
+
x = torch.zeros(
|
| 182 |
+
(tokens, 4096), dtype=output_dtype, device=runtime.device
|
| 183 |
+
)
|
| 184 |
+
ids = torch.arange(8, dtype=torch.int32, device=runtime.device)
|
| 185 |
+
ids = ids.expand(tokens, 8).contiguous()
|
| 186 |
+
weights = torch.full((tokens, 8), 0.125, device=runtime.device)
|
| 187 |
+
runtime(x, weights, ids)
|
| 188 |
+
|
| 189 |
+
# Each layer owns compiled descriptors and scratch; do not deduplicate
|
| 190 |
+
# across layers solely because K/shape match.
|
| 191 |
+
return B12xWarmupUnit(
|
| 192 |
+
name="TrellisMX",
|
| 193 |
+
key=(type(self), self.layer_index, self.rank),
|
| 194 |
+
compile=compile,
|
| 195 |
+
)
|
runtime/patches/vllm_utils_trellismx.py
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# SPDX-License-Identifier: Apache-2.0
|
| 2 |
+
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
| 3 |
+
"""CPU-only validation for the versioned TrellisMX routed overlay."""
|
| 4 |
+
|
| 5 |
+
import hashlib
|
| 6 |
+
import json
|
| 7 |
+
import struct
|
| 8 |
+
from dataclasses import dataclass
|
| 9 |
+
from functools import lru_cache
|
| 10 |
+
from pathlib import Path
|
| 11 |
+
|
| 12 |
+
import regex as re
|
| 13 |
+
|
| 14 |
+
_PREFIX = re.compile(
|
| 15 |
+
r"(?:model\.language_model|language_model\.model|model)"
|
| 16 |
+
r"\.layers\.(\d+)\.(?:mtp_block\.)?mlp\.experts"
|
| 17 |
+
)
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
def routed_layer(prefix: str) -> int | None:
|
| 21 |
+
match = _PREFIX.fullmatch(prefix)
|
| 22 |
+
if match is None:
|
| 23 |
+
return None
|
| 24 |
+
layer = int(match[1])
|
| 25 |
+
return layer if 3 <= layer <= 45 else None
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
def sha256(path: Path) -> str:
|
| 29 |
+
digest = hashlib.sha256()
|
| 30 |
+
with path.open("rb") as source:
|
| 31 |
+
for chunk in iter(lambda: source.read(8 * 1024 * 1024), b""):
|
| 32 |
+
digest.update(chunk)
|
| 33 |
+
return digest.hexdigest()
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
def _local_file(root: Path, relative: str) -> Path:
|
| 37 |
+
path = root / relative
|
| 38 |
+
# Symlinked local checkpoints are supported, but traversal in manifests is not.
|
| 39 |
+
if Path(relative).is_absolute() or ".." in Path(relative).parts:
|
| 40 |
+
raise ValueError(f"Unsafe TrellisMX path: {relative}")
|
| 41 |
+
if not path.is_file():
|
| 42 |
+
raise ValueError(f"Missing TrellisMX artifact: {path}")
|
| 43 |
+
return path
|
| 44 |
+
|
| 45 |
+
|
| 46 |
+
@dataclass(frozen=True)
|
| 47 |
+
class Overlay:
|
| 48 |
+
root: Path
|
| 49 |
+
records: dict
|
| 50 |
+
transform_hash: str
|
| 51 |
+
|
| 52 |
+
def sidecar(self, layer: int, rank: int, *, verify: bool = True) -> Path:
|
| 53 |
+
record = self.records[layer, rank]
|
| 54 |
+
path = _local_file(self.root, record["path"])
|
| 55 |
+
if verify and sha256(path) != record["sha256"]:
|
| 56 |
+
raise ValueError(f"TrellisMX weight hash mismatch: {path}")
|
| 57 |
+
with path.open("rb") as stream:
|
| 58 |
+
length_bytes = stream.read(8)
|
| 59 |
+
if len(length_bytes) != 8:
|
| 60 |
+
raise ValueError(f"Truncated safetensors header: {path}")
|
| 61 |
+
length = struct.unpack("<Q", length_bytes)[0]
|
| 62 |
+
if not 2 <= length <= 16 * 1024 * 1024:
|
| 63 |
+
raise ValueError(f"Invalid safetensors header size: {path}")
|
| 64 |
+
header = json.loads(stream.read(length))
|
| 65 |
+
metadata = header.get("__metadata__", {})
|
| 66 |
+
expected = {
|
| 67 |
+
"schema": "glm53-p8-coupled-h512-h128-tp4-rank.v1",
|
| 68 |
+
"layer": str(layer),
|
| 69 |
+
"rank": str(rank),
|
| 70 |
+
"world_size": "4",
|
| 71 |
+
"bits": str(record["bits"]),
|
| 72 |
+
"alphabet": "e4m3",
|
| 73 |
+
"scale": "ue8m0-k32",
|
| 74 |
+
"law": "procedural-mcg-alpha2",
|
| 75 |
+
"source_design_sha256": record["source_design_sha256"],
|
| 76 |
+
}
|
| 77 |
+
for name, value in expected.items():
|
| 78 |
+
if metadata.get(name) != value:
|
| 79 |
+
raise ValueError(f"TrellisMX {path}: incompatible {name}")
|
| 80 |
+
return path
|
| 81 |
+
|
| 82 |
+
|
| 83 |
+
@lru_cache(maxsize=4)
|
| 84 |
+
def load_overlay(directory: str) -> Overlay:
|
| 85 |
+
root = Path(directory).absolute()
|
| 86 |
+
manifest = json.loads(_local_file(root, "trellismx-manifest.json").read_text())
|
| 87 |
+
if manifest.get("schema") != "trellismx.hf-overlay-release.v1":
|
| 88 |
+
raise ValueError("Unsupported TrellisMX overlay schema")
|
| 89 |
+
if manifest.get("carrier") != {
|
| 90 |
+
"repo_id": "local-inference-lab/GLM-5.3-Flash-NVFP4",
|
| 91 |
+
"revision": "520de24eabf507659eaef7c70f14fd584527facc",
|
| 92 |
+
}:
|
| 93 |
+
raise ValueError("Unsupported TrellisMX carrier identity")
|
| 94 |
+
allocation = manifest.get("allocation", {})
|
| 95 |
+
expected_layers = {int(n) for n in allocation}
|
| 96 |
+
if expected_layers not in (set(range(3, 45)), set(range(3, 46))):
|
| 97 |
+
raise ValueError("GLM TrellisMX requires layers 3..44 and optional MTP45")
|
| 98 |
+
if any(type(bits) is not int or bits not in (4, 5) for bits in allocation.values()):
|
| 99 |
+
raise ValueError("This TrellisMX adapter supports K4/K5 only")
|
| 100 |
+
designs = {
|
| 101 |
+
sha256(_local_file(root, f"design/design-{index}.json")) for index in range(3)
|
| 102 |
+
}
|
| 103 |
+
transform = _local_file(root, "design/transform.json")
|
| 104 |
+
transform_config = json.loads(transform.read_text())
|
| 105 |
+
if (
|
| 106 |
+
transform_config.get("boundary") != "coupled-h512-h128-suh-svh-v1"
|
| 107 |
+
or transform_config.get("sign_draw") != 0
|
| 108 |
+
or transform_config.get("activation") != "silu-cap10"
|
| 109 |
+
):
|
| 110 |
+
raise ValueError("Unsupported TrellisMX coupled transform")
|
| 111 |
+
records = {}
|
| 112 |
+
for record in manifest.get("files", []):
|
| 113 |
+
key = record["layer"], record["rank"]
|
| 114 |
+
if key in records or key[0] not in expected_layers or key[1] not in range(4):
|
| 115 |
+
raise ValueError("Duplicate or invalid TrellisMX layer/rank")
|
| 116 |
+
expected_path = f"sidecars/p8-layer-{key[0]:03d}-tp4-rank-{key[1]}.safetensors"
|
| 117 |
+
if record["path"] != expected_path:
|
| 118 |
+
raise ValueError("Unexpected TrellisMX sidecar path")
|
| 119 |
+
if record["bits"] != allocation[str(key[0])]:
|
| 120 |
+
raise ValueError("TrellisMX allocation and sidecar rate disagree")
|
| 121 |
+
if record["source_design_sha256"] not in designs:
|
| 122 |
+
raise ValueError("TrellisMX source design outside allowlist")
|
| 123 |
+
if not re.fullmatch(r"[0-9a-f]{64}", record["sha256"]):
|
| 124 |
+
raise ValueError("Invalid TrellisMX weight hash")
|
| 125 |
+
if _local_file(root, record["path"]).stat().st_size != record["bytes"]:
|
| 126 |
+
raise ValueError("TrellisMX sidecar size mismatch")
|
| 127 |
+
records[key] = record
|
| 128 |
+
if set(records) != {(layer, rank) for layer in expected_layers for rank in range(4)}:
|
| 129 |
+
raise ValueError("Incomplete TrellisMX TP4 inventory")
|
| 130 |
+
return Overlay(root, records, sha256(transform))
|
runtime/serve-codecv2-mtp.sh
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
set -euo pipefail
|
| 3 |
+
runtime_dir=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)
|
| 4 |
+
checkpoint_dir=$(dirname -- "$runtime_dir")
|
| 5 |
+
python3 "$runtime_dir/install-mtp-runtime.py"
|
| 6 |
+
export MODEL_ROOT=${MODEL_ROOT:-$checkpoint_dir/carrier}
|
| 7 |
+
export VLLM_TRELLISMX_CHECKPOINT=${VLLM_TRELLISMX_CHECKPOINT:-$checkpoint_dir}
|
| 8 |
+
exec /release/serve-rp2.sh "$@"
|
sidecars/p8-layer-045-tp4-rank-0.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1c699c04c08180eadad2f40a6c371295d1c46618db7029e7b7587e73a93c6dd6
|
| 3 |
+
size 963497720
|
sidecars/p8-layer-045-tp4-rank-1.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4ac3d8d98b153f23d65c690cf0e24a98974ba52611642619544ad0311d3277c1
|
| 3 |
+
size 963497720
|
sidecars/p8-layer-045-tp4-rank-2.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2750ae58989848196200dfa2991f40824ff6b52066f53319afd8dcc113c9bde6
|
| 3 |
+
size 963497720
|
sidecars/p8-layer-045-tp4-rank-3.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c9071b28f75dfbbb6b16017721ddc185d2eb9ab021d81bdc1ad4d632cb846ddb
|
| 3 |
+
size 963497720
|
trellismx-manifest.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
| 1 |
{
|
| 2 |
"schema": "trellismx.hf-overlay-release.v1",
|
| 3 |
-
"checkpoint": "codec-v2-uniform-k4",
|
| 4 |
"packaging": "routed-overlay-requires-pinned-stock-carrier",
|
| 5 |
"source_manifest_sha256": "dd4b391c015f9738cf7e4ed2c43b57cebc859f703e3502ca76ec3eb98d7397d8",
|
| 6 |
"carrier": {
|
|
@@ -49,20 +49,18 @@
|
|
| 49 |
"6": 4,
|
| 50 |
"7": 4,
|
| 51 |
"8": 4,
|
| 52 |
-
"9": 4
|
|
|
|
| 53 |
},
|
| 54 |
"payload": {
|
| 55 |
-
"
|
| 56 |
-
"
|
| 57 |
-
"
|
| 58 |
-
"
|
| 59 |
-
"
|
| 60 |
-
"safetensors_header_bytes": 564480,
|
| 61 |
-
"same_size_or_smaller": false,
|
| 62 |
-
"stored_bpw_including_metadata": 4.253994694462529,
|
| 63 |
"weight_payload_bpw": 4.25,
|
| 64 |
-
"weight_payload_bytes":
|
| 65 |
-
"note": "codec-v2 re-encode of every routed layer; see design/design-1.json"
|
| 66 |
},
|
| 67 |
"files": [
|
| 68 |
{
|
|
@@ -1576,7 +1574,62 @@
|
|
| 1576 |
"rank": 3,
|
| 1577 |
"bits": 4,
|
| 1578 |
"source_design_sha256": "82687b2cace99c681d01e84c5bef5cf5874091231072f03755b307f4a69b9c4d"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1579 |
}
|
| 1580 |
],
|
| 1581 |
-
"runtime_image": "verdictai/trellismx@sha256:609a5fc1cd7d994ba32d9c03626c414d315947eb9f13fab474a15bc8dfbe0129"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1582 |
}
|
|
|
|
| 1 |
{
|
| 2 |
"schema": "trellismx.hf-overlay-release.v1",
|
| 3 |
+
"checkpoint": "codec-v2-uniform-k4-mtp45-k4",
|
| 4 |
"packaging": "routed-overlay-requires-pinned-stock-carrier",
|
| 5 |
"source_manifest_sha256": "dd4b391c015f9738cf7e4ed2c43b57cebc859f703e3502ca76ec3eb98d7397d8",
|
| 6 |
"carrier": {
|
|
|
|
| 49 |
"6": 4,
|
| 50 |
"7": 4,
|
| 51 |
"8": 4,
|
| 52 |
+
"9": 4,
|
| 53 |
+
"45": 4
|
| 54 |
},
|
| 55 |
"payload": {
|
| 56 |
+
"coupled_metadata_bytes": 155042176,
|
| 57 |
+
"file_bytes": 165721576928,
|
| 58 |
+
"logical_elements": 311653564416,
|
| 59 |
+
"safetensors_header_bytes": 578656,
|
| 60 |
+
"stored_bpw_including_metadata": 4.2539947133553015,
|
|
|
|
|
|
|
|
|
|
| 61 |
"weight_payload_bpw": 4.25,
|
| 62 |
+
"weight_payload_bytes": 165565956096,
|
| 63 |
+
"note": "codec-v2 re-encode of every routed layer; see design/design-1.json; MTP45 K4 included; per-module details in MTP-VERIFY-REPORT.json"
|
| 64 |
},
|
| 65 |
"files": [
|
| 66 |
{
|
|
|
|
| 1574 |
"rank": 3,
|
| 1575 |
"bits": 4,
|
| 1576 |
"source_design_sha256": "82687b2cace99c681d01e84c5bef5cf5874091231072f03755b307f4a69b9c4d"
|
| 1577 |
+
},
|
| 1578 |
+
{
|
| 1579 |
+
"layer": 45,
|
| 1580 |
+
"rank": 0,
|
| 1581 |
+
"bits": 4,
|
| 1582 |
+
"path": "sidecars/p8-layer-045-tp4-rank-0.safetensors",
|
| 1583 |
+
"bytes": 963497720,
|
| 1584 |
+
"sha256": "1c699c04c08180eadad2f40a6c371295d1c46618db7029e7b7587e73a93c6dd6",
|
| 1585 |
+
"source_design_sha256": "dc3f7e59c570d09f2dcf93921b4ce060bf36dfdcc1634c6cd615229bbc055d10"
|
| 1586 |
+
},
|
| 1587 |
+
{
|
| 1588 |
+
"layer": 45,
|
| 1589 |
+
"rank": 1,
|
| 1590 |
+
"bits": 4,
|
| 1591 |
+
"path": "sidecars/p8-layer-045-tp4-rank-1.safetensors",
|
| 1592 |
+
"bytes": 963497720,
|
| 1593 |
+
"sha256": "4ac3d8d98b153f23d65c690cf0e24a98974ba52611642619544ad0311d3277c1",
|
| 1594 |
+
"source_design_sha256": "dc3f7e59c570d09f2dcf93921b4ce060bf36dfdcc1634c6cd615229bbc055d10"
|
| 1595 |
+
},
|
| 1596 |
+
{
|
| 1597 |
+
"layer": 45,
|
| 1598 |
+
"rank": 2,
|
| 1599 |
+
"bits": 4,
|
| 1600 |
+
"path": "sidecars/p8-layer-045-tp4-rank-2.safetensors",
|
| 1601 |
+
"bytes": 963497720,
|
| 1602 |
+
"sha256": "2750ae58989848196200dfa2991f40824ff6b52066f53319afd8dcc113c9bde6",
|
| 1603 |
+
"source_design_sha256": "dc3f7e59c570d09f2dcf93921b4ce060bf36dfdcc1634c6cd615229bbc055d10"
|
| 1604 |
+
},
|
| 1605 |
+
{
|
| 1606 |
+
"layer": 45,
|
| 1607 |
+
"rank": 3,
|
| 1608 |
+
"bits": 4,
|
| 1609 |
+
"path": "sidecars/p8-layer-045-tp4-rank-3.safetensors",
|
| 1610 |
+
"bytes": 963497720,
|
| 1611 |
+
"sha256": "c9071b28f75dfbbb6b16017721ddc185d2eb9ab021d81bdc1ad4d632cb846ddb",
|
| 1612 |
+
"source_design_sha256": "dc3f7e59c570d09f2dcf93921b4ce060bf36dfdcc1634c6cd615229bbc055d10"
|
| 1613 |
}
|
| 1614 |
],
|
| 1615 |
+
"runtime_image": "verdictai/trellismx@sha256:609a5fc1cd7d994ba32d9c03626c414d315947eb9f13fab474a15bc8dfbe0129",
|
| 1616 |
+
"mtp": {
|
| 1617 |
+
"layer": 45,
|
| 1618 |
+
"bits": 4,
|
| 1619 |
+
"scale_policy": "signed-unit coupled vectors; CPU torch seed 530045; no prior r27 MTP scales",
|
| 1620 |
+
"native_runtime_patch": "runtime/serve-codecv2-mtp.sh"
|
| 1621 |
+
},
|
| 1622 |
+
"main_model_payload": {
|
| 1623 |
+
"bytes_under_identity_files": -151878720,
|
| 1624 |
+
"coupled_metadata_bytes": 151436544,
|
| 1625 |
+
"file_bytes": 161867586048,
|
| 1626 |
+
"identity_k4_file_bytes": 161715707328,
|
| 1627 |
+
"logical_elements": 304405807104,
|
| 1628 |
+
"safetensors_header_bytes": 564480,
|
| 1629 |
+
"same_size_or_smaller": false,
|
| 1630 |
+
"stored_bpw_including_metadata": 4.253994694462529,
|
| 1631 |
+
"weight_payload_bpw": 4.25,
|
| 1632 |
+
"weight_payload_bytes": 161715585024,
|
| 1633 |
+
"note": "codec-v2 re-encode of every routed layer; see design/design-1.json"
|
| 1634 |
+
}
|
| 1635 |
}
|