"""Fail-closed public release metadata and Hugging Face file-domain helpers.""" from __future__ import annotations import hashlib import json import os from pathlib import Path import re import subprocess import textwrap from typing import Any, Mapping PP_PARTITION = [9, 10, 10, 10, 10, 10, 10, 9] HESSIAN_REPO = "brandonmusic/GLM-5.2-BMM-Law-SQG-Hessians" _EXCLUDED_PREFIXES = ((".cache", "huggingface"), (".materialize_state",)) _ALLOWED_SUFFIXES = { ".conf", ".json", ".jinja", ".md", ".model", ".py", ".safetensors", ".sh", ".template", ".tiktoken", ".txt", ".yaml", ".yml", } _ALLOWED_NAMES = { ".gitattributes", "LICENSE", "NOTICE", "SHA256SUMS", "SQG_REPRODUCIBILITY_SHA256SUMS", "SQG_REPRODUCIBILITY_SOURCE.tar.gz", "SQG_RUNTIME_OVERLAY.tar.gz", "SQG_RUNTIME_OVERLAY_SHA256SUMS", } _REQUIRED_RELEASE_FILES = { "README.md", "LICENSE", "compose.yaml", "serve.sh", "HESSIAN_DATASET_LAYOUT.md", "REPRODUCE_SQG_W4A8.md", "REPRODUCTION_SOURCE_SHA256.json", "RELEASE_PROVENANCE.json", "config.json", "quantization_config.json", "model.safetensors.index.json", "FULL_SQG_NATIVE_MANIFEST.json", "FULL_SQG_NATIVE_ACCEPTANCE.json", } _IMMUTABLE_IMAGE_RE = re.compile( r"^[a-z0-9]+(?:[._-][a-z0-9]+)*(?:/[a-z0-9]+(?:[._-][a-z0-9]+)*)+" r"@sha256:[0-9a-f]{64}$" ) def validate_runtime_image_ref(image: str) -> str: if not isinstance(image, str) or not _IMMUTABLE_IMAGE_RE.fullmatch(image): raise ValueError( "VERDICTAI image must be a registry repository pinned by sha256 digest" ) return image def canonical_bytes(value: Any) -> bytes: return (json.dumps(value, indent=2, sort_keys=True) + "\n").encode() def sha256_file(path: Path) -> str: digest = hashlib.sha256() with path.open("rb") as handle: while block := handle.read(32 << 20): digest.update(block) return digest.hexdigest() def load_object(path: Path) -> dict[str, Any]: if path.is_symlink() or not path.is_file(): raise FileNotFoundError(path) value = json.loads(path.read_text(encoding="utf-8")) if not isinstance(value, dict): raise ValueError(f"expected JSON object: {path}") return value def atomic_json(path: Path, value: Mapping[str, Any]) -> None: path = path.resolve() path.parent.mkdir(parents=True, exist_ok=True) temporary = path.with_name(f".{path.name}.tmp-{os.getpid()}") try: with temporary.open("xb") as handle: handle.write(canonical_bytes(value)) handle.flush() os.fsync(handle.fileno()) os.replace(temporary, path) finally: temporary.unlink(missing_ok=True) def atomic_text(path: Path, value: str, *, executable: bool = False) -> None: path = path.resolve() path.parent.mkdir(parents=True, exist_ok=True) temporary = path.with_name(f".{path.name}.tmp-{os.getpid()}") try: with temporary.open("x", encoding="utf-8") as handle: handle.write(value) handle.flush() os.fsync(handle.fileno()) temporary.chmod(0o755 if executable else 0o644) os.replace(temporary, path) finally: temporary.unlink(missing_ok=True) def artifact(path: Path) -> dict[str, Any]: return { "path": str(path.resolve()), "bytes": path.stat().st_size, "sha256": sha256_file(path), } def hessian_binding(receipt: Path, *, final: bool) -> dict[str, Any]: del final # Every model phase now requires the final accepted dataset. value = load_object(receipt) if ( value.get("complete") is not True or value.get("public") is not True or value.get("repo") != HESSIAN_REPO or value.get("phase") != "final_publication_verified" or value.get("initial_upload_verified") is not True or value.get("build_verified") is not True or value.get("main_verified") is not True or value.get("accepted_tag_verified") is not True or value.get("revision") != "main" or value.get("main_commit") != value.get("commit") or value.get("build_remote_object_domain_sha256") != value.get("main_remote_object_domain_sha256") ): raise ValueError("final public Hessian dataset binding is absent") commit = value.get("commit") accepted_tag = value.get("accepted_tag") if not isinstance(commit, str) or len(commit) != 40: raise ValueError("verified public Hessian dataset commit is absent") if not isinstance(accepted_tag, str) or not accepted_tag.startswith("accepted-"): raise ValueError("verified public Hessian dataset tag is absent") return { "repo": value["repo"], # Dataset documentation may advance after payload acceptance. Bind the # model to the immutable accepted tag and commit, not mutable main. "revision": accepted_tag, "archive_main_revision_at_acceptance": value["revision"], "commit": commit, "receipt": artifact(receipt), "public": True, # Both staging and promotion now bind the accepted public dataset. "final": True, } def _excluded(relative: Path) -> bool: return any( relative.parts[: len(prefix)] == prefix for prefix in _EXCLUDED_PREFIXES ) def file_census(root: Path) -> dict[str, int]: """Return only reviewed public files, excluding resumable operational state.""" root = root.resolve() result: dict[str, int] = {} for path in sorted(root.rglob("*")): relative = path.relative_to(root) if _excluded(relative): continue if path.is_symlink(): raise ValueError(f"public model release contains a symlink: {relative}") if path.is_dir(): continue if not path.is_file() or len(relative.parts) != 1: raise ValueError(f"public model release contains a nested file: {relative}") if path.name not in _ALLOWED_NAMES and path.suffix not in _ALLOWED_SUFFIXES: raise ValueError(f"public model release file is not allowlisted: {relative}") result[relative.as_posix()] = path.stat().st_size missing = _REQUIRED_RELEASE_FILES - set(result) if missing: raise ValueError(f"public model release lacks required files: {sorted(missing)}") return result def prepare_release_files( *, model: Path, model_repo: str, model_revision: str, accepted: Mapping[str, Mapping[str, Any]], hessian: Mapping[str, Any], image: str, phase: str, owner_speed_publication: bool = False, ) -> dict[str, Any]: if phase not in {"build_branch", "main"}: raise ValueError(f"unknown release phase: {phase}") image = validate_runtime_image_ref(image) model = model.resolve() manifest = load_object(model / "FULL_SQG_NATIVE_MANIFEST.json") census = manifest.get("census") if ( manifest.get("activation_endpoint") != "full-w4a8" or manifest.get("codebook") != "sqg_xor_cheb_t12" or not isinstance(census, Mapping) or census.get("mcg_marker_tensors") != 0 or census.get("total_sqg_matrices") != 58_748 or manifest.get("construction", {}).get("acceptance_runtime") != "tp1_pp8" ): raise ValueError("model is not the all-SQG full-W4A8 PP8/TP1 release") if owner_speed_publication: kld = { "complete": False, "passed": False, "skipped": True, "reason": "owner_directed_speed_publication_no_kld", } runtime = { "complete": False, "passed": False, "skipped": True, "reason": "owner_directed_speed_publication_no_runtime_acceptance", } else: kld = accepted["kld"] runtime = accepted["runtime_smoke"] dataset_ref = hessian.get("commit") or hessian["revision"] if owner_speed_publication: status = ( "owner-directed unvalidated release on main; KLD and runtime acceptance were skipped" if phase == "main" else "owner-directed unvalidated build branch; KLD and runtime acceptance were skipped" ) else: status = ( "accepted release on main" if phase == "main" else "accepted build branch; main promotion awaits final dataset publication" ) provenance: dict[str, Any] = { "schema": "glm52-sqg-w4a8-public-release-provenance-v1", "complete": True, "phase": phase, "status": status, "model_repo": model_repo, "model_revision": model_revision, "official_bf16_repo": manifest["official_bf16_repo"], "official_bf16_revision": manifest["official_bf16_revision"], "activation_endpoint": "full-w4a8", "tensor_parallel_size": 1, "pipeline_parallel_size": 8, "pipeline_partition": PP_PARTITION, "codebook": manifest["codebook"], "census": dict(census), "hessian_dataset": dict(hessian), "accepted": {name: dict(value) for name, value in accepted.items()}, "owner_speed_publication": owner_speed_publication, "quality_acceptance_run": not owner_speed_publication, "runtime_acceptance_run": not owner_speed_publication, "runtime_image": image, "license_policy": "derived weights remain subject to upstream GLM-5.2 terms", } atomic_json(model / "RELEASE_PROVENANCE.json", provenance) atomic_text( model / "README.md", textwrap.dedent( f"""\ --- license: other license_name: glm-5.2-license license_link: https://huggingface.co/zai-org/GLM-5.2/blob/{manifest['official_bf16_revision']}/LICENSE library_name: vllm pipeline_tag: text-generation base_model: zai-org/GLM-5.2 datasets: - {hessian['repo']} - brandonmusic/GLM-5.2-BMM-Law-SQG-Hessians-Canonical tags: [glm, sqg, w4a8, bmm-law, mixture-of-experts, blackwell] inference: false --- # GLM-5.2 SQG W4A8 This is an all-SQG, BF16-source-only GLM-5.2 checkpoint. Routed expert tensors use independent per-tensor K3/K4 assignments (384 of each per layer); the 380 selected non-routed matrices use SQG K6. It contains 58,748 SQG matrices and zero MCG marker tensors. Routed experts use the full-W4A8 endpoint: both `h` and `SiLU(gate) * up` are A8. Dense K6 matrices are calibrated in that construction context but materialized and served as native SQG W6A16. Routed gate/up calibration uses the frozen BMM Law blend `H13 = 0.75 * H13_layer + 0.25 * H13_expert`, with routed applied-router-gate-square weighting and the sealed per-tensor rate map. Down projections are encoded from the candidate-conditioned W4A8 path using the recorded `(H, B)` cross-term construction. Hadamard rotations and input-side scales remain on the activation side so native SQG E4M3 weight labels reach FP8 MMA unchanged. Release state: **{status}**. ## Release regime - Tensor parallel: **1** - Pipeline parallel: **8**, partition `{PP_PARTITION}` - Final-logit KLD: **{'not run (owner-directed speed publication)' if kld.get('skipped') else kld.get('mean_kld')}** - Runtime acceptance: **{'not run (owner-directed speed publication)' if runtime.get('skipped') else str(runtime.get('generated_tokens')) + ' generated tokens'}** - Official BF16 revision: `{manifest['official_bf16_revision']}` - [Calibration/Hessian dataset](https://huggingface.co/datasets/{hessian['repo']}/tree/{dataset_ref}): `{dataset_ref}` - [Canonical deduplicated calibration view](https://huggingface.co/datasets/brandonmusic/GLM-5.2-BMM-Law-SQG-Hessians-Canonical/tree/canonical-v1): `canonical-v1` This format requires the custom SQG W4A8 vLLM runtime. It is not a drop-in Transformers checkpoint. Do not start it as TP8, DCP4, an SM120-only MCG image, or an A16 fallback. The accepted topology is PP8/TP1 and the supplied launch artifacts enforce it. ## Run Set `VERDICTAI_IMAGE` to the immutable image tested for this release: ```bash export VERDICTAI_IMAGE={image} ./serve.sh start ./serve.sh logs ``` The API binds to `127.0.0.1:8000` by default. Exact construction, runtime, acceptance, and dataset identities are recorded in `RELEASE_PROVENANCE.json` and the two `FULL_SQG_NATIVE_*.json` files. ## Reproduce the quantization See [`REPRODUCE_SQG_W4A8.md`](./REPRODUCE_SQG_W4A8.md) for the complete BF16-source capture, H13/H2/B calibration, per-tensor K3/K4 routed encoding, dense-K6 encoding, materialization, and publication procedure. Exact sealed scripts and source hashes are in the linked calibration dataset under `reproduction/` and `derived/lineage/reproducibility/`. [`HESSIAN_DATASET_LAYOUT.md`](./HESSIAN_DATASET_LAYOUT.md) identifies the ready-to-use H13 files, canonical activation views, dense-H inputs, and intentionally repeated archival paths so users do not download all 3.1 TB unnecessarily. ## License This repository does not relicense GLM-5.2 or third-party runtime code. The derived checkpoint remains subject to upstream GLM-5.2 terms; see `LICENSE`. """ ), ) atomic_text( model / "LICENSE", textwrap.dedent( f"""\ GLM-5.2 DERIVED CHECKPOINT NOTICE This repository contains a quantized derivative of zai-org/GLM-5.2 at revision {manifest['official_bf16_revision']}. No new license is asserted over the upstream model or third-party runtime code. Use, redistribution, and modification remain subject to the upstream GLM-5.2 license and applicable third-party notices: https://huggingface.co/zai-org/GLM-5.2/blob/{manifest['official_bf16_revision']}/LICENSE """ ), ) atomic_text( model / "compose.yaml", textwrap.dedent( """\ services: glm52: image: ${VERDICTAI_IMAGE:?set VERDICTAI_IMAGE to the accepted immutable image} restart: unless-stopped gpus: all ipc: host shm_size: "32g" ulimits: memlock: -1 nofile: 1048576 ports: - "${BIND_ADDRESS:-127.0.0.1}:${PORT:-8000}:8000" environment: CUDA_VISIBLE_DEVICES: "${CUDA_VISIBLE_DEVICES:-0,1,2,3,4,5,6,7}" VLLM_PP_LAYER_PARTITION: "9,10,10,10,10,10,10,9" VLLM_WORKER_MULTIPROC_METHOD: spawn VLLM_DEEP_GEMM_WARMUP: skip volumes: - "${MODEL_DIR:?set MODEL_DIR to this repository}:/model:ro" - "${CACHE_DIR:-./.runtime-cache}:/cache:rw" command: - vllm - serve - /model - --served-model-name - ${SERVED_MODEL_NAME:-GLM-5.2-SQG-W4A8} - --host - 0.0.0.0 - --port - "8000" - --trust-remote-code - --tensor-parallel-size - "1" - --pipeline-parallel-size - "8" - --distributed-executor-backend - mp - --quantization - exl3 - --dtype - bfloat16 - --kv-cache-dtype - bfloat16 - --load-format - safetensors - --attention-backend - FLASHMLA_SPARSE - --enforce-eager - --disable-custom-all-reduce - --gpu-memory-utilization - ${GPU_MEMORY_UTILIZATION:-0.90} - --max-model-len - ${MAX_MODEL_LEN:-262144} - --speculative-config - '{"method":"mtp","num_speculative_tokens":1}' healthcheck: test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8000/v1/models >/dev/null"] interval: 30s timeout: 10s retries: 60 start_period: 20m """ ), ) atomic_text( model / "serve.sh", textwrap.dedent( """\ #!/usr/bin/env bash set -Eeuo pipefail script_dir="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)" export MODEL_DIR="${MODEL_DIR:-$script_dir}" export CACHE_DIR="${CACHE_DIR:-$script_dir/.runtime-cache}" : "${VERDICTAI_IMAGE:?set VERDICTAI_IMAGE to the accepted immutable verdictai image}" command -v nvidia-smi >/dev/null || { echo "nvidia-smi is required" >&2; exit 2; } gpu_count="$(nvidia-smi --query-gpu=index --format=csv,noheader | wc -l)" [[ "$gpu_count" -eq 8 ]] || { echo "PP8/TP1 requires exactly 8 visible GPUs; found $gpu_count" >&2; exit 2; } for required in config.json quantization_config.json model.safetensors.index.json \ FULL_SQG_NATIVE_MANIFEST.json FULL_SQG_NATIVE_ACCEPTANCE.json; do [[ -f "$MODEL_DIR/$required" ]] || { echo "missing $MODEL_DIR/$required" >&2; exit 2; } done python3 - "$MODEL_DIR" "$VERDICTAI_IMAGE" <<'PY' import json, pathlib, sys root = pathlib.Path(sys.argv[1]) requested_image = sys.argv[2] q = json.loads((root / "quantization_config.json").read_text()) m = json.loads((root / "FULL_SQG_NATIVE_MANIFEST.json").read_text()) p = json.loads((root / "RELEASE_PROVENANCE.json").read_text()) assert q["quant_method"] == "exl3" and q["quant_algo"] == "SQG" assert q["codebook"] == "sqg_xor_cheb_t12" assert q["activation_endpoint"] == "full-w4a8" assert q["allow_a16_fallback"] is False assert m["census"]["mcg_marker_tensors"] == 0 assert m["construction"]["acceptance_runtime"] == "tp1_pp8" assert p["runtime_image"] == requested_image PY command="${1:-start}" compose=(docker compose -f "$script_dir/compose.yaml") case "$command" in start) "${compose[@]}" up -d --force-recreate ;; stop) "${compose[@]}" down ;; restart) "${compose[@]}" down; "${compose[@]}" up -d --force-recreate ;; logs) "${compose[@]}" logs --tail 100 -f glm52 ;; status) "${compose[@]}" ps ;; *) echo "usage: $0 [start|stop|restart|logs|status]" >&2; exit 2 ;; esac """ ), executable=True, ) return provenance def upload_folder( *, hf_cli: str, repo: str, revision: str, model: Path, workers: int, environment: Mapping[str, str], ) -> None: subprocess.run( [ hf_cli, "upload-large-folder", repo, str(model.resolve()), "--type", "model", "--revision", revision, "--num-workers", str(workers), "--no-bars", "--exclude", ".materialize_state/**", "--exclude", ".cache/huggingface/**", ], check=True, env=dict(environment), ) def verify_model_census( api: Any, repo: str, revision: str, local: Mapping[str, int] ) -> Any: remote_info = api.model_info(repo, revision=revision, files_metadata=True) if getattr(remote_info, "private", True): raise ValueError("accepted model repository is not public") remote = {item.rfilename: int(item.size) for item in remote_info.siblings} missing = { name: size for name, size in local.items() if name not in remote or (name != ".gitattributes" and remote[name] != size) } if missing: raise ValueError(f"remote model file census differs for {len(missing)} files") unexpected = set(remote) - set(local) - {".gitattributes"} if unexpected: raise ValueError( f"remote model contains files outside the release allowlist: " f"{sorted(unexpected)[:3]}" ) forbidden = sorted( name for name in remote if name.startswith(".materialize_state/") or name.startswith(".cache/huggingface/") ) if forbidden: raise ValueError(f"remote model contains operational state: {forbidden[:3]}") return remote_info __all__ = [ "artifact", "file_census", "hessian_binding", "prepare_release_files", "upload_folder", "verify_model_census", ]