#!/usr/bin/env python3 """CPU preflight of installed quantizer APIs, packed shapes and audit rejection.""" import importlib.metadata import argparse import json import shutil import struct import tempfile from pathlib import Path from audit_export import positive_scale_count, tensor_files from quantize import calibration_attention_outputs def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--export-dir", type=Path, help="Preserve tiny export for a serving-image schema compatibility check") parser.add_argument("--awq", action="store_true", help="Also run the actual AWQ sequential recipe; requires a reserved CUDA GPU") args = parser.parse_args() if args.export_dir is not None and args.export_dir.exists(): parser.error("Export directory already exists") import torch from llmcompressor import oneshot from llmcompressor.modeling.moe.linearize import linearize_moe from llmcompressor.modifiers.quantization import QuantizationModifier from transformers import MellumConfig, MellumForCausalLM, PreTrainedTokenizerFast from tokenizers import Tokenizer from tokenizers.models import WordLevel torch.set_num_threads(4) torch.set_num_interop_threads(1) if args.awq and not torch.cuda.is_available(): parser.error("AWQ 0.14 pins calibration memory and requires an NVIDIA driver/GPU") print("Versions:", json.dumps({p: importlib.metadata.version(p) for p in ("llmcompressor", "compressed-tensors", "transformers", "torch")})) with tempfile.TemporaryDirectory() as tmp: root = Path(tmp) def tensor(values, dtype): raw = struct.pack("<" + {"F32": "f", "F16": "e", "BF16": "H"}[dtype] * len(values), *values) path = root / "scales" path.write_bytes(raw) return path, {"dtype": dtype, "data_offsets": [0, len(raw)]}, 0 for dtype, values in (("F32", [.5, 1]), ("F16", [.5, 1]), ("BF16", [0x3f00, 0x3f80])): assert positive_scale_count(tensor(values, dtype)) == 2 for values in ([-.5, 1], [0, 1], [float("nan"), 1], [float("inf"), 1]): try: positive_scale_count(tensor(values, "F32")) except ValueError: pass else: raise AssertionError("Invalid scale accepted") header = json.dumps({"x": {"dtype": "F32", "shape": [2], "data_offsets": [0, 8]}}).encode() (root / "model.safetensors").write_bytes(struct.pack("": 0, "": 1, **{f"t{i}": i for i in range(2, 128)}}, unk_token="")), unk_token="", pad_token="", model_max_length=128) tokenizer.save_pretrained(source) cfg._attn_implementation = "eager" model = MellumForCausalLM(cfg).to(torch.bfloat16).eval() model.name_or_path = str(source) model.config._name_or_path = str(source) linearize_moe(model) # The synthetic in-memory conversion creates Linear modules in the # default dtype; production load_context loads their source BF16 weights. model.to(torch.bfloat16) assert sum(".experts." in n and isinstance(m, torch.nn.Linear) for n, m in model.named_modules()) == 6 inputs = torch.randint(0, cfg.vocab_size, (1, 16)) with torch.no_grad(): reference = model(input_ids=inputs, use_cache=False).logits with calibration_attention_outputs(model) as hooks: assert len(hooks) == 1 hooked = model(input_ids=inputs, use_cache=False).logits restored = model(input_ids=inputs, use_cache=False).logits assert torch.equal(reference, hooked) and torch.equal(reference, restored) print("Calibration attention hook: bit-identical full-model logits and removal verified") output = root / "toy-export" if args.awq: dataset = torch.utils.data.DataLoader([ {"input_ids": torch.randint(0, cfg.vocab_size, (16,))} for _ in range(4)], batch_size=1) with calibration_attention_outputs(model): oneshot(model=model, processor=tokenizer, dataset=dataset, recipe=str(Path(__file__).with_name("recipe.yaml")), pipeline="sequential", sequential_targets=["MellumDecoderLayer"], sequential_offload_device="cpu", sequential_prefetch=False, max_seq_length=16, num_calibration_samples=4, output_dir=str(output), save_compressed=True) else: recipe = QuantizationModifier(config_groups={"group_0": { "targets": ["Linear"], "weights": { "num_bits": 4, "type": "int", "symmetric": True, "group_size": 32, "strategy": "group", "dynamic": False, "observer": "mse"}, "input_activations": None, "output_activations": None}}, ignore=["lm_head", "re:.*mlp[.]gate$"]) oneshot(model=model, processor=tokenizer, recipe=recipe, pipeline="datafree", output_dir=str(output), save_compressed=True) headers = tensor_files(output) shapes = {"model.layers.0.mlp.experts.0.gate_proj.weight_packed": [32, 8], "model.layers.0.mlp.experts.0.gate_proj.weight_scale": [32, 2], "model.layers.0.mlp.experts.0.down_proj.weight_packed": [64, 4], "model.layers.0.mlp.experts.0.down_proj.weight_scale": [64, 1]} for name, shape in shapes.items(): assert headers[name][1]["shape"] == shape, (name, headers[name][1]) if name.endswith("weight_scale"): positive_scale_count(headers[name]) print("Packed expert export shapes:", json.dumps({n: headers[n][1] for n in shapes})) if args.export_dir is not None: shutil.copytree(output, args.export_dir) mode = "CUDA AWQ" if args.awq else "CPU" print(f"{mode} preflight passed: compressed shapes, hook equality/removal and audit rejection") if __name__ == "__main__": main()