Mellum2.1-12B-A2.5B-Thinking-AWQ-W4A16-G32 / calibration-run-receipt.json
blake-lucas's picture
Publish experimental Mellum2.1 AWQ W4A16 group-32 derivative
b43fd3a verified
Raw History Blame Contribute Delete
3.13 kB
{
"schema_version": 1,
"model_id": "JetBrains/Mellum2.1-12B-A2.5B-Thinking",
"source_revision": "92ddae9fc7665e9f801d141d2e5a6b2caf2460c4",
"quantizer_image_id": "sha256:e4f0dbe60e59050796a15b7b0c4646130a93acfd4e2431cde05923e401470b10",
"source_sha256": {
"quantize.py": "890aef9cf2ceb5c50d663a0014a51326ff960b2098fc30f10efdf2d130d0fc33",
"audit_export.py": "9c2476d66193a505f43f8982a07c053fdabfe01205af051f6d6f7c37085ac199",
"check_quantizer.py": "c8bbb692f1e82e4193df78710df3f7b6d617793d66c8dfa8f0ab4836d05572c2",
"recipe.yaml": "a73469b793eb4af8eacaa3241cecb6b699d3ccc88374e575458cf35500f6ab3a",
"requirements.quantize.txt": "3dddfbf7bf5a7cdd012a56fd80f47ae895db1468f654b196f7ca10dab2f75617",
"Dockerfile.quantize": "a6f281f8c21f025bb6e825e4efed7ce982af03e573e5a9d1214918bc6a44ec13"
},
"source_sha256_basis": "Runtime files verified against installed frozen image; Dockerfile from preserved original build context",
"run": {
"started_at": "2026-10-08T22:00:48.764973527Z",
"finished_at": "2026-10-09T01:28:31.258642816Z",
"calibration_started_at": "2026-10-08T22:01:37.723800+00:00",
"calibration_finished_at": "2026-10-09T01:11:45.235800+00:00",
"exit_code": 0,
"oom_killed": false,
"calibration_started_at_basis": "Compression lifecycle initialization log; calibration follows initialization",
"calibration_finished_at_basis": "Final propagation complete and compression lifecycle finalized log"
},
"resources": {
"initial": {
"ram_bytes": 51539607552,
"ram_swap_bytes": 60129542144
},
"final": {
"ram_bytes": 55834574848,
"ram_swap_bytes": 77309411328
},
"overrides": [
{
"at": "2026-10-08T22:25:34.159628+00:00",
"before": {
"ram_bytes": 51539607552,
"ram_swap_bytes": 60129542144
},
"after": {
"ram_bytes": 55834574848,
"ram_swap_bytes": 64424509440
},
"at_basis": "Receipt file mtime; original receipt omitted explicit timestamp"
},
{
"at": "2026-10-08T23:01:29.787460+00:00",
"before": {
"ram_bytes": 55834574848,
"ram_swap_bytes": 64424509440
},
"after": {
"ram_bytes": 55834574848,
"ram_swap_bytes": 68719476736
}
},
{
"at": "2026-10-08T23:56:24.515470+00:00",
"before": {
"ram_bytes": 55834574848,
"ram_swap_bytes": 68719476736
},
"after": {
"ram_bytes": 55834574848,
"ram_swap_bytes": 77309411328
}
}
],
"temporary_host_swap_added_bytes": 17179869184,
"persistent_os_swap_changes": false
},
"telemetry": {
"sampling_interval_seconds": 5,
"device_peak_mib": 8219,
"resident_baseline_mib": 1198,
"limitations": "Sampled total device memory includes resident retrieval services; not a per-process CUDA allocator peak. Background workload and host-memory pressure may affect timings.",
"samples": 2453
},
"quantization": {
"effective_observer": "memoryless_mse"
}
}