File size: 1,674 Bytes
8964653
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
{
  "schema_version": 1,
  "artifact": "EigenLabs/Qwen3.5-35B-A3B-MLX-VL-4bit-g64",
  "checks": {
    "source_cache_verification": {
      "status": "pass",
      "checked_files": 27
    },
    "safetensors_index_exact_coverage": {
      "status": "pass",
      "indexed_tensors": 2136,
      "shards": 5
    },
    "all_quantized_modules_are_affine_w4_g64": {
      "status": "pass",
      "quantized_modules": 525,
      "eight_bit_modules": 0
    },
    "inline_mtp_structure": {
      "status": "pass",
      "source_tensors": 785,
      "serialized_tensors": 46,
      "source_parameters": 844640768
    },
    "inline_mtp_affine_spot_checks": {
      "status": "pass",
      "fc_4bit_mean_abs_error": 0.0005984869785606861,
      "q_proj_4bit_mean_abs_error": 0.0014587895711883903,
      "expert0_gate_4bit_mean_abs_error": 0.0012867144541814923,
      "router_4bit_mean_abs_error": 0.0008649227092973888,
      "shifted_norm_max_abs_error": 0.0
    },
    "chat_template_source_parity": {
      "status": "pass",
      "sha256": "a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715"
    },
    "target_only_benchmark": {
      "status": "pass",
      "mlx_vlm_version": "0.6.15",
      "prompt_tokens": 1701,
      "generation_tokens": 128,
      "measured_iterations": 3,
      "median_prefill_tokens_per_second": 1642.0583817114323,
      "median_decode_tokens_per_second": 113.67484303705231,
      "peak_memory_gb": 23.252961806,
      "mtp_active": false
    },
    "image_inference": {
      "status": "not_run"
    },
    "mtp_speculative_parity": {
      "status": "not_run"
    },
    "video_inference": {
      "status": "not_run"
    }
  }
}