Feature Extraction
MLX
Safetensors
ministral3
apple-silicon
quantized
mixed-precision
axquant
axq
development
experimental
mistral3
6bit
6-bit
embedding
sentence-similarity
4-bit precision
Instructions to use AutomatosX/AX-Nemotron-3-Embed-1B-MLX-AXQ-6bit with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- MLX
How to use AutomatosX/AX-Nemotron-3-Embed-1B-MLX-AXQ-6bit with MLX:
# Download the model from the Hub pip install huggingface_hub[hf_xet] huggingface-cli download --local-dir AX-Nemotron-3-Embed-1B-MLX-AXQ-6bit AutomatosX/AX-Nemotron-3-Embed-1B-MLX-AXQ-6bit
- Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- LM Studio
- Atomic Chat
Download model-manifest.json from AutomatosX/AX-Nemotron-3-Embed-1B-MLX-AXQ-6bit: direct link, hf CLI and curl.
- Browser
- Download file 58.2 kB
-
https://huggingface.co/AutomatosX/AX-Nemotron-3-Embed-1B-MLX-AXQ-6bit/resolve/main/model-manifest.json
- Command line
-
hf download hf://AutomatosX/AX-Nemotron-3-Embed-1B-MLX-AXQ-6bit/model-manifest.json
-
curl -L -o model-manifest.json https://huggingface.co/AutomatosX/AX-Nemotron-3-Embed-1B-MLX-AXQ-6bit/resolve/main/model-manifest.json
58.2 kB
| { | |
| "schema_version": "ax.native_model.v1", | |
| "model_family": "mistral3", | |
| "tensor_format": "safetensors", | |
| "layer_count": 16, | |
| "hidden_size": 2048, | |
| "intermediate_size": 6144, | |
| "attention_head_count": 24, | |
| "attention_head_dim": 128, | |
| "kv_head_count": 8, | |
| "vocab_size": 131072, | |
| "tie_word_embeddings": true, | |
| "rope_theta": 1000000, | |
| "no_rope_layer_interval": 0, | |
| "intermediate_size_mlp": 0, | |
| "attn_output_gate": false, | |
| "rms_norm_eps": 0.00001, | |
| "moe_norm_topk_prob": false, | |
| "hidden_size_per_layer_input": 0, | |
| "tensors": [ | |
| { | |
| "name": "model.norm.weight", | |
| "role": "final_norm", | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 41508, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.embed_tokens.weight", | |
| "role": "token_embedding", | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 8 | |
| }, | |
| "shape": [ | |
| 131072, | |
| 512 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 548835876, | |
| "length_bytes": 268435456 | |
| }, | |
| { | |
| "name": "model.layers.0.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 0, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 537301540, | |
| "length_bytes": 1048576 | |
| }, | |
| { | |
| "name": "model.layers.0.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 0, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 504197668, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.0.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 0, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 434061860, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.0.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 0, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 839950884, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.0.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 0, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 3072, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 445600292, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.0.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 0, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 458773028, | |
| "length_bytes": 1048576 | |
| }, | |
| { | |
| "name": "model.layers.0.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 0, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 768 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 529367588, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.0.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 0, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 406405668, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.0.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 0, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 467685924, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.1.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 1, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 520323620, | |
| "length_bytes": 1048576 | |
| }, | |
| { | |
| "name": "model.layers.1.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 1, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 491282980, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.1.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 1, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 506233380, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.1.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 1, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 265794084, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.1.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 1, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 3072, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 526221860, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.1.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 1, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 482300452, | |
| "length_bytes": 1048576 | |
| }, | |
| { | |
| "name": "model.layers.1.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 1, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 768 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 833659428, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.1.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 1, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 473977380, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.1.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 1, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 301916708, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.2.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 2, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 448746020, | |
| "length_bytes": 1048576 | |
| }, | |
| { | |
| "name": "model.layers.2.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 2, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 489706020, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.2.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 2, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 576 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 415121956, | |
| "length_bytes": 4718592 | |
| }, | |
| { | |
| "name": "model.layers.2.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 2, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 259039780, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.2.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 2, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 3072, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 459821604, | |
| "length_bytes": 4718592 | |
| }, | |
| { | |
| "name": "model.layers.2.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 2, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 223912484, | |
| "length_bytes": 1572864 | |
| }, | |
| { | |
| "name": "model.layers.2.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 2, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 768 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 398082596, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.2.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 2, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 248029732, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.2.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 2, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 450187812, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.3.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 3, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 547263012, | |
| "length_bytes": 1572864 | |
| }, | |
| { | |
| "name": "model.layers.3.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 3, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 328528420, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.3.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 3, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 576 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 438780452, | |
| "length_bytes": 4718592 | |
| }, | |
| { | |
| "name": "model.layers.3.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 3, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 832213540, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.3.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 3, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 3072, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 393036324, | |
| "length_bytes": 4718592 | |
| }, | |
| { | |
| "name": "model.layers.3.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 3, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 466113060, | |
| "length_bytes": 1572864 | |
| }, | |
| { | |
| "name": "model.layers.3.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 3, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 768 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 367014436, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.3.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 3, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 373305892, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.3.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 3, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 273400356, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.4.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 4, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 362418724, | |
| "length_bytes": 1572864 | |
| }, | |
| { | |
| "name": "model.layers.4.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 4, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 351273508, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.4.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 4, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 576 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 498561572, | |
| "length_bytes": 4718592 | |
| }, | |
| { | |
| "name": "model.layers.4.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 4, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 325771812, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.4.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 4, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 3072, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 235971108, | |
| "length_bytes": 4718592 | |
| }, | |
| { | |
| "name": "model.layers.4.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 4, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 356061732, | |
| "length_bytes": 1572864 | |
| }, | |
| { | |
| "name": "model.layers.4.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 4, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 768 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 328991268, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.4.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 4, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 825528868, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.4.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 4, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 335282724, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.5.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 5, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 321446436, | |
| "length_bytes": 1572864 | |
| }, | |
| { | |
| "name": "model.layers.5.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 5, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 290181668, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.5.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 5, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 576 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 308208164, | |
| "length_bytes": 4718592 | |
| }, | |
| { | |
| "name": "model.layers.5.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 5, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 382022180, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.5.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 5, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 3072, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 259043876, | |
| "length_bytes": 4718592 | |
| }, | |
| { | |
| "name": "model.layers.5.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 5, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 319087140, | |
| "length_bytes": 1572864 | |
| }, | |
| { | |
| "name": "model.layers.5.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 5, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 768 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 290972196, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.5.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 5, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 426918436, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.5.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 5, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 492270116, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.6.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 6, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 281723428, | |
| "length_bytes": 1572864 | |
| }, | |
| { | |
| "name": "model.layers.6.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 6, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 363995684, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.6.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 6, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 576 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 357700132, | |
| "length_bytes": 4718592 | |
| }, | |
| { | |
| "name": "model.layers.6.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 6, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 444613156, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.6.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 6, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 3072, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 284279332, | |
| "length_bytes": 4718592 | |
| }, | |
| { | |
| "name": "model.layers.6.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 6, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 205562404, | |
| "length_bytes": 1572864 | |
| }, | |
| { | |
| "name": "model.layers.6.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 6, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 768 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 540971556, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.6.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 6, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 241672740, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.6.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 6, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 817664548, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.7.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 7, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 279691812, | |
| "length_bytes": 1572864 | |
| }, | |
| { | |
| "name": "model.layers.7.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 7, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 196907556, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.7.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 7, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 576 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 218341924, | |
| "length_bytes": 4718592 | |
| }, | |
| { | |
| "name": "model.layers.7.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 7, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 290177572, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.7.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 7, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 3072, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 254321188, | |
| "length_bytes": 4718592 | |
| }, | |
| { | |
| "name": "model.layers.7.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 7, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 437207588, | |
| "length_bytes": 1572864 | |
| }, | |
| { | |
| "name": "model.layers.7.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 7, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 768 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 198484516, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.7.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 7, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 208708132, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.7.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 7, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 265798180, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.8.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 8, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 317514276, | |
| "length_bytes": 1572864 | |
| }, | |
| { | |
| "name": "model.layers.8.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 8, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 166232612, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.8.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 8, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 576 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 189239844, | |
| "length_bytes": 4718592 | |
| }, | |
| { | |
| "name": "model.layers.8.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 8, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 166228516, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.8.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 8, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 3072, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 351277604, | |
| "length_bytes": 4718592 | |
| }, | |
| { | |
| "name": "model.layers.8.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 8, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 194744868, | |
| "length_bytes": 1572864 | |
| }, | |
| { | |
| "name": "model.layers.8.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 8, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 768 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 173314596, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.8.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 8, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 182358564, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.8.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 8, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 166236708, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.9.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 9, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 390873636, | |
| "length_bytes": 1572864 | |
| }, | |
| { | |
| "name": "model.layers.9.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 9, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 125657636, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.9.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 9, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 576 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 146436644, | |
| "length_bytes": 4718592 | |
| }, | |
| { | |
| "name": "model.layers.9.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 9, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 125653540, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.9.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 9, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 3072, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 521372196, | |
| "length_bytes": 4718592 | |
| }, | |
| { | |
| "name": "model.layers.9.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 9, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 412697124, | |
| "length_bytes": 1572864 | |
| }, | |
| { | |
| "name": "model.layers.9.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 9, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 768 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 128938532, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.9.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 9, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 136802852, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.9.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 9, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 159937060, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.10.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 10, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 121197092, | |
| "length_bytes": 1048576 | |
| }, | |
| { | |
| "name": "model.layers.10.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 10, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 102187556, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.10.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 10, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 117789220, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.10.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 10, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 363991588, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.10.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 10, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 3072, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 122376740, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.10.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 10, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 538350116, | |
| "length_bytes": 1048576 | |
| }, | |
| { | |
| "name": "model.layers.10.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 10, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 768 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 483414564, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.10.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 10, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 111104548, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.10.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 10, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 102191652, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.11.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 11, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 157708836, | |
| "length_bytes": 1048576 | |
| }, | |
| { | |
| "name": "model.layers.11.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 11, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 89596452, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.11.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 11, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 89993764, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.11.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 11, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 536904228, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.11.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 11, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 3072, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 312926756, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.11.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 11, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 94319140, | |
| "length_bytes": 1048576 | |
| }, | |
| { | |
| "name": "model.layers.11.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 11, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 768 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 95367716, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.11.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 11, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 341574180, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.11.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 11, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 151286308, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.12.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 12, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 85008932, | |
| "length_bytes": 1048576 | |
| }, | |
| { | |
| "name": "model.layers.12.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 12, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 51708452, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.12.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 12, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 74129956, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.12.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 12, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 102183460, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.12.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 12, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 3072, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 86450724, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.12.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 12, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 77406756, | |
| "length_bytes": 1048576 | |
| }, | |
| { | |
| "name": "model.layers.12.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 12, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 768 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 509379108, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.12.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 12, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 67445284, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.12.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 12, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 56431140, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.13.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 13, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 50659876, | |
| "length_bytes": 1048576 | |
| }, | |
| { | |
| "name": "model.layers.13.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 13, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 33481252, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.13.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 13, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 215196196, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.13.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 13, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 33477156, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.13.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 13, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 3072, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 143094308, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.13.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 13, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 109269540, | |
| "length_bytes": 1048576 | |
| }, | |
| { | |
| "name": "model.layers.13.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 13, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 768 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 35058212, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.13.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 13, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 43708964, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.13.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 13, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 78455332, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.14.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 14, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 519275044, | |
| "length_bytes": 1048576 | |
| }, | |
| { | |
| "name": "model.layers.14.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 14, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 24166948, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.14.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 14, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 64299556, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.14.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 14, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 50655780, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.14.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 14, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 3072, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 53285412, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.14.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 14, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 443564580, | |
| "length_bytes": 1048576 | |
| }, | |
| { | |
| "name": "model.layers.14.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 14, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 768 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 229679652, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.14.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 14, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 26530340, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.14.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 14, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 420626980, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.15.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 15, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 363999780, | |
| "length_bytes": 1048576 | |
| }, | |
| { | |
| "name": "model.layers.15.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 15, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 62722596, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.15.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 15, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 384 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 15778340, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.15.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 15, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 2048 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 45604, | |
| "length_bytes": 4096 | |
| }, | |
| { | |
| "name": "model.layers.15.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 15, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 3072, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 21021220, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.15.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 15, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 18924068, | |
| "length_bytes": 1048576 | |
| }, | |
| { | |
| "name": "model.layers.15.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 15, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 2048, | |
| 768 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 836132, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.15.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 15, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 8700452, | |
| "length_bytes": 6291456 | |
| }, | |
| { | |
| "name": "model.layers.15.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 15, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 6144, | |
| 256 | |
| ], | |
| "file": "model.safetensors", | |
| "offset_bytes": 849130020, | |
| "length_bytes": 6291456 | |
| } | |
| ] | |
| } |