Feature Extraction
MLX
Safetensors
ministral3
apple-silicon
quantized
mixed-precision
axquant
axq
development
experimental
mistral3
6bit
6-bit
embedding
sentence-similarity
4-bit precision
Instructions to use AutomatosX/AX-Nemotron-3-Embed-8B-MLX-AXQ-6bit with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- MLX
How to use AutomatosX/AX-Nemotron-3-Embed-8B-MLX-AXQ-6bit with MLX:
# Download the model from the Hub pip install huggingface_hub[hf_xet] huggingface-cli download --local-dir AX-Nemotron-3-Embed-8B-MLX-AXQ-6bit AutomatosX/AX-Nemotron-3-Embed-8B-MLX-AXQ-6bit
- Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- LM Studio
- Atomic Chat
Download model-manifest.json from AutomatosX/AX-Nemotron-3-Embed-8B-MLX-AXQ-6bit: direct link, hf CLI and curl.
- Browser
- Download file 128 kB
-
https://huggingface.co/AutomatosX/AX-Nemotron-3-Embed-8B-MLX-AXQ-6bit/resolve/main/model-manifest.json
- Command line
-
hf download hf://AutomatosX/AX-Nemotron-3-Embed-8B-MLX-AXQ-6bit/model-manifest.json
-
curl -L -o model-manifest.json https://huggingface.co/AutomatosX/AX-Nemotron-3-Embed-8B-MLX-AXQ-6bit/resolve/main/model-manifest.json
128 kB
| { | |
| "schema_version": "ax.native_model.v1", | |
| "model_family": "mistral3", | |
| "tensor_format": "safetensors", | |
| "layer_count": 34, | |
| "hidden_size": 4096, | |
| "intermediate_size": 14336, | |
| "attention_head_count": 32, | |
| "attention_head_dim": 128, | |
| "kv_head_count": 8, | |
| "vocab_size": 131072, | |
| "tie_word_embeddings": true, | |
| "rope_theta": 1000000, | |
| "no_rope_layer_interval": 0, | |
| "intermediate_size_mlp": 0, | |
| "attn_output_gate": false, | |
| "rms_norm_eps": 0.00001, | |
| "moe_norm_topk_prob": false, | |
| "hidden_size_per_layer_input": 0, | |
| "tensors": [ | |
| { | |
| "name": "model.norm.weight", | |
| "role": "final_norm", | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 8778, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.embed_tokens.weight", | |
| "role": "token_embedding", | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 8 | |
| }, | |
| "shape": [ | |
| 131072, | |
| 1024 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4689411227, | |
| "length_bytes": 536870912 | |
| }, | |
| { | |
| "name": "model.layers.0.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 0, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4653235355, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.0.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 0, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4526480539, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.0.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 0, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4258421915, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.0.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 0, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1771117723, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.0.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 0, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4295253147, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.0.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 0, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1953225883, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.0.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 0, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 1792 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4614700187, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.0.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 0, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4170726555, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.0.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 0, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4356070555, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.1.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 1, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4582587547, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.1.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 1, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4495801499, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.1.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 1, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4529634459, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.1.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 1, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3489422491, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.1.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 1, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4602117275, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.1.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 1, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4428553371, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.1.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 1, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 1792 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3934370971, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.1.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 1, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4387265691, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.1.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 1, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3780975771, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.2.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 2, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4311506075, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.2.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 2, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4491730075, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.2.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 2, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4208868507, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.2.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 2, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3361619099, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.2.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 2, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4333001883, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.2.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 2, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3168812187, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.2.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 2, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 1792 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4090371227, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.2.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 2, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3328588955, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.2.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 2, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1880325275, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.3.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 3, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3774020763, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.3.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 3, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3875642523, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.3.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 3, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4277820571, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.3.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 3, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4225252507, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.3.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 3, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4076608667, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.3.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 3, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4352924827, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.3.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 3, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 1792 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3415104667, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.3.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 3, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1553136795, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.3.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 3, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3523378331, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.4.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 4, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4025212059, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.4.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 4, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3571612827, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.4.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 4, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2171755675, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.4.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 4, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3865525403, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.4.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 4, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3249945755, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.4.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 4, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2691112091, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.4.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 4, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 1792 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3875650715, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.4.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 4, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 5271764123, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.4.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 4, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3905010843, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.5.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 5, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3847429275, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.5.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 5, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3621182619, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.5.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 5, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3814005915, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.5.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 5, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4055235739, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.5.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 5, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3390987419, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.5.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 5, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3836943515, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.5.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 5, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 1792 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2502745243, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.5.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 5, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3262528667, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.5.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 5, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4496989339, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.6.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 6, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3572276379, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.6.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 6, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3780967579, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.6.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 6, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4012629147, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.6.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 6, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3028401307, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.6.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 6, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3591806107, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.6.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 6, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1808358555, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.6.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 6, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 1792 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4660051099, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.6.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 6, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3295558811, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.6.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 6, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 5231787163, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.7.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 7, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3555884187, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.7.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 7, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3032210587, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.7.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 7, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3110354075, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.7.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 7, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3109821595, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.7.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 7, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2316467355, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.7.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 7, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4274674843, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.7.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 7, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 1792 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3035896987, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.7.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 7, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3077315739, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.7.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 7, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3489561755, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.8.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 8, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3829079195, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.8.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 8, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2919586971, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.8.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 8, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3012148379, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.8.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 8, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2919447707, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.8.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 8, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2968632475, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.8.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 8, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3028540571, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.8.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 8, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 1792 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2022956187, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.8.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 8, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2981215387, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.8.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 8, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2919726235, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.9.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 9, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4072676507, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.9.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 9, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2673409179, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.9.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 9, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2803834011, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.9.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 9, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2673400987, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.9.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 9, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4589403291, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.9.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 9, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4200086683, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.9.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 9, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 1792 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2707102875, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.9.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 9, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2742492315, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.9.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 9, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2873572507, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.10.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 10, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2644827291, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.10.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 10, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2502212763, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.10.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 10, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2587942043, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.10.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 10, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4028357787, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.10.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 10, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2097536155, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.10.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 10, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3171957915, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.10.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 10, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 1792 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4433009819, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.10.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 10, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2555698331, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.10.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 10, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1714355355, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.11.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 11, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2846432411, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.11.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 11, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2349104283, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.11.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 11, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2349636763, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.11.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 11, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2859932827, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.11.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 11, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3608059035, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.11.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 11, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2378996891, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.11.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 11, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 1792 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2382142619, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.11.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 11, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3220585627, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.11.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 11, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2816679067, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.12.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 12, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2331016347, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.12.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 12, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1946254491, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.12.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 12, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2234670235, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.12.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 12, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2458164379, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.12.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 12, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2334686363, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.12.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 12, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2249620635, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.12.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 12, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 1792 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4542217371, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.12.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 12, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2204785819, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.12.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 12, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2139225243, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.13.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 13, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2114190491, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.13.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 13, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1949547675, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.13.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 13, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2648104091, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.13.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 13, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1949539483, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.13.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 13, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2790202523, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.13.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 13, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2535775387, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.13.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 13, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 1792 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1960565915, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.13.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 13, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2063457435, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.13.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 13, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2252766363, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.14.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 14, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4579441819, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.14.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 14, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1863802011, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.14.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 14, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2191678619, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.14.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 14, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2114051227, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.14.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 14, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2126642331, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.14.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 14, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1943108763, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.14.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 14, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 1792 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3191094427, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.14.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 14, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1909685403, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.14.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 14, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4225260699, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.15.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 15, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4028365979, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.15.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 15, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2171731099, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.15.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 15, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1751194779, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.15.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 15, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1747385499, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.15.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 15, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1851219099, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.15.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 15, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1843223707, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.15.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 15, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 1792 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1771125915, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.15.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 15, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1811635355, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.15.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 15, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3676634267, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.16.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 16, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1743715483, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.16.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 16, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3621174427, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.16.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 16, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2860465307, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.16.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 16, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2958916763, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.16.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 16, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 5307939995, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.16.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 16, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1711078555, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.16.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 16, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 1792 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4462369947, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.16.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 16, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1586166939, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.16.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 16, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1989926043, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.17.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 17, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3588660379, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.17.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 17, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1394261147, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.17.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 17, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1444871323, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.17.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 17, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3621166235, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.17.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 17, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3575553179, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.17.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 17, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1501625499, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.17.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 17, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 1792 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1405811867, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.17.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 17, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2286058651, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.17.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 17, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1505295515, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.18.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 18, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1377483931, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.18.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 18, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1325964443, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.18.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 18, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1359002779, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.18.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 18, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3875634331, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.18.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 18, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1381678235, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.18.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 18, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1371716763, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.18.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 18, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 1792 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1329642651, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.18.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 18, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3122936987, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.18.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 18, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3361627291, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.19.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 19, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 5226282139, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.19.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 19, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4168490139, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.19.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 19, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1294376091, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.19.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 19, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1197767835, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.19.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 19, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1313381531, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.19.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 19, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1307221147, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.19.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 19, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 1792 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1235131547, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.19.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 19, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1264491675, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.19.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 19, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 32, | |
| "bits": 4 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 512 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1197907099, | |
| "length_bytes": 29360128 | |
| }, | |
| { | |
| "name": "model.layers.20.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 20, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1179548827, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.20.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 20, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1804680347, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.20.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 20, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2538921115, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.20.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 20, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2171747483, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.20.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 20, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1185184923, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.20.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 20, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1176140955, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.20.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 20, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 2688 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1457585307, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.20.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 20, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1129610395, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.20.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 20, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4120255643, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.21.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 21, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 5264948379, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.21.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 21, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 953564315, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.21.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 21, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1092385947, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.21.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 21, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 953556123, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.21.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 21, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1113357467, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.21.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 21, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1105755291, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.21.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 21, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 2688 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1003248795, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.21.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 21, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1047297179, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.21.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 21, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 955538587, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.22.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 22, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 937295003, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.22.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 22, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3861322907, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.22.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 22, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 921296027, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.22.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 22, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 850238619, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.22.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 22, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 940965019, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.22.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 22, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 934141083, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.22.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 22, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 2688 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3632594075, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.22.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 22, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 876723355, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.22.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 22, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1663237275, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.23.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 23, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 846568603, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.23.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 23, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2247253147, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.23.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 23, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 833723547, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.23.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 23, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 741702811, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.23.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 23, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 850246811, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.23.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 23, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3552738459, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.23.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 23, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 2688 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 786275483, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.23.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 23, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3729980571, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.23.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 23, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 741711003, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.24.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 24, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 725449883, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.24.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 24, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 937286811, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.24.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 24, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 708934811, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.24.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 24, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 607476891, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.24.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 24, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 729119899, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.24.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 24, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 721779867, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.24.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 24, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 2688 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 612990107, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.24.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 24, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 664370331, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.24.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 24, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1619197083, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.25.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 25, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4321991835, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.25.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 25, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 496319643, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.25.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 25, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 547708059, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.25.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 25, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1398463643, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.25.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 25, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2660818075, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.25.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 25, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 604331163, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.25.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 25, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 2688 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2458172571, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.25.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 25, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 560290971, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.25.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 25, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 498162843, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.26.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 26, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2168585371, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.26.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 26, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3035888795, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.26.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 26, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 481508507, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.26.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 26, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3865795739, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.26.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 26, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2694519963, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.26.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 26, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2684165275, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.26.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 26, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 2688 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 390544539, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.26.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 26, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 436419739, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.26.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 26, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2600656027, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.27.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 27, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 385563803, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.27.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 27, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2171739291, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.27.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 27, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 324353179, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.27.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 27, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 953547931, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.27.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 27, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3156229275, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.27.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 27, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 381238427, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.27.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 27, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 2688 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 280312987, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.27.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 27, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2411633819, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.27.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 27, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3964779675, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.28.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 28, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 272972955, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.28.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 28, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2964954267, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.28.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 28, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 2362219675, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.28.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 28, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 920763547, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.28.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 28, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3709664411, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.28.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 28, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4043570331, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.28.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 28, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 2688 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 337198235, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.28.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 28, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 228801691, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.28.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 28, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 182926491, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.29.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 29, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1615527067, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.29.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 29, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 4012620955, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.29.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 29, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3559029915, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.29.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 29, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 1047288987, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.29.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 29, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 170343579, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.29.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 29, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 167197851, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.29.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 29, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 2688 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 123026587, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.29.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 29, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3445251227, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.29.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 29, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 78986395, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.30.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 30, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3106675867, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.30.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 30, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 259170890, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.30.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 30, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 47791259, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.30.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 30, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 622928458, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.30.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 30, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 64568475, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.30.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 30, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 60374171, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.30.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 30, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 2688 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 576791114, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.30.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 30, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00001-of-00002.safetensors", | |
| "offset_bytes": 3751067, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.30.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 30, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 170566218, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.31.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 31, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 539820618, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.31.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 31, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 328254026, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.31.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 31, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 559358538, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.31.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 31, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 490267210, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.31.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 31, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 333767242, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.31.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 31, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 265339466, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.31.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 31, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 2688 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 394846794, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.31.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 31, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 446227018, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.31.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 31, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 346350154, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.32.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 32, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 268485194, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.32.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 32, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 559350346, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.32.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 32, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 544277066, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.32.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 32, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 166756938, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.32.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 32, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 315671114, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.32.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 32, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 261538378, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.32.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 32, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 2688 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 214606410, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.32.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 32, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 271630922, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.32.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 32, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 495780426, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.33.self_attn.k_proj.weight", | |
| "role": "attention_k", | |
| "layer_index": 33, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 490275402, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.33.input_layernorm.weight", | |
| "role": "attention_norm", | |
| "layer_index": 33, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 25162, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.33.self_attn.o_proj.weight", | |
| "role": "attention_o", | |
| "layer_index": 33, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 137658954, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.33.post_attention_layernorm.weight", | |
| "role": "attention_post_norm", | |
| "layer_index": 33, | |
| "dtype": "bf16", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 16970, | |
| "length_bytes": 8192 | |
| }, | |
| { | |
| "name": "model.layers.33.self_attn.q_proj.weight", | |
| "role": "attention_q", | |
| "layer_index": 33, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 768 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 154174026, | |
| "length_bytes": 12582912 | |
| }, | |
| { | |
| "name": "model.layers.33.self_attn.v_proj.weight", | |
| "role": "attention_v", | |
| "layer_index": 33, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 1024, | |
| 768 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 150372938, | |
| "length_bytes": 3145728 | |
| }, | |
| { | |
| "name": "model.layers.33.mlp.down_proj.weight", | |
| "role": "ffn_down", | |
| "layer_index": 33, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 4096, | |
| 2688 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 47743562, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.33.mlp.gate_proj.weight", | |
| "role": "ffn_gate", | |
| "layer_index": 33, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 93618762, | |
| "length_bytes": 44040192 | |
| }, | |
| { | |
| "name": "model.layers.33.mlp.up_proj.weight", | |
| "role": "ffn_up", | |
| "layer_index": 33, | |
| "dtype": "u32", | |
| "source_quantized": true, | |
| "quantization": { | |
| "mode": "affine", | |
| "group_size": 64, | |
| "bits": 6 | |
| }, | |
| "shape": [ | |
| 14336, | |
| 768 | |
| ], | |
| "file": "model-00002-of-00002.safetensors", | |
| "offset_bytes": 33354, | |
| "length_bytes": 44040192 | |
| } | |
| ] | |
| } |