Mellum2.1-12B-A2.5B-Thinking-AWQ-W4A16-G32 / quantization-recipe.yaml
blake-lucas's picture
Publish experimental Mellum2.1 AWQ W4A16 group-32 derivative
b43fd3a verified
Raw History Blame Contribute Delete
1.51 kB
# W4A16 metadata is deliberate. Explicit W4A8 metadata enters the fork's
# CPU-only W4A8 MoE adapter; Marlin's INT8 activation override is a runtime choice.
default_stage:
default_modifiers:
AWQModifier:
mappings:
- smooth_layer: re:model.*input_layernorm$
balance_layers:
- re:model.*self_attn[.]q_proj$
- re:model.*self_attn[.]k_proj$
- re:model.*self_attn[.]v_proj$
- smooth_layer: re:model.*post_attention_layernorm$
balance_layers:
# The BF16 router must be balanced too, preserving its logits when
# the shared input normalization is smoothed.
- re:model.*mlp[.]gate$
- re:model.*mlp[.]experts.*gate_proj$
- re:model.*mlp[.]experts.*up_proj$
- smooth_layer: re:model.*mlp[.]experts.*up_proj$
balance_layers:
- re:model.*mlp[.]experts.*down_proj$
duo_scaling: both
n_grid: 20
offload_device: cpu
QuantizationModifier:
config_groups:
group_0:
targets: [Linear]
weights:
num_bits: 4
type: int
symmetric: true
group_size: 32
strategy: group
dynamic: false
observer: mse
input_activations: null
output_activations: null
ignore:
- lm_head
- re:.*embed_tokens.*
- re:.*mlp[.]gate$
- re:.*mtp.*
bypass_divisibility_checks: false