# W4A16 metadata is deliberate. Explicit W4A8 metadata enters the fork's # CPU-only W4A8 MoE adapter; Marlin's INT8 activation override is a runtime choice. default_stage: default_modifiers: AWQModifier: mappings: - smooth_layer: re:model.*input_layernorm$ balance_layers: - re:model.*self_attn[.]q_proj$ - re:model.*self_attn[.]k_proj$ - re:model.*self_attn[.]v_proj$ - smooth_layer: re:model.*post_attention_layernorm$ balance_layers: # The BF16 router must be balanced too, preserving its logits when # the shared input normalization is smoothed. - re:model.*mlp[.]gate$ - re:model.*mlp[.]experts.*gate_proj$ - re:model.*mlp[.]experts.*up_proj$ - smooth_layer: re:model.*mlp[.]experts.*up_proj$ balance_layers: - re:model.*mlp[.]experts.*down_proj$ duo_scaling: both n_grid: 20 offload_device: cpu QuantizationModifier: config_groups: group_0: targets: [Linear] weights: num_bits: 4 type: int symmetric: true group_size: 32 strategy: group dynamic: false observer: mse input_activations: null output_activations: null ignore: - lm_head - re:.*embed_tokens.* - re:.*mlp[.]gate$ - re:.*mtp.* bypass_divisibility_checks: false