default_stage: default_modifiers: AWQModifier: config_groups: mlp_experts_projections: targets: ['re:.*block_sparse_moe\.experts\.[0-9]+\.(w1|w2|w3)$'] weights: num_bits: 4 type: int symmetric: true group_size: 32 strategy: group block_structure: null dynamic: false actorder: null scale_dtype: null zp_dtype: null observer: minmax observer_kwargs: {} input_activations: null output_activations: null format: null targets: [Linear] ignore: [lm_head, embed_tokens, 're:.*self_attn.*', 're:.*block_sparse_moe\.gate$', 're:.*\.layers\.(58|59|60|61)\.block_sparse_moe\.experts\.[0-9]+\.(w1|w2|w3)$'] bypass_divisibility_checks: false mappings: - smooth_layer: re:.*post_attention_layernorm$ balance_layers: ['re:.*w1$', 're:.*w3$'] activation_hook_target: null - smooth_layer: re:.*w3$ balance_layers: ['re:.*w2$'] activation_hook_target: null offload_device: !!python/object/apply:torch.device [cpu] duo_scaling: true n_grid: 20