default_stage: quantization_modifiers: QuantizationModifier: targets: [Linear] ignore: [lm_head, 're:.*gate$', 're:.*self_attn.*', 're:.*layers\.[0-2]\..*', 're:.*shared_experts.*'] scheme: FP8_BLOCK bypass_divisibility_checks: false requires_calibration_data: false