From 3ccd0d4392df6843c6674ca97074e0513690aa0c Mon Sep 17 00:00:00 2001 From: xuebwang-amd Date: Mon, 17 Aug 2026 08:28:16 +0000 Subject: [PATCH] [Quantization] Remove the dead `ocp_mx_scheme` branch from `moe_kernel_quantize_input` `moe_kernel_quantize_input` carries an `ocp_mx_scheme` parameter and a leading branch that short-circuits `*_a_fp8` schemes into `_fp8_quantize_dequantize`. Nothing reaches it: of the 26 call sites in the tree, none supplies `ocp_mx_scheme=`, the maximum positional arity at any of them is 5 against a parameter at positional index 6, and there is no `**kwargs` splat, aliased import, `functools.partial` or `getattr` reference anywhere. The one arm with behavior was `.endswith("a_fp8")`, and the only reachable a_fp8 scheme is `w_mxfp4_a_fp8` -- `w_mxfp6_e3m2_a_fp8` and `w_mxfp6_e2m3_a_fp8` raise `NotImplementedError` earlier in `QuarkOCP_MX_MoEMethod.get_fused_moe_quant_config`. For that scheme the emulation experts set `quantization_emulation = True` and `quant_dtype = current_platform.fp8_dtype()`, so the surviving dispatch reaches the identical `_fp8_quantize_dequantize(A, A_scale)`. The other two arms were a bare `pass` and a comment. Remove both, per the in-code TODO and @fxmarty-amd's review on #43983: "`moe_kernel_quantize_input` should rely solely on `quant_dtype` now". `FusedMoEQuantConfig.ocp_mx_scheme` and its other consumers (`OCP_MXQuantizationEmulationTritonExperts`, the `fused_batched_moe` NYI asserts, the deprecation guard in `fused_experts_impl`) are untouched. Signed-off-by: xuebwang-amd Co-authored-by: Claude Opus 5 --- vllm/model_executor/layers/fused_moe/utils.py | 16 ---------------- 1 file changed, 16 deletions(-) diff --git a/vllm/model_executor/layers/fused_moe/utils.py b/vllm/model_executor/layers/fused_moe/utils.py index cce8ccd073fc..f11e29f1a64c 100644 --- a/vllm/model_executor/layers/fused_moe/utils.py +++ b/vllm/model_executor/layers/fused_moe/utils.py @@ -280,25 +280,9 @@ def moe_kernel_quantize_input( per_act_token_quant: bool, block_shape: list[int] | None = None, is_scale_swizzled: bool = True, - ocp_mx_scheme: str | None = None, quantization_emulation: bool = False, mx_alignment: int = 0, ) -> tuple[torch.Tensor, torch.Tensor | None]: - # Handle OCP MX scheme that requires QDQ (quantize-dequantize) for emulation - if ocp_mx_scheme is not None: - if ocp_mx_scheme in {"w_mxfp4", "w_mxfp4_a_mxfp4"}: - pass # No QDQ needed for these schemes - elif ocp_mx_scheme.endswith("a_fp8"): - # Perform QDQ (quantize and dequantize) on activation for emulation - # purpose, because there is no native kernel for weight in ocp_mx_scheme - # and activation in FP8. The implementation is based on existing - # non-emulation ops. - # TODO: Remove this `ocp_mx_scheme is not None` block and rely solely - # on `quantization_emulation`. - return _fp8_quantize_dequantize(A, A_scale) - # else: For other schemes (e.g., *_a_mxfp6_e3m2, *_a_mxfp6_e2m3), - # weights are already dequantized, and we proceed with normal - # activation quantization below. if quant_dtype == current_platform.fp8_dtype(): if quantization_emulation: return _fp8_quantize_dequantize(A, A_scale)