vllm.model_executor.layers.fused_moe.fused_moe_modular_method ¶

logger `module-attribute` ¶

logger = init_logger(__name__)

FusedMoEModularMethod ¶

Bases: FusedMoEMethodBase, CustomOp

Source code in vllm/model_executor/layers/fused_moe/fused_moe_modular_method.py

@CustomOp.register("modular_fused_moe")
class FusedMoEModularMethod(FusedMoEMethodBase, CustomOp):
    def __init__(
        self, old_quant_method: FusedMoEMethodBase, experts: FusedMoEModularKernel
    ):
        super().__init__(old_quant_method.moe)
        self.moe_quant_config = old_quant_method.moe_quant_config
        self.fused_experts = experts
        self.disable_expert_map = getattr(
            old_quant_method,
            "disable_expert_map",
            not self.fused_experts.supports_expert_map(),
        )
        self.old_quant_method = old_quant_method
        logger.debug("Swapping out %s", self.old_quant_method.__class__.__name__)

    @staticmethod
    def make(
        moe_layer: torch.nn.Module,
        old_quant_method: FusedMoEMethodBase,
        prepare_finalize: FusedMoEPrepareAndFinalize,
        shared_experts: torch.nn.Module | None,
    ) -> "FusedMoEModularMethod":
        parallel_config = getattr(
            getattr(moe_layer, "vllm_config", None),
            "parallel_config",
            None,
        )
        return FusedMoEModularMethod(
            old_quant_method,
            FusedMoEModularKernel(
                prepare_finalize,
                old_quant_method.select_gemm_impl(prepare_finalize, moe_layer),
                shared_experts,
                getattr(moe_layer, "shared_experts_stream", None),
                parallel_config=parallel_config,
            ),
        )

    @property
    def topk_indices_dtype(self) -> torch.dtype | None:
        return self.fused_experts.prepare_finalize.topk_indices_dtype()

    @property
    def supports_eplb(self) -> bool:
        return self.old_quant_method.supports_eplb

    @property
    def allow_inplace(self) -> bool:
        return self.old_quant_method.allow_inplace

    @property
    def method_name(self) -> str:
        return self.old_quant_method.method_name

    def create_weights(
        self,
        layer: torch.nn.Module,
        num_experts: int,
        hidden_size: int,
        intermediate_size_per_partition: int,
        params_dtype: torch.dtype,
        **extra_weight_attrs,
    ):
        raise NotImplementedError

    def get_fused_moe_quant_config(
        self, layer: torch.nn.Module
    ) -> FusedMoEQuantConfig | None:
        return self.moe_quant_config

    def apply(
        self,
        layer: "FusedMoE",  # type: ignore[name-defined] # noqa: F821
        x: torch.Tensor,
        router_logits: torch.Tensor,
    ) -> torch.Tensor | tuple[torch.Tensor, torch.Tensor]:
        topk_weights, topk_ids, zero_expert_result = layer.select_experts(
            hidden_states=x,
            router_logits=router_logits,
        )

        result = self.fused_experts(
            hidden_states=x,
            w1=layer.w13_weight,
            w2=layer.w2_weight,
            topk_weights=topk_weights,
            topk_ids=topk_ids,
            inplace=self.allow_inplace,
            activation=layer.activation,
            global_num_experts=layer.global_num_experts,
            apply_router_weight_on_input=layer.apply_router_weight_on_input,
            expert_map=None if self.disable_expert_map else layer.expert_map,
        )

        if layer.zero_expert_num != 0 and layer.zero_expert_type is not None:
            assert not isinstance(result, tuple), (
                "Shared + zero experts are mutually exclusive not yet supported"
            )
            return result, zero_expert_result
        else:
            return result

allow_inplace `property` ¶

allow_inplace: bool

disable_expert_map `instance-attribute` ¶

disable_expert_map = getattr(
    old_quant_method,
    "disable_expert_map",
    not supports_expert_map(),
)

fused_experts `instance-attribute` ¶

fused_experts = experts

method_name `property` ¶

method_name: str

moe_quant_config `instance-attribute` ¶

moe_quant_config = moe_quant_config

old_quant_method `instance-attribute` ¶

old_quant_method = old_quant_method

supports_eplb `property` ¶

supports_eplb: bool

topk_indices_dtype `property` ¶

topk_indices_dtype: dtype | None

init ¶

__init__(
    old_quant_method: FusedMoEMethodBase,
    experts: FusedMoEModularKernel,
)

Source code in vllm/model_executor/layers/fused_moe/fused_moe_modular_method.py

def __init__(
    self, old_quant_method: FusedMoEMethodBase, experts: FusedMoEModularKernel
):
    super().__init__(old_quant_method.moe)
    self.moe_quant_config = old_quant_method.moe_quant_config
    self.fused_experts = experts
    self.disable_expert_map = getattr(
        old_quant_method,
        "disable_expert_map",
        not self.fused_experts.supports_expert_map(),
    )
    self.old_quant_method = old_quant_method
    logger.debug("Swapping out %s", self.old_quant_method.__class__.__name__)

apply ¶

apply(
    layer: FusedMoE, x: Tensor, router_logits: Tensor
) -> Tensor | tuple[Tensor, Tensor]

Source code in vllm/model_executor/layers/fused_moe/fused_moe_modular_method.py

def apply(
    self,
    layer: "FusedMoE",  # type: ignore[name-defined] # noqa: F821
    x: torch.Tensor,
    router_logits: torch.Tensor,
) -> torch.Tensor | tuple[torch.Tensor, torch.Tensor]:
    topk_weights, topk_ids, zero_expert_result = layer.select_experts(
        hidden_states=x,
        router_logits=router_logits,
    )

    result = self.fused_experts(
        hidden_states=x,
        w1=layer.w13_weight,
        w2=layer.w2_weight,
        topk_weights=topk_weights,
        topk_ids=topk_ids,
        inplace=self.allow_inplace,
        activation=layer.activation,
        global_num_experts=layer.global_num_experts,
        apply_router_weight_on_input=layer.apply_router_weight_on_input,
        expert_map=None if self.disable_expert_map else layer.expert_map,
    )

    if layer.zero_expert_num != 0 and layer.zero_expert_type is not None:
        assert not isinstance(result, tuple), (
            "Shared + zero experts are mutually exclusive not yet supported"
        )
        return result, zero_expert_result
    else:
        return result

create_weights ¶

create_weights(
    layer: Module,
    num_experts: int,
    hidden_size: int,
    intermediate_size_per_partition: int,
    params_dtype: dtype,
    **extra_weight_attrs,
)

Source code in vllm/model_executor/layers/fused_moe/fused_moe_modular_method.py

def create_weights(
    self,
    layer: torch.nn.Module,
    num_experts: int,
    hidden_size: int,
    intermediate_size_per_partition: int,
    params_dtype: torch.dtype,
    **extra_weight_attrs,
):
    raise NotImplementedError

get_fused_moe_quant_config ¶

get_fused_moe_quant_config(
    layer: Module,
) -> FusedMoEQuantConfig | None

Source code in vllm/model_executor/layers/fused_moe/fused_moe_modular_method.py

def get_fused_moe_quant_config(
    self, layer: torch.nn.Module
) -> FusedMoEQuantConfig | None:
    return self.moe_quant_config

make `staticmethod` ¶

make(
    moe_layer: Module,
    old_quant_method: FusedMoEMethodBase,
    prepare_finalize: FusedMoEPrepareAndFinalize,
    shared_experts: Module | None,
) -> FusedMoEModularMethod

Source code in vllm/model_executor/layers/fused_moe/fused_moe_modular_method.py

@staticmethod
def make(
    moe_layer: torch.nn.Module,
    old_quant_method: FusedMoEMethodBase,
    prepare_finalize: FusedMoEPrepareAndFinalize,
    shared_experts: torch.nn.Module | None,
) -> "FusedMoEModularMethod":
    parallel_config = getattr(
        getattr(moe_layer, "vllm_config", None),
        "parallel_config",
        None,
    )
    return FusedMoEModularMethod(
        old_quant_method,
        FusedMoEModularKernel(
            prepare_finalize,
            old_quant_method.select_gemm_impl(prepare_finalize, moe_layer),
            shared_experts,
            getattr(moe_layer, "shared_experts_stream", None),
            parallel_config=parallel_config,
        ),
    )

vllm.model_executor.layers.fused_moe.fused_moe_modular_method ¶

logger module-attribute ¶

FusedMoEModularMethod ¶

allow_inplace property ¶

disable_expert_map instance-attribute ¶

fused_experts instance-attribute ¶

method_name property ¶

moe_quant_config instance-attribute ¶

old_quant_method instance-attribute ¶

supports_eplb property ¶

topk_indices_dtype property ¶

__init__ ¶

apply ¶

create_weights ¶

get_fused_moe_quant_config ¶

make staticmethod ¶

logger `module-attribute` ¶

allow_inplace `property` ¶

disable_expert_map `instance-attribute` ¶

fused_experts `instance-attribute` ¶

method_name `property` ¶

moe_quant_config `instance-attribute` ¶

old_quant_method `instance-attribute` ¶

supports_eplb `property` ¶

topk_indices_dtype `property` ¶

init ¶

make `staticmethod` ¶