vllm.model_executor.layers.fused_moe.triton_cutlass_moe ¶

TritonOrCutlassExperts ¶

Bases: FallbackExperts

Cutlass with fallback to Triton for low latency shapes on SM100.

Source code in vllm/model_executor/layers/fused_moe/triton_cutlass_moe.py

class TritonOrCutlassExperts(FallbackExperts):
    """Cutlass with fallback to Triton for low latency shapes on SM100."""

    def __init__(
        self,
        moe_config: FusedMoEConfig,
        quant_config: FusedMoEQuantConfig,
    ):
        self.is_sm100 = current_platform.has_device_capability(100)
        super().__init__(
            experts=CutlassExpertsFp8(moe_config, quant_config),
            fallback_experts=TritonExperts(moe_config, quant_config),
        )

    @staticmethod
    def get_clses() -> tuple[
        type[mk.FusedMoEPermuteExpertsUnpermute],
        type[mk.FusedMoEPermuteExpertsUnpermute],
    ]:
        return (CutlassExpertsFp8, TritonExperts)

    def workspace_shapes(
        self,
        M: int,
        N: int,
        K: int,
        topk: int,
        global_num_experts: int,
        local_num_experts: int,
        expert_tokens_meta: mk.ExpertTokensMetadata | None,
        activation: str,
    ) -> tuple[tuple[int, ...], tuple[int, ...], tuple[int, ...]]:
        # Small batch fallback for sm100.
        if self.is_sm100 and M <= 8:
            return self.fallback_experts.workspace_shapes(
                M,
                N,
                K,
                topk,
                global_num_experts,
                local_num_experts,
                expert_tokens_meta,
                activation,
            )
        else:
            return self.experts.workspace_shapes(
                M,
                N,
                K,
                topk,
                global_num_experts,
                local_num_experts,
                expert_tokens_meta,
                activation,
            )

    def _select_experts_impl(
        self,
        hidden_states: torch.Tensor,
        w1: torch.Tensor,
        w2: torch.Tensor,
    ) -> mk.FusedMoEPermuteExpertsUnpermute:
        # Small batch fallback for sm100.
        if self.is_sm100 and hidden_states.shape[0] <= 8:
            return self.fallback_experts
        else:
            return self.experts

is_sm100 `instance-attribute` ¶

is_sm100 = has_device_capability(100)

init ¶

__init__(
    moe_config: FusedMoEConfig,
    quant_config: FusedMoEQuantConfig,
)

Source code in vllm/model_executor/layers/fused_moe/triton_cutlass_moe.py

def __init__(
    self,
    moe_config: FusedMoEConfig,
    quant_config: FusedMoEQuantConfig,
):
    self.is_sm100 = current_platform.has_device_capability(100)
    super().__init__(
        experts=CutlassExpertsFp8(moe_config, quant_config),
        fallback_experts=TritonExperts(moe_config, quant_config),
    )

_select_experts_impl ¶

_select_experts_impl(
    hidden_states: Tensor, w1: Tensor, w2: Tensor
) -> FusedMoEPermuteExpertsUnpermute

Source code in vllm/model_executor/layers/fused_moe/triton_cutlass_moe.py

def _select_experts_impl(
    self,
    hidden_states: torch.Tensor,
    w1: torch.Tensor,
    w2: torch.Tensor,
) -> mk.FusedMoEPermuteExpertsUnpermute:
    # Small batch fallback for sm100.
    if self.is_sm100 and hidden_states.shape[0] <= 8:
        return self.fallback_experts
    else:
        return self.experts

get_clses `staticmethod` ¶

get_clses() -> tuple[
    type[FusedMoEPermuteExpertsUnpermute],
    type[FusedMoEPermuteExpertsUnpermute],
]

Source code in vllm/model_executor/layers/fused_moe/triton_cutlass_moe.py

@staticmethod
def get_clses() -> tuple[
    type[mk.FusedMoEPermuteExpertsUnpermute],
    type[mk.FusedMoEPermuteExpertsUnpermute],
]:
    return (CutlassExpertsFp8, TritonExperts)

workspace_shapes ¶

workspace_shapes(
    M: int,
    N: int,
    K: int,
    topk: int,
    global_num_experts: int,
    local_num_experts: int,
    expert_tokens_meta: ExpertTokensMetadata | None,
    activation: str,
) -> tuple[
    tuple[int, ...], tuple[int, ...], tuple[int, ...]
]

Source code in vllm/model_executor/layers/fused_moe/triton_cutlass_moe.py

def workspace_shapes(
    self,
    M: int,
    N: int,
    K: int,
    topk: int,
    global_num_experts: int,
    local_num_experts: int,
    expert_tokens_meta: mk.ExpertTokensMetadata | None,
    activation: str,
) -> tuple[tuple[int, ...], tuple[int, ...], tuple[int, ...]]:
    # Small batch fallback for sm100.
    if self.is_sm100 and M <= 8:
        return self.fallback_experts.workspace_shapes(
            M,
            N,
            K,
            topk,
            global_num_experts,
            local_num_experts,
            expert_tokens_meta,
            activation,
        )
    else:
        return self.experts.workspace_shapes(
            M,
            N,
            K,
            topk,
            global_num_experts,
            local_num_experts,
            expert_tokens_meta,
            activation,
        )

vllm.model_executor.layers.fused_moe.triton_cutlass_moe ¶

TritonOrCutlassExperts ¶

is_sm100 instance-attribute ¶

__init__ ¶

_select_experts_impl ¶

get_clses staticmethod ¶

workspace_shapes ¶

is_sm100 `instance-attribute` ¶

init ¶

get_clses `staticmethod` ¶