Skip to content
KernelIndex
Search⌘K

submission 754160

KatherineRWilson · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 118 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-moe-mxfp4-754160?include=source"
interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, fp32, fp8_e8m0, int32, mxfp4

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
AMD MXFP4 MoEsuite of 7 cases
AMD Instinct MI355X
178.6µs
#429 of 782
2026-04-07

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:f585e6805b6286bf73635bbc3cdf7a603cf7754b7964b775424c217e9288b8a1
license declaredunknown
license concludedunknown
authorsKatherineRWilson
imported2026-08-26

Kernel source

submission.py118 lines
from utils import make_match_reference
from task import input_t, output_t
import torch
import torch.nn.functional as F
import math

import aiter
from aiter import ActivationType, QuantType, dtypes
from aiter.fused_moe import fused_moe
from aiter.utility import fp4_utils
from aiter.ops.shuffle import shuffle_weight


# ──────────────────────────────────────────────────────────────────────
# Constants
# ──────────────────────────────────────────────────────────────────────
MXFP4_BLOCK_SIZE = 32
PAD_ALIGN = 256                     # 先保留 256 对齐,后面上云后可尝试改小测试性能


def _pad_to(x: int, align: int) -> int:
    """Padding helper"""
    return (x + align - 1) // align * align


def _quant_and_shuffle(x: torch.Tensor, shuffle_weight_layout=(16, 16)):
    """统一量化 + shuffle 函数,减少重复代码"""
    weight_fp4, scale_e8m0 = aiter.get_torch_quant(QuantType.per_1x32)(x, quant_dtype=dtypes.fp4x2)
    
    scale_shuffled = fp4_utils.e8m0_shuffle(scale_e8m0)
    
    if shuffle_weight_layout is not None:
        weight_shuffled = shuffle_weight(weight_fp4, layout=shuffle_weight_layout)
    else:
        weight_shuffled = weight_fp4
    
    return weight_fp4, weight_shuffled, scale_e8m0, scale_shuffled


def generate_input(
    dhidden: int,
    dexpert: int,
    nroutedexperts: int,
    nexpertspertoken: int,
    nsharedexperts: int,
    bs: int,
    seed: int,
) -> input_t:
    torch.manual_seed(seed)
    
    # 1. 生成基础 Tensor
    hidden_states = torch.randn((bs, dhidden), dtype=torch.bfloat16, device="cuda")
    # 模拟 Top-K 路由
    topk_weights = torch.randn((bs, nexpertspertoken), dtype=torch.bfloat16, device="cuda")
    topk_ids = torch.randint(0, nroutedexperts, (bs, nexpertspertoken), dtype=torch.int32, device="cuda")
    
    # 2. 生成权重 (Gate/Up 和 Down)
    # Gate/Up 通常是合并存储的,所以维度是 [nroutedexperts, 2 * dexpert, dhidden]
    w1_raw = torch.randn((nroutedexperts, 2 * dexpert, dhidden), dtype=torch.bfloat16, device="cuda")
    w2_raw = torch.randn((nroutedexperts, dhidden, dexpert), dtype=torch.bfloat16, device="cuda")
    
    # 3. 执行 MXFP4 量化和 Shuffle (这是性能关键)
    # Layout (16, 16) 是 MI300 的常用配置
    _, w1_shuffled, w1_scale, w1_scale_sh = _quant_and_shuffle(w1_raw, (16, 16))
    _, w2_shuffled, w2_scale, w2_scale_sh = _quant_and_shuffle(w2_raw, (16, 16))
    
    config = {
        "d_hidden": dhidden,
        "d_hidden_pad": dhidden, # 如果有 Padding 逻辑可以改这里
        "d_expert": dexpert,
        "d_expert_pad": dexpert,
    }
    
    return (
        hidden_states,
        w1_raw, w2_raw, w1_scale, w2_scale, # Raw 数据 (用于 Reference 对比)
        w1_shuffled, w2_shuffled,
        w1_scale_sh, w2_scale_sh,
        topk_weights, topk_ids,
        config
    )


# ──────────────────────────────────────────────────────────────────────
# 优化后的 ref_kernel(最终版)
# ──────────────────────────────────────────────────────────────────────
def ref_kernel(data: input_t) -> output_t:
    """
    摒弃 Padding 和 Tiling,追求 MI355X 极限吞吐
    """
    (
        hidden_states, gate_up_weight, down_weight, 
        gate_up_weight_scale, down_weight_scale, 
        gate_up_weight_shuffled, down_weight_shuffled, 
        gate_up_weight_scale_shuffled, down_weight_scale_shuffled, 
        topk_weights, topk_ids, config
    ) = data 

    hidden_states = hidden_states.contiguous()

    return fused_moe(
        hidden_states,
        gate_up_weight_shuffled,
        down_weight_shuffled,
        topk_weights,
        topk_ids,
        expert_mask=None,
        activation=ActivationType.Silu,
        quant_type=QuantType.per_1x32,
        w1_scale=gate_up_weight_scale_shuffled,
        w2_scale=down_weight_scale_shuffled,
        hidden_pad=config.get("d_hidden_pad", 0) - config.get("d_hidden", 0),
        intermediate_pad=config.get("d_expert_pad", 0) - config.get("d_expert", 0),
    )

custom_kernel = ref_kernel

check_implementation = make_match_reference(ref_kernel, rtol=5e-2, atol=5e-2)
scrolls · 118 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON