submission 754160
KatherineRWilson · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 118 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-moe-mxfp4-754160?include=source"interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, fp32, fp8_e8m0, int32, mxfp4
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:f585e6805b6286bf73635bbc3cdf7a603cf7754b7964b775424c217e9288b8a1
license declaredunknown
license concludedunknown
authorsKatherineRWilson
imported2026-08-26
Kernel source
submission.py118 lines
from utils import make_match_reference
from task import input_t, output_t
import torch
import torch.nn.functional as F
import math
import aiter
from aiter import ActivationType, QuantType, dtypes
from aiter.fused_moe import fused_moe
from aiter.utility import fp4_utils
from aiter.ops.shuffle import shuffle_weight
# ──────────────────────────────────────────────────────────────────────
# Constants
# ──────────────────────────────────────────────────────────────────────
MXFP4_BLOCK_SIZE = 32
PAD_ALIGN = 256 # 先保留 256 对齐,后面上云后可尝试改小测试性能
def _pad_to(x: int, align: int) -> int:
"""Padding helper"""
return (x + align - 1) // align * align
def _quant_and_shuffle(x: torch.Tensor, shuffle_weight_layout=(16, 16)):
"""统一量化 + shuffle 函数,减少重复代码"""
weight_fp4, scale_e8m0 = aiter.get_torch_quant(QuantType.per_1x32)(x, quant_dtype=dtypes.fp4x2)
scale_shuffled = fp4_utils.e8m0_shuffle(scale_e8m0)
if shuffle_weight_layout is not None:
weight_shuffled = shuffle_weight(weight_fp4, layout=shuffle_weight_layout)
else:
weight_shuffled = weight_fp4
return weight_fp4, weight_shuffled, scale_e8m0, scale_shuffled
def generate_input(
dhidden: int,
dexpert: int,
nroutedexperts: int,
nexpertspertoken: int,
nsharedexperts: int,
bs: int,
seed: int,
) -> input_t:
torch.manual_seed(seed)
# 1. 生成基础 Tensor
hidden_states = torch.randn((bs, dhidden), dtype=torch.bfloat16, device="cuda")
# 模拟 Top-K 路由
topk_weights = torch.randn((bs, nexpertspertoken), dtype=torch.bfloat16, device="cuda")
topk_ids = torch.randint(0, nroutedexperts, (bs, nexpertspertoken), dtype=torch.int32, device="cuda")
# 2. 生成权重 (Gate/Up 和 Down)
# Gate/Up 通常是合并存储的,所以维度是 [nroutedexperts, 2 * dexpert, dhidden]
w1_raw = torch.randn((nroutedexperts, 2 * dexpert, dhidden), dtype=torch.bfloat16, device="cuda")
w2_raw = torch.randn((nroutedexperts, dhidden, dexpert), dtype=torch.bfloat16, device="cuda")
# 3. 执行 MXFP4 量化和 Shuffle (这是性能关键)
# Layout (16, 16) 是 MI300 的常用配置
_, w1_shuffled, w1_scale, w1_scale_sh = _quant_and_shuffle(w1_raw, (16, 16))
_, w2_shuffled, w2_scale, w2_scale_sh = _quant_and_shuffle(w2_raw, (16, 16))
config = {
"d_hidden": dhidden,
"d_hidden_pad": dhidden, # 如果有 Padding 逻辑可以改这里
"d_expert": dexpert,
"d_expert_pad": dexpert,
}
return (
hidden_states,
w1_raw, w2_raw, w1_scale, w2_scale, # Raw 数据 (用于 Reference 对比)
w1_shuffled, w2_shuffled,
w1_scale_sh, w2_scale_sh,
topk_weights, topk_ids,
config
)
# ──────────────────────────────────────────────────────────────────────
# 优化后的 ref_kernel(最终版)
# ──────────────────────────────────────────────────────────────────────
def ref_kernel(data: input_t) -> output_t:
"""
摒弃 Padding 和 Tiling,追求 MI355X 极限吞吐
"""
(
hidden_states, gate_up_weight, down_weight,
gate_up_weight_scale, down_weight_scale,
gate_up_weight_shuffled, down_weight_shuffled,
gate_up_weight_scale_shuffled, down_weight_scale_shuffled,
topk_weights, topk_ids, config
) = data
hidden_states = hidden_states.contiguous()
return fused_moe(
hidden_states,
gate_up_weight_shuffled,
down_weight_shuffled,
topk_weights,
topk_ids,
expert_mask=None,
activation=ActivationType.Silu,
quant_type=QuantType.per_1x32,
w1_scale=gate_up_weight_scale_shuffled,
w2_scale=down_weight_scale_shuffled,
hidden_pad=config.get("d_hidden_pad", 0) - config.get("d_hidden", 0),
intermediate_pad=config.get("d_expert_pad", 0) - config.get("d_expert", 0),
)
custom_kernel = ref_kernel
check_implementation = make_match_reference(ref_kernel, rtol=5e-2, atol=5e-2)scrolls · 118 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON