submission 754134
zhuang000123 · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 63 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-moe-mxfp4-754134?include=source"interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, fp32, fp8_e8m0, int32, mxfp4
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:7ce9478e6eb913736035d80eb02ffc381b20e8d176b2a2e8a5ab2c89881d8385
license declaredunknown
license concludedunknown
authorszhuang000123
imported2026-08-15
Kernel source
submission.py63 lines
#!POPCORN leaderboard amd-moe-mxfp4
#!POPCORN gpu MI355X
import torch
from typing import Dict
from task import input_t, output_t
from aiter import ActivationType, QuantType
from aiter.fused_moe import fused_moe
import aiter.fused_moe as _fmoe_mod
_KN1_LARGE = "moe_ck2stages_gemm1_256x128x128x128_1x4_MulABScaleShuffled_v3_Nswizzle0_Quant3_MulRoutedWeight0_silu_FP4X2_FP4X2_B16"
_KN1_MED = "moe_ck2stages_gemm1_256x32x128x128_1x4_MulABScaleShuffled_v3_Nswizzle0_Quant3_MulRoutedWeight0_silu_FP4X2_FP4X2_B16"
_KN1_MED64 = "moe_ck2stages_gemm1_256x64x128x128_1x4_MulABScaleShuffled_v3_Nswizzle0_Quant3_MulRoutedWeight0_silu_FP4X2_FP4X2_B16"
_KN2_SMALL = "moe_ck2stages_gemm2_64x32x32x128_1x1_MulABScaleExpertWeightShuffled_v1_Nswizzle0_Quant3_MulRoutedWeight1_FP4X2_FP4X2_B16"
_KN2_FLYDSL = "flydsl_moe2_afp4_wfp4_bf16_t64x256x256_reduce"
_injected = False
def _inject_configs():
global _injected
if _injected: return
_injected = True
cfg = _fmoe_mod.cfg_2stages
if cfg is None: return
c = ("ActivationType.Silu","torch.bfloat16","torch.float4_e2m1fn_x2","torch.float4_e2m1fn_x2","QuantType.per_1x32",True,False)
t = {"kernelName1":"","kernelName2":"","run_1stage":0}
# bs=16: CKTile (proven fastest for small batch)
cfg[(256,16,7168,256,257,9)+c] = {"block_m":16,"ksplit":4,**t} # E=257 90µs
cfg[(256,16,7168,512,33,9)+c] = {"block_m":16,"ksplit":2,**t} # E=33 60µs
# E=257 bs=128/512: LARGE stage1 + FlyDSL t64x128 reduce stage2 (NEW!)
cfg[(256,128,7168,256,257,9)+c] = {"block_m":32,"ksplit":0,
"kernelName1":_KN1_LARGE,"kernelName2":_KN2_FLYDSL,"run_1stage":0} # 144µs (was 160)
cfg[(256,512,7168,256,257,9)+c] = {"block_m":32,"ksplit":0,
"kernelName1":_KN1_LARGE,"kernelName2":_KN2_FLYDSL,"run_1stage":0} # 170µs (was 178)
# E=33 bs=128: MED64 stage1 + FlyDSL t64x256 reduce (test: target <105µs)
cfg[(256,128,7168,512,33,9)+c] = {"block_m":32,"ksplit":0,
"kernelName1":_KN1_MED64,"kernelName2":_KN2_FLYDSL,"run_1stage":0}
# E=33 bs=512/d=512: MED stage1 + FlyDSL t64x128 reduce (169µs, lead ✅)
cfg[(256,512,7168,512,33,9)+c] = {"block_m":32,"ksplit":0,
"kernelName1":_KN1_MED,"kernelName2":_KN2_FLYDSL,"run_1stage":0}
# E=33 d=2048: CK auto block_m=64 (338µs, FlyDSL all inf)
cfg[(256,512,7168,2048,33,9)+c] = {"block_m":64,"ksplit":0,**t}
_fmoe_mod.get_2stage_cfgs.cache_clear()
_call_count = 0
def custom_kernel(data: input_t) -> output_t:
global _call_count
(hs,w1,w2,w1s,w2s,w1sh,w2sh,w1ssh,w2ssh,tw,ti,cfg) = data
hp = cfg["d_hidden_pad"]-cfg["d_hidden"]; ip = cfg["d_expert_pad"]-cfg["d_expert"]
output = fused_moe(hs,w1sh,w2sh,tw,ti,expert_mask=None,activation=ActivationType.Silu,
quant_type=QuantType.per_1x32,doweight_stage1=False,
w1_scale=w1ssh,w2_scale=w2ssh,a1_scale=None,a2_scale=None,
hidden_pad=hp,intermediate_pad=ip)
_call_count += 1
if _call_count == 1: _inject_configs()
return output
scrolls · 63 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON