submission 659958
yuzhou2 · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 183 lines, June 9 Researcher Reciprocity License v1.0.
submission_current.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-moe-mxfp4-659958?include=source"interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, fp32, fp8_e8m0, int32, mxfp4
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:ec4411f55b13f78f6f1fb83a53f172f4040369ee93e538672e0f92ff351efa78
license declaredunknown
license concludedunknown
authorsyuzhou2
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
fp4
"""MoE MXFP4 v134: v94 + FlyDSL stage2 t64x256x256_reduce for E=33 d=512.Kernel source
submission_current.py183 lines
"""MoE MXFP4 v134: v94 + FlyDSL stage2 t64x256x256_reduce for E=33 d=512.
Based on v94. Single change: replace CK XDL small s2 kernel with
flydsl_moe2_afp4_wfp4_bf16_t64x256x256_reduce for bs=512 E=33 d=512.
Confirmed -7.9% on target (169 vs 185us). Geomean ~145us warm (~-1.4%).
CSV routing:
- CKTile block_m=16 (ksplit=2): TEST shapes + bs=16 all
- CKTile block_m=32 (ksplit=2): bs=128 all
- CK XDL (ksplit=0): bs=512 E=33 d=512 (Small s1 + FlyDSL t64 reduce s2)
- CK XDL (ksplit=0): bs=512 E=257 d=256 (Medium s1 + Small s2)
- Heuristic w/ block_size_M=64: bs=512 E=33 d=2048
"""
import os
import importlib
import torch
from task import input_t, output_t
_CSV_SETUP_DONE = False
def _setup_csv():
global _CSV_SETUP_DONE
if _CSV_SETUP_DONE:
return
_CSV_SETUP_DONE = True
os.environ["AITER_USE_OPUS_MOE_SORTING"] = "1"
spec = importlib.util.find_spec("aiter")
if spec and spec.submodule_search_locations:
aiter_pkg = spec.submodule_search_locations[0]
else:
import aiter as _a
aiter_pkg = os.path.dirname(_a.__file__)
config_dir = os.path.join(aiter_pkg, "configs")
default_csv = os.path.join(config_dir, "tuned_fmoe.csv")
csv_path = "/tmp/moe_hybrid_v134.csv"
header = "cu_num,token,model_dim,inter_dim,expert,topk,act_type,dtype,q_dtype_a,q_dtype_w,q_type,use_g1u1,doweight_stage1,block_m,ksplit,us1,kernelName1,err1,us2,kernelName2,err2,us,run_1stage,tflops,bw,_tag"
s1_small = "moe_ck2stages_gemm1_64x32x32x128_1x1_MulABScaleShuffled_v3_Nswizzle0_Quant3_MulRoutedWeight0_silu_FP4X2_FP4X2_B16"
s2_small = "moe_ck2stages_gemm2_64x32x32x128_1x1_MulABScaleExpertWeightShuffled_v1_Nswizzle0_Quant3_MulRoutedWeight1_FP4X2_FP4X2_B16"
s1_med = "moe_ck2stages_gemm1_256x32x128x128_1x4_MulABScaleShuffled_v3_Nswizzle0_Quant3_MulRoutedWeight0_silu_FP4X2_FP4X2_B16"
lines = [header]
cktile_bm16_shapes = [
(8, 4096, 1024, 257, 9),
(32, 7168, 2048, 33, 9),
(128, 4096, 1536, 65, 7),
(16, 7168, 256, 257, 9),
(16, 7168, 512, 33, 9),
]
for token, mdim, idim, expert, topk in cktile_bm16_shapes:
line = (
f"256,{token},{mdim},{idim},{expert},{topk},"
f"ActivationType.Silu,torch.bfloat16,"
f"torch.float4_e2m1fn_x2,torch.float4_e2m1fn_x2,"
f"QuantType.per_1x32,1,0,"
f"16,2,"
f"0.5,,0.0,0.5,,0.0,"
f"1.0,0,0.0,0.0,"
)
lines.append(line)
cktile_bm32_shapes = [
(128, 7168, 256, 257, 9),
(128, 7168, 512, 33, 9),
]
for token, mdim, idim, expert, topk in cktile_bm32_shapes:
line = (
f"256,{token},{mdim},{idim},{expert},{topk},"
f"ActivationType.Silu,torch.bfloat16,"
f"torch.float4_e2m1fn_x2,torch.float4_e2m1fn_x2,"
f"QuantType.per_1x32,1,0,"
f"32,2,"
f"0.5,,0.0,0.5,,0.0,"
f"1.0,0,0.0,0.0,"
)
lines.append(line)
# CK XDL: bs=512 E=33 d=512 — FlyDSL t64 REDUCE for stage2 <- CHANGED
line = (
f"256,512,7168,512,33,9,"
f"ActivationType.Silu,torch.bfloat16,"
f"torch.float4_e2m1fn_x2,torch.float4_e2m1fn_x2,"
f"QuantType.per_1x32,1,0,"
f"32,0,"
f"50.0,{s1_small},0.0,30.0,flydsl_moe2_afp4_wfp4_bf16_t64x256x256_reduce,0.0,"
f"140.0,0,0.0,0.0,"
)
lines.append(line)
# CK XDL: bs=512 E=257 d=256 (Medium s1 + Small s2)
line = (
f"256,512,7168,256,257,9,"
f"ActivationType.Silu,torch.bfloat16,"
f"torch.float4_e2m1fn_x2,torch.float4_e2m1fn_x2,"
f"QuantType.per_1x32,1,0,"
f"32,0,"
f"100.0,{s1_med},0.0,70.0,{s2_small},0.0,"
f"170.0,0,0.0,0.0,"
)
lines.append(line)
with open(csv_path, "w") as f:
f.write("\n".join(lines) + "\n")
merge_paths = [csv_path]
if os.path.exists(default_csv):
merge_paths.append(default_csv)
os.environ["AITER_CONFIG_FMOE"] = ":".join(merge_paths)
_fused_moe = None
_ActivationType = None
_QuantType = None
_shape_cache: dict = {}
def _ensure_imports():
global _fused_moe, _ActivationType, _QuantType
if _fused_moe is not None:
return
_setup_csv()
from aiter import ActivationType, QuantType
from aiter.fused_moe import fused_moe
_fused_moe = fused_moe
_ActivationType = ActivationType
_QuantType = QuantType
def custom_kernel(data: input_t) -> output_t:
global _fused_moe, _ActivationType, _QuantType
_ensure_imports()
(
hidden_states,
gate_up_weight, down_weight,
gate_up_weight_scale, down_weight_scale,
gate_up_weight_shuffled, down_weight_shuffled,
gate_up_weight_scale_shuffled, down_weight_scale_shuffled,
topk_weights, topk_ids,
config,
) = data
d_hidden_pad = config["d_hidden_pad"]
d_expert_pad = config["d_expert_pad"]
d_expert = config["d_expert"]
cache_key = (hidden_states.shape[0], gate_up_weight_shuffled.shape[0],
config["d_hidden"], d_expert, d_hidden_pad, d_expert_pad)
cached = _shape_cache.get(cache_key)
if cached is None:
hidden_pad = d_hidden_pad - config["d_hidden"]
intermediate_pad = d_expert_pad - d_expert
bsm = 64 if d_expert >= 2048 else None
cached = (hidden_pad, intermediate_pad, bsm)
_shape_cache[cache_key] = cached
hidden_pad, intermediate_pad, bsm = cached
output = _fused_moe(
hidden_states,
gate_up_weight_shuffled,
down_weight_shuffled,
topk_weights,
topk_ids,
expert_mask=None,
activation=_ActivationType.Silu,
quant_type=_QuantType.per_1x32,
doweight_stage1=False,
w1_scale=gate_up_weight_scale_shuffled,
w2_scale=down_weight_scale_shuffled,
a1_scale=None,
a2_scale=None,
block_size_M=bsm,
hidden_pad=hidden_pad,
intermediate_pad=intermediate_pad,
)
return output
scrolls · 183 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON