submission 702057
LiangSu8899 · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 81 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-moe-mxfp4-702057?include=source"interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, fp32, fp8_e8m0, int32, mxfp4
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:e604379bfec2d3116c3f90cd4da7fb7f06312b34c4cb0cf862477d0a065764e2
license declaredunknown
license concludedunknown
authorsLiangSu8899
imported2026-08-15
Kernel source
submission.py81 lines
"""MoE v137: v132 + dp=0 (auto) for E>128 instead of dp=2.
dp=0 auto-selects sorting strategy and is MUCH faster for E257:
E257/bs128: 220µs vs 269µs (-18%)
E257/bs512: 254µs vs 439µs (-42%)
Combined with v132's CKTile block_m tuning for E<=128.
"""
import os
import torch
from task import input_t, output_t
from aiter import ActivationType, QuantType
import aiter.fused_moe as _fm
from aiter.fused_moe import fused_moe
_first = True
def custom_kernel(data: input_t) -> output_t:
global _first
(hidden_states, guw, dw, guws, dws,
guw_s, dw_s, guws_s, dws_s,
topk_weights, topk_ids, config) = data
hp = config['d_hidden_pad'] - config['d_hidden']
ip = config['d_expert_pad'] - config['d_expert']
E = guw_s.shape[0]
token_num = hidden_states.shape[0]
topk = topk_ids.shape[1]
# JIT warmup
if _first:
_first = False
os.environ["AITER_KSPLIT"] = "0"
dp = 0 # auto is best for warmup too
fused_moe(
hidden_states, guw_s, dw_s, topk_weights, topk_ids,
expert_mask=None, activation=ActivationType.Silu,
quant_type=QuantType.per_1x32, doweight_stage1=False,
w1_scale=guws_s, w2_scale=dws_s,
a1_scale=None, a2_scale=None,
hidden_pad=hp, intermediate_pad=ip,
moe_sorting_dispatch_policy=dp)
torch.cuda.synchronize()
# CKTile for E<=128 small batch
tokens_per_expert = token_num * topk / E
block_m = None
if E <= 128 and tokens_per_expert <= 40:
want_ksplit = "2"
if tokens_per_expert > 10:
block_m = 32
else:
want_ksplit = "0"
current = os.environ.get("AITER_KSPLIT", "0")
if current != want_ksplit:
os.environ["AITER_KSPLIT"] = want_ksplit
_fm.get_ksplit.cache_clear()
_fm.cfg_2stages = None
# dp=0 (auto) for ALL shapes - auto-select is better than forced mp
if E > 128:
dp = 0 # auto - much better than dp=2 for E257
elif token_num <= 128:
dp = 1
else:
dp = 0
kwargs = dict(
expert_mask=None, activation=ActivationType.Silu,
quant_type=QuantType.per_1x32, doweight_stage1=False,
w1_scale=guws_s, w2_scale=dws_s,
a1_scale=None, a2_scale=None,
hidden_pad=hp, intermediate_pad=ip,
moe_sorting_dispatch_policy=dp)
if block_m is not None:
kwargs['block_size_M'] = block_m
return fused_moe(
hidden_states, guw_s, dw_s, topk_weights, topk_ids,
**kwargs)
scrolls · 81 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON