submission 686584
mingkai_37292 · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 158 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-moe-mxfp4-686584?include=source"interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, fp32, fp8_e8m0, int32, mxfp4
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:a253cb17353d121d21e2c41b00b9972897747957cda19654bd4ed99810d0c3dd
license declaredunknown
license concludedunknown
authorsmingkai_37292
imported2026-08-26
Kernel source
submission.py158 lines
#!POPCORN leaderboard amd-moe-mxfp4
#!POPCORN gpu MI355X
import os
import importlib.util
# Enable OPUS MOE sorting for potentially faster token dispatch
os.environ["AITER_USE_OPUS_MOE_SORTING"] = "1"
# ---------------------------------------------------------------------------
# Build a combined tuning CSV that includes ALL existing aiter CSV entries
# (primary + model_configs) PLUS our auto-tuner-discovered optimal entries
# for 32-expert shapes.
#
# Setting AITER_CONFIG_FMOE bypasses the library's merge pipeline entirely,
# so we must include entries from model_configs CSVs ourselves and deduplicate.
# ---------------------------------------------------------------------------
_COMBINED_CSV = "/tmp/amd_moe_mxfp4_combined.csv"
def _build_combined_csv():
"""Merge ALL existing aiter CSVs + our 32-expert entries, deduplicated."""
spec = importlib.util.find_spec("aiter")
if not spec or not spec.origin:
return False
import pandas as pd
aiter_dir = os.path.dirname(spec.origin)
configs_dir = os.path.join(aiter_dir, "configs")
mc_dir = os.path.join(configs_dir, "model_configs")
csv_paths = [
os.path.join(configs_dir, "tuned_fmoe.csv"),
os.path.join(mc_dir, "dsv3_fp4_tuned_fmoe.csv"),
os.path.join(mc_dir, "a8w8_blockscale_tuned_fmoe_qwen3_235b.csv"),
]
dfs = []
for p in csv_paths:
if os.path.exists(p):
try:
dfs.append(pd.read_csv(p))
except Exception:
pass
if not dfs:
return False
# --- optimal 32-expert configs from AITER_ONLINE_TUNE exploration ------
K1_S = "moe_ck2stages_gemm1_64x32x32x128_1x1_MulABScaleShuffled_v3_Nswizzle0_Quant3_MulRoutedWeight0_silu_FP4X2_FP4X2_B16"
K2_S = "moe_ck2stages_gemm2_64x32x32x128_1x1_MulABScaleExpertWeightShuffled_v1_Nswizzle0_Quant3_MulRoutedWeight1_FP4X2_FP4X2_B16"
K1_L = "moe_ck2stages_gemm1_256x32x128x128_1x4_MulABScaleShuffled_v3_Nswizzle0_Quant3_MulRoutedWeight0_silu_FP4X2_FP4X2_B16"
K1_XL = "moe_ck2stages_gemm1_256x128x128x128_1x4_MulABScaleShuffled_v3_Nswizzle0_Quant3_MulRoutedWeight0_silu_FP4X2_FP4X2_B16"
K2_XL = "moe_ck2stages_gemm2_256x128x128x128_1x4_MulABScaleExpertWeightShuffled_v3_Nswizzle0_Quant3_MulRoutedWeight1_FP4X2_FP4X2_B16"
my_entries = pd.DataFrame([
# bs=16 E=33 d=512: block_m=32 64x32x32 (correct block_m)
dict(cu_num=256, token=16, model_dim=7168, inter_dim=512, expert=33, topk=9,
act_type="ActivationType.Silu", dtype="torch.bfloat16",
q_dtype_a="torch.float4_e2m1fn_x2", q_dtype_w="torch.float4_e2m1fn_x2",
q_type="QuantType.per_1x32", use_g1u1=1, doweight_stage1=0,
block_m=32, ksplit=0, kernelName1=K1_S, kernelName2=K2_S, run_1stage=0, us=47.41),
# bs=128 E=33 d=512: block_m=32 64x32x32 (beats heuristic block_m=64, -10%)
dict(cu_num=256, token=128, model_dim=7168, inter_dim=512, expert=33, topk=9,
act_type="ActivationType.Silu", dtype="torch.bfloat16",
q_dtype_a="torch.float4_e2m1fn_x2", q_dtype_w="torch.float4_e2m1fn_x2",
q_type="QuantType.per_1x32", use_g1u1=1, doweight_stage1=0,
block_m=32, ksplit=0, kernelName1=K1_S, kernelName2=K2_S, run_1stage=0, us=58.58),
# bs=512 E=33 d=512: block_m=32 256x32x128 (beats 64x32x32, -15%)
dict(cu_num=256, token=512, model_dim=7168, inter_dim=512, expert=33, topk=9,
act_type="ActivationType.Silu", dtype="torch.bfloat16",
q_dtype_a="torch.float4_e2m1fn_x2", q_dtype_w="torch.float4_e2m1fn_x2",
q_type="QuantType.per_1x32", use_g1u1=1, doweight_stage1=0,
block_m=32, ksplit=0, kernelName1=K1_L, kernelName2=K2_S, run_1stage=0, us=129.09),
# bs=512 E=33 d=2048: block_m=128 256x128x128 (best for large expert dim)
dict(cu_num=256, token=512, model_dim=7168, inter_dim=2048, expert=33, topk=9,
act_type="ActivationType.Silu", dtype="torch.bfloat16",
q_dtype_a="torch.float4_e2m1fn_x2", q_dtype_w="torch.float4_e2m1fn_x2",
q_type="QuantType.per_1x32", use_g1u1=1, doweight_stage1=0,
block_m=128, ksplit=0, kernelName1=K1_XL, kernelName2=K2_XL, run_1stage=0, us=267.06),
])
dfs.append(my_entries)
combined = pd.concat(dfs, ignore_index=True)
# Filter out tagged rows (same logic as aiter's get_cfg_2stages)
if "_tag" in combined.columns:
combined = combined[combined["_tag"].fillna("") == ""]
# Deduplicate: sort by us ascending, keep lowest-latency entry per key
index_cols = ["cu_num", "token", "model_dim", "inter_dim", "expert", "topk",
"act_type", "dtype", "q_dtype_a", "q_dtype_w", "q_type",
"use_g1u1", "doweight_stage1"]
idx = [c for c in index_cols if c in combined.columns]
if idx and "us" in combined.columns:
combined["us"] = pd.to_numeric(combined["us"], errors="coerce")
combined = combined.sort_values("us")
combined = combined.drop_duplicates(subset=idx, keep="first")
combined.to_csv(_COMBINED_CSV, index=False)
return True
try:
if _build_combined_csv():
os.environ["AITER_CONFIG_FMOE"] = _COMBINED_CSV
except Exception:
pass # fall back to library defaults
# ---------------------------------------------------------------------------
import torch
from typing import Dict
from task import input_t, output_t
from aiter import ActivationType, QuantType
from aiter.fused_moe import fused_moe
def custom_kernel(data: input_t) -> output_t:
(
hidden_states,
gate_up_weight,
down_weight,
gate_up_weight_scale,
down_weight_scale,
gate_up_weight_shuffled,
down_weight_shuffled,
gate_up_weight_scale_shuffled,
down_weight_scale_shuffled,
topk_weights,
topk_ids,
config,
) = data
hidden_pad = config["d_hidden_pad"] - config["d_hidden"]
intermediate_pad = config["d_expert_pad"] - config["d_expert"]
output = fused_moe(
hidden_states,
gate_up_weight_shuffled,
down_weight_shuffled,
topk_weights,
topk_ids,
expert_mask=None,
activation=ActivationType.Silu,
quant_type=QuantType.per_1x32,
doweight_stage1=False,
w1_scale=gate_up_weight_scale_shuffled,
w2_scale=down_weight_scale_shuffled,
a1_scale=None,
a2_scale=None,
hidden_pad=hidden_pad,
intermediate_pad=intermediate_pad,
)
return output
scrolls · 158 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON