submission 513826
ooousay · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 92 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-moe-mxfp4-513826?include=source"interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, fp32, fp8_e8m0, int32, mxfp4
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:3d20508538f22c47256f2f27694d7777a222adedb371a9a00bb6e579f52590c2
license declaredunknown
license concludedunknown
authorsooousay
imported2026-08-15
Kernel source
submission.py92 lines
import os
# OPUS variant of MoE sorting kernel — consistent 3-6% improvement
os.environ["AITER_USE_OPUS_MOE_SORTING"] = "1"
import torch
from aiter import ActivationType, QuantType
from aiter.fused_moe import fused_moe, get_2stage_cfgs
from task import input_t, output_t
_current_mode = None
def custom_kernel(data: input_t) -> output_t:
global _current_mode
(
hidden_states,
gate_up_weight,
down_weight,
gate_up_weight_scale,
down_weight_scale,
gate_up_weight_shuffled,
down_weight_shuffled,
gate_up_weight_scale_shuffled,
down_weight_scale_shuffled,
topk_weights,
topk_ids,
config,
) = data
M = hidden_states.shape[0]
E = gate_up_weight_shuffled.shape[0]
# Config-adaptive kernel selection via env vars.
#
# Three modes:
# 1. "cktile" — E≤64, M≤128: KSPLIT=2 triggers cktile kernels via heuristic
# path (no CSV for E=33). CK Tile + split_k=2 massively improves CU
# utilization for small batch (bs=16: -30%, bs=128: -10%).
#
# 2. "cktile_e257" — E>64, M≤128: Bypass CSV tuning + KSPLIT=2 to force
# cktile path for E=257 small/medium batch. Benefits:
# - CK Tile splitk keeps input bf16 (skips pre-quantization kernel)
# - No inter-stage requantization (skips another kernel)
# - split_k=2 improves CU utilization for sparse expert activation
# - Proven: bs=16/E=257 went from 129µs → 96.5µs (-25%)
#
# 3. "default" — Everything else: CSV-tuned CK kernels (optimal for
# large batch where compute dominates launch overhead).
if E <= 64 and M <= 128:
mode = "cktile"
elif E > 64:
mode = "cktile_e257"
else:
mode = "default"
if mode != _current_mode:
if mode == "cktile":
os.environ["AITER_KSPLIT"] = "2"
os.environ.pop("AITER_BYPASS_TUNE_CONFIG", None)
elif mode == "cktile_e257":
os.environ["AITER_KSPLIT"] = "2"
os.environ["AITER_BYPASS_TUNE_CONFIG"] = "1"
else:
os.environ.pop("AITER_KSPLIT", None)
os.environ.pop("AITER_BYPASS_TUNE_CONFIG", None)
get_2stage_cfgs.cache_clear()
_current_mode = mode
hidden_pad = config["d_hidden_pad"] - config["d_hidden"]
intermediate_pad = config["d_expert_pad"] - config["d_expert"]
output = fused_moe(
hidden_states,
gate_up_weight_shuffled,
down_weight_shuffled,
topk_weights,
topk_ids,
expert_mask=None,
activation=ActivationType.Silu,
quant_type=QuantType.per_1x32,
doweight_stage1=False,
w1_scale=gate_up_weight_scale_shuffled,
w2_scale=down_weight_scale_shuffled,
a1_scale=None,
a2_scale=None,
hidden_pad=hidden_pad,
intermediate_pad=intermediate_pad,
)
return output
scrolls · 92 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 513713.
import os- # OPUS variant of MoE sorting kernel: 3-6% improvement+ # OPUS variant of MoE sorting kernel — consistent 3-6% improvementos.environ["AITER_USE_OPUS_MOE_SORTING"] = "1"import torch- import aiter- from aiter import ActivationType, QuantType, dtypes+ from aiter import ActivationType, QuantTypefrom aiter.fused_moe import fused_moe, get_2stage_cfgsfrom task import input_t, output_t- _current_ksplit = None+ _current_mode = Nonedef custom_kernel(data: input_t) -> output_t:- global _current_ksplit+ global _current_mode(hidden_states,gate_up_weight,⋯ 12 unchanged linesM = hidden_states.shape[0]E = gate_up_weight_shuffled.shape[0]- # Adaptive ksplit: cktile kernels with split-K=2 are much faster- # for small batch sizes with few experts (E=33, M<=128).- # For large batches, default ksplit=0 is better (less reduction overhead).- want_ksplit = "2" if (E <= 64 and M <= 128) else None+ # Config-adaptive kernel selection via env vars.+ #+ # Three modes:+ # 1. "cktile" — E≤64, M≤128: KSPLIT=2 triggers cktile kernels via heuristic+ # path (no CSV for E=33). CK Tile + split_k=2 massively improves CU+ # utilization for small batch (bs=16: -30%, bs=128: -10%).+ #+ # 2. "cktile_e257" — E>64, M≤128: Bypass CSV tuning + KSPLIT=2 to force+ # cktile path for E=257 small/medium batch. Benefits:+ # - CK Tile splitk keeps input bf16 (skips pre-quantization kernel)+ # - No inter-stage requantization (skips another kernel)+ # - split_k=2 improves CU utilization for sparse expert activation+ # - Proven: bs=16/E=257 went from 129µs → 96.5µs (-25%)+ #+ # 3. "default" — Everything else: CSV-tuned CK kernels (optimal for+ # large batch where compute dominates launch overhead).- if want_ksplit != _current_ksplit:- if want_ksplit:- os.environ["AITER_KSPLIT"] = want_ksplit+ if E <= 64 and M <= 128:+ mode = "cktile"+ elif E > 64:+ mode = "cktile_e257"+ else:+ mode = "default"++ if mode != _current_mode:+ if mode == "cktile":+ os.environ["AITER_KSPLIT"] = "2"+ os.environ.pop("AITER_BYPASS_TUNE_CONFIG", None)+ elif mode == "cktile_e257":+ os.environ["AITER_KSPLIT"] = "2"+ os.environ["AITER_BYPASS_TUNE_CONFIG"] = "1"else:os.environ.pop("AITER_KSPLIT", None)+ os.environ.pop("AITER_BYPASS_TUNE_CONFIG", None)get_2stage_cfgs.cache_clear()- _current_ksplit = want_ksplit+ _current_mode = modehidden_pad = config["d_hidden_pad"] - config["d_hidden"]intermediate_pad = config["d_expert_pad"] - config["d_expert"]
scrolls · 73 diff lines total
Best evidence level for this revision: reported
JSON