submission 567194
Aniket Sadashiva · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 93 lines, June 9 Researcher Reciprocity License v1.0.
submission__v4.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-moe-mxfp4-567194?include=source"interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, fp32, fp8_e8m0, int32, mxfp4
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:1ed097feae2f3d057ec1fafdc77ea0a6f95163a6ea9dee10d37d6c96277dda34
license declaredunknown
license concludedunknown
authorsAniket Sadashiva
imported2026-08-15
Kernel source
submission__v4.py93 lines
"""
submission__v4: runtime hybrid + selective ksplit.
Use ksplit=2 only on the small 33-expert / 512-intermediate shapes where it
materially helps; keep submission__v1 behavior everywhere else.
"""
import os
from task import input_t, output_t
os.environ["AITER_USE_OPUS_MOE_SORTING"] = "1"
os.environ["AITER_USE_NT"] = "0"
from aiter import ActivationType, QuantType
import aiter.fused_moe as fused_moe_mod
_LAST_RUNTIME_STATE = None
def _set_runtime_state(use_opus: bool, use_nt: bool, ksplit: int | None = None) -> None:
global _LAST_RUNTIME_STATE
state = (use_opus, use_nt, ksplit)
if state == _LAST_RUNTIME_STATE:
return
fused_moe_mod._USE_OPUS_MOE_SORTING = use_opus
os.environ["AITER_USE_NT"] = "1" if use_nt else "0"
if ksplit in (None, 0):
os.environ.pop("AITER_KSPLIT", None)
else:
os.environ["AITER_KSPLIT"] = str(ksplit)
fused_moe_mod.use_nt.cache_clear()
fused_moe_mod.get_ksplit.cache_clear()
fused_moe_mod.get_2stage_cfgs.cache_clear()
_LAST_RUNTIME_STATE = state
def _policy(config: dict) -> tuple[bool, bool, int | None]:
total_experts = config["n_routed_experts"] + config["n_shared_experts"]
if total_experts == 257:
return True, False, None
if config["d_expert"] == 512 and config["bs"] <= 128:
return False, True, 2
return False, True, None
def custom_kernel(data: input_t) -> output_t:
(
hidden_states,
gate_up_weight,
down_weight,
gate_up_weight_scale,
down_weight_scale,
gate_up_weight_shuffled,
down_weight_shuffled,
gate_up_weight_scale_shuffled,
down_weight_scale_shuffled,
topk_weights,
topk_ids,
config,
) = data
use_opus, use_nt, ksplit = _policy(config)
_set_runtime_state(use_opus=use_opus, use_nt=use_nt, ksplit=ksplit)
hidden_pad = config["d_hidden_pad"] - config["d_hidden"]
intermediate_pad = config["d_expert_pad"] - config["d_expert"]
return fused_moe_mod.fused_moe(
hidden_states,
gate_up_weight_shuffled,
down_weight_shuffled,
topk_weights,
topk_ids,
expert_mask=None,
activation=ActivationType.Silu,
quant_type=QuantType.per_1x32,
doweight_stage1=False,
w1_scale=gate_up_weight_scale_shuffled,
w2_scale=down_weight_scale_shuffled,
a1_scale=None,
a2_scale=None,
hidden_pad=hidden_pad,
intermediate_pad=intermediate_pad,
)
scrolls · 93 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 566632.
"""- v5: Safe approach - only env vars, no custom function params.- Shape branching with env var tuning only.+ submission__v4: runtime hybrid + selective ksplit.++ Use ksplit=2 only on the small 33-expert / 512-intermediate shapes where it+ materially helps; keep submission__v1 behavior everywhere else."""import os+from task import input_t, output_t- # Set env vars BEFORE importing aiter (some are read at import time)os.environ["AITER_USE_OPUS_MOE_SORTING"] = "1"os.environ["AITER_USE_NT"] = "0"- import aiterfrom aiter import ActivationType, QuantType- from aiter.fused_moe import fused_moe+ import aiter.fused_moe as fused_moe_mod- def _call_default(data):+ _LAST_RUNTIME_STATE = None+++ def _set_runtime_state(use_opus: bool, use_nt: bool, ksplit: int | None = None) -> None:+ global _LAST_RUNTIME_STATE++ state = (use_opus, use_nt, ksplit)+ if state == _LAST_RUNTIME_STATE:+ return++ fused_moe_mod._USE_OPUS_MOE_SORTING = use_opus+ os.environ["AITER_USE_NT"] = "1" if use_nt else "0"++ if ksplit in (None, 0):+ os.environ.pop("AITER_KSPLIT", None)+ else:+ os.environ["AITER_KSPLIT"] = str(ksplit)++ fused_moe_mod.use_nt.cache_clear()+ fused_moe_mod.get_ksplit.cache_clear()+ fused_moe_mod.get_2stage_cfgs.cache_clear()+ _LAST_RUNTIME_STATE = state+++ def _policy(config: dict) -> tuple[bool, bool, int | None]:+ total_experts = config["n_routed_experts"] + config["n_shared_experts"]++ if total_experts == 257:+ return True, False, None++ if config["d_expert"] == 512 and config["bs"] <= 128:+ return False, True, 2++ return False, True, None+++ def custom_kernel(data: input_t) -> output_t:(- hidden_states, gate_up_weight, down_weight,- gate_up_weight_scale, down_weight_scale,- gate_up_weight_shuffled, down_weight_shuffled,- gate_up_weight_scale_shuffled, down_weight_scale_shuffled,- topk_weights, topk_ids, config,+ hidden_states,+ gate_up_weight,+ down_weight,+ gate_up_weight_scale,+ down_weight_scale,+ gate_up_weight_shuffled,+ down_weight_shuffled,+ gate_up_weight_scale_shuffled,+ down_weight_scale_shuffled,+ topk_weights,+ topk_ids,+ config,) = data+ use_opus, use_nt, ksplit = _policy(config)+ _set_runtime_state(use_opus=use_opus, use_nt=use_nt, ksplit=ksplit)+hidden_pad = config["d_hidden_pad"] - config["d_hidden"]intermediate_pad = config["d_expert_pad"] - config["d_expert"]- return fused_moe(+ return fused_moe_mod.fused_moe(hidden_states,gate_up_weight_shuffled,down_weight_shuffled,⋯ 10 unchanged lineshidden_pad=hidden_pad,intermediate_pad=intermediate_pad,)--- def custom_kernel(data: input_t) -> output_t:- return _call_default(data)
scrolls · 99 diff lines total
Best evidence level for this revision: reported
JSON