Skip to content
KernelIndex
Search⌘K

submission 567594

Aniket Sadashiva · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 141 lines, June 9 Researcher Reciprocity License v1.0.

submission__v6.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-moe-mxfp4-567594?include=source"
interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, fp32, fp8_e8m0, int32, mxfp4

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
AMD MXFP4 MoEsuite of 7 cases
AMD Instinct MI355X
167.7µs
#297 of 782
2026-03-16

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:dd2612a574616ee6892fb984db087b599c12ba83dcd5b44b041e4f81cadee545
license declaredunknown
license concludedunknown
authorsAniket Sadashiva
imported2026-08-15

Kernel source

submission__v6.py141 lines
"""
submission__v6: merge tuned large E=33 configs with selective ksplit on small E=33.

Large E=33 shapes use injected MI355X-tuned CK configs.
Small E=33 / d_expert=512 / bs<=128 shapes use ksplit=2.
E=257 keeps the runtime-hybrid policy from submission__v1.
"""
import os

from task import input_t, output_t

# Inject only the large E=33 rows so the small 33x512 shapes still fall back to
# heuristic selection, where we can force the split-k CKTile path.
_csv_path = "/home/runner/aiter/aiter/configs/model_configs/e33_fp4_tuned_fmoe.csv"
_csv_header = (
    "cu_num,token,model_dim,inter_dim,expert,topk,act_type,dtype,q_dtype_a,"
    "q_dtype_w,q_type,use_g1u1,doweight_stage1,block_m,ksplit,us1,kernelName1,"
    "err1,us2,kernelName2,err2,us,run_1stage,tflops,bw,_tag"
)
_common = (
    "ActivationType.Silu,torch.bfloat16,torch.float4_e2m1fn_x2,"
    "torch.float4_e2m1fn_x2,QuantType.per_1x32,1,0"
)
_k1_512_med = (
    "moe_ck2stages_gemm1_256x32x128x128_1x4_MulABScaleShuffled_v3_"
    "Nswizzle0_Quant3_MulRoutedWeight0_silu_FP4X2_FP4X2_B16"
)
_k2_small = (
    "moe_ck2stages_gemm2_64x32x32x128_1x1_MulABScaleExpertWeightShuffled_v1_"
    "Nswizzle0_Quant3_MulRoutedWeight1_FP4X2_FP4X2_B16"
)
_k1_2048 = (
    "moe_ck2stages_gemm1_256x128x128x128_1x4_MulABScaleShuffled_v3_"
    "Nswizzle0_Quant3_MulRoutedWeight0_silu_FP4X2_FP4X2_B16"
)
_k2_2048 = (
    "moe_ck2stages_gemm2_256x128x128x128_1x4_MulABScaleExpertWeightShuffled_v3_"
    "Nswizzle0_Quant3_MulRoutedWeight1_FP4X2_FP4X2_B16"
)
_rows = [
    (
        f"256,512,7168,512,33,9,{_common},32,0,0,{_k1_512_med},0.0%,0,"
        f"{_k2_small},0.0%,129.79,0,781.78,2884.18,"
    ),
    (
        f"256,512,7168,2048,33,9,{_common},128,0,0,{_k1_2048},0.0%,0,"
        f"{_k2_2048},0.0%,275.08,0,1475.47,5323.27,"
    ),
]

try:
    with open(_csv_path, "w") as f:
        f.write(_csv_header + "\n")
        for row in _rows:
            f.write(row + "\n")
except Exception:
    pass

os.environ["AITER_USE_OPUS_MOE_SORTING"] = "1"
os.environ["AITER_USE_NT"] = "0"

from aiter import ActivationType, QuantType
import aiter.fused_moe as fused_moe_mod


_LAST_RUNTIME_STATE = None


def _set_runtime_state(use_opus: bool, use_nt: bool, ksplit: int | None = None) -> None:
    global _LAST_RUNTIME_STATE

    state = (use_opus, use_nt, ksplit)
    if state == _LAST_RUNTIME_STATE:
        return

    fused_moe_mod._USE_OPUS_MOE_SORTING = use_opus
    os.environ["AITER_USE_NT"] = "1" if use_nt else "0"

    if ksplit in (None, 0):
        os.environ.pop("AITER_KSPLIT", None)
    else:
        os.environ["AITER_KSPLIT"] = str(ksplit)

    fused_moe_mod.use_nt.cache_clear()
    fused_moe_mod.get_ksplit.cache_clear()
    fused_moe_mod.get_2stage_cfgs.cache_clear()
    _LAST_RUNTIME_STATE = state


def _policy(config: dict) -> tuple[bool, bool, int | None]:
    total_experts = config["n_routed_experts"] + config["n_shared_experts"]

    if total_experts == 257:
        return True, False, None

    if config["d_expert"] == 512 and config["bs"] <= 128:
        return False, True, 2

    return True, True, None


def custom_kernel(data: input_t) -> output_t:
    (
        hidden_states,
        gate_up_weight,
        down_weight,
        gate_up_weight_scale,
        down_weight_scale,
        gate_up_weight_shuffled,
        down_weight_shuffled,
        gate_up_weight_scale_shuffled,
        down_weight_scale_shuffled,
        topk_weights,
        topk_ids,
        config,
    ) = data

    use_opus, use_nt, ksplit = _policy(config)
    _set_runtime_state(use_opus=use_opus, use_nt=use_nt, ksplit=ksplit)

    hidden_pad = config["d_hidden_pad"] - config["d_hidden"]
    intermediate_pad = config["d_expert_pad"] - config["d_expert"]

    return fused_moe_mod.fused_moe(
        hidden_states,
        gate_up_weight_shuffled,
        down_weight_shuffled,
        topk_weights,
        topk_ids,
        expert_mask=None,
        activation=ActivationType.Silu,
        quant_type=QuantType.per_1x32,
        doweight_stage1=False,
        w1_scale=gate_up_weight_scale_shuffled,
        w2_scale=down_weight_scale_shuffled,
        a1_scale=None,
        a2_scale=None,
        hidden_pad=hidden_pad,
        intermediate_pad=intermediate_pad,
    )
scrolls · 141 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 567194.

"""
- submission__v4: runtime hybrid + selective ksplit.
+ submission__v6: merge tuned large E=33 configs with selective ksplit on small E=33.
- Use ksplit=2 only on the small 33-expert / 512-intermediate shapes where it
- materially helps; keep submission__v1 behavior everywhere else.
+ Large E=33 shapes use injected MI355X-tuned CK configs.
+ Small E=33 / d_expert=512 / bs<=128 shapes use ksplit=2.
+ E=257 keeps the runtime-hybrid policy from submission__v1.
"""
import os
from task import input_t, output_t
+ # Inject only the large E=33 rows so the small 33x512 shapes still fall back to
+ # heuristic selection, where we can force the split-k CKTile path.
+ _csv_path = "/home/runner/aiter/aiter/configs/model_configs/e33_fp4_tuned_fmoe.csv"
+ _csv_header = (
+ "cu_num,token,model_dim,inter_dim,expert,topk,act_type,dtype,q_dtype_a,"
+ "q_dtype_w,q_type,use_g1u1,doweight_stage1,block_m,ksplit,us1,kernelName1,"
+ "err1,us2,kernelName2,err2,us,run_1stage,tflops,bw,_tag"
+ )
+ _common = (
+ "ActivationType.Silu,torch.bfloat16,torch.float4_e2m1fn_x2,"
+ "torch.float4_e2m1fn_x2,QuantType.per_1x32,1,0"
+ )
+ _k1_512_med = (
+ "moe_ck2stages_gemm1_256x32x128x128_1x4_MulABScaleShuffled_v3_"
+ "Nswizzle0_Quant3_MulRoutedWeight0_silu_FP4X2_FP4X2_B16"
+ )
+ _k2_small = (
+ "moe_ck2stages_gemm2_64x32x32x128_1x1_MulABScaleExpertWeightShuffled_v1_"
+ "Nswizzle0_Quant3_MulRoutedWeight1_FP4X2_FP4X2_B16"
+ )
+ _k1_2048 = (
+ "moe_ck2stages_gemm1_256x128x128x128_1x4_MulABScaleShuffled_v3_"
+ "Nswizzle0_Quant3_MulRoutedWeight0_silu_FP4X2_FP4X2_B16"
+ )
+ _k2_2048 = (
+ "moe_ck2stages_gemm2_256x128x128x128_1x4_MulABScaleExpertWeightShuffled_v3_"
+ "Nswizzle0_Quant3_MulRoutedWeight1_FP4X2_FP4X2_B16"
+ )
+ _rows = [
+ (
+ f"256,512,7168,512,33,9,{_common},32,0,0,{_k1_512_med},0.0%,0,"
+ f"{_k2_small},0.0%,129.79,0,781.78,2884.18,"
+ ),
+ (
+ f"256,512,7168,2048,33,9,{_common},128,0,0,{_k1_2048},0.0%,0,"
+ f"{_k2_2048},0.0%,275.08,0,1475.47,5323.27,"
+ ),
+ ]
+
+ try:
+ with open(_csv_path, "w") as f:
+ f.write(_csv_header + "\n")
+ for row in _rows:
+ f.write(row + "\n")
+ except Exception:
+ pass
+
os.environ["AITER_USE_OPUS_MOE_SORTING"] = "1"
os.environ["AITER_USE_NT"] = "0"
⋯ 34 unchanged lines
if config["d_expert"] == 512 and config["bs"] <= 128:
return False, True, 2
- return False, True, None
+ return True, True, None
def custom_kernel(data: input_t) -> output_t:
scrolls · 73 diff lines total

Best evidence level for this revision: reported

JSON