Skip to content
KernelIndex
Search⌘K

submission 686584

mingkai_37292 · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 158 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-moe-mxfp4-686584?include=source"
interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, fp32, fp8_e8m0, int32, mxfp4

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
AMD MXFP4 MoEsuite of 7 cases
AMD Instinct MI355X
175.8µs
#353 of 782
2026-04-01

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:a253cb17353d121d21e2c41b00b9972897747957cda19654bd4ed99810d0c3dd
license declaredunknown
license concludedunknown
authorsmingkai_37292
imported2026-08-26

Kernel source

submission.py158 lines
#!POPCORN leaderboard amd-moe-mxfp4
#!POPCORN gpu MI355X

import os
import importlib.util

# Enable OPUS MOE sorting for potentially faster token dispatch
os.environ["AITER_USE_OPUS_MOE_SORTING"] = "1"

# ---------------------------------------------------------------------------
# Build a combined tuning CSV that includes ALL existing aiter CSV entries
# (primary + model_configs) PLUS our auto-tuner-discovered optimal entries
# for 32-expert shapes.
#
# Setting AITER_CONFIG_FMOE bypasses the library's merge pipeline entirely,
# so we must include entries from model_configs CSVs ourselves and deduplicate.
# ---------------------------------------------------------------------------

_COMBINED_CSV = "/tmp/amd_moe_mxfp4_combined.csv"

def _build_combined_csv():
    """Merge ALL existing aiter CSVs + our 32-expert entries, deduplicated."""
    spec = importlib.util.find_spec("aiter")
    if not spec or not spec.origin:
        return False

    import pandas as pd

    aiter_dir = os.path.dirname(spec.origin)
    configs_dir = os.path.join(aiter_dir, "configs")
    mc_dir = os.path.join(configs_dir, "model_configs")

    csv_paths = [
        os.path.join(configs_dir, "tuned_fmoe.csv"),
        os.path.join(mc_dir, "dsv3_fp4_tuned_fmoe.csv"),
        os.path.join(mc_dir, "a8w8_blockscale_tuned_fmoe_qwen3_235b.csv"),
    ]

    dfs = []
    for p in csv_paths:
        if os.path.exists(p):
            try:
                dfs.append(pd.read_csv(p))
            except Exception:
                pass

    if not dfs:
        return False

    # --- optimal 32-expert configs from AITER_ONLINE_TUNE exploration ------
    K1_S  = "moe_ck2stages_gemm1_64x32x32x128_1x1_MulABScaleShuffled_v3_Nswizzle0_Quant3_MulRoutedWeight0_silu_FP4X2_FP4X2_B16"
    K2_S  = "moe_ck2stages_gemm2_64x32x32x128_1x1_MulABScaleExpertWeightShuffled_v1_Nswizzle0_Quant3_MulRoutedWeight1_FP4X2_FP4X2_B16"
    K1_L  = "moe_ck2stages_gemm1_256x32x128x128_1x4_MulABScaleShuffled_v3_Nswizzle0_Quant3_MulRoutedWeight0_silu_FP4X2_FP4X2_B16"
    K1_XL = "moe_ck2stages_gemm1_256x128x128x128_1x4_MulABScaleShuffled_v3_Nswizzle0_Quant3_MulRoutedWeight0_silu_FP4X2_FP4X2_B16"
    K2_XL = "moe_ck2stages_gemm2_256x128x128x128_1x4_MulABScaleExpertWeightShuffled_v3_Nswizzle0_Quant3_MulRoutedWeight1_FP4X2_FP4X2_B16"

    my_entries = pd.DataFrame([
        # bs=16  E=33 d=512:  block_m=32  64x32x32  (correct block_m)
        dict(cu_num=256, token=16,  model_dim=7168, inter_dim=512,  expert=33, topk=9,
             act_type="ActivationType.Silu", dtype="torch.bfloat16",
             q_dtype_a="torch.float4_e2m1fn_x2", q_dtype_w="torch.float4_e2m1fn_x2",
             q_type="QuantType.per_1x32", use_g1u1=1, doweight_stage1=0,
             block_m=32, ksplit=0, kernelName1=K1_S, kernelName2=K2_S, run_1stage=0, us=47.41),
        # bs=128 E=33 d=512:  block_m=32  64x32x32  (beats heuristic block_m=64, -10%)
        dict(cu_num=256, token=128, model_dim=7168, inter_dim=512,  expert=33, topk=9,
             act_type="ActivationType.Silu", dtype="torch.bfloat16",
             q_dtype_a="torch.float4_e2m1fn_x2", q_dtype_w="torch.float4_e2m1fn_x2",
             q_type="QuantType.per_1x32", use_g1u1=1, doweight_stage1=0,
             block_m=32, ksplit=0, kernelName1=K1_S, kernelName2=K2_S, run_1stage=0, us=58.58),
        # bs=512 E=33 d=512:  block_m=32  256x32x128 (beats 64x32x32, -15%)
        dict(cu_num=256, token=512, model_dim=7168, inter_dim=512,  expert=33, topk=9,
             act_type="ActivationType.Silu", dtype="torch.bfloat16",
             q_dtype_a="torch.float4_e2m1fn_x2", q_dtype_w="torch.float4_e2m1fn_x2",
             q_type="QuantType.per_1x32", use_g1u1=1, doweight_stage1=0,
             block_m=32, ksplit=0, kernelName1=K1_L, kernelName2=K2_S, run_1stage=0, us=129.09),
        # bs=512 E=33 d=2048: block_m=128 256x128x128 (best for large expert dim)
        dict(cu_num=256, token=512, model_dim=7168, inter_dim=2048, expert=33, topk=9,
             act_type="ActivationType.Silu", dtype="torch.bfloat16",
             q_dtype_a="torch.float4_e2m1fn_x2", q_dtype_w="torch.float4_e2m1fn_x2",
             q_type="QuantType.per_1x32", use_g1u1=1, doweight_stage1=0,
             block_m=128, ksplit=0, kernelName1=K1_XL, kernelName2=K2_XL, run_1stage=0, us=267.06),
    ])
    dfs.append(my_entries)

    combined = pd.concat(dfs, ignore_index=True)

    # Filter out tagged rows (same logic as aiter's get_cfg_2stages)
    if "_tag" in combined.columns:
        combined = combined[combined["_tag"].fillna("") == ""]

    # Deduplicate: sort by us ascending, keep lowest-latency entry per key
    index_cols = ["cu_num", "token", "model_dim", "inter_dim", "expert", "topk",
                  "act_type", "dtype", "q_dtype_a", "q_dtype_w", "q_type",
                  "use_g1u1", "doweight_stage1"]
    idx = [c for c in index_cols if c in combined.columns]

    if idx and "us" in combined.columns:
        combined["us"] = pd.to_numeric(combined["us"], errors="coerce")
        combined = combined.sort_values("us")
        combined = combined.drop_duplicates(subset=idx, keep="first")

    combined.to_csv(_COMBINED_CSV, index=False)
    return True

try:
    if _build_combined_csv():
        os.environ["AITER_CONFIG_FMOE"] = _COMBINED_CSV
except Exception:
    pass  # fall back to library defaults
# ---------------------------------------------------------------------------

import torch
from typing import Dict
from task import input_t, output_t

from aiter import ActivationType, QuantType
from aiter.fused_moe import fused_moe


def custom_kernel(data: input_t) -> output_t:
    (
        hidden_states,
        gate_up_weight,
        down_weight,
        gate_up_weight_scale,
        down_weight_scale,
        gate_up_weight_shuffled,
        down_weight_shuffled,
        gate_up_weight_scale_shuffled,
        down_weight_scale_shuffled,
        topk_weights,
        topk_ids,
        config,
    ) = data

    hidden_pad = config["d_hidden_pad"] - config["d_hidden"]
    intermediate_pad = config["d_expert_pad"] - config["d_expert"]

    output = fused_moe(
        hidden_states,
        gate_up_weight_shuffled,
        down_weight_shuffled,
        topk_weights,
        topk_ids,
        expert_mask=None,
        activation=ActivationType.Silu,
        quant_type=QuantType.per_1x32,
        doweight_stage1=False,
        w1_scale=gate_up_weight_scale_shuffled,
        w2_scale=down_weight_scale_shuffled,
        a1_scale=None,
        a2_scale=None,
        hidden_pad=hidden_pad,
        intermediate_pad=intermediate_pad,
    )

    return output
scrolls · 158 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON