Skip to content
KernelIndex
Search⌘K

submission 727806

nanbeilvdougao · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 82 lines, June 9 Researcher Reciprocity License v1.0.

submission_20260405_1.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-moe-mxfp4-727806?include=source"
interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, fp32, fp8_e8m0, int32, mxfp4

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
AMD MXFP4 MoEsuite of 7 cases
AMD Instinct MI355X
183.9µs
#529 of 782
2026-04-04

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:406a8e495b2ad289dc452d980949694bf0e568cccc062647a573d95a5f249b9a
license declaredunknown
license concludedunknown
authorsnanbeilvdougao
imported2026-08-26

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

fp4overlay_dir = Path(tempfile.mkdtemp(prefix="amd-moe-mxfp4-overlay-"))

Kernel source

submission_20260405_1.py82 lines
import csv
import importlib.util
import os
import tempfile
from pathlib import Path


TARGET_ROWS = [
    {
        "cu_num": "256", "token": "128", "model_dim": "7168", "inter_dim": "512", "expert": "33", "topk": "9",
        "act_type": "ActivationType.Silu", "dtype": "torch.bfloat16", "q_dtype_a": "torch.float4_e2m1fn_x2", "q_dtype_w": "torch.float4_e2m1fn_x2",
        "q_type": "QuantType.per_1x32", "use_g1u1": "1", "doweight_stage1": "0", "block_m": "32", "ksplit": "0",
        "us1": "238.357", "kernelName1": "moe_ck2stages_gemm1_64x32x32x128_1x1_MulABScaleShuffled_v3_Nswizzle0_Quant3_MulRoutedWeight0_silu_FP4X2_FP4X2_B16",
        "err1": "0.0%", "us2": "124.6709", "kernelName2": "moe_ck2stages_gemm2_256x32x128x128_1x4_MulABScaleExpertWeightShuffled_v3_Nswizzle0_Quant3_MulRoutedWeight1_FP4X2_FP4X2_B16",
        "err2": "2.8%", "us": "363.0279", "run_1stage": "0", "tflops": "62.11", "bw": "11653.68",
    },
    {
        "cu_num": "256", "token": "512", "model_dim": "7168", "inter_dim": "2048", "expert": "33", "topk": "9",
        "act_type": "ActivationType.Silu", "dtype": "torch.bfloat16", "q_dtype_a": "torch.float4_e2m1fn_x2", "q_dtype_w": "torch.float4_e2m1fn_x2",
        "q_type": "QuantType.per_1x32", "use_g1u1": "1", "doweight_stage1": "0", "block_m": "64", "ksplit": "0",
        "us1": "0.0", "kernelName1": "moe_ck2stages_gemm1_256x64x128x128_1x4_MulABScaleShuffled_v3_Nswizzle0_Quant3_MulRoutedWeight0_silu_FP4X2_FP4X2_B16",
        "err1": "0.0%", "us2": "0.0", "kernelName2": "moe_ck2stages_gemm2_256x64x128x128_1x4_MulABScaleExpertWeightShuffled_v3_Nswizzle0_Quant3_MulRoutedWeight1_FP4X2_FP4X2_B16",
        "err2": "0.0%", "us": "0.0", "run_1stage": "0", "tflops": "0.0", "bw": "0.0",
    },
]

HEADER = ["cu_num","token","model_dim","inter_dim","expert","topk","act_type","dtype","q_dtype_a","q_dtype_w","q_type","use_g1u1","doweight_stage1","block_m","ksplit","us1","kernelName1","err1","us2","kernelName2","err2","us","run_1stage","tflops","bw"]
PREFERRED_FP4_CONFIGS = ["dsv3_fp4_tuned_fmoe.csv", "kimik2_fp4_tuned_fmoe.csv"]


def _configure_aiter_overlay() -> None:
    spec = importlib.util.find_spec("aiter")
    if spec is None or not spec.submodule_search_locations:
        return
    pkg_dir = Path(next(iter(spec.submodule_search_locations)))
    config_dir = pkg_dir / "configs"
    merge_paths = []
    existing = os.environ.get("AITER_CONFIG_FMOE", "")
    if existing:
        merge_paths.extend([p for p in existing.split(":") if p])
    else:
        base_file = config_dir / "tuned_fmoe.csv"
        if base_file.is_file():
            merge_paths.append(str(base_file))
        model_configs_dir = config_dir / "model_configs"
        if model_configs_dir.is_dir():
            for name in PREFERRED_FP4_CONFIGS:
                path = model_configs_dir / name
                if path.is_file():
                    merge_paths.append(str(path))
    overlay_dir = Path(tempfile.mkdtemp(prefix="amd-moe-mxfp4-overlay-"))
    overlay_path = overlay_dir / "submission_20260405_1_overlay.csv"
    with overlay_path.open("w", newline="") as f:
        writer = csv.DictWriter(f, fieldnames=HEADER)
        writer.writeheader()
        writer.writerows(TARGET_ROWS)
    merge_paths.append(str(overlay_path))
    os.environ["AITER_CONFIG_FMOE"] = ":".join(merge_paths)


_configure_aiter_overlay()

from task import input_t, output_t
from aiter import ActivationType, QuantType
from aiter.fused_moe import fused_moe


def custom_kernel(data: input_t) -> output_t:
    (
        hidden_states, _gate_up_weight, _down_weight, _gate_up_weight_scale, _down_weight_scale,
        gate_up_weight_shuffled, down_weight_shuffled, gate_up_weight_scale_shuffled, down_weight_scale_shuffled,
        topk_weights, topk_ids, config,
    ) = data
    hidden_pad = config["d_hidden_pad"] - config["d_hidden"]
    intermediate_pad = config["d_expert_pad"] - config["d_expert"]
    return fused_moe(
        hidden_states, gate_up_weight_shuffled, down_weight_shuffled, topk_weights, topk_ids,
        expert_mask=None, activation=ActivationType.Silu, quant_type=QuantType.per_1x32,
        doweight_stage1=False, w1_scale=gate_up_weight_scale_shuffled, w2_scale=down_weight_scale_shuffled,
        a1_scale=None, a2_scale=None, hidden_pad=hidden_pad, intermediate_pad=intermediate_pad,
    )
scrolls · 82 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON