Skip to content
KernelIndex
Search⌘K

submission 662960

yuzhou2_44760 · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 184 lines, June 9 Researcher Reciprocity License v1.0.

submission_v163_cktile_d2048.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-moe-mxfp4-662960?include=source"
interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, fp32, fp8_e8m0, int32, mxfp4

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
AMD MXFP4 MoEsuite of 7 cases
AMD Instinct MI355X
166.7µs
#292 of 782
2026-03-29

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:90726d3cfa5d0ada60e0132693668e45667297bfe0fc48e0a5ffb57a835d421f
license declaredunknown
license concludedunknown
authorsyuzhou2_44760
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

fp4"""MoE MXFP4 v163: v134 + CKTile ksplit=2 for d=2048 (was CK XDL heuristic).

Kernel source

submission_v163_cktile_d2048.py184 lines
"""MoE MXFP4 v163: v134 + CKTile ksplit=2 for d=2048 (was CK XDL heuristic).

Based on v134. Single change: route bs=512 E=33 d=2048 through CKTile
block_m=16 ksplit=2 instead of CK XDL heuristic with block_size_M=64.
CKTile handles FP4 quantization INTERNALLY, eliminating separate
fused_dynamic_mxfp4_quant_moe_sort kernel launches for both stages.

CSV routing:
  - CKTile block_m=16 (ksplit=2): TEST shapes + bs=16 all + bs=512 E=33 d=2048
  - CKTile block_m=32 (ksplit=2): bs=128 all
  - CK XDL (ksplit=0): bs=512 E=33 d=512 (Small s1 + FlyDSL t64 reduce s2)
  - CK XDL (ksplit=0): bs=512 E=257 d=256 (Medium s1 + Small s2)
"""
import os
import importlib
import torch
from task import input_t, output_t

_CSV_SETUP_DONE = False

def _setup_csv():
    global _CSV_SETUP_DONE
    if _CSV_SETUP_DONE:
        return
    _CSV_SETUP_DONE = True

    os.environ["AITER_USE_OPUS_MOE_SORTING"] = "1"

    spec = importlib.util.find_spec("aiter")
    if spec and spec.submodule_search_locations:
        aiter_pkg = spec.submodule_search_locations[0]
    else:
        import aiter as _a
        aiter_pkg = os.path.dirname(_a.__file__)

    config_dir = os.path.join(aiter_pkg, "configs")
    default_csv = os.path.join(config_dir, "tuned_fmoe.csv")

    csv_path = "/tmp/moe_hybrid_v163.csv"

    header = "cu_num,token,model_dim,inter_dim,expert,topk,act_type,dtype,q_dtype_a,q_dtype_w,q_type,use_g1u1,doweight_stage1,block_m,ksplit,us1,kernelName1,err1,us2,kernelName2,err2,us,run_1stage,tflops,bw,_tag"

    s1_small = "moe_ck2stages_gemm1_64x32x32x128_1x1_MulABScaleShuffled_v3_Nswizzle0_Quant3_MulRoutedWeight0_silu_FP4X2_FP4X2_B16"
    s2_small = "moe_ck2stages_gemm2_64x32x32x128_1x1_MulABScaleExpertWeightShuffled_v1_Nswizzle0_Quant3_MulRoutedWeight1_FP4X2_FP4X2_B16"
    s1_med = "moe_ck2stages_gemm1_256x32x128x128_1x4_MulABScaleShuffled_v3_Nswizzle0_Quant3_MulRoutedWeight0_silu_FP4X2_FP4X2_B16"

    lines = [header]

    cktile_bm16_shapes = [
        (8,   4096, 1024, 257, 9),
        (32,  7168, 2048, 33,  9),
        (128, 4096, 1536, 65,  7),
        (16,  7168, 256,  257, 9),
        (16,  7168, 512,  33,  9),
        (512, 7168, 2048, 33,  9),  # bs=512 d=2048 — CKTile ksplit=2 (was CK XDL heuristic)
    ]
    for token, mdim, idim, expert, topk in cktile_bm16_shapes:
        line = (
            f"256,{token},{mdim},{idim},{expert},{topk},"
            f"ActivationType.Silu,torch.bfloat16,"
            f"torch.float4_e2m1fn_x2,torch.float4_e2m1fn_x2,"
            f"QuantType.per_1x32,1,0,"
            f"16,2,"
            f"0.5,,0.0,0.5,,0.0,"
            f"1.0,0,0.0,0.0,"
        )
        lines.append(line)

    cktile_bm32_shapes = [
        (128, 7168, 256,  257, 9),
        (128, 7168, 512,  33,  9),
    ]
    for token, mdim, idim, expert, topk in cktile_bm32_shapes:
        line = (
            f"256,{token},{mdim},{idim},{expert},{topk},"
            f"ActivationType.Silu,torch.bfloat16,"
            f"torch.float4_e2m1fn_x2,torch.float4_e2m1fn_x2,"
            f"QuantType.per_1x32,1,0,"
            f"32,2,"
            f"0.5,,0.0,0.5,,0.0,"
            f"1.0,0,0.0,0.0,"
        )
        lines.append(line)

    # CK XDL: bs=512 E=33 d=512 — FlyDSL t64 REDUCE for stage2 <- CHANGED
    line = (
        f"256,512,7168,512,33,9,"
        f"ActivationType.Silu,torch.bfloat16,"
        f"torch.float4_e2m1fn_x2,torch.float4_e2m1fn_x2,"
        f"QuantType.per_1x32,1,0,"
        f"32,0,"
        f"50.0,{s1_small},0.0,30.0,flydsl_moe2_afp4_wfp4_bf16_t64x256x256_reduce,0.0,"
        f"140.0,0,0.0,0.0,"
    )
    lines.append(line)

    # CK XDL: bs=512 E=257 d=256 (Medium s1 + Small s2)
    line = (
        f"256,512,7168,256,257,9,"
        f"ActivationType.Silu,torch.bfloat16,"
        f"torch.float4_e2m1fn_x2,torch.float4_e2m1fn_x2,"
        f"QuantType.per_1x32,1,0,"
        f"32,0,"
        f"100.0,{s1_med},0.0,70.0,{s2_small},0.0,"
        f"170.0,0,0.0,0.0,"
    )
    lines.append(line)

    with open(csv_path, "w") as f:
        f.write("\n".join(lines) + "\n")

    merge_paths = [csv_path]
    if os.path.exists(default_csv):
        merge_paths.append(default_csv)

    os.environ["AITER_CONFIG_FMOE"] = ":".join(merge_paths)


_fused_moe = None
_ActivationType = None
_QuantType = None
_shape_cache: dict = {}

def _ensure_imports():
    global _fused_moe, _ActivationType, _QuantType
    if _fused_moe is not None:
        return
    _setup_csv()
    from aiter import ActivationType, QuantType
    from aiter.fused_moe import fused_moe
    _fused_moe = fused_moe
    _ActivationType = ActivationType
    _QuantType = QuantType


def custom_kernel(data: input_t) -> output_t:
    global _fused_moe, _ActivationType, _QuantType
    _ensure_imports()

    (
        hidden_states,
        gate_up_weight, down_weight,
        gate_up_weight_scale, down_weight_scale,
        gate_up_weight_shuffled, down_weight_shuffled,
        gate_up_weight_scale_shuffled, down_weight_scale_shuffled,
        topk_weights, topk_ids,
        config,
    ) = data

    d_hidden_pad = config["d_hidden_pad"]
    d_expert_pad = config["d_expert_pad"]
    d_expert = config["d_expert"]
    cache_key = (hidden_states.shape[0], gate_up_weight_shuffled.shape[0],
                 config["d_hidden"], d_expert, d_hidden_pad, d_expert_pad)

    cached = _shape_cache.get(cache_key)
    if cached is None:
        hidden_pad = d_hidden_pad - config["d_hidden"]
        intermediate_pad = d_expert_pad - d_expert
        bsm = None
        cached = (hidden_pad, intermediate_pad, bsm)
        _shape_cache[cache_key] = cached
    hidden_pad, intermediate_pad, bsm = cached

    output = _fused_moe(
        hidden_states,
        gate_up_weight_shuffled,
        down_weight_shuffled,
        topk_weights,
        topk_ids,
        expert_mask=None,
        activation=_ActivationType.Silu,
        quant_type=_QuantType.per_1x32,
        doweight_stage1=False,
        w1_scale=gate_up_weight_scale_shuffled,
        w2_scale=down_weight_scale_shuffled,
        a1_scale=None,
        a2_scale=None,
        block_size_M=bsm,
        hidden_pad=hidden_pad,
        intermediate_pad=intermediate_pad,
    )
    return output
scrolls · 184 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON