Skip to content
KernelIndex
Search⌘K

submission 513826

ooousay · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 92 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-moe-mxfp4-513826?include=source"
interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, fp32, fp8_e8m0, int32, mxfp4

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
AMD MXFP4 MoEsuite of 7 cases
AMD Instinct MI355X
155.4µs
#244 of 782
2026-03-06

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:3d20508538f22c47256f2f27694d7777a222adedb371a9a00bb6e579f52590c2
license declaredunknown
license concludedunknown
authorsooousay
imported2026-08-15

Kernel source

submission.py92 lines
import os
# OPUS variant of MoE sorting kernel — consistent 3-6% improvement
os.environ["AITER_USE_OPUS_MOE_SORTING"] = "1"

import torch
from aiter import ActivationType, QuantType
from aiter.fused_moe import fused_moe, get_2stage_cfgs
from task import input_t, output_t

_current_mode = None


def custom_kernel(data: input_t) -> output_t:
    global _current_mode
    (
        hidden_states,
        gate_up_weight,
        down_weight,
        gate_up_weight_scale,
        down_weight_scale,
        gate_up_weight_shuffled,
        down_weight_shuffled,
        gate_up_weight_scale_shuffled,
        down_weight_scale_shuffled,
        topk_weights,
        topk_ids,
        config,
    ) = data

    M = hidden_states.shape[0]
    E = gate_up_weight_shuffled.shape[0]

    # Config-adaptive kernel selection via env vars.
    #
    # Three modes:
    # 1. "cktile" — E≤64, M≤128: KSPLIT=2 triggers cktile kernels via heuristic
    #    path (no CSV for E=33). CK Tile + split_k=2 massively improves CU
    #    utilization for small batch (bs=16: -30%, bs=128: -10%).
    #
    # 2. "cktile_e257" — E>64, M≤128: Bypass CSV tuning + KSPLIT=2 to force
    #    cktile path for E=257 small/medium batch. Benefits:
    #    - CK Tile splitk keeps input bf16 (skips pre-quantization kernel)
    #    - No inter-stage requantization (skips another kernel)
    #    - split_k=2 improves CU utilization for sparse expert activation
    #    - Proven: bs=16/E=257 went from 129µs → 96.5µs (-25%)
    #
    # 3. "default" — Everything else: CSV-tuned CK kernels (optimal for
    #    large batch where compute dominates launch overhead).

    if E <= 64 and M <= 128:
        mode = "cktile"
    elif E > 64:
        mode = "cktile_e257"
    else:
        mode = "default"

    if mode != _current_mode:
        if mode == "cktile":
            os.environ["AITER_KSPLIT"] = "2"
            os.environ.pop("AITER_BYPASS_TUNE_CONFIG", None)
        elif mode == "cktile_e257":
            os.environ["AITER_KSPLIT"] = "2"
            os.environ["AITER_BYPASS_TUNE_CONFIG"] = "1"
        else:
            os.environ.pop("AITER_KSPLIT", None)
            os.environ.pop("AITER_BYPASS_TUNE_CONFIG", None)
        get_2stage_cfgs.cache_clear()
        _current_mode = mode

    hidden_pad = config["d_hidden_pad"] - config["d_hidden"]
    intermediate_pad = config["d_expert_pad"] - config["d_expert"]

    output = fused_moe(
        hidden_states,
        gate_up_weight_shuffled,
        down_weight_shuffled,
        topk_weights,
        topk_ids,
        expert_mask=None,
        activation=ActivationType.Silu,
        quant_type=QuantType.per_1x32,
        doweight_stage1=False,
        w1_scale=gate_up_weight_scale_shuffled,
        w2_scale=down_weight_scale_shuffled,
        a1_scale=None,
        a2_scale=None,
        hidden_pad=hidden_pad,
        intermediate_pad=intermediate_pad,
    )

    return output
scrolls · 92 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 513713.

import os
- # OPUS variant of MoE sorting kernel: 3-6% improvement
+ # OPUS variant of MoE sorting kernel — consistent 3-6% improvement
os.environ["AITER_USE_OPUS_MOE_SORTING"] = "1"
import torch
- import aiter
- from aiter import ActivationType, QuantType, dtypes
+ from aiter import ActivationType, QuantType
from aiter.fused_moe import fused_moe, get_2stage_cfgs
from task import input_t, output_t
- _current_ksplit = None
+ _current_mode = None
def custom_kernel(data: input_t) -> output_t:
- global _current_ksplit
+ global _current_mode
(
hidden_states,
gate_up_weight,
⋯ 12 unchanged lines
M = hidden_states.shape[0]
E = gate_up_weight_shuffled.shape[0]
- # Adaptive ksplit: cktile kernels with split-K=2 are much faster
- # for small batch sizes with few experts (E=33, M<=128).
- # For large batches, default ksplit=0 is better (less reduction overhead).
- want_ksplit = "2" if (E <= 64 and M <= 128) else None
+ # Config-adaptive kernel selection via env vars.
+ #
+ # Three modes:
+ # 1. "cktile" — E≤64, M≤128: KSPLIT=2 triggers cktile kernels via heuristic
+ # path (no CSV for E=33). CK Tile + split_k=2 massively improves CU
+ # utilization for small batch (bs=16: -30%, bs=128: -10%).
+ #
+ # 2. "cktile_e257" — E>64, M≤128: Bypass CSV tuning + KSPLIT=2 to force
+ # cktile path for E=257 small/medium batch. Benefits:
+ # - CK Tile splitk keeps input bf16 (skips pre-quantization kernel)
+ # - No inter-stage requantization (skips another kernel)
+ # - split_k=2 improves CU utilization for sparse expert activation
+ # - Proven: bs=16/E=257 went from 129µs → 96.5µs (-25%)
+ #
+ # 3. "default" — Everything else: CSV-tuned CK kernels (optimal for
+ # large batch where compute dominates launch overhead).
- if want_ksplit != _current_ksplit:
- if want_ksplit:
- os.environ["AITER_KSPLIT"] = want_ksplit
+ if E <= 64 and M <= 128:
+ mode = "cktile"
+ elif E > 64:
+ mode = "cktile_e257"
+ else:
+ mode = "default"
+
+ if mode != _current_mode:
+ if mode == "cktile":
+ os.environ["AITER_KSPLIT"] = "2"
+ os.environ.pop("AITER_BYPASS_TUNE_CONFIG", None)
+ elif mode == "cktile_e257":
+ os.environ["AITER_KSPLIT"] = "2"
+ os.environ["AITER_BYPASS_TUNE_CONFIG"] = "1"
else:
os.environ.pop("AITER_KSPLIT", None)
+ os.environ.pop("AITER_BYPASS_TUNE_CONFIG", None)
get_2stage_cfgs.cache_clear()
- _current_ksplit = want_ksplit
+ _current_mode = mode
hidden_pad = config["d_hidden_pad"] - config["d_hidden"]
intermediate_pad = config["d_expert_pad"] - config["d_expert"]
scrolls · 73 diff lines total

Best evidence level for this revision: reported

JSON