Skip to content
KernelIndex
Search⌘K

submission 551278

fchange3413 · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 134 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-mxfp4-mm-551278?include=source"
interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, mxfp4

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
AMD MXFP4 GEMMsuite of 6 cases
AMD Instinct MI355X
22.3µs
#746 of 1143
2026-03-14

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:7cf3dbeb2d956bfd8494189f78411a11ab1872275b24a90a43c097bbe659f654
license declaredunknown
license concludedunknown
authorsfchange3413
imported2026-08-26

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

split-k"splitK": 0,

Kernel source

submission.py134 lines
#!POPCORN leaderboard amd-mxfp4-mm
#!POPCORN gpu MI355X

import csv
import importlib.util
import os
import tempfile
from pathlib import Path

import torch

from task import input_t, output_t


def mangle_kernel_name(short_name: str) -> str:
    return f"_ZN5aiter{len(short_name)}{short_name}E"


STATIC_OVERRIDES = [
    {
        "M": 4,
        "N": 2880,
        "K": 512,
        "kernelName": mangle_kernel_name("f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128"),
        "splitK": 0,
    },
    {
        "M": 16,
        "N": 2112,
        "K": 7168,
        "kernelName": mangle_kernel_name("f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128"),
        "splitK": 2,
    },
    {
        "M": 32,
        "N": 4096,
        "K": 512,
        "kernelName": mangle_kernel_name("f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128"),
        "splitK": 0,
    },
    {
        "M": 32,
        "N": 2880,
        "K": 512,
        "kernelName": mangle_kernel_name("f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128"),
        "splitK": 0,
    },
]


def ensure_override_csv() -> str:
    existing = os.environ.get("AITER_CONFIG_GEMM_A4W4")
    if existing and os.pathsep in existing:
        return existing

    cu_num = torch.cuda.get_device_properties(0).multi_processor_count
    workdir = Path(tempfile.gettempdir()) / "amd_mxfp4_mm"
    workdir.mkdir(parents=True, exist_ok=True)
    csv_path = workdir / "a4w4_overrides.csv"

    with csv_path.open("w", newline="") as f:
        writer = csv.DictWriter(
            f,
            fieldnames=[
                "cu_num",
                "M",
                "N",
                "K",
                "kernelId",
                "splitK",
                "us",
                "kernelName",
                "tflops",
                "bw",
                "errRatio",
            ],
        )
        writer.writeheader()
        for row in STATIC_OVERRIDES:
            writer.writerow(
                {
                    "cu_num": cu_num,
                    "M": row["M"],
                    "N": row["N"],
                    "K": row["K"],
                    "kernelId": -1,
                    "us": 0.0,
                    "kernelName": row["kernelName"],
                    "splitK": row["splitK"],
                    "tflops": 0.0,
                    "bw": 0.0,
                    "errRatio": 0.0,
                }
            )

    spec = importlib.util.find_spec("aiter")
    if spec is None or not spec.submodule_search_locations:
        raise RuntimeError("Unable to locate installed aiter package for config merge.")
    package_dir = Path(next(iter(spec.submodule_search_locations)))
    default_csv = package_dir / "configs" / "a4w4_blockscale_tuned_gemm.csv"

    merged = str(default_csv) + os.pathsep + str(csv_path)
    os.environ["AITER_CONFIG_GEMM_A4W4"] = merged
    return merged


ensure_override_csv()

import aiter
from aiter import dtypes
from aiter.ops.triton.quant import dynamic_mxfp4_quant
from aiter.utility.fp4_utils import e8m0_shuffle


def quant_mxfp4(x: torch.Tensor):
    x_fp4, bs_e8m0 = dynamic_mxfp4_quant(x)
    bs_e8m0 = e8m0_shuffle(bs_e8m0)
    return x_fp4.view(dtypes.fp4x2), bs_e8m0.view(dtypes.fp8_e8m0)


def custom_kernel(data: input_t) -> output_t:
    a, _b, _b_q, b_shuffle, b_scale_sh = data
    a = a.contiguous()
    a_q, a_scale_sh = quant_mxfp4(a)

    return aiter.gemm_a4w4(
        a_q,
        b_shuffle,
        a_scale_sh,
        b_scale_sh,
        dtype=dtypes.bf16,
        bpreshuffle=True,
    )
scrolls · 134 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON