Skip to content
KernelIndex
Search⌘K

submission 643450

SwordHoly · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 103 lines, June 9 Researcher Reciprocity License v1.0.

v10_tuned_csv.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-mxfp4-mm-643450?include=source"
interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, mxfp4

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
AMD MXFP4 GEMMsuite of 6 cases
AMD Instinct MI355X
22.5µs
#755 of 1143
2026-03-27

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:3733ec8f792e19079fa571ef79d10380e52efc4257f8c1f87352501ada74cb35
license declaredunknown
license concludedunknown
authorsSwordHoly
imported2026-08-26

Kernel source

v10_tuned_csv.py103 lines
"""
GEMM v10: Custom tuned CSV with entries for ALL benchmark shapes.
The existing a4w4_blockscale_tuned_gemm.csv misses N=2880 (small M) and N=4096/K=512.

From v8 diagnostic, available BpreShuffle kernels:
  32x128, 32x256, 64x128, 64x256, 64x640, 64x768,
  96x128, 96x256, 96x384, 96x512, 96x640,
  128x128, 128x256, 128x384, 128x512,
  160x128, 160x256, 160x384,
  192x128, 192x256,
  224x128, 224x256,
  256x128, 256x256

Existing CSV matches for our shapes:
  M=8,N=2112,K=7168: 32x128, 12.8µs
  M=64,N=7168,K=2048: 32x128, 6.8µs
  M=256,N=3072,K=1536: 32x128, 6.2µs

Missing shapes: M=4/32 with N=2880/K=512, M=32 with N=4096/K=512, M=16 with N=2112/K=7168

Strategy: Merge existing tuned CSV with custom entries for missing shapes.
Try different kernels for small M: 32x128 vs 64x128.
"""
from task import input_t, output_t


def custom_kernel(data: input_t) -> output_t:
    import os
    import sys
    import torch
    import aiter
    from aiter import dtypes
    from aiter.ops.triton.quant import dynamic_mxfp4_quant
    from aiter.utility.fp4_utils import e8m0_shuffle

    if not hasattr(custom_kernel, "_setup_done"):
        custom_kernel._setup_done = True

        # Find and extend the existing tuned CSV
        try:
            aiter_root = os.path.dirname(os.path.dirname(aiter.__file__))
            base_csv = os.path.join(aiter_root, "aiter", "configs", "a4w4_blockscale_tuned_gemm.csv")

            custom_csv = "/tmp/a4w4_tuned_v10.csv"
            if not os.path.exists(custom_csv):
                # Read existing tuned CSV
                with open(base_csv) as f:
                    existing_lines = f.readlines()

                # Add custom entries for missing shapes
                # Format: cu_num,M,N,K,kernelId,splitK,us,kernelName,tflops,bw,errRatio
                k32 = "_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E"
                k64 = "_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_64x128E"

                custom_entries = [
                    # M=4, N=2880, K=512 - very small M, use 64x128 (better for small M)
                    f"256,4,2880,512,29,0,10.0,{k64},0,0,0",
                    # M=16, N=2112, K=7168 - small M, large K, splitK=2
                    f"256,16,2112,7168,29,2,10.0,{k64},0,0,0",
                    # M=32, N=4096, K=512 - medium M, small K
                    f"256,32,4096,512,29,0,10.0,{k64},0,0,0",
                    # M=32, N=2880, K=512 - medium M, small K
                    f"256,32,2880,512,29,0,10.0,{k64},0,0,0",
                    # M=256, N=2880, K=512 (test shape)
                    f"256,256,2880,512,29,0,10.0,{k64},0,0,0",
                ]

                # Write merged CSV (custom entries first for priority)
                with open(custom_csv, "w") as f:
                    f.write(existing_lines[0])  # header
                    for entry in custom_entries:
                        f.write(entry + "\n")
                    for line in existing_lines[1:]:
                        f.write(line)

                print(f"[v10] Merged CSV: {len(existing_lines)} existing + {len(custom_entries)} custom entries", file=sys.stderr)

            os.environ["AITER_CONFIG_GEMM_A4W4"] = custom_csv
            print(f"[v10] Set AITER_CONFIG_GEMM_A4W4={custom_csv}", file=sys.stderr)

        except Exception as e:
            print(f"[v10] Setup error: {e}", file=sys.stderr)

    def _quant_mxfp4(x, shuffle=True):
        x_fp4, bs_e8m0 = dynamic_mxfp4_quant(x)
        if shuffle:
            bs_e8m0 = e8m0_shuffle(bs_e8m0)
        return x_fp4.view(dtypes.fp4x2), bs_e8m0.view(dtypes.fp8_e8m0)

    A, B, B_q, B_shuffle, B_scale_sh = data
    A = A.contiguous()
    A_q, A_scale_sh = _quant_mxfp4(A, shuffle=True)

    out = aiter.gemm_a4w4(
        A_q,
        B_shuffle,
        A_scale_sh,
        B_scale_sh,
        dtype=dtypes.bf16,
        bpreshuffle=True,
    )
    return out
scrolls · 103 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON