Skip to content
KernelIndex
Search⌘K

submission 529480

josusanmartin · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 154 lines, June 9 Researcher Reciprocity License v1.0.

submission_20260311_auto_cached_quant_h8_2112_7168_h8_32x128_s3_6a1a37d4.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-mxfp4-mm-529480?include=source"
interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, mxfp4

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
AMD MXFP4 GEMMsuite of 6 cases
AMD Instinct MI355X
13.2µs
#413 of 1143
2026-03-11

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:3f2be44b11d7f3c5e2bc63da7c49474203c2b77de567f07cd12d22f2add88dda
license declaredunknown
license concludedunknown
authorsjosusanmartin
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

fp4Autogenerated MXFP4 candidate: cached_quant_h8_2112_7168_h8_32x128_s3
split-k{'cu_num': '256', 'M': '4', 'N': '2880', 'K': '512', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '2'},

Kernel source

submission_20260311_auto_cached_quant_h8_2112_7168_h8_32x128_s3_6a1a37d4.py154 lines
#!POPCORN leaderboard amd-mxfp4-mm
#!POPCORN gpu MI355X

"""
Autogenerated MXFP4 candidate: cached_quant_h8_2112_7168_h8_32x128_s3
Strategy: cached_quant
Generated by autosubmit_amd_mxfp4_mm.py.
"""
from __future__ import annotations

import csv
import os
from pathlib import Path

import torch

_AITER_BASE = Path("/home/runner/aiter")
_CUSTOM_CONFIG_PATH = "/tmp/submission_20260311_auto_cached_quant_h8_2112_7168_h8_32x128_s3_6a1a37d4.csv"
_CUSTOM_ROWS = (
    {'cu_num': '256', 'M': '4', 'N': '2880', 'K': '512', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '2'},
    {'cu_num': '256', 'M': '16', 'N': '2112', 'K': '7168', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '3'},
    {'cu_num': '256', 'M': '32', 'N': '2880', 'K': '512', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '2'},
    {'cu_num': '256', 'M': '32', 'N': '4096', 'K': '512', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '2'},
    {'cu_num': '256', 'M': '8', 'N': '2112', 'K': '7168', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '3'},
)
_CUSTOM_KEYS = {(row["M"], row["N"], row["K"]) for row in _CUSTOM_ROWS}


def _looks_like_config(path: Path) -> bool:
    try:
        with path.open(newline="") as f:
            reader = csv.DictReader(f)
            return reader.fieldnames is not None and "kernelName" in reader.fieldnames and "splitK" in reader.fieldnames
    except Exception:
        return False


def _find_default_config() -> Path | None:
    candidates = (
        _AITER_BASE / "aiter" / "configs" / "a4w4_blockscale_tuned_gemm.csv",
        _AITER_BASE / "hsa" / "configs" / "a4w4_tuned_gemm.csv",
        _AITER_BASE / "configs" / "a4w4_tuned_gemm.csv",
    )
    for path in candidates:
        if path.exists() and _looks_like_config(path):
            return path
    for path in sorted(_AITER_BASE.rglob("*a4w4*.csv")):
        if _looks_like_config(path):
            return path
    return None


def _write_merged_config():
    base_path = _find_default_config()
    fieldnames = ["cu_num", "M", "N", "K", "kernelName", "splitK"]
    rows = []

    if base_path is not None:
        with base_path.open(newline="") as f:
            reader = csv.DictReader(f)
            if reader.fieldnames:
                fieldnames = list(reader.fieldnames)
            for row in reader:
                if (row.get("M", ""), row.get("N", ""), row.get("K", "")) in _CUSTOM_KEYS:
                    continue
                rows.append(row)

    rows.extend(_CUSTOM_ROWS)
    with open(_CUSTOM_CONFIG_PATH, "w", newline="") as f:
        writer = csv.DictWriter(f, fieldnames=fieldnames)
        writer.writeheader()
        writer.writerows(rows)


_write_merged_config()
os.environ["AITER_CONFIG_GEMM_A4W4"] = _CUSTOM_CONFIG_PATH

import aiter
from aiter import QuantType, dtypes
from aiter.utility import fp4_utils

from task import input_t, output_t

_DTYPES = dtypes
_FP4_UTILS = fp4_utils
_TRITON_QUANT = aiter.get_triton_quant(QuantType.per_1x32)
_GEMM = aiter.gemm_a4w4
_A_QUANT_BUFFERS = {}
_OUT_CACHE = {}

_QUANT_BLOCK_SIZE_BY_MK = {
    (4, 512): 32,
    (32, 512): 32,
}


def _ceil_div(x: int, y: int) -> int:
    return (x + y - 1) // y


def _get_quant_buffers(m: int, k: int, device):
    scale_n = _ceil_div(k, 32)
    scale_n_pad = _ceil_div(scale_n, 8) * 8
    scale_m_pad = _ceil_div(m, 256) * 256
    key = (device.type, device.index, m, k, scale_m_pad, scale_n_pad)
    buffers = _A_QUANT_BUFFERS.get(key)
    if buffers is None:
        a_q = torch.empty((m, k // 2), dtype=torch.uint8, device=device)
        a_scale = torch.empty((scale_m_pad, scale_n_pad), dtype=torch.uint8, device=device)
        buffers = (a_q, a_scale, scale_n, scale_m_pad, scale_n_pad)
        _A_QUANT_BUFFERS[key] = buffers
    return buffers


def _quantize_a(a):
    m = int(a.shape[0])
    k = int(a.shape[1])
    try:
        a_q, a_scale, scale_n, scale_m_pad, scale_n_pad = _get_quant_buffers(m, k, a.device)
        block_size = _QUANT_BLOCK_SIZE_BY_MK.get((m, k), 128)
        grid = (_ceil_div(m, block_size), scale_n_pad)
        _FP4_UTILS._dynamic_mxfp4_quant_kernel_asm_layout[grid](
            a,
            a_q,
            a_scale,
            *a.stride(),
            *a_q.stride(),
            *a_scale.stride(),
            M=m,
            N=k,
            scaleN=scale_n,
            scaleM_pad=scale_m_pad,
            scaleN_pad=scale_n_pad,
            BLOCK_SIZE=block_size,
            MXFP4_QUANT_BLOCK_SIZE=32,
            SCALING_MODE=0,
            SHUFFLE=True,
        )
        return a_q.view(_DTYPES.fp4x2), a_scale.view(_DTYPES.fp8_e8m0)
    except Exception:
        return _TRITON_QUANT(a, shuffle=True)

def custom_kernel(data: input_t) -> output_t:
    a, _b, _b_q, b_shuffle, b_scale_sh = data
    a_q, a_scale_sh = _quantize_a(a)
    return _GEMM(
        a_q,
        b_shuffle,
        a_scale_sh,
        b_scale_sh,
        dtype=dtypes.bf16,
        bpreshuffle=True,
    )
scrolls · 154 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 528645.

⋯ 1 unchanged lines
#!POPCORN gpu MI355X
"""
- Version 38: Merge config with ALL optimal splitK values.
- Reads original aiter CSV, merges custom splitK for M=4, M=16, M=32 shapes.
- Preserves original tuned configs for M=64, M=256 etc.
+ Autogenerated MXFP4 candidate: cached_quant_h8_2112_7168_h8_32x128_s3
+ Strategy: cached_quant
+ Generated by autosubmit_amd_mxfp4_mm.py.
"""
+ from __future__ import annotations
+
+ import csv
import os
- import glob
- import pandas as pd
+ from pathlib import Path
- _CUSTOM_CONFIG_PATH = "/tmp/custom_a4w4_all_config.csv"
+ import torch
- _aiter_base = "/home/runner/aiter"
- _possible_paths = [
- f"{_aiter_base}/aiter/configs/a4w4_blockscale_tuned_gemm.csv",
- f"{_aiter_base}/hsa/configs/a4w4_tuned_gemm.csv",
- f"{_aiter_base}/configs/a4w4_tuned_gemm.csv",
- ]
+ _AITER_BASE = Path("/home/runner/aiter")
+ _CUSTOM_CONFIG_PATH = "/tmp/submission_20260311_auto_cached_quant_h8_2112_7168_h8_32x128_s3_6a1a37d4.csv"
+ _CUSTOM_ROWS = (
+ {'cu_num': '256', 'M': '4', 'N': '2880', 'K': '512', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '2'},
+ {'cu_num': '256', 'M': '16', 'N': '2112', 'K': '7168', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '3'},
+ {'cu_num': '256', 'M': '32', 'N': '2880', 'K': '512', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '2'},
+ {'cu_num': '256', 'M': '32', 'N': '4096', 'K': '512', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '2'},
+ {'cu_num': '256', 'M': '8', 'N': '2112', 'K': '7168', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '3'},
+ )
+ _CUSTOM_KEYS = {(row["M"], row["N"], row["K"]) for row in _CUSTOM_ROWS}
- for pattern in [f"{_aiter_base}/**/a4w4*tuned*.csv", f"{_aiter_base}/**/*a4w4*.csv"]:
- _possible_paths.extend(glob.glob(pattern, recursive=True))
- _original_df = None
- for path in _possible_paths:
- if os.path.exists(path):
- try:
- df = pd.read_csv(path)
- if 'kernelName' in df.columns and 'splitK' in df.columns:
- _original_df = df
- break
- except Exception:
- continue
+ def _looks_like_config(path: Path) -> bool:
+ try:
+ with path.open(newline="") as f:
+ reader = csv.DictReader(f)
+ return reader.fieldnames is not None and "kernelName" in reader.fieldnames and "splitK" in reader.fieldnames
+ except Exception:
+ return False
- cu_num = 256
- _kernel_32x128 = "_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E"
- custom_rows = [
- {"cu_num": cu_num, "M": 4, "N": 2880, "K": 512,
- "kernelName": _kernel_32x128, "splitK": 2},
- {"cu_num": cu_num, "M": 16, "N": 2112, "K": 7168,
- "kernelName": _kernel_32x128, "splitK": 3},
- {"cu_num": cu_num, "M": 32, "N": 2880, "K": 512,
- "kernelName": _kernel_32x128, "splitK": 2},
- {"cu_num": cu_num, "M": 32, "N": 4096, "K": 512,
- "kernelName": _kernel_32x128, "splitK": 2},
- ]
+ def _find_default_config() -> Path | None:
+ candidates = (
+ _AITER_BASE / "aiter" / "configs" / "a4w4_blockscale_tuned_gemm.csv",
+ _AITER_BASE / "hsa" / "configs" / "a4w4_tuned_gemm.csv",
+ _AITER_BASE / "configs" / "a4w4_tuned_gemm.csv",
+ )
+ for path in candidates:
+ if path.exists() and _looks_like_config(path):
+ return path
+ for path in sorted(_AITER_BASE.rglob("*a4w4*.csv")):
+ if _looks_like_config(path):
+ return path
+ return None
- custom_df = pd.DataFrame(custom_rows)
- if _original_df is not None:
- for _, row in custom_df.iterrows():
- mask = (_original_df['M'] == row['M']) & \
- (_original_df['N'] == row['N']) & \
- (_original_df['K'] == row['K'])
- _original_df = _original_df[~mask]
- combined_df = pd.concat([_original_df, custom_df], ignore_index=True)
- else:
- combined_df = custom_df
+ def _write_merged_config():
+ base_path = _find_default_config()
+ fieldnames = ["cu_num", "M", "N", "K", "kernelName", "splitK"]
+ rows = []
- combined_df.to_csv(_CUSTOM_CONFIG_PATH, index=False)
+ if base_path is not None:
+ with base_path.open(newline="") as f:
+ reader = csv.DictReader(f)
+ if reader.fieldnames:
+ fieldnames = list(reader.fieldnames)
+ for row in reader:
+ if (row.get("M", ""), row.get("N", ""), row.get("K", "")) in _CUSTOM_KEYS:
+ continue
+ rows.append(row)
+
+ rows.extend(_CUSTOM_ROWS)
+ with open(_CUSTOM_CONFIG_PATH, "w", newline="") as f:
+ writer = csv.DictWriter(f, fieldnames=fieldnames)
+ writer.writeheader()
+ writer.writerows(rows)
+
+
+ _write_merged_config()
os.environ["AITER_CONFIG_GEMM_A4W4"] = _CUSTOM_CONFIG_PATH
- import torch
import aiter
from aiter import QuantType, dtypes
+ from aiter.utility import fp4_utils
+
from task import input_t, output_t
- _quant_func = aiter.get_triton_quant(QuantType.per_1x32)
- _bf16 = dtypes.bf16
+ _DTYPES = dtypes
+ _FP4_UTILS = fp4_utils
+ _TRITON_QUANT = aiter.get_triton_quant(QuantType.per_1x32)
+ _GEMM = aiter.gemm_a4w4
+ _A_QUANT_BUFFERS = {}
+ _OUT_CACHE = {}
+ _QUANT_BLOCK_SIZE_BY_MK = {
+ (4, 512): 32,
+ (32, 512): 32,
+ }
+
+ def _ceil_div(x: int, y: int) -> int:
+ return (x + y - 1) // y
+
+
+ def _get_quant_buffers(m: int, k: int, device):
+ scale_n = _ceil_div(k, 32)
+ scale_n_pad = _ceil_div(scale_n, 8) * 8
+ scale_m_pad = _ceil_div(m, 256) * 256
+ key = (device.type, device.index, m, k, scale_m_pad, scale_n_pad)
+ buffers = _A_QUANT_BUFFERS.get(key)
+ if buffers is None:
+ a_q = torch.empty((m, k // 2), dtype=torch.uint8, device=device)
+ a_scale = torch.empty((scale_m_pad, scale_n_pad), dtype=torch.uint8, device=device)
+ buffers = (a_q, a_scale, scale_n, scale_m_pad, scale_n_pad)
+ _A_QUANT_BUFFERS[key] = buffers
+ return buffers
+
+
+ def _quantize_a(a):
+ m = int(a.shape[0])
+ k = int(a.shape[1])
+ try:
+ a_q, a_scale, scale_n, scale_m_pad, scale_n_pad = _get_quant_buffers(m, k, a.device)
+ block_size = _QUANT_BLOCK_SIZE_BY_MK.get((m, k), 128)
+ grid = (_ceil_div(m, block_size), scale_n_pad)
+ _FP4_UTILS._dynamic_mxfp4_quant_kernel_asm_layout[grid](
+ a,
+ a_q,
+ a_scale,
+ *a.stride(),
+ *a_q.stride(),
+ *a_scale.stride(),
+ M=m,
+ N=k,
+ scaleN=scale_n,
+ scaleM_pad=scale_m_pad,
+ scaleN_pad=scale_n_pad,
+ BLOCK_SIZE=block_size,
+ MXFP4_QUANT_BLOCK_SIZE=32,
+ SCALING_MODE=0,
+ SHUFFLE=True,
+ )
+ return a_q.view(_DTYPES.fp4x2), a_scale.view(_DTYPES.fp8_e8m0)
+ except Exception:
+ return _TRITON_QUANT(a, shuffle=True)
+
def custom_kernel(data: input_t) -> output_t:
- A, B, B_q, B_shuffle, B_scale_sh = data
- A_q, A_scale_sh = _quant_func(A, shuffle=True)
- return aiter.gemm_a4w4(
- A_q, B_shuffle, A_scale_sh, B_scale_sh,
- dtype=_bf16, bpreshuffle=True,
+ a, _b, _b_q, b_shuffle, b_scale_sh = data
+ a_q, a_scale_sh = _quantize_a(a)
+ return _GEMM(
+ a_q,
+ b_shuffle,
+ a_scale_sh,
+ b_scale_sh,
+ dtype=dtypes.bf16,
+ bpreshuffle=True,
)
scrolls · 208 diff lines total

Best evidence level for this revision: reported

JSON