submission 529480
josusanmartin · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 154 lines, June 9 Researcher Reciprocity License v1.0.
submission_20260311_auto_cached_quant_h8_2112_7168_h8_32x128_s3_6a1a37d4.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-mxfp4-mm-529480?include=source"interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, mxfp4
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:3f2be44b11d7f3c5e2bc63da7c49474203c2b77de567f07cd12d22f2add88dda
license declaredunknown
license concludedunknown
authorsjosusanmartin
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
Kernel source
submission_20260311_auto_cached_quant_h8_2112_7168_h8_32x128_s3_6a1a37d4.py154 lines
#!POPCORN leaderboard amd-mxfp4-mm
#!POPCORN gpu MI355X
"""
Autogenerated MXFP4 candidate: cached_quant_h8_2112_7168_h8_32x128_s3
Strategy: cached_quant
Generated by autosubmit_amd_mxfp4_mm.py.
"""
from __future__ import annotations
import csv
import os
from pathlib import Path
import torch
_AITER_BASE = Path("/home/runner/aiter")
_CUSTOM_CONFIG_PATH = "/tmp/submission_20260311_auto_cached_quant_h8_2112_7168_h8_32x128_s3_6a1a37d4.csv"
_CUSTOM_ROWS = (
{'cu_num': '256', 'M': '4', 'N': '2880', 'K': '512', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '2'},
{'cu_num': '256', 'M': '16', 'N': '2112', 'K': '7168', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '3'},
{'cu_num': '256', 'M': '32', 'N': '2880', 'K': '512', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '2'},
{'cu_num': '256', 'M': '32', 'N': '4096', 'K': '512', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '2'},
{'cu_num': '256', 'M': '8', 'N': '2112', 'K': '7168', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '3'},
)
_CUSTOM_KEYS = {(row["M"], row["N"], row["K"]) for row in _CUSTOM_ROWS}
def _looks_like_config(path: Path) -> bool:
try:
with path.open(newline="") as f:
reader = csv.DictReader(f)
return reader.fieldnames is not None and "kernelName" in reader.fieldnames and "splitK" in reader.fieldnames
except Exception:
return False
def _find_default_config() -> Path | None:
candidates = (
_AITER_BASE / "aiter" / "configs" / "a4w4_blockscale_tuned_gemm.csv",
_AITER_BASE / "hsa" / "configs" / "a4w4_tuned_gemm.csv",
_AITER_BASE / "configs" / "a4w4_tuned_gemm.csv",
)
for path in candidates:
if path.exists() and _looks_like_config(path):
return path
for path in sorted(_AITER_BASE.rglob("*a4w4*.csv")):
if _looks_like_config(path):
return path
return None
def _write_merged_config():
base_path = _find_default_config()
fieldnames = ["cu_num", "M", "N", "K", "kernelName", "splitK"]
rows = []
if base_path is not None:
with base_path.open(newline="") as f:
reader = csv.DictReader(f)
if reader.fieldnames:
fieldnames = list(reader.fieldnames)
for row in reader:
if (row.get("M", ""), row.get("N", ""), row.get("K", "")) in _CUSTOM_KEYS:
continue
rows.append(row)
rows.extend(_CUSTOM_ROWS)
with open(_CUSTOM_CONFIG_PATH, "w", newline="") as f:
writer = csv.DictWriter(f, fieldnames=fieldnames)
writer.writeheader()
writer.writerows(rows)
_write_merged_config()
os.environ["AITER_CONFIG_GEMM_A4W4"] = _CUSTOM_CONFIG_PATH
import aiter
from aiter import QuantType, dtypes
from aiter.utility import fp4_utils
from task import input_t, output_t
_DTYPES = dtypes
_FP4_UTILS = fp4_utils
_TRITON_QUANT = aiter.get_triton_quant(QuantType.per_1x32)
_GEMM = aiter.gemm_a4w4
_A_QUANT_BUFFERS = {}
_OUT_CACHE = {}
_QUANT_BLOCK_SIZE_BY_MK = {
(4, 512): 32,
(32, 512): 32,
}
def _ceil_div(x: int, y: int) -> int:
return (x + y - 1) // y
def _get_quant_buffers(m: int, k: int, device):
scale_n = _ceil_div(k, 32)
scale_n_pad = _ceil_div(scale_n, 8) * 8
scale_m_pad = _ceil_div(m, 256) * 256
key = (device.type, device.index, m, k, scale_m_pad, scale_n_pad)
buffers = _A_QUANT_BUFFERS.get(key)
if buffers is None:
a_q = torch.empty((m, k // 2), dtype=torch.uint8, device=device)
a_scale = torch.empty((scale_m_pad, scale_n_pad), dtype=torch.uint8, device=device)
buffers = (a_q, a_scale, scale_n, scale_m_pad, scale_n_pad)
_A_QUANT_BUFFERS[key] = buffers
return buffers
def _quantize_a(a):
m = int(a.shape[0])
k = int(a.shape[1])
try:
a_q, a_scale, scale_n, scale_m_pad, scale_n_pad = _get_quant_buffers(m, k, a.device)
block_size = _QUANT_BLOCK_SIZE_BY_MK.get((m, k), 128)
grid = (_ceil_div(m, block_size), scale_n_pad)
_FP4_UTILS._dynamic_mxfp4_quant_kernel_asm_layout[grid](
a,
a_q,
a_scale,
*a.stride(),
*a_q.stride(),
*a_scale.stride(),
M=m,
N=k,
scaleN=scale_n,
scaleM_pad=scale_m_pad,
scaleN_pad=scale_n_pad,
BLOCK_SIZE=block_size,
MXFP4_QUANT_BLOCK_SIZE=32,
SCALING_MODE=0,
SHUFFLE=True,
)
return a_q.view(_DTYPES.fp4x2), a_scale.view(_DTYPES.fp8_e8m0)
except Exception:
return _TRITON_QUANT(a, shuffle=True)
def custom_kernel(data: input_t) -> output_t:
a, _b, _b_q, b_shuffle, b_scale_sh = data
a_q, a_scale_sh = _quantize_a(a)
return _GEMM(
a_q,
b_shuffle,
a_scale_sh,
b_scale_sh,
dtype=dtypes.bf16,
bpreshuffle=True,
)
scrolls · 154 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 528645.
⋯ 1 unchanged lines#!POPCORN gpu MI355X"""- Version 38: Merge config with ALL optimal splitK values.- Reads original aiter CSV, merges custom splitK for M=4, M=16, M=32 shapes.- Preserves original tuned configs for M=64, M=256 etc.+ Autogenerated MXFP4 candidate: cached_quant_h8_2112_7168_h8_32x128_s3+ Strategy: cached_quant+ Generated by autosubmit_amd_mxfp4_mm.py."""+ from __future__ import annotations++ import csvimport os- import glob- import pandas as pd+ from pathlib import Path- _CUSTOM_CONFIG_PATH = "/tmp/custom_a4w4_all_config.csv"+ import torch- _aiter_base = "/home/runner/aiter"- _possible_paths = [- f"{_aiter_base}/aiter/configs/a4w4_blockscale_tuned_gemm.csv",- f"{_aiter_base}/hsa/configs/a4w4_tuned_gemm.csv",- f"{_aiter_base}/configs/a4w4_tuned_gemm.csv",- ]+ _AITER_BASE = Path("/home/runner/aiter")+ _CUSTOM_CONFIG_PATH = "/tmp/submission_20260311_auto_cached_quant_h8_2112_7168_h8_32x128_s3_6a1a37d4.csv"+ _CUSTOM_ROWS = (+ {'cu_num': '256', 'M': '4', 'N': '2880', 'K': '512', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '2'},+ {'cu_num': '256', 'M': '16', 'N': '2112', 'K': '7168', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '3'},+ {'cu_num': '256', 'M': '32', 'N': '2880', 'K': '512', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '2'},+ {'cu_num': '256', 'M': '32', 'N': '4096', 'K': '512', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '2'},+ {'cu_num': '256', 'M': '8', 'N': '2112', 'K': '7168', 'kernelName': '_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E', 'splitK': '3'},+ )+ _CUSTOM_KEYS = {(row["M"], row["N"], row["K"]) for row in _CUSTOM_ROWS}- for pattern in [f"{_aiter_base}/**/a4w4*tuned*.csv", f"{_aiter_base}/**/*a4w4*.csv"]:- _possible_paths.extend(glob.glob(pattern, recursive=True))- _original_df = None- for path in _possible_paths:- if os.path.exists(path):- try:- df = pd.read_csv(path)- if 'kernelName' in df.columns and 'splitK' in df.columns:- _original_df = df- break- except Exception:- continue+ def _looks_like_config(path: Path) -> bool:+ try:+ with path.open(newline="") as f:+ reader = csv.DictReader(f)+ return reader.fieldnames is not None and "kernelName" in reader.fieldnames and "splitK" in reader.fieldnames+ except Exception:+ return False- cu_num = 256- _kernel_32x128 = "_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E"- custom_rows = [- {"cu_num": cu_num, "M": 4, "N": 2880, "K": 512,- "kernelName": _kernel_32x128, "splitK": 2},- {"cu_num": cu_num, "M": 16, "N": 2112, "K": 7168,- "kernelName": _kernel_32x128, "splitK": 3},- {"cu_num": cu_num, "M": 32, "N": 2880, "K": 512,- "kernelName": _kernel_32x128, "splitK": 2},- {"cu_num": cu_num, "M": 32, "N": 4096, "K": 512,- "kernelName": _kernel_32x128, "splitK": 2},- ]+ def _find_default_config() -> Path | None:+ candidates = (+ _AITER_BASE / "aiter" / "configs" / "a4w4_blockscale_tuned_gemm.csv",+ _AITER_BASE / "hsa" / "configs" / "a4w4_tuned_gemm.csv",+ _AITER_BASE / "configs" / "a4w4_tuned_gemm.csv",+ )+ for path in candidates:+ if path.exists() and _looks_like_config(path):+ return path+ for path in sorted(_AITER_BASE.rglob("*a4w4*.csv")):+ if _looks_like_config(path):+ return path+ return None- custom_df = pd.DataFrame(custom_rows)- if _original_df is not None:- for _, row in custom_df.iterrows():- mask = (_original_df['M'] == row['M']) & \- (_original_df['N'] == row['N']) & \- (_original_df['K'] == row['K'])- _original_df = _original_df[~mask]- combined_df = pd.concat([_original_df, custom_df], ignore_index=True)- else:- combined_df = custom_df+ def _write_merged_config():+ base_path = _find_default_config()+ fieldnames = ["cu_num", "M", "N", "K", "kernelName", "splitK"]+ rows = []- combined_df.to_csv(_CUSTOM_CONFIG_PATH, index=False)+ if base_path is not None:+ with base_path.open(newline="") as f:+ reader = csv.DictReader(f)+ if reader.fieldnames:+ fieldnames = list(reader.fieldnames)+ for row in reader:+ if (row.get("M", ""), row.get("N", ""), row.get("K", "")) in _CUSTOM_KEYS:+ continue+ rows.append(row)++ rows.extend(_CUSTOM_ROWS)+ with open(_CUSTOM_CONFIG_PATH, "w", newline="") as f:+ writer = csv.DictWriter(f, fieldnames=fieldnames)+ writer.writeheader()+ writer.writerows(rows)+++ _write_merged_config()os.environ["AITER_CONFIG_GEMM_A4W4"] = _CUSTOM_CONFIG_PATH- import torchimport aiterfrom aiter import QuantType, dtypes+ from aiter.utility import fp4_utils+from task import input_t, output_t- _quant_func = aiter.get_triton_quant(QuantType.per_1x32)- _bf16 = dtypes.bf16+ _DTYPES = dtypes+ _FP4_UTILS = fp4_utils+ _TRITON_QUANT = aiter.get_triton_quant(QuantType.per_1x32)+ _GEMM = aiter.gemm_a4w4+ _A_QUANT_BUFFERS = {}+ _OUT_CACHE = {}+ _QUANT_BLOCK_SIZE_BY_MK = {+ (4, 512): 32,+ (32, 512): 32,+ }++ def _ceil_div(x: int, y: int) -> int:+ return (x + y - 1) // y+++ def _get_quant_buffers(m: int, k: int, device):+ scale_n = _ceil_div(k, 32)+ scale_n_pad = _ceil_div(scale_n, 8) * 8+ scale_m_pad = _ceil_div(m, 256) * 256+ key = (device.type, device.index, m, k, scale_m_pad, scale_n_pad)+ buffers = _A_QUANT_BUFFERS.get(key)+ if buffers is None:+ a_q = torch.empty((m, k // 2), dtype=torch.uint8, device=device)+ a_scale = torch.empty((scale_m_pad, scale_n_pad), dtype=torch.uint8, device=device)+ buffers = (a_q, a_scale, scale_n, scale_m_pad, scale_n_pad)+ _A_QUANT_BUFFERS[key] = buffers+ return buffers+++ def _quantize_a(a):+ m = int(a.shape[0])+ k = int(a.shape[1])+ try:+ a_q, a_scale, scale_n, scale_m_pad, scale_n_pad = _get_quant_buffers(m, k, a.device)+ block_size = _QUANT_BLOCK_SIZE_BY_MK.get((m, k), 128)+ grid = (_ceil_div(m, block_size), scale_n_pad)+ _FP4_UTILS._dynamic_mxfp4_quant_kernel_asm_layout[grid](+ a,+ a_q,+ a_scale,+ *a.stride(),+ *a_q.stride(),+ *a_scale.stride(),+ M=m,+ N=k,+ scaleN=scale_n,+ scaleM_pad=scale_m_pad,+ scaleN_pad=scale_n_pad,+ BLOCK_SIZE=block_size,+ MXFP4_QUANT_BLOCK_SIZE=32,+ SCALING_MODE=0,+ SHUFFLE=True,+ )+ return a_q.view(_DTYPES.fp4x2), a_scale.view(_DTYPES.fp8_e8m0)+ except Exception:+ return _TRITON_QUANT(a, shuffle=True)+def custom_kernel(data: input_t) -> output_t:- A, B, B_q, B_shuffle, B_scale_sh = data- A_q, A_scale_sh = _quant_func(A, shuffle=True)- return aiter.gemm_a4w4(- A_q, B_shuffle, A_scale_sh, B_scale_sh,- dtype=_bf16, bpreshuffle=True,+ a, _b, _b_q, b_shuffle, b_scale_sh = data+ a_q, a_scale_sh = _quantize_a(a)+ return _GEMM(+ a_q,+ b_shuffle,+ a_scale_sh,+ b_scale_sh,+ dtype=dtypes.bf16,+ bpreshuffle=True,)
scrolls · 208 diff lines total
Best evidence level for this revision: reported
JSON