Skip to content
KernelIndex
Search⌘K

submission 779854

ajay_a · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 48 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-matmul-v2-779854?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp16

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
FP16 matmulsuite of 8 cases
NVIDIA B200
115.3µs
#19 of 53
2026-04-23

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:a03e7a044ea512a4d36fa5d49e4970b3085d4c541a4957281aff25d0783171cd
license declaredunknown
license concludedunknown
authorsajay_a
imported2026-08-15

Kernel source

submission.py48 lines
#!POPCORN leaderboard matmul_v2
#!POPCORN gpu B200

# at::matmul C++ wrapper. Same winning pattern as conv2d_v2:
# - Dispatch to cuBLAS via ATen (same function the bot's reference uses)
# - Force TF32 off + deterministic to match reference precision exactly
# - cudnn.benchmark=True so the cuBLASlt heuristic can pick the best algo
from task import input_t, output_t
import torch

torch.backends.cudnn.allow_tf32 = False
torch.backends.cudnn.deterministic = True
torch.backends.cudnn.benchmark = True
torch.backends.cuda.matmul.allow_tf32 = False

from torch.utils.cpp_extension import load_inline


_CUDA_SRC = r"""
#include <ATen/ATen.h>
#include <torch/torch.h>

// Dispatches to cuBLAS for dense matmul; write result into preallocated out.
void matmul_fwd(const torch::Tensor& A,
                const torch::Tensor& B,
                torch::Tensor& out) {
    at::matmul_out(out, A, B);
}
"""

_CPP_SRC = "void matmul_fwd(const torch::Tensor&, const torch::Tensor&, torch::Tensor&);"

_mod = load_inline(
    name="matmul_cublas_wrap",
    cpp_sources=_CPP_SRC,
    cuda_sources=_CUDA_SRC,
    functions=["matmul_fwd"],
    extra_cuda_cflags=["-O3", "-arch=sm_100"],
    extra_cflags=["-O3"],
    verbose=False,
)


def custom_kernel(data: input_t) -> output_t:
    A, B, out = data
    _mod.matmul_fwd(A, B, out)
    return out
scrolls · 48 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 779851.

#!POPCORN leaderboard matmul_v2
#!POPCORN gpu B200
- # B200-tuned Triton matmul. Fixed config (BM=256, BN=256, BK=64, num_warps=8, num_stages=3)
- # wins on 4Kx5Kx4K shape vs autotune ceremony. input_precision="tf32x3" gives fp32 accuracy
- # via triple-TF32 emulation — needed because the bot compares against fp32-precision reference.
+ # at::matmul C++ wrapper. Same winning pattern as conv2d_v2:
+ # - Dispatch to cuBLAS via ATen (same function the bot's reference uses)
+ # - Force TF32 off + deterministic to match reference precision exactly
+ # - cudnn.benchmark=True so the cuBLASlt heuristic can pick the best algo
from task import input_t, output_t
- import triton
- import triton.language as tl
+ import torch
+ torch.backends.cudnn.allow_tf32 = False
+ torch.backends.cudnn.deterministic = True
+ torch.backends.cudnn.benchmark = True
+ torch.backends.cuda.matmul.allow_tf32 = False
- @triton.jit
- def matmul_kernel(
- A, B, C, M, N, K,
- sA0, sA1, sB0, sB1, sC0, sC1,
- BM: tl.constexpr, BN: tl.constexpr, BK: tl.constexpr, GM: tl.constexpr,
- ):
- pid = tl.program_id(0)
- num_m = tl.cdiv(M, BM)
- num_n = tl.cdiv(N, BN)
- num_in_g = GM * num_n
- gid = pid // num_in_g
- fm = gid * GM
- gsm = min(num_m - fm, GM)
- pm = fm + ((pid % num_in_g) % gsm)
- pn = (pid % num_in_g) // gsm
+ from torch.utils.cpp_extension import load_inline
- om = pm * BM + tl.arange(0, BM)
- on = pn * BN + tl.arange(0, BN)
- ok = tl.arange(0, BK)
- ap = A + om[:, None] * sA0 + ok[None, :] * sA1
- bp = B + ok[:, None] * sB0 + on[None, :] * sB1
+ _CUDA_SRC = r"""
+ #include <ATen/ATen.h>
+ #include <torch/torch.h>
- acc = tl.zeros((BM, BN), dtype=tl.float32)
- for k in range(0, tl.cdiv(K, BK)):
- a_mask = (ok[None, :] + k * BK) < K
- b_mask = (ok[:, None] + k * BK) < K
- a = tl.load(ap, mask=a_mask, other=0.0)
- b = tl.load(bp, mask=b_mask, other=0.0)
- acc = tl.dot(a, b, acc=acc, input_precision="ieee")
- ap += BK * sA1
- bp += BK * sB0
+ // Dispatches to cuBLAS for dense matmul; write result into preallocated out.
+ void matmul_fwd(const torch::Tensor& A,
+ const torch::Tensor& B,
+ torch::Tensor& out) {
+ at::matmul_out(out, A, B);
+ }
+ """
- cm = (om[:, None] < M) & (on[None, :] < N)
- tl.store(C + om[:, None] * sC0 + on[None, :] * sC1,
- acc.to(C.dtype.element_ty), mask=cm)
+ _CPP_SRC = "void matmul_fwd(const torch::Tensor&, const torch::Tensor&, torch::Tensor&);"
+ _mod = load_inline(
+ name="matmul_cublas_wrap",
+ cpp_sources=_CPP_SRC,
+ cuda_sources=_CUDA_SRC,
+ functions=["matmul_fwd"],
+ extra_cuda_cflags=["-O3", "-arch=sm_100"],
+ extra_cflags=["-O3"],
+ verbose=False,
+ )
+
def custom_kernel(data: input_t) -> output_t:
- A, B, output = data
- M, K = A.shape
- _, N = B.shape
- BM, BN, BK = 128, 128, 64
- GM = 8
- grid = (triton.cdiv(M, BM) * triton.cdiv(N, BN),)
- matmul_kernel[grid](
- A, B, output, M, N, K,
- A.stride(0), A.stride(1),
- B.stride(0), B.stride(1),
- output.stride(0), output.stride(1),
- BM=BM, BN=BN, BK=BK, GM=GM,
- num_warps=8, num_stages=3,
- )
- return output
+ A, B, out = data
+ _mod.matmul_fwd(A, B, out)
+ return out
scrolls · 99 diff lines total

Best evidence level for this revision: reported

JSON