submission 756979
ethan0027 · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 96 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-matmul-v2-756979?include=source"interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp16
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:0bb016abd313d5d0dd54373ee5a2c65734c8345c73ce0f3e6439d6a6afbc4b51
license declaredunknown
license concludedunknown
authorsethan0027
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
Kernel source
submission.py96 lines
# EVOLVE-BLOCK-START
import torch
import triton
import triton.language as tl
from typing import Tuple
@triton.jit
def matmul_kernel(
a_ptr, b_ptr, c_ptr,
M, N, K,
stride_am, stride_ak,
stride_bk, stride_bn,
stride_cm, stride_cn,
BLOCK_M: tl.constexpr,
BLOCK_N: tl.constexpr,
BLOCK_K: tl.constexpr,
):
pid_m = tl.program_id(0)
pid_n = tl.program_id(1)
# Offsets for the current tile
offs_m = pid_m * BLOCK_M + tl.arange(0, BLOCK_M)
offs_n = pid_n * BLOCK_N + tl.arange(0, BLOCK_N)
mask_m = offs_m < M
mask_n = offs_n < N
# Accumulator in fp32 for numerical stability
acc = tl.zeros((BLOCK_M, BLOCK_N), dtype=tl.float32)
# Iterate over K dimension in tiles
for k in range(0, K, BLOCK_K):
offs_k = k + tl.arange(0, BLOCK_K)
mask_k = offs_k < K
# Load A tile (M x K)
a_ptrs = a_ptr + (offs_m[:, None] * stride_am + offs_k[None, :] * stride_ak)
a = tl.load(a_ptrs, mask=mask_m[:, None] & mask_k[None, :], other=0.0)
# Load B tile (K x N)
b_ptrs = b_ptr + (offs_k[:, None] * stride_bk + offs_n[None, :] * stride_bn)
b = tl.load(b_ptrs, mask=mask_k[:, None] & mask_n[None, :], other=0.0)
# Compute tile multiplication and accumulate
acc = tl.dot(a, b, acc)
# Write the result back to C
c_ptrs = c_ptr + (offs_m[:, None] * stride_cm + offs_n[None, :] * stride_cn)
c = acc.to(tl.float16)
tl.store(c_ptrs, c, mask=mask_m[:, None] & mask_n[None, :])
def custom_kernel(data: Tuple[torch.Tensor, torch.Tensor, torch.Tensor]) -> torch.Tensor:
a, b, c = data
# Ensure tensors are on CUDA and contiguous
if not a.is_cuda:
a = a.cuda()
if not b.is_cuda:
b = b.cuda()
if not c.is_cuda:
c = c.cuda()
a = a.contiguous()
b = b.contiguous()
c = c.contiguous()
M, K = a.shape
K2, N = b.shape
assert K == K2, "Inner dimensions must match"
# Strides in elements (row‑major layout)
stride_am, stride_ak = a.stride()
stride_bk, stride_bn = b.stride()
stride_cm, stride_cn = c.stride()
# Block sizes tuned for Hopper FP16 tensor cores
BLOCK_M = 128
BLOCK_N = 128
BLOCK_K = 32
grid = (triton.cdiv(M, BLOCK_M), triton.cdiv(N, BLOCK_N))
matmul_kernel[grid](
a,
b,
c,
M, N, K,
stride_am, stride_ak,
stride_bk, stride_bn,
stride_cm, stride_cn,
BLOCK_M=BLOCK_M,
BLOCK_N=BLOCK_N,
BLOCK_K=BLOCK_K,
)
return c
# EVOLVE-BLOCK-END
scrolls · 96 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON