submission 780468
wenyuan1459 · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 111 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-conv2d-v2-780468?include=source"interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:0b021a2c54741562b093daf2953d5d2f7121f798ec5a37514705f67e81e3fc7b
license declaredunknown
license concludedunknown
authorswenyuan1459
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
autotune
@triton.autotune(mma
acc += tl.dot(inp, wt, input_precision="ieee")num-warps = 8
triton.Config({'BLOCK_M': 128, 'BLOCK_N': 128, 'BLOCK_K': 32}, num_stages=3, num_warps=8),stages = 3
triton.Config({'BLOCK_M': 128, 'BLOCK_N': 128, 'BLOCK_K': 32}, num_stages=3, num_warps=8),Kernel source
submission.py111 lines
#!POPCORN leaderboard conv2d_v2
#!POPCORN gpu H100
import torch
import triton
import triton.language as tl
from task import input_t, output_t
@triton.autotune(
configs=[
triton.Config({'BLOCK_M': 128, 'BLOCK_N': 128, 'BLOCK_K': 32}, num_stages=3, num_warps=8),
triton.Config({'BLOCK_M': 128, 'BLOCK_N': 64, 'BLOCK_K': 32}, num_stages=4, num_warps=4),
triton.Config({'BLOCK_M': 64, 'BLOCK_N': 128, 'BLOCK_K': 32}, num_stages=4, num_warps=4),
triton.Config({'BLOCK_M': 64, 'BLOCK_N': 64, 'BLOCK_K': 32}, num_stages=4, num_warps=4),
triton.Config({'BLOCK_M': 128, 'BLOCK_N': 128, 'BLOCK_K': 64}, num_stages=3, num_warps=8),
triton.Config({'BLOCK_M': 128, 'BLOCK_N': 64, 'BLOCK_K': 64}, num_stages=3, num_warps=8),
triton.Config({'BLOCK_M': 64, 'BLOCK_N': 64, 'BLOCK_K': 64}, num_stages=3, num_warps=4),
triton.Config({'BLOCK_M': 256, 'BLOCK_N': 64, 'BLOCK_K': 32}, num_stages=3, num_warps=8),
triton.Config({'BLOCK_M': 64, 'BLOCK_N': 256, 'BLOCK_K': 32}, num_stages=3, num_warps=8),
triton.Config({'BLOCK_M': 256, 'BLOCK_N': 128, 'BLOCK_K': 32}, num_stages=3, num_warps=8),
],
key=['M', 'N', 'K_total'],
)
@triton.jit
def conv2d_implicit_gemm_kernel(
input_ptr, weight_ptr, output_ptr,
OH, OW,
M, N, K_total,
KK,
K,
stride_ib, stride_ic, stride_ih, stride_iw,
stride_wn, stride_wc, stride_wh, stride_ww,
stride_ob, stride_oc, stride_oh, stride_ow,
BLOCK_M: tl.constexpr, BLOCK_N: tl.constexpr, BLOCK_K: tl.constexpr,
):
pid = tl.program_id(0)
num_m_blocks = tl.cdiv(M, BLOCK_M)
num_n_blocks = tl.cdiv(N, BLOCK_N)
GROUP_M: tl.constexpr = 8
num_pid_in_group = GROUP_M * num_n_blocks
group_id = pid // num_pid_in_group
first_pid_m = group_id * GROUP_M
group_size_m = min(num_m_blocks - first_pid_m, GROUP_M)
pid_m = first_pid_m + ((pid % num_pid_in_group) % group_size_m)
pid_n = (pid % num_pid_in_group) // group_size_m
m_offs = pid_m * BLOCK_M + tl.arange(0, BLOCK_M)
n_offs = pid_n * BLOCK_N + tl.arange(0, BLOCK_N)
OHOW = OH * OW
b = m_offs // OHOW
rem = m_offs % OHOW
oh = rem // OW
ow = rem % OW
acc = tl.zeros((BLOCK_M, BLOCK_N), dtype=tl.float32)
m_mask = m_offs < M
n_mask = n_offs < N
base_input = input_ptr + b * stride_ib + oh * stride_ih + ow * stride_iw
base_weight = weight_ptr + n_offs * stride_wn
for k_start in range(0, K_total, BLOCK_K):
k_offs = k_start + tl.arange(0, BLOCK_K)
ci = k_offs // KK
k_rem = k_offs % KK
kh = k_rem // K
kw = k_rem % K
input_ptrs = base_input[:, None] + ci[None, :] * stride_ic + kh[None, :] * stride_ih + kw[None, :] * stride_iw
k_mask = k_offs < K_total
inp = tl.load(input_ptrs, mask=m_mask[:, None] & k_mask[None, :], other=0.0)
weight_ptrs = base_weight[None, :] + ci[:, None] * stride_wc + kh[:, None] * stride_wh + kw[:, None] * stride_ww
wt = tl.load(weight_ptrs, mask=k_mask[:, None] & n_mask[None, :], other=0.0)
acc += tl.dot(inp, wt, input_precision="ieee")
output_ptrs = output_ptr + b[:, None] * stride_ob + n_offs[None, :] * stride_oc + oh[:, None] * stride_oh + ow[:, None] * stride_ow
tl.store(output_ptrs, acc, mask=m_mask[:, None] & n_mask[None, :])
def custom_kernel(data: input_t) -> output_t:
input_tensor, kernel, output = data
B, C_in, H, W = input_tensor.shape
C_out, C_in_k, K, _ = kernel.shape
OH = H - K + 1
OW = W - K + 1
M = B * OH * OW
N = C_out
K_total = C_in_k * K * K
grid = lambda meta: (triton.cdiv(M, meta['BLOCK_M']) * triton.cdiv(N, meta['BLOCK_N']),)
conv2d_implicit_gemm_kernel[grid](
input_tensor, kernel, output,
OH, OW,
M, N, K_total,
K * K,
K,
input_tensor.stride(0), input_tensor.stride(1), input_tensor.stride(2), input_tensor.stride(3),
kernel.stride(0), kernel.stride(1), kernel.stride(2), kernel.stride(3),
output.stride(0), output.stride(1), output.stride(2), output.stride(3),
)
return output
scrolls · 111 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON