Skip to content
KernelIndex
Search⌘K

submission 780468

wenyuan1459 · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 111 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-conv2d-v2-780468?include=source"
interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
2D convolutionsuite of 5 cases
NVIDIA H100
96.9ms
#14 of 35
2026-04-29

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:0b021a2c54741562b093daf2953d5d2f7121f798ec5a37514705f67e81e3fc7b
license declaredunknown
license concludedunknown
authorswenyuan1459
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

autotune@triton.autotune(
mmaacc += tl.dot(inp, wt, input_precision="ieee")
num-warps = 8triton.Config({'BLOCK_M': 128, 'BLOCK_N': 128, 'BLOCK_K': 32}, num_stages=3, num_warps=8),
stages = 3triton.Config({'BLOCK_M': 128, 'BLOCK_N': 128, 'BLOCK_K': 32}, num_stages=3, num_warps=8),

Kernel source

submission.py111 lines
#!POPCORN leaderboard conv2d_v2
#!POPCORN gpu H100

import torch
import triton
import triton.language as tl
from task import input_t, output_t


@triton.autotune(
    configs=[
        triton.Config({'BLOCK_M': 128, 'BLOCK_N': 128, 'BLOCK_K': 32}, num_stages=3, num_warps=8),
        triton.Config({'BLOCK_M': 128, 'BLOCK_N': 64, 'BLOCK_K': 32}, num_stages=4, num_warps=4),
        triton.Config({'BLOCK_M': 64, 'BLOCK_N': 128, 'BLOCK_K': 32}, num_stages=4, num_warps=4),
        triton.Config({'BLOCK_M': 64, 'BLOCK_N': 64, 'BLOCK_K': 32}, num_stages=4, num_warps=4),
        triton.Config({'BLOCK_M': 128, 'BLOCK_N': 128, 'BLOCK_K': 64}, num_stages=3, num_warps=8),
        triton.Config({'BLOCK_M': 128, 'BLOCK_N': 64, 'BLOCK_K': 64}, num_stages=3, num_warps=8),
        triton.Config({'BLOCK_M': 64, 'BLOCK_N': 64, 'BLOCK_K': 64}, num_stages=3, num_warps=4),
        triton.Config({'BLOCK_M': 256, 'BLOCK_N': 64, 'BLOCK_K': 32}, num_stages=3, num_warps=8),
        triton.Config({'BLOCK_M': 64, 'BLOCK_N': 256, 'BLOCK_K': 32}, num_stages=3, num_warps=8),
        triton.Config({'BLOCK_M': 256, 'BLOCK_N': 128, 'BLOCK_K': 32}, num_stages=3, num_warps=8),
    ],
    key=['M', 'N', 'K_total'],
)
@triton.jit
def conv2d_implicit_gemm_kernel(
    input_ptr, weight_ptr, output_ptr,
    OH, OW,
    M, N, K_total,
    KK,
    K,
    stride_ib, stride_ic, stride_ih, stride_iw,
    stride_wn, stride_wc, stride_wh, stride_ww,
    stride_ob, stride_oc, stride_oh, stride_ow,
    BLOCK_M: tl.constexpr, BLOCK_N: tl.constexpr, BLOCK_K: tl.constexpr,
):
    pid = tl.program_id(0)
    num_m_blocks = tl.cdiv(M, BLOCK_M)
    num_n_blocks = tl.cdiv(N, BLOCK_N)

    GROUP_M: tl.constexpr = 8
    num_pid_in_group = GROUP_M * num_n_blocks
    group_id = pid // num_pid_in_group
    first_pid_m = group_id * GROUP_M
    group_size_m = min(num_m_blocks - first_pid_m, GROUP_M)
    pid_m = first_pid_m + ((pid % num_pid_in_group) % group_size_m)
    pid_n = (pid % num_pid_in_group) // group_size_m

    m_offs = pid_m * BLOCK_M + tl.arange(0, BLOCK_M)
    n_offs = pid_n * BLOCK_N + tl.arange(0, BLOCK_N)

    OHOW = OH * OW
    b = m_offs // OHOW
    rem = m_offs % OHOW
    oh = rem // OW
    ow = rem % OW

    acc = tl.zeros((BLOCK_M, BLOCK_N), dtype=tl.float32)

    m_mask = m_offs < M
    n_mask = n_offs < N

    base_input = input_ptr + b * stride_ib + oh * stride_ih + ow * stride_iw
    base_weight = weight_ptr + n_offs * stride_wn

    for k_start in range(0, K_total, BLOCK_K):
        k_offs = k_start + tl.arange(0, BLOCK_K)

        ci = k_offs // KK
        k_rem = k_offs % KK
        kh = k_rem // K
        kw = k_rem % K

        input_ptrs = base_input[:, None] + ci[None, :] * stride_ic + kh[None, :] * stride_ih + kw[None, :] * stride_iw
        k_mask = k_offs < K_total
        inp = tl.load(input_ptrs, mask=m_mask[:, None] & k_mask[None, :], other=0.0)

        weight_ptrs = base_weight[None, :] + ci[:, None] * stride_wc + kh[:, None] * stride_wh + kw[:, None] * stride_ww
        wt = tl.load(weight_ptrs, mask=k_mask[:, None] & n_mask[None, :], other=0.0)

        acc += tl.dot(inp, wt, input_precision="ieee")

    output_ptrs = output_ptr + b[:, None] * stride_ob + n_offs[None, :] * stride_oc + oh[:, None] * stride_oh + ow[:, None] * stride_ow
    tl.store(output_ptrs, acc, mask=m_mask[:, None] & n_mask[None, :])


def custom_kernel(data: input_t) -> output_t:
    input_tensor, kernel, output = data
    B, C_in, H, W = input_tensor.shape
    C_out, C_in_k, K, _ = kernel.shape
    OH = H - K + 1
    OW = W - K + 1

    M = B * OH * OW
    N = C_out
    K_total = C_in_k * K * K

    grid = lambda meta: (triton.cdiv(M, meta['BLOCK_M']) * triton.cdiv(N, meta['BLOCK_N']),)

    conv2d_implicit_gemm_kernel[grid](
        input_tensor, kernel, output,
        OH, OW,
        M, N, K_total,
        K * K,
        K,
        input_tensor.stride(0), input_tensor.stride(1), input_tensor.stride(2), input_tensor.stride(3),
        kernel.stride(0), kernel.stride(1), kernel.stride(2), kernel.stride(3),
        output.stride(0), output.stride(1), output.stride(2), output.stride(3),
    )
    return output
scrolls · 111 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON