submission 513368
JordanNanos · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 37 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-conv2d-v2-513368?include=source"interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:376e44aee306b9e769b421c3ad65c978dccd5c878db65871dde3e38d324983e3
license declaredunknown
license concludedunknown
authorsJordanNanos
imported2026-08-15
Kernel source
submission.py37 lines
import torch
import torch.nn.functional as F
from task import input_t, output_t
def custom_kernel(data: input_t) -> output_t:
"""
Fast 2D convolution using cuDNN with benchmark mode enabled.
Disable TF32 to match reference precision (rtol=1e-3, atol=1e-3).
Enable benchmark mode so cuDNN can select the fastest algorithm.
"""
input_tensor, kernel, output = data
# Save state
old_tf32_cudnn = torch.backends.cudnn.allow_tf32
old_tf32_matmul = torch.backends.cuda.matmul.allow_tf32
old_benchmark = torch.backends.cudnn.benchmark
# Disable TF32 for precision, enable benchmark for speed
torch.backends.cudnn.allow_tf32 = False
torch.backends.cuda.matmul.allow_tf32 = False
torch.backends.cudnn.benchmark = True
try:
result = F.conv2d(
input_tensor,
kernel,
stride=1,
padding=0,
)
output.copy_(result)
finally:
torch.backends.cudnn.allow_tf32 = old_tf32_cudnn
torch.backends.cuda.matmul.allow_tf32 = old_tf32_matmul
torch.backends.cudnn.benchmark = old_benchmark
return outputscrolls · 37 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 512856.
⋯ 3 unchanged linesdef custom_kernel(data: input_t) -> output_t:+ """+ Fast 2D convolution using cuDNN with benchmark mode enabled.+ Disable TF32 to match reference precision (rtol=1e-3, atol=1e-3).+ Enable benchmark mode so cuDNN can select the fastest algorithm.+ """input_tensor, kernel, output = data- batch, channels, height, width = input_tensor.shape- out_channels, in_channels, kh, kw = kernel.shape+ # Save state+ old_tf32_cudnn = torch.backends.cudnn.allow_tf32+ old_tf32_matmul = torch.backends.cuda.matmul.allow_tf32+ old_benchmark = torch.backends.cudnn.benchmark- out_h = height - kh + 1- out_w = width - kw + 1+ # Disable TF32 for precision, enable benchmark for speed+ torch.backends.cudnn.allow_tf32 = False+ torch.backends.cuda.matmul.allow_tf32 = False+ torch.backends.cudnn.benchmark = True- # im2col: (batch, in_channels*kh*kw, out_h*out_w)- input_unfolded = F.unfold(input_tensor, kernel_size=(kh, kw), stride=1, padding=0)+ try:+ result = F.conv2d(+ input_tensor,+ kernel,+ stride=1,+ padding=0,+ )+ output.copy_(result)+ finally:+ torch.backends.cudnn.allow_tf32 = old_tf32_cudnn+ torch.backends.cuda.matmul.allow_tf32 = old_tf32_matmul+ torch.backends.cudnn.benchmark = old_benchmark- # kernel: (out_channels, in_channels*kh*kw)- kernel_matrix = kernel.view(out_channels, -1)-- # matmul broadcast: (out_channels, K) x (batch, K, out_h*out_w) -> (batch, out_channels, out_h*out_w)- result = torch.matmul(kernel_matrix, input_unfolded)-- output.copy_(result.view(batch, out_channels, out_h, out_w))return outputNo newline at end of file
scrolls · 49 diff lines total
Best evidence level for this revision: reported
JSON