submission 612191
dannywillowliu-uchi · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 79 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-conv2d-v2-612191?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:381bc4a3a809d6ffbee19a68baf391da669fde9b12cf6dbe266682dc4ca90418
license declaredunknown
license concludedunknown
authorsdannywillowliu-uchi
imported2026-08-15
Kernel source
submission.py79 lines
import os
os.environ['CUBLAS_WORKSPACE_CONFIG'] = ':4096:8'
from task import input_t, output_t
import torch
import torch.nn.functional as F
torch.backends.cudnn.allow_tf32 = False
torch.backends.cuda.matmul.allow_tf32 = False
torch.backends.cudnn.benchmark = True
# Try cudnn_frontend (Python bindings for cuDNN v9 graph API)
_use_cudnn_fe = False
try:
import cudnn
_use_cudnn_fe = True
except ImportError:
pass
if _use_cudnn_fe:
import cudnn
_graph = None
_cache = {}
def _build_graph(B, C, H, W, K, kH, kW):
oH = H - kH + 1
oW = W - kW + 1
graph = cudnn.pygraph(
io_data_type=cudnn.data_type.FLOAT,
intermediate_data_type=cudnn.data_type.FLOAT,
compute_data_type=cudnn.data_type.FLOAT,
)
X = graph.tensor(name="X", dim=[B, C, H, W], stride=[C*H*W, H*W, W, 1])
W_tensor = graph.tensor(name="W", dim=[K, C, kH, kW], stride=[C*kH*kW, kH*kW, kW, 1])
Y = graph.conv_fprop(
image=X,
weight=W_tensor,
pre_padding=[0, 0],
post_padding=[0, 0],
stride=[1, 1],
dilation=[1, 1],
)
Y.set_name("Y")
Y.set_output(True).set_dim([B, K, oH, oW]).set_stride([K*oH*oW, oH*oW, oW, 1])
graph.validate()
graph.build_operation_graph()
graph.create_execution_plans([cudnn.heur_mode.A])
graph.check_support()
graph.build_plans()
return graph, X, W_tensor, Y
def custom_kernel(data: input_t) -> output_t:
global _graph, _cache
input_tensor, kernel, output = data
B, C, H, W = input_tensor.shape
K, _, kH, kW = kernel.shape
key = (B, C, H, W, K, kH, kW)
if key not in _cache:
_cache[key] = _build_graph(B, C, H, W, K, kH, kW)
graph, X, W_tensor, Y = _cache[key]
workspace = torch.empty(graph.get_workspace_size(), device="cuda", dtype=torch.uint8)
graph.execute(
{X: input_tensor, W_tensor: kernel, Y: output},
workspace
)
return output
else:
def custom_kernel(data: input_t) -> output_t:
input_tensor, kernel, output = data
return F.conv2d(input_tensor, kernel, stride=1, padding=0)
scrolls · 79 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 612159.
⋯ 7 unchanged linestorch.backends.cuda.matmul.allow_tf32 = Falsetorch.backends.cudnn.benchmark = True- def custom_kernel(data: input_t) -> output_t:- input_tensor, kernel, output = data- # F.conv2d allocates a new tensor. Can we avoid copy?- # Try: return the F.conv2d result directly (ignore output tensor)- return F.conv2d(input_tensor, kernel, stride=1, padding=0)+ # Try cudnn_frontend (Python bindings for cuDNN v9 graph API)+ _use_cudnn_fe = False+ try:+ import cudnn+ _use_cudnn_fe = True+ except ImportError:+ pass++ if _use_cudnn_fe:+ import cudnn++ _graph = None+ _cache = {}++ def _build_graph(B, C, H, W, K, kH, kW):+ oH = H - kH + 1+ oW = W - kW + 1++ graph = cudnn.pygraph(+ io_data_type=cudnn.data_type.FLOAT,+ intermediate_data_type=cudnn.data_type.FLOAT,+ compute_data_type=cudnn.data_type.FLOAT,+ )++ X = graph.tensor(name="X", dim=[B, C, H, W], stride=[C*H*W, H*W, W, 1])+ W_tensor = graph.tensor(name="W", dim=[K, C, kH, kW], stride=[C*kH*kW, kH*kW, kW, 1])++ Y = graph.conv_fprop(+ image=X,+ weight=W_tensor,+ pre_padding=[0, 0],+ post_padding=[0, 0],+ stride=[1, 1],+ dilation=[1, 1],+ )+ Y.set_name("Y")+ Y.set_output(True).set_dim([B, K, oH, oW]).set_stride([K*oH*oW, oH*oW, oW, 1])++ graph.validate()+ graph.build_operation_graph()+ graph.create_execution_plans([cudnn.heur_mode.A])+ graph.check_support()+ graph.build_plans()++ return graph, X, W_tensor, Y++ def custom_kernel(data: input_t) -> output_t:+ global _graph, _cache+ input_tensor, kernel, output = data+ B, C, H, W = input_tensor.shape+ K, _, kH, kW = kernel.shape++ key = (B, C, H, W, K, kH, kW)+ if key not in _cache:+ _cache[key] = _build_graph(B, C, H, W, K, kH, kW)++ graph, X, W_tensor, Y = _cache[key]++ workspace = torch.empty(graph.get_workspace_size(), device="cuda", dtype=torch.uint8)+ graph.execute(+ {X: input_tensor, W_tensor: kernel, Y: output},+ workspace+ )+ return output+ else:+ def custom_kernel(data: input_t) -> output_t:+ input_tensor, kernel, output = data+ return F.conv2d(input_tensor, kernel, stride=1, padding=0)
scrolls · 77 diff lines total
Best evidence level for this revision: reported
JSON