Skip to content
KernelIndex
Search⌘K

submission 612191

dannywillowliu-uchi · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 79 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-conv2d-v2-612191?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
2D convolutionsuite of 5 cases
NVIDIA B200
6.48ms
#3 of 28
2026-03-22

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:381bc4a3a809d6ffbee19a68baf391da669fde9b12cf6dbe266682dc4ca90418
license declaredunknown
license concludedunknown
authorsdannywillowliu-uchi
imported2026-08-15

Kernel source

submission.py79 lines
import os
os.environ['CUBLAS_WORKSPACE_CONFIG'] = ':4096:8'
from task import input_t, output_t
import torch
import torch.nn.functional as F

torch.backends.cudnn.allow_tf32 = False
torch.backends.cuda.matmul.allow_tf32 = False
torch.backends.cudnn.benchmark = True

# Try cudnn_frontend (Python bindings for cuDNN v9 graph API)
_use_cudnn_fe = False
try:
    import cudnn
    _use_cudnn_fe = True
except ImportError:
    pass

if _use_cudnn_fe:
    import cudnn
    
    _graph = None
    _cache = {}
    
    def _build_graph(B, C, H, W, K, kH, kW):
        oH = H - kH + 1
        oW = W - kW + 1
        
        graph = cudnn.pygraph(
            io_data_type=cudnn.data_type.FLOAT,
            intermediate_data_type=cudnn.data_type.FLOAT,
            compute_data_type=cudnn.data_type.FLOAT,
        )
        
        X = graph.tensor(name="X", dim=[B, C, H, W], stride=[C*H*W, H*W, W, 1])
        W_tensor = graph.tensor(name="W", dim=[K, C, kH, kW], stride=[C*kH*kW, kH*kW, kW, 1])
        
        Y = graph.conv_fprop(
            image=X,
            weight=W_tensor,
            pre_padding=[0, 0],
            post_padding=[0, 0],
            stride=[1, 1],
            dilation=[1, 1],
        )
        Y.set_name("Y")
        Y.set_output(True).set_dim([B, K, oH, oW]).set_stride([K*oH*oW, oH*oW, oW, 1])
        
        graph.validate()
        graph.build_operation_graph()
        graph.create_execution_plans([cudnn.heur_mode.A])
        graph.check_support()
        graph.build_plans()
        
        return graph, X, W_tensor, Y
    
    def custom_kernel(data: input_t) -> output_t:
        global _graph, _cache
        input_tensor, kernel, output = data
        B, C, H, W = input_tensor.shape
        K, _, kH, kW = kernel.shape
        
        key = (B, C, H, W, K, kH, kW)
        if key not in _cache:
            _cache[key] = _build_graph(B, C, H, W, K, kH, kW)
        
        graph, X, W_tensor, Y = _cache[key]
        
        workspace = torch.empty(graph.get_workspace_size(), device="cuda", dtype=torch.uint8)
        graph.execute(
            {X: input_tensor, W_tensor: kernel, Y: output},
            workspace
        )
        return output
else:
    def custom_kernel(data: input_t) -> output_t:
        input_tensor, kernel, output = data
        return F.conv2d(input_tensor, kernel, stride=1, padding=0)
scrolls · 79 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 612159.

⋯ 7 unchanged lines
torch.backends.cuda.matmul.allow_tf32 = False
torch.backends.cudnn.benchmark = True
- def custom_kernel(data: input_t) -> output_t:
- input_tensor, kernel, output = data
- # F.conv2d allocates a new tensor. Can we avoid copy?
- # Try: return the F.conv2d result directly (ignore output tensor)
- return F.conv2d(input_tensor, kernel, stride=1, padding=0)
+ # Try cudnn_frontend (Python bindings for cuDNN v9 graph API)
+ _use_cudnn_fe = False
+ try:
+ import cudnn
+ _use_cudnn_fe = True
+ except ImportError:
+ pass
+
+ if _use_cudnn_fe:
+ import cudnn
+
+ _graph = None
+ _cache = {}
+
+ def _build_graph(B, C, H, W, K, kH, kW):
+ oH = H - kH + 1
+ oW = W - kW + 1
+
+ graph = cudnn.pygraph(
+ io_data_type=cudnn.data_type.FLOAT,
+ intermediate_data_type=cudnn.data_type.FLOAT,
+ compute_data_type=cudnn.data_type.FLOAT,
+ )
+
+ X = graph.tensor(name="X", dim=[B, C, H, W], stride=[C*H*W, H*W, W, 1])
+ W_tensor = graph.tensor(name="W", dim=[K, C, kH, kW], stride=[C*kH*kW, kH*kW, kW, 1])
+
+ Y = graph.conv_fprop(
+ image=X,
+ weight=W_tensor,
+ pre_padding=[0, 0],
+ post_padding=[0, 0],
+ stride=[1, 1],
+ dilation=[1, 1],
+ )
+ Y.set_name("Y")
+ Y.set_output(True).set_dim([B, K, oH, oW]).set_stride([K*oH*oW, oH*oW, oW, 1])
+
+ graph.validate()
+ graph.build_operation_graph()
+ graph.create_execution_plans([cudnn.heur_mode.A])
+ graph.check_support()
+ graph.build_plans()
+
+ return graph, X, W_tensor, Y
+
+ def custom_kernel(data: input_t) -> output_t:
+ global _graph, _cache
+ input_tensor, kernel, output = data
+ B, C, H, W = input_tensor.shape
+ K, _, kH, kW = kernel.shape
+
+ key = (B, C, H, W, K, kH, kW)
+ if key not in _cache:
+ _cache[key] = _build_graph(B, C, H, W, K, kH, kW)
+
+ graph, X, W_tensor, Y = _cache[key]
+
+ workspace = torch.empty(graph.get_workspace_size(), device="cuda", dtype=torch.uint8)
+ graph.execute(
+ {X: input_tensor, W_tensor: kernel, Y: output},
+ workspace
+ )
+ return output
+ else:
+ def custom_kernel(data: input_t) -> output_t:
+ input_tensor, kernel, output = data
+ return F.conv2d(input_tensor, kernel, stride=1, padding=0)
scrolls · 77 diff lines total

Best evidence level for this revision: reported

JSON