Skip to content
KernelIndex
Search⌘K

submission 614130

dannywillowliu-uchi · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 18 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-conv2d-v2-614130?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
2D convolutionsuite of 5 cases
NVIDIA A100
18.8ms
#5 of 40
2026-03-23

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:996dee4cb749d98ff09f86c80ca9b5f0fd24f07f5b709e3ab55a2a930212e493
license declaredunknown
license concludedunknown
authorsdannywillowliu-uchi
imported2026-08-15

Kernel source

submission.py18 lines
import os
os.environ['CUBLAS_WORKSPACE_CONFIG'] = ':4096:8'
from task import input_t, output_t
import torch
import torch.nn.functional as F

torch.backends.cudnn.allow_tf32 = False
torch.backends.cuda.matmul.allow_tf32 = False
torch.backends.cudnn.benchmark = True
try:
    torch.backends.cudnn.benchmark_limit = 0  # No limit on algorithm search
except:
    pass

def custom_kernel(data: input_t) -> output_t:
    input_tensor, kernel, output = data
    return F.conv2d(input_tensor, kernel, stride=1, padding=0)

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 612204.

⋯ 6 unchanged lines
torch.backends.cudnn.allow_tf32 = False
torch.backends.cuda.matmul.allow_tf32 = False
torch.backends.cudnn.benchmark = True
-
try:
- import cudnn
- _has_cudnn_fe = True
- except ImportError:
- _has_cudnn_fe = False
+ torch.backends.cudnn.benchmark_limit = 0 # No limit on algorithm search
+ except:
+ pass
- if _has_cudnn_fe:
- import cudnn
-
- _cache = {}
- _workspace = None
- _max_ws = 0
-
- def _build(B, C, H, W, K, kH, kW):
- global _workspace, _max_ws
- oH = H - kH + 1
- oW = W - kW + 1
-
- graph = cudnn.pygraph(
- io_data_type=cudnn.data_type.FLOAT,
- intermediate_data_type=cudnn.data_type.FLOAT,
- compute_data_type=cudnn.data_type.FLOAT,
- )
-
- X = graph.tensor(name="X", dim=[B, C, H, W], stride=[C*H*W, H*W, W, 1])
- W_t = graph.tensor(name="W", dim=[K, C, kH, kW], stride=[C*kH*kW, kH*kW, kW, 1])
-
- Y = graph.conv_fprop(image=X, weight=W_t, pre_padding=[0, 0], post_padding=[0, 0], stride=[1, 1], dilation=[1, 1])
- Y.set_name("Y").set_output(True).set_dim([B, K, oH, oW]).set_stride([K*oH*oW, oH*oW, oW, 1])
-
- graph.validate()
- graph.build_operation_graph()
-
- # Try multiple heuristic modes for better algorithm selection
- graph.create_execution_plans([cudnn.heur_mode.A, cudnn.heur_mode.B])
- graph.check_support()
- graph.build_plans(cudnn.build_plan_policy.ALL)
-
- ws_needed = graph.get_workspace_size()
- if ws_needed > _max_ws:
- _workspace = torch.empty(ws_needed, device="cuda", dtype=torch.uint8)
- _max_ws = ws_needed
-
- return graph, X, W_t, Y
-
- # Pre-warm with the benchmark shape
- _wi = torch.randn(1, 128, 256, 256, device="cuda")
- _wk = torch.randn(128, 128, 32, 32, device="cuda")
- _wo = torch.empty(1, 128, 225, 225, device="cuda")
- g, x, w, y = _build(1, 128, 256, 256, 128, 32, 32)
- _cache[(1, 128, 256, 256, 128, 32, 32)] = (g, x, w, y)
- g.execute({x: _wi, w: _wk, y: _wo}, _workspace)
- torch.cuda.synchronize()
- del _wi, _wk, _wo
-
- # Also pre-warm test shapes
- for B, C, H, W, K, kH, kW in [(1,16,32,32,16,4,4), (2,16,32,32,16,4,4), (1,32,64,64,32,4,4), (2,32,64,64,32,8,8), (1,64,128,128,64,8,8)]:
- k = (B, C, H, W, K, kH, kW)
- oH, oW = H-kH+1, W-kW+1
- ti = torch.randn(B, C, H, W, device="cuda")
- tk = torch.randn(K, C, kH, kW, device="cuda")
- to_ = torch.empty(B, K, oH, oW, device="cuda")
- g2, x2, w2, y2 = _build(B, C, H, W, K, kH, kW)
- _cache[k] = (g2, x2, w2, y2)
- g2.execute({x2: ti, w2: tk, y2: to_}, _workspace)
- torch.cuda.synchronize()
- del ti, tk, to_
-
- def custom_kernel(data: input_t) -> output_t:
- input_tensor, kernel, output = data
- B, C, H, W = input_tensor.shape
- K, _, kH, kW = kernel.shape
-
- key = (B, C, H, W, K, kH, kW)
- if key not in _cache:
- _cache[key] = _build(B, C, H, W, K, kH, kW)
-
- g, x, w, y = _cache[key]
- g.execute({x: input_tensor, w: kernel, y: output}, _workspace)
- return output
- else:
- def custom_kernel(data: input_t) -> output_t:
- input_tensor, kernel, output = data
- return F.conv2d(input_tensor, kernel, stride=1, padding=0)
+ def custom_kernel(data: input_t) -> output_t:
+ input_tensor, kernel, output = data
+ return F.conv2d(input_tensor, kernel, stride=1, padding=0)
scrolls · 95 diff lines total

Best evidence level for this revision: reported

JSON