submission 614130
dannywillowliu-uchi · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 18 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-conv2d-v2-614130?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:996dee4cb749d98ff09f86c80ca9b5f0fd24f07f5b709e3ab55a2a930212e493
license declaredunknown
license concludedunknown
authorsdannywillowliu-uchi
imported2026-08-15
Kernel source
submission.py18 lines
import os
os.environ['CUBLAS_WORKSPACE_CONFIG'] = ':4096:8'
from task import input_t, output_t
import torch
import torch.nn.functional as F
torch.backends.cudnn.allow_tf32 = False
torch.backends.cuda.matmul.allow_tf32 = False
torch.backends.cudnn.benchmark = True
try:
torch.backends.cudnn.benchmark_limit = 0 # No limit on algorithm search
except:
pass
def custom_kernel(data: input_t) -> output_t:
input_tensor, kernel, output = data
return F.conv2d(input_tensor, kernel, stride=1, padding=0)
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 612204.
⋯ 6 unchanged linestorch.backends.cudnn.allow_tf32 = Falsetorch.backends.cuda.matmul.allow_tf32 = Falsetorch.backends.cudnn.benchmark = True-try:- import cudnn- _has_cudnn_fe = True- except ImportError:- _has_cudnn_fe = False+ torch.backends.cudnn.benchmark_limit = 0 # No limit on algorithm search+ except:+ pass- if _has_cudnn_fe:- import cudnn-- _cache = {}- _workspace = None- _max_ws = 0-- def _build(B, C, H, W, K, kH, kW):- global _workspace, _max_ws- oH = H - kH + 1- oW = W - kW + 1-- graph = cudnn.pygraph(- io_data_type=cudnn.data_type.FLOAT,- intermediate_data_type=cudnn.data_type.FLOAT,- compute_data_type=cudnn.data_type.FLOAT,- )-- X = graph.tensor(name="X", dim=[B, C, H, W], stride=[C*H*W, H*W, W, 1])- W_t = graph.tensor(name="W", dim=[K, C, kH, kW], stride=[C*kH*kW, kH*kW, kW, 1])-- Y = graph.conv_fprop(image=X, weight=W_t, pre_padding=[0, 0], post_padding=[0, 0], stride=[1, 1], dilation=[1, 1])- Y.set_name("Y").set_output(True).set_dim([B, K, oH, oW]).set_stride([K*oH*oW, oH*oW, oW, 1])-- graph.validate()- graph.build_operation_graph()-- # Try multiple heuristic modes for better algorithm selection- graph.create_execution_plans([cudnn.heur_mode.A, cudnn.heur_mode.B])- graph.check_support()- graph.build_plans(cudnn.build_plan_policy.ALL)-- ws_needed = graph.get_workspace_size()- if ws_needed > _max_ws:- _workspace = torch.empty(ws_needed, device="cuda", dtype=torch.uint8)- _max_ws = ws_needed-- return graph, X, W_t, Y-- # Pre-warm with the benchmark shape- _wi = torch.randn(1, 128, 256, 256, device="cuda")- _wk = torch.randn(128, 128, 32, 32, device="cuda")- _wo = torch.empty(1, 128, 225, 225, device="cuda")- g, x, w, y = _build(1, 128, 256, 256, 128, 32, 32)- _cache[(1, 128, 256, 256, 128, 32, 32)] = (g, x, w, y)- g.execute({x: _wi, w: _wk, y: _wo}, _workspace)- torch.cuda.synchronize()- del _wi, _wk, _wo-- # Also pre-warm test shapes- for B, C, H, W, K, kH, kW in [(1,16,32,32,16,4,4), (2,16,32,32,16,4,4), (1,32,64,64,32,4,4), (2,32,64,64,32,8,8), (1,64,128,128,64,8,8)]:- k = (B, C, H, W, K, kH, kW)- oH, oW = H-kH+1, W-kW+1- ti = torch.randn(B, C, H, W, device="cuda")- tk = torch.randn(K, C, kH, kW, device="cuda")- to_ = torch.empty(B, K, oH, oW, device="cuda")- g2, x2, w2, y2 = _build(B, C, H, W, K, kH, kW)- _cache[k] = (g2, x2, w2, y2)- g2.execute({x2: ti, w2: tk, y2: to_}, _workspace)- torch.cuda.synchronize()- del ti, tk, to_-- def custom_kernel(data: input_t) -> output_t:- input_tensor, kernel, output = data- B, C, H, W = input_tensor.shape- K, _, kH, kW = kernel.shape-- key = (B, C, H, W, K, kH, kW)- if key not in _cache:- _cache[key] = _build(B, C, H, W, K, kH, kW)-- g, x, w, y = _cache[key]- g.execute({x: input_tensor, w: kernel, y: output}, _workspace)- return output- else:- def custom_kernel(data: input_t) -> output_t:- input_tensor, kernel, output = data- return F.conv2d(input_tensor, kernel, stride=1, padding=0)+ def custom_kernel(data: input_t) -> output_t:+ input_tensor, kernel, output = data+ return F.conv2d(input_tensor, kernel, stride=1, padding=0)
scrolls · 95 diff lines total
Best evidence level for this revision: reported
JSON