submission 113056
shigao · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 146 lines, June 9 Researcher Reciprocity License v1.0.
template.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-conv2d-v2-113056?include=source"interfacepython
Compatibility
measured onNVIDIA L4
declared hardwareNVIDIA L4
architecturessm_89
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:37c5e8d7d6f61365fe390ca5319e78112e477a1a1ffe884dd0380001738deb61
license declaredunknown
license concludedunknown
authorsshigao
imported2026-08-15
Kernel source
template.py146 lines
import torch
from torch.utils.cpp_extension import load_inline
# --- 1. CUDA Kernel Source ---
cuda_source = r"""
#include <torch/extension.h>
#include <cuda_runtime.h>
__global__ void conv2d_kernel(
const float* __restrict__ input,
const float* __restrict__ weight,
float* __restrict__ output,
int batch,
int in_channels,
int out_channels,
int height,
int width,
int kernel_size) {
// 计算输出坐标 (W, H)
int w_out = blockIdx.x * blockDim.x + threadIdx.x;
int h_out = blockIdx.y * blockDim.y + threadIdx.y;
// 计算 Batch 和 Output Channel
// blockIdx.z 负责覆盖 (Batch * Out_Channels)
int b_idx = blockIdx.z / out_channels;
int c_out = blockIdx.z % out_channels;
int out_h_size = height - kernel_size + 1;
int out_w_size = width - kernel_size + 1;
// 越界检查
if (w_out >= out_w_size || h_out >= out_h_size || b_idx >= batch) {
return;
}
float acc = 0.0f;
// 核心卷积循环:遍历输入通道
for (int c_in = 0; c_in < in_channels; ++c_in) {
for (int kh = 0; kh < kernel_size; ++kh) {
int in_h = h_out + kh;
for (int kw = 0; kw < kernel_size; ++kw) {
int in_w = w_out + kw;
// Input memory layout: [batch, in_channels, height, width]
int in_idx = ((b_idx * in_channels + c_in) * height + in_h) * width + in_w;
// Weight memory layout: [out_channels, in_channels, k, k]
// 题目要求 kernel shape 是 (channels, channels, k, k),即 (out, in, k, k)
int k_idx = ((c_out * in_channels + c_in) * kernel_size + kh) * kernel_size + kw;
acc += input[in_idx] * weight[k_idx];
}
}
}
// Output memory layout: [batch, out_channels, out_h, out_w]
int out_idx = ((b_idx * out_channels + c_out) * out_h_size + h_out) * out_w_size + w_out;
output[out_idx] = acc;
}
// C++ Launcher
torch::Tensor conv2d_forward_wrapper(torch::Tensor input, torch::Tensor kernel) {
// 检查是否为 CUDA
TORCH_CHECK(input.is_cuda(), "Input must be a CUDA tensor");
TORCH_CHECK(kernel.is_cuda(), "Kernel must be a CUDA tensor");
const int batch = input.size(0);
const int in_channels = input.size(1);
const int height = input.size(2);
const int width = input.size(3);
const int out_channels = kernel.size(0);
const int kernel_size = kernel.size(2);
const int out_h = height - kernel_size + 1;
const int out_w = width - kernel_size + 1;
// 分配输出内存
auto options = torch::TensorOptions().dtype(input.dtype()).device(input.device());
auto output = torch::empty({batch, out_channels, out_h, out_w}, options);
// 配置 Kernel 启动参数
dim3 block(16, 16);
dim3 grid(
(out_w + block.x - 1) / block.x,
(out_h + block.y - 1) / block.y,
batch * out_channels
);
conv2d_kernel<<<grid, block>>>(
input.data_ptr<float>(),
kernel.data_ptr<float>(),
output.data_ptr<float>(),
batch,
in_channels,
out_channels,
height,
width,
kernel_size
);
return output;
}
"""
cpp_source = (
"torch::Tensor conv2d_forward_wrapper(torch::Tensor input, torch::Tensor kernel);"
)
# --- 2. 编译扩展 ---
conv2d_ext = load_inline(
name="conv2d_v3_fixed",
cpp_sources=[cpp_source],
cuda_sources=[cuda_source],
functions=["conv2d_forward_wrapper"],
extra_cflags=["-std=c++17", "-O3"],
extra_cuda_cflags=[
"-O3",
"--use_fast_math",
"-std=c++17",
"--ptxas-options=-O3",
],
verbose=False,
)
# --- 3. Python Wrapper (修复了解包错误) ---
def custom_kernel(input_tuple):
"""
Input: Tuple (input, kernel, ...possible bias/output...)
Output: convolved result
"""
# 【修复】不要使用 a, b = input_tuple,因为 input_tuple 长度可能大于 2
# 我们只取前两个作为 input 和 kernel
input_tensor = input_tuple[0]
kernel = input_tuple[1]
# 确保连续内存
input_tensor = input_tensor.contiguous()
kernel = kernel.contiguous()
# 调用 C++ 扩展
output = conv2d_ext.conv2d_forward_wrapper(input_tensor, kernel)
return outputscrolls · 146 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON