submission 113078
shigao · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 140 lines, June 9 Researcher Reciprocity License v1.0.
template.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-conv2d-v2-113078?include=source"interfacepython
Compatibility
measured onNVIDIA L4
declared hardwareNVIDIA L4
architecturessm_89
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:6cd9c630168b0b45b441680d192344c8ea4f146f463bdb76f8e4f3fafa068775
license declaredunknown
license concludedunknown
authorsshigao
imported2026-08-15
Kernel source
template.py140 lines
import torch
from torch.utils.cpp_extension import load_inline
# --- 1. CUDA Kernel Source (核心逻辑不变,依然很快) ---
cuda_source = r"""
#include <torch/extension.h>
#include <cuda_runtime.h>
__global__ void conv2d_kernel(
const float* __restrict__ input,
const float* __restrict__ weight,
float* __restrict__ output,
int batch,
int in_channels,
int out_channels,
int height,
int width,
int kernel_size) {
int w_out = blockIdx.x * blockDim.x + threadIdx.x;
int h_out = blockIdx.y * blockDim.y + threadIdx.y;
int b_idx = blockIdx.z / out_channels;
int c_out = blockIdx.z % out_channels;
int out_h_size = height - kernel_size + 1;
int out_w_size = width - kernel_size + 1;
if (w_out >= out_w_size || h_out >= out_h_size || b_idx >= batch) {
return;
}
float acc = 0.0f;
for (int c_in = 0; c_in < in_channels; ++c_in) {
for (int kh = 0; kh < kernel_size; ++kh) {
int in_h = h_out + kh;
for (int kw = 0; kw < kernel_size; ++kw) {
int in_w = w_out + kw;
int in_idx = ((b_idx * in_channels + c_in) * height + in_h) * width + in_w;
int k_idx = ((c_out * in_channels + c_in) * kernel_size + kh) * kernel_size + kw;
acc += input[in_idx] * weight[k_idx];
}
}
}
int out_idx = ((b_idx * out_channels + c_out) * out_h_size + h_out) * out_w_size + w_out;
output[out_idx] = acc;
}
// --- C++ Launcher 修改点 ---
// 这里的返回值改为 void,因为我们直接写在传入的 output 张量里
void conv2d_forward_wrapper(
torch::Tensor input,
torch::Tensor kernel,
torch::Tensor output) { // 接收 output 作为参数
const int batch = input.size(0);
const int in_channels = input.size(1);
const int height = input.size(2);
const int width = input.size(3);
const int out_channels = kernel.size(0);
const int kernel_size = kernel.size(2);
const int out_h = height - kernel_size + 1;
const int out_w = width - kernel_size + 1;
// 不再需要 torch::empty 分配内存
// 直接使用 output.data_ptr<float>()
dim3 block(16, 16);
dim3 grid(
(out_w + block.x - 1) / block.x,
(out_h + block.y - 1) / block.y,
batch * out_channels
);
conv2d_kernel<<<grid, block>>>(
input.data_ptr<float>(),
kernel.data_ptr<float>(),
output.data_ptr<float>(), // 直接传指针
batch,
in_channels,
out_channels,
height,
width,
kernel_size
);
}
"""
cpp_source = (
"void conv2d_forward_wrapper("
" torch::Tensor input,"
" torch::Tensor kernel,"
" torch::Tensor output" # 对应修改签名
");"
)
# --- 2. 编译扩展 ---
conv2d_ext = load_inline(
name="conv2d_v4_inplace",
cpp_sources=[cpp_source],
cuda_sources=[cuda_source],
functions=["conv2d_forward_wrapper"],
extra_cflags=["-std=c++17", "-O3"],
extra_cuda_cflags=[
"-O3",
"--use_fast_math",
"-std=c++17",
"--ptxas-options=-O3",
],
verbose=False,
)
# --- 3. Python Wrapper (完全匹配官方 generate_input) ---
def custom_kernel(input_tuple):
"""
Input: Tuple (input, kernel, output)
Output: convolved result (returned specifically to satisfy any return checks)
"""
# 1. 正确解包 3 个 Tensor
input_tensor, kernel, output = input_tuple
# 2. 确保连续性 (Contiguous)
# 注意:如果 input/kernel 已经是 contiguous 的,这步操作开销几乎为 0
if not input_tensor.is_contiguous():
input_tensor = input_tensor.contiguous()
if not kernel.is_contiguous():
kernel = kernel.contiguous()
# output 一般由 factory method 生成,默认是 contiguous 的,但为了安全也可以检查
if not output.is_contiguous():
output = output.contiguous()
# 3. 调用 C++ (In-place 操作)
conv2d_ext.conv2d_forward_wrapper(input_tensor, kernel, output)
# 4. 必须返回 output,因为 leaderboard 通常会检查返回值
return outputscrolls · 140 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 113056.
import torchfrom torch.utils.cpp_extension import load_inline- # --- 1. CUDA Kernel Source ---+ # --- 1. CUDA Kernel Source (核心逻辑不变,依然很快) ---cuda_source = r"""#include <torch/extension.h>#include <cuda_runtime.h>⋯ 9 unchanged linesint width,int kernel_size) {- // 计算输出坐标 (W, H)int w_out = blockIdx.x * blockDim.x + threadIdx.x;int h_out = blockIdx.y * blockDim.y + threadIdx.y;- // 计算 Batch 和 Output Channel- // blockIdx.z 负责覆盖 (Batch * Out_Channels)int b_idx = blockIdx.z / out_channels;int c_out = blockIdx.z % out_channels;int out_h_size = height - kernel_size + 1;int out_w_size = width - kernel_size + 1;- // 越界检查if (w_out >= out_w_size || h_out >= out_h_size || b_idx >= batch) {return;}float acc = 0.0f;- // 核心卷积循环:遍历输入通道for (int c_in = 0; c_in < in_channels; ++c_in) {for (int kh = 0; kh < kernel_size; ++kh) {int in_h = h_out + kh;for (int kw = 0; kw < kernel_size; ++kw) {int in_w = w_out + kw;-- // Input memory layout: [batch, in_channels, height, width]int in_idx = ((b_idx * in_channels + c_in) * height + in_h) * width + in_w;-- // Weight memory layout: [out_channels, in_channels, k, k]- // 题目要求 kernel shape 是 (channels, channels, k, k),即 (out, in, k, k)int k_idx = ((c_out * in_channels + c_in) * kernel_size + kh) * kernel_size + kw;-acc += input[in_idx] * weight[k_idx];}}}- // Output memory layout: [batch, out_channels, out_h, out_w]int out_idx = ((b_idx * out_channels + c_out) * out_h_size + h_out) * out_w_size + w_out;output[out_idx] = acc;}- // C++ Launcher- torch::Tensor conv2d_forward_wrapper(torch::Tensor input, torch::Tensor kernel) {- // 检查是否为 CUDA- TORCH_CHECK(input.is_cuda(), "Input must be a CUDA tensor");- TORCH_CHECK(kernel.is_cuda(), "Kernel must be a CUDA tensor");+ // --- C++ Launcher 修改点 ---+ // 这里的返回值改为 void,因为我们直接写在传入的 output 张量里+ void conv2d_forward_wrapper(+ torch::Tensor input,+ torch::Tensor kernel,+ torch::Tensor output) { // 接收 output 作为参数const int batch = input.size(0);const int in_channels = input.size(1);const int height = input.size(2);const int width = input.size(3);- const int out_channels = kernel.size(0);+ const int out_channels = kernel.size(0);const int kernel_size = kernel.size(2);const int out_h = height - kernel_size + 1;const int out_w = width - kernel_size + 1;- // 分配输出内存- auto options = torch::TensorOptions().dtype(input.dtype()).device(input.device());- auto output = torch::empty({batch, out_channels, out_h, out_w}, options);+ // 不再需要 torch::empty 分配内存+ // 直接使用 output.data_ptr<float>()- // 配置 Kernel 启动参数dim3 block(16, 16);dim3 grid((out_w + block.x - 1) / block.x,⋯ 4 unchanged linesconv2d_kernel<<<grid, block>>>(input.data_ptr<float>(),kernel.data_ptr<float>(),- output.data_ptr<float>(),+ output.data_ptr<float>(), // 直接传指针batch,in_channels,out_channels,⋯ 1 unchanged lineswidth,kernel_size);-- return output;}"""cpp_source = (- "torch::Tensor conv2d_forward_wrapper(torch::Tensor input, torch::Tensor kernel);"+ "void conv2d_forward_wrapper("+ " torch::Tensor input,"+ " torch::Tensor kernel,"+ " torch::Tensor output" # 对应修改签名+ ");")# --- 2. 编译扩展 ---conv2d_ext = load_inline(- name="conv2d_v3_fixed",+ name="conv2d_v4_inplace",cpp_sources=[cpp_source],cuda_sources=[cuda_source],functions=["conv2d_forward_wrapper"],⋯ 7 unchanged linesverbose=False,)- # --- 3. Python Wrapper (修复了解包错误) ---+ # --- 3. Python Wrapper (完全匹配官方 generate_input) ---def custom_kernel(input_tuple):"""- Input: Tuple (input, kernel, ...possible bias/output...)- Output: convolved result+ Input: Tuple (input, kernel, output)+ Output: convolved result (returned specifically to satisfy any return checks)"""- # 【修复】不要使用 a, b = input_tuple,因为 input_tuple 长度可能大于 2- # 我们只取前两个作为 input 和 kernel- input_tensor = input_tuple[0]- kernel = input_tuple[1]+ # 1. 正确解包 3 个 Tensor+ input_tensor, kernel, output = input_tuple- # 确保连续内存- input_tensor = input_tensor.contiguous()- kernel = kernel.contiguous()+ # 2. 确保连续性 (Contiguous)+ # 注意:如果 input/kernel 已经是 contiguous 的,这步操作开销几乎为 0+ if not input_tensor.is_contiguous():+ input_tensor = input_tensor.contiguous()+ if not kernel.is_contiguous():+ kernel = kernel.contiguous()+ # output 一般由 factory method 生成,默认是 contiguous 的,但为了安全也可以检查+ if not output.is_contiguous():+ output = output.contiguous()- # 调用 C++ 扩展- output = conv2d_ext.conv2d_forward_wrapper(input_tensor, kernel)+ # 3. 调用 C++ (In-place 操作)+ conv2d_ext.conv2d_forward_wrapper(input_tensor, kernel, output)+ # 4. 必须返回 output,因为 leaderboard 通常会检查返回值return outputNo newline at end of file
scrolls · 164 diff lines total
Best evidence level for this revision: reported
JSON