Skip to content
KernelIndex
Search⌘K

submission 113078

shigao · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 140 lines, June 9 Researcher Reciprocity License v1.0.

template.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-conv2d-v2-113078?include=source"
interfacepython
Compatibility
measured onNVIDIA L4
declared hardwareNVIDIA L4
architecturessm_89
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
2D convolutionsuite of 5 cases
NVIDIA L4
1.24s
#17 of 21
2025-11-29

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:6cd9c630168b0b45b441680d192344c8ea4f146f463bdb76f8e4f3fafa068775
license declaredunknown
license concludedunknown
authorsshigao
imported2026-08-15

Kernel source

template.py140 lines
import torch
from torch.utils.cpp_extension import load_inline

# --- 1. CUDA Kernel Source (核心逻辑不变,依然很快) ---
cuda_source = r"""
#include <torch/extension.h>
#include <cuda_runtime.h>

__global__ void conv2d_kernel(
    const float* __restrict__ input,
    const float* __restrict__ weight,
    float* __restrict__ output,
    int batch,
    int in_channels,
    int out_channels,
    int height,
    int width,
    int kernel_size) {
    
    int w_out = blockIdx.x * blockDim.x + threadIdx.x;
    int h_out = blockIdx.y * blockDim.y + threadIdx.y;
    
    int b_idx = blockIdx.z / out_channels;
    int c_out = blockIdx.z % out_channels;

    int out_h_size = height - kernel_size + 1;
    int out_w_size = width - kernel_size + 1;

    if (w_out >= out_w_size || h_out >= out_h_size || b_idx >= batch) {
        return;
    }

    float acc = 0.0f;
    
    for (int c_in = 0; c_in < in_channels; ++c_in) {
        for (int kh = 0; kh < kernel_size; ++kh) {
            int in_h = h_out + kh;
            for (int kw = 0; kw < kernel_size; ++kw) {
                int in_w = w_out + kw;
                int in_idx = ((b_idx * in_channels + c_in) * height + in_h) * width + in_w;
                int k_idx = ((c_out * in_channels + c_in) * kernel_size + kh) * kernel_size + kw;
                acc += input[in_idx] * weight[k_idx];
            }
        }
    }

    int out_idx = ((b_idx * out_channels + c_out) * out_h_size + h_out) * out_w_size + w_out;
    output[out_idx] = acc;
}

// --- C++ Launcher 修改点 ---
// 这里的返回值改为 void,因为我们直接写在传入的 output 张量里
void conv2d_forward_wrapper(
    torch::Tensor input, 
    torch::Tensor kernel, 
    torch::Tensor output) { // 接收 output 作为参数

    const int batch = input.size(0);
    const int in_channels = input.size(1);
    const int height = input.size(2);
    const int width = input.size(3);

    const int out_channels = kernel.size(0);
    const int kernel_size = kernel.size(2);

    const int out_h = height - kernel_size + 1;
    const int out_w = width - kernel_size + 1;

    // 不再需要 torch::empty 分配内存
    // 直接使用 output.data_ptr<float>()

    dim3 block(16, 16);
    dim3 grid(
        (out_w + block.x - 1) / block.x,
        (out_h + block.y - 1) / block.y,
        batch * out_channels 
    );

    conv2d_kernel<<<grid, block>>>(
        input.data_ptr<float>(),
        kernel.data_ptr<float>(),
        output.data_ptr<float>(), // 直接传指针
        batch,
        in_channels,
        out_channels,
        height,
        width,
        kernel_size
    );
}
"""

cpp_source = (
    "void conv2d_forward_wrapper("
    "    torch::Tensor input,"
    "    torch::Tensor kernel,"
    "    torch::Tensor output" # 对应修改签名
    ");"
)

# --- 2. 编译扩展 ---
conv2d_ext = load_inline(
    name="conv2d_v4_inplace",
    cpp_sources=[cpp_source],
    cuda_sources=[cuda_source],
    functions=["conv2d_forward_wrapper"],
    extra_cflags=["-std=c++17", "-O3"],
    extra_cuda_cflags=[
        "-O3",
        "--use_fast_math", 
        "-std=c++17",
        "--ptxas-options=-O3",
    ],
    verbose=False,
)

# --- 3. Python Wrapper (完全匹配官方 generate_input) ---
def custom_kernel(input_tuple):
    """
    Input: Tuple (input, kernel, output)
    Output: convolved result (returned specifically to satisfy any return checks)
    """
    # 1. 正确解包 3 个 Tensor
    input_tensor, kernel, output = input_tuple
    
    # 2. 确保连续性 (Contiguous)
    # 注意:如果 input/kernel 已经是 contiguous 的,这步操作开销几乎为 0
    if not input_tensor.is_contiguous():
        input_tensor = input_tensor.contiguous()
    if not kernel.is_contiguous():
        kernel = kernel.contiguous()
    # output 一般由 factory method 生成,默认是 contiguous 的,但为了安全也可以检查
    if not output.is_contiguous():
        output = output.contiguous()
    
    # 3. 调用 C++ (In-place 操作)
    conv2d_ext.conv2d_forward_wrapper(input_tensor, kernel, output)
    
    # 4. 必须返回 output,因为 leaderboard 通常会检查返回值
    return output
scrolls · 140 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 113056.

import torch
from torch.utils.cpp_extension import load_inline
- # --- 1. CUDA Kernel Source ---
+ # --- 1. CUDA Kernel Source (核心逻辑不变,依然很快) ---
cuda_source = r"""
#include <torch/extension.h>
#include <cuda_runtime.h>
⋯ 9 unchanged lines
int width,
int kernel_size) {
- // 计算输出坐标 (W, H)
int w_out = blockIdx.x * blockDim.x + threadIdx.x;
int h_out = blockIdx.y * blockDim.y + threadIdx.y;
- // 计算 Batch 和 Output Channel
- // blockIdx.z 负责覆盖 (Batch * Out_Channels)
int b_idx = blockIdx.z / out_channels;
int c_out = blockIdx.z % out_channels;
int out_h_size = height - kernel_size + 1;
int out_w_size = width - kernel_size + 1;
- // 越界检查
if (w_out >= out_w_size || h_out >= out_h_size || b_idx >= batch) {
return;
}
float acc = 0.0f;
- // 核心卷积循环:遍历输入通道
for (int c_in = 0; c_in < in_channels; ++c_in) {
for (int kh = 0; kh < kernel_size; ++kh) {
int in_h = h_out + kh;
for (int kw = 0; kw < kernel_size; ++kw) {
int in_w = w_out + kw;
-
- // Input memory layout: [batch, in_channels, height, width]
int in_idx = ((b_idx * in_channels + c_in) * height + in_h) * width + in_w;
-
- // Weight memory layout: [out_channels, in_channels, k, k]
- // 题目要求 kernel shape 是 (channels, channels, k, k),即 (out, in, k, k)
int k_idx = ((c_out * in_channels + c_in) * kernel_size + kh) * kernel_size + kw;
-
acc += input[in_idx] * weight[k_idx];
}
}
}
- // Output memory layout: [batch, out_channels, out_h, out_w]
int out_idx = ((b_idx * out_channels + c_out) * out_h_size + h_out) * out_w_size + w_out;
output[out_idx] = acc;
}
- // C++ Launcher
- torch::Tensor conv2d_forward_wrapper(torch::Tensor input, torch::Tensor kernel) {
- // 检查是否为 CUDA
- TORCH_CHECK(input.is_cuda(), "Input must be a CUDA tensor");
- TORCH_CHECK(kernel.is_cuda(), "Kernel must be a CUDA tensor");
+ // --- C++ Launcher 修改点 ---
+ // 这里的返回值改为 void,因为我们直接写在传入的 output 张量里
+ void conv2d_forward_wrapper(
+ torch::Tensor input,
+ torch::Tensor kernel,
+ torch::Tensor output) { // 接收 output 作为参数
const int batch = input.size(0);
const int in_channels = input.size(1);
const int height = input.size(2);
const int width = input.size(3);
- const int out_channels = kernel.size(0);
+ const int out_channels = kernel.size(0);
const int kernel_size = kernel.size(2);
const int out_h = height - kernel_size + 1;
const int out_w = width - kernel_size + 1;
- // 分配输出内存
- auto options = torch::TensorOptions().dtype(input.dtype()).device(input.device());
- auto output = torch::empty({batch, out_channels, out_h, out_w}, options);
+ // 不再需要 torch::empty 分配内存
+ // 直接使用 output.data_ptr<float>()
- // 配置 Kernel 启动参数
dim3 block(16, 16);
dim3 grid(
(out_w + block.x - 1) / block.x,
⋯ 4 unchanged lines
conv2d_kernel<<<grid, block>>>(
input.data_ptr<float>(),
kernel.data_ptr<float>(),
- output.data_ptr<float>(),
+ output.data_ptr<float>(), // 直接传指针
batch,
in_channels,
out_channels,
⋯ 1 unchanged lines
width,
kernel_size
);
-
- return output;
}
"""
cpp_source = (
- "torch::Tensor conv2d_forward_wrapper(torch::Tensor input, torch::Tensor kernel);"
+ "void conv2d_forward_wrapper("
+ " torch::Tensor input,"
+ " torch::Tensor kernel,"
+ " torch::Tensor output" # 对应修改签名
+ ");"
)
# --- 2. 编译扩展 ---
conv2d_ext = load_inline(
- name="conv2d_v3_fixed",
+ name="conv2d_v4_inplace",
cpp_sources=[cpp_source],
cuda_sources=[cuda_source],
functions=["conv2d_forward_wrapper"],
⋯ 7 unchanged lines
verbose=False,
)
- # --- 3. Python Wrapper (修复了解包错误) ---
+ # --- 3. Python Wrapper (完全匹配官方 generate_input) ---
def custom_kernel(input_tuple):
"""
- Input: Tuple (input, kernel, ...possible bias/output...)
- Output: convolved result
+ Input: Tuple (input, kernel, output)
+ Output: convolved result (returned specifically to satisfy any return checks)
"""
- # 【修复】不要使用 a, b = input_tuple,因为 input_tuple 长度可能大于 2
- # 我们只取前两个作为 input 和 kernel
- input_tensor = input_tuple[0]
- kernel = input_tuple[1]
+ # 1. 正确解包 3 个 Tensor
+ input_tensor, kernel, output = input_tuple
- # 确保连续内存
- input_tensor = input_tensor.contiguous()
- kernel = kernel.contiguous()
+ # 2. 确保连续性 (Contiguous)
+ # 注意:如果 input/kernel 已经是 contiguous 的,这步操作开销几乎为 0
+ if not input_tensor.is_contiguous():
+ input_tensor = input_tensor.contiguous()
+ if not kernel.is_contiguous():
+ kernel = kernel.contiguous()
+ # output 一般由 factory method 生成,默认是 contiguous 的,但为了安全也可以检查
+ if not output.is_contiguous():
+ output = output.contiguous()
- # 调用 C++ 扩展
- output = conv2d_ext.conv2d_forward_wrapper(input_tensor, kernel)
+ # 3. 调用 C++ (In-place 操作)
+ conv2d_ext.conv2d_forward_wrapper(input_tensor, kernel, output)
+ # 4. 必须返回 output,因为 leaderboard 通常会检查返回值
return output
No newline at end of file
scrolls · 164 diff lines total

Best evidence level for this revision: reported

JSON