Skip to content
KernelIndex
Search⌘K

submission 68986

albert9823 · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 61 lines, June 9 Researcher Reciprocity License v1.0.

submission_v8.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-68986?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA A100
2.40ms
#9 of 137
2025-11-10

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:be2665eb53f9032241704611da4b55580f1ce2a29e31e44117aedaa9f8f68330
license declaredunknown
license concludedunknown
authorsalbert9823
imported2026-08-15

Kernel source

submission_v8.py61 lines
import torch
from torch.utils.cpp_extension import load_inline
from typing import List
from task import input_t, output_t

cuda_source = """
template<int NUM_THREADS, int N_PER_THREAD>
__global__ void __launch_bounds__(NUM_THREADS) rgb2grayKernel_float3(const float* __restrict__ rgb, float* __restrict__ gray, const int size) {
    const int tidx = blockIdx.x * blockDim.x + threadIdx.x;
#pragma unroll
    for(int i = 0; i < N_PER_THREAD; i++) {
        const int idx = tidx + i*blockDim.x*gridDim.x;
        //if (idx < size) {
            const float3* rgb3 = reinterpret_cast<const float3 *>(rgb);
            const float3 pixel = rgb3[idx];
            gray[idx] = 0.2989f * pixel.x + 0.5870f * pixel.y + 0.1140f * pixel.z;
        //}
    }
}

torch::Tensor rgb2gray(torch::Tensor rgb) {

    auto gray = torch::empty({rgb.size(0),rgb.size(1)},rgb.options());
    const int N = gray.numel();  

// H100 and A100 (256)
// 128,256,512, N_PER_THREAD = 4
    const int N_PER_THREAD = 4;
    const int threads = 1024; 
    const int blocks = (N + N_PER_THREAD*threads - 1) / (N_PER_THREAD*threads);  

    rgb2grayKernel_float3<threads,N_PER_THREAD><<<blocks, threads>>>(
        rgb.data_ptr<float>(),
        gray.data_ptr<float>(),
        N
    );
    
    return gray;
}
"""

cpp_source = """
#include <torch/extension.h>

torch::Tensor rgb2gray(torch::Tensor rgb);
"""

rgb2gray_module = load_inline(
    name='rgb2gray',
    cpp_sources=cpp_source,
    cuda_sources=cuda_source,
    functions=['rgb2gray'],
    verbose=True,
)

def custom_kernel(data: input_t) -> output_t:
    x, out = data                           # x: (H,W,3), out: (H,W)
    return rgb2gray_module.rgb2gray(x)         # extension expects a single Tensor
    #out.copy_(y)
    #return out
scrolls · 61 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 66945.

- import torch, triton, triton.language as tl
+ import torch
+ from torch.utils.cpp_extension import load_inline
+ from typing import List
from task import input_t, output_t
- @triton.jit
- def rgb2gray_kernel(
- x_ptr, y_ptr,
- H, W,
- stride_h, stride_w, stride_c,
- y_stride_h, y_stride_w,
- WR, WG, WB, # weights as scalars
- BLOCK_H: tl.constexpr,
- BLOCK_W: tl.constexpr,
- ):
- pid_h = tl.program_id(0)
- pid_w = tl.program_id(1)
+ cuda_source = """
+ template<int NUM_THREADS, int N_PER_THREAD>
+ __global__ void __launch_bounds__(NUM_THREADS) rgb2grayKernel_float3(const float* __restrict__ rgb, float* __restrict__ gray, const int size) {
+ const int tidx = blockIdx.x * blockDim.x + threadIdx.x;
+ #pragma unroll
+ for(int i = 0; i < N_PER_THREAD; i++) {
+ const int idx = tidx + i*blockDim.x*gridDim.x;
+ //if (idx < size) {
+ const float3* rgb3 = reinterpret_cast<const float3 *>(rgb);
+ const float3 pixel = rgb3[idx];
+ gray[idx] = 0.2989f * pixel.x + 0.5870f * pixel.y + 0.1140f * pixel.z;
+ //}
+ }
+ }
- h0 = pid_h * BLOCK_H
- w0 = pid_w * BLOCK_W
+ torch::Tensor rgb2gray(torch::Tensor rgb) {
- hs = h0 + tl.arange(0, BLOCK_H) # (BH,)
- ws = w0 + tl.arange(0, BLOCK_W) # (BW,)
- Hs = hs[:, None] # (BH,1)
- Ws = ws[None, :] # (1,BW)
+ auto gray = torch::empty({rgb.size(0),rgb.size(1)},rgb.options());
+ const int N = gray.numel();
- mask = (Hs < H) & (Ws < W)
- base = Hs * stride_h + Ws * stride_w # (BH,BW)
+ // H100 and A100 (256)
+ // 128,256,512, N_PER_THREAD = 4
+ const int N_PER_THREAD = 4;
+ const int threads = 1024;
+ const int blocks = (N + N_PER_THREAD*threads - 1) / (N_PER_THREAD*threads);
- # Vectorized 4-lane per-pixel load (last lane is dummy)
- ch = tl.arange(0, 4) # (4,)
- ptrs = x_ptr + base[:, :, None] + ch[None, None, :] * stride_c # (BH,BW,4)
+ rgb2grayKernel_float3<threads,N_PER_THREAD><<<blocks, threads>>>(
+ rgb.data_ptr<float>(),
+ gray.data_ptr<float>(),
+ N
+ );
+
+ return gray;
+ }
+ """
- rgb4 = tl.load(
- ptrs,
- mask=mask[:, :, None] & (ch[None, None, :] < 3),
- other=0.0,
- cache_modifier=".ca",
- ) # (BH,BW,4) = [R,G,B,0]
+ cpp_source = """
+ #include <torch/extension.h>
- # Build weights per lane without slicing
- # w_lane[k] = [WR, WG, WB, 0][k]
- w_lane = (tl.where(ch == 0, WR, 0.0)
- + tl.where(ch == 1, WG, 0.0)
- + tl.where(ch == 2, WB, 0.0)) # shape (4,)
- w_broadcast = w_lane[None, None, :] # (1,1,4)
+ torch::Tensor rgb2gray(torch::Tensor rgb);
+ """
- # Weighted sum across channel axis
- y = tl.sum(rgb4 * w_broadcast, axis=2) # (BH,BW)
+ rgb2gray_module = load_inline(
+ name='rgb2gray',
+ cpp_sources=cpp_source,
+ cuda_sources=cuda_source,
+ functions=['rgb2gray'],
+ verbose=True,
+ )
- tl.store(y_ptr + Hs * y_stride_h + Ws * y_stride_w, y, mask=mask)
-
def custom_kernel(data: input_t) -> output_t:
- x, out = data
- H, W, C = x.shape
- s_h, s_w, s_c = x.stride()
- ys_h, ys_w = out.stride()
-
- BLOCK_H, BLOCK_W = 64, 64
- grid = (triton.cdiv(H, BLOCK_H), triton.cdiv(W, BLOCK_W))
-
- rgb2gray_kernel[grid](
- x, out,
- H, W,
- s_h, s_w, s_c,
- ys_h, ys_w,
- 0.2989, 0.5870, 0.1140, # WR, WG, WB
- BLOCK_H=BLOCK_H, BLOCK_W=BLOCK_W,
- num_warps=8, num_stages=2
- )
- return out
-
+ x, out = data # x: (H,W,3), out: (H,W)
+ return rgb2gray_module.rgb2gray(x) # extension expects a single Tensor
+ #out.copy_(y)
+ #return out
scrolls · 119 diff lines total

Best evidence level for this revision: reported

JSON