submission 68986
albert9823 · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 61 lines, June 9 Researcher Reciprocity License v1.0.
submission_v8.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-68986?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:be2665eb53f9032241704611da4b55580f1ce2a29e31e44117aedaa9f8f68330
license declaredunknown
license concludedunknown
authorsalbert9823
imported2026-08-15
Kernel source
submission_v8.py61 lines
import torch
from torch.utils.cpp_extension import load_inline
from typing import List
from task import input_t, output_t
cuda_source = """
template<int NUM_THREADS, int N_PER_THREAD>
__global__ void __launch_bounds__(NUM_THREADS) rgb2grayKernel_float3(const float* __restrict__ rgb, float* __restrict__ gray, const int size) {
const int tidx = blockIdx.x * blockDim.x + threadIdx.x;
#pragma unroll
for(int i = 0; i < N_PER_THREAD; i++) {
const int idx = tidx + i*blockDim.x*gridDim.x;
//if (idx < size) {
const float3* rgb3 = reinterpret_cast<const float3 *>(rgb);
const float3 pixel = rgb3[idx];
gray[idx] = 0.2989f * pixel.x + 0.5870f * pixel.y + 0.1140f * pixel.z;
//}
}
}
torch::Tensor rgb2gray(torch::Tensor rgb) {
auto gray = torch::empty({rgb.size(0),rgb.size(1)},rgb.options());
const int N = gray.numel();
// H100 and A100 (256)
// 128,256,512, N_PER_THREAD = 4
const int N_PER_THREAD = 4;
const int threads = 1024;
const int blocks = (N + N_PER_THREAD*threads - 1) / (N_PER_THREAD*threads);
rgb2grayKernel_float3<threads,N_PER_THREAD><<<blocks, threads>>>(
rgb.data_ptr<float>(),
gray.data_ptr<float>(),
N
);
return gray;
}
"""
cpp_source = """
#include <torch/extension.h>
torch::Tensor rgb2gray(torch::Tensor rgb);
"""
rgb2gray_module = load_inline(
name='rgb2gray',
cpp_sources=cpp_source,
cuda_sources=cuda_source,
functions=['rgb2gray'],
verbose=True,
)
def custom_kernel(data: input_t) -> output_t:
x, out = data # x: (H,W,3), out: (H,W)
return rgb2gray_module.rgb2gray(x) # extension expects a single Tensor
#out.copy_(y)
#return out
scrolls · 61 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 66945.
- import torch, triton, triton.language as tl+ import torch+ from torch.utils.cpp_extension import load_inline+ from typing import Listfrom task import input_t, output_t- @triton.jit- def rgb2gray_kernel(- x_ptr, y_ptr,- H, W,- stride_h, stride_w, stride_c,- y_stride_h, y_stride_w,- WR, WG, WB, # weights as scalars- BLOCK_H: tl.constexpr,- BLOCK_W: tl.constexpr,- ):- pid_h = tl.program_id(0)- pid_w = tl.program_id(1)+ cuda_source = """+ template<int NUM_THREADS, int N_PER_THREAD>+ __global__ void __launch_bounds__(NUM_THREADS) rgb2grayKernel_float3(const float* __restrict__ rgb, float* __restrict__ gray, const int size) {+ const int tidx = blockIdx.x * blockDim.x + threadIdx.x;+ #pragma unroll+ for(int i = 0; i < N_PER_THREAD; i++) {+ const int idx = tidx + i*blockDim.x*gridDim.x;+ //if (idx < size) {+ const float3* rgb3 = reinterpret_cast<const float3 *>(rgb);+ const float3 pixel = rgb3[idx];+ gray[idx] = 0.2989f * pixel.x + 0.5870f * pixel.y + 0.1140f * pixel.z;+ //}+ }+ }- h0 = pid_h * BLOCK_H- w0 = pid_w * BLOCK_W+ torch::Tensor rgb2gray(torch::Tensor rgb) {- hs = h0 + tl.arange(0, BLOCK_H) # (BH,)- ws = w0 + tl.arange(0, BLOCK_W) # (BW,)- Hs = hs[:, None] # (BH,1)- Ws = ws[None, :] # (1,BW)+ auto gray = torch::empty({rgb.size(0),rgb.size(1)},rgb.options());+ const int N = gray.numel();- mask = (Hs < H) & (Ws < W)- base = Hs * stride_h + Ws * stride_w # (BH,BW)+ // H100 and A100 (256)+ // 128,256,512, N_PER_THREAD = 4+ const int N_PER_THREAD = 4;+ const int threads = 1024;+ const int blocks = (N + N_PER_THREAD*threads - 1) / (N_PER_THREAD*threads);- # Vectorized 4-lane per-pixel load (last lane is dummy)- ch = tl.arange(0, 4) # (4,)- ptrs = x_ptr + base[:, :, None] + ch[None, None, :] * stride_c # (BH,BW,4)+ rgb2grayKernel_float3<threads,N_PER_THREAD><<<blocks, threads>>>(+ rgb.data_ptr<float>(),+ gray.data_ptr<float>(),+ N+ );++ return gray;+ }+ """- rgb4 = tl.load(- ptrs,- mask=mask[:, :, None] & (ch[None, None, :] < 3),- other=0.0,- cache_modifier=".ca",- ) # (BH,BW,4) = [R,G,B,0]+ cpp_source = """+ #include <torch/extension.h>- # Build weights per lane without slicing- # w_lane[k] = [WR, WG, WB, 0][k]- w_lane = (tl.where(ch == 0, WR, 0.0)- + tl.where(ch == 1, WG, 0.0)- + tl.where(ch == 2, WB, 0.0)) # shape (4,)- w_broadcast = w_lane[None, None, :] # (1,1,4)+ torch::Tensor rgb2gray(torch::Tensor rgb);+ """- # Weighted sum across channel axis- y = tl.sum(rgb4 * w_broadcast, axis=2) # (BH,BW)+ rgb2gray_module = load_inline(+ name='rgb2gray',+ cpp_sources=cpp_source,+ cuda_sources=cuda_source,+ functions=['rgb2gray'],+ verbose=True,+ )- tl.store(y_ptr + Hs * y_stride_h + Ws * y_stride_w, y, mask=mask)-def custom_kernel(data: input_t) -> output_t:- x, out = data- H, W, C = x.shape- s_h, s_w, s_c = x.stride()- ys_h, ys_w = out.stride()-- BLOCK_H, BLOCK_W = 64, 64- grid = (triton.cdiv(H, BLOCK_H), triton.cdiv(W, BLOCK_W))-- rgb2gray_kernel[grid](- x, out,- H, W,- s_h, s_w, s_c,- ys_h, ys_w,- 0.2989, 0.5870, 0.1140, # WR, WG, WB- BLOCK_H=BLOCK_H, BLOCK_W=BLOCK_W,- num_warps=8, num_stages=2- )- return out-+ x, out = data # x: (H,W,3), out: (H,W)+ return rgb2gray_module.rgb2gray(x) # extension expects a single Tensor+ #out.copy_(y)+ #return out
scrolls · 119 diff lines total
Best evidence level for this revision: reported
JSON