submission 38017
mebenstein · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 84 lines, June 9 Researcher Reciprocity License v1.0.
greyscale.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-38017?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:8f58a814a31be13de3ec2a8fbb84dc9cbcf8be9bea2957dc62a2280a66198f20
license declaredunknown
license concludedunknown
authorsmebenstein
imported2026-08-15
Kernel source
greyscale.py84 lines
import torch
import time
import os
os.environ["TORCH_CUDA_ARCH_LIST"] = "8.0"
from torch.utils.cpp_extension import load_inline
cuda_source = """
#define WARP_SIZE 32
#define N_WARPS 4
#define THREADS (WARP_SIZE * N_WARPS)
constexpr float3 factors = {0.2989f, 0.5870f, 0.1140f};
__global__ void __launch_bounds__(THREADS) greyscale_kernel(const float* x, float* out, size_t n) {
const unsigned int tid = threadIdx.x;
const size_t offset = blockIdx.x * THREADS + threadIdx.y * WARP_SIZE;
const size_t out_idx = offset + tid;
size_t idx = offset * 3 + tid;
const int row = tid % 3;
float values[3];
#pragma unroll
for(int i = 0; i < 3; ++i){
if(idx < n*3)
values[i] = __ldcs(x + idx);
idx += WARP_SIZE;
}
float3 color;
color.x = values[row];
color.y = __shfl_sync(0xffffffff, values[(row + 2)%3], (tid + 1) % 32);
color.z = __shfl_sync(0xffffffff, values[(row + 1)%3], (tid + 2) % 32);
float result = color.x * factors.x + color.y * factors.y + color.z * factors.z;
result = __shfl_sync(0xffffffff, result, (tid * 3) % 32);
if(out_idx < n)
__stcs(out + out_idx, result);
// x y z x y z x y z x y z x y z x y z x y z x y z x y z x y z x y
// z x y z x y z x y z x y z x y z x y z x y z x y z x y z x y z x
// y z x y z x y z x y z x y z x y z x y z x y z x y z x y z x y z
// x,y,z,x y,z,x,y z,x,y,z ... x,y,z,x y,z,x,y
// z,x,y,z x,y,z,x y,z,x,y ... z,x,y,z x,y,z,x
// y,z,x,y z,x,y,z x,y,z,x ... y,z,x,y z,x,y,z
}
torch::Tensor greyscale(torch::Tensor x, torch::Tensor out) {
int threads = THREADS;
int blocks = (out.numel() + threads - 1) / threads;
greyscale_kernel<<<blocks, dim3{WARP_SIZE,N_WARPS}>>>(
x.data_ptr<float>(), out.data_ptr<float>(), out.numel()
);
return out;
}
"""
cpp_source = """
torch::Tensor greyscale(torch::Tensor x, torch::Tensor out);
"""
# Compile inline
module = load_inline(
name="sum_f32_to_f64",
cpp_sources=cpp_source,
cuda_sources=cuda_source,
functions=["greyscale"],
verbose=False,
with_cuda=True,
extra_cuda_cflags=['-arch=compute_80', '-O3', '-arch=native']
)
from task import input_t, output_t
def custom_kernel(data: input_t) -> output_t:
inp, out = data
return module.greyscale(inp, out)scrolls · 84 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON