Skip to content
KernelIndex
Search⌘K

submission 38017

mebenstein · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 84 lines, June 9 Researcher Reciprocity License v1.0.

greyscale.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-38017?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA A100
2.55ms
#28 of 137
2025-09-14

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:8f58a814a31be13de3ec2a8fbb84dc9cbcf8be9bea2957dc62a2280a66198f20
license declaredunknown
license concludedunknown
authorsmebenstein
imported2026-08-15

Kernel source

greyscale.py84 lines

import torch
import time
import os
os.environ["TORCH_CUDA_ARCH_LIST"] = "8.0"

from torch.utils.cpp_extension import load_inline


cuda_source = """
#define WARP_SIZE 32
#define N_WARPS 4
#define THREADS (WARP_SIZE * N_WARPS)

constexpr float3 factors = {0.2989f, 0.5870f, 0.1140f};

__global__ void __launch_bounds__(THREADS) greyscale_kernel(const float* x, float* out, size_t n) {
    const unsigned int tid = threadIdx.x;
    const size_t offset = blockIdx.x * THREADS + threadIdx.y * WARP_SIZE;
    const size_t out_idx = offset + tid;
    size_t idx = offset * 3 + tid;
    const int row = tid % 3;

    float values[3];

    #pragma unroll
    for(int i = 0; i < 3; ++i){
        if(idx < n*3)
            values[i] = __ldcs(x + idx);
        idx += WARP_SIZE;
    }

    float3 color;
    color.x = values[row];
    color.y = __shfl_sync(0xffffffff, values[(row + 2)%3], (tid + 1) % 32);
    color.z = __shfl_sync(0xffffffff, values[(row + 1)%3], (tid + 2) % 32);

    float result = color.x * factors.x + color.y * factors.y + color.z * factors.z;
    result = __shfl_sync(0xffffffff, result, (tid * 3) % 32);

    if(out_idx < n)
        __stcs(out + out_idx, result);
    
    // x y z x y z x y z x y z x y z x y z x y z x y z x y z x y z x y
    // z x y z x y z x y z x y z x y z x y z x y z x y z x y z x y z x
    // y z x y z x y z x y z x y z x y z x y z x y z x y z x y z x y z

    // x,y,z,x y,z,x,y z,x,y,z ... x,y,z,x y,z,x,y
    // z,x,y,z x,y,z,x y,z,x,y ... z,x,y,z x,y,z,x
    // y,z,x,y z,x,y,z x,y,z,x ... y,z,x,y z,x,y,z
}

torch::Tensor greyscale(torch::Tensor x, torch::Tensor out) {
    int threads = THREADS;
    int blocks = (out.numel() + threads - 1) / threads;

    greyscale_kernel<<<blocks, dim3{WARP_SIZE,N_WARPS}>>>(
        x.data_ptr<float>(), out.data_ptr<float>(), out.numel()
    );

    return out;
}
"""

cpp_source = """
torch::Tensor greyscale(torch::Tensor x, torch::Tensor out);
"""

# Compile inline
module = load_inline(
    name="sum_f32_to_f64",
    cpp_sources=cpp_source,
    cuda_sources=cuda_source,
    functions=["greyscale"],
    verbose=False,
    with_cuda=True,
    extra_cuda_cflags=['-arch=compute_80', '-O3', '-arch=native']
)

from task import input_t, output_t

def custom_kernel(data: input_t) -> output_t:
    inp, out = data
    return module.greyscale(inp, out)
scrolls · 84 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON