Skip to content
KernelIndex
Search⌘K

submission 512879

JordanNanos · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 80 lines, June 9 Researcher Reciprocity License v1.0.

grayscale_py_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-512879?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA B200
602.4µs
#34 of 84
2026-03-05

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:0f8947e0758b3304ec9dc543485453f450585e746555c8d003347b8d3431e99c
license declaredunknown
license concludedunknown
authorsJordanNanos
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

vector-width = float4const float4* in4 = reinterpret_cast<const float4*>(input + base_pixel * 3);

Kernel source

grayscale_py_submission.py80 lines
import torch
from task import input_t, output_t
from torch.utils.cpp_extension import load_inline

cuda_source = """
#include <cuda_runtime.h>
#include <torch/extension.h>

__global__ void grayscale_v4_kernel(
    const float* __restrict__ input,
    float* __restrict__ output,
    int n_pixels
) {
    int tid = blockIdx.x * blockDim.x + threadIdx.x;
    int base_pixel = tid * 4;

    const float wr = 0.2989f;
    const float wg = 0.5870f;
    const float wb = 0.1140f;

    if (base_pixel + 3 < n_pixels) {
        const float4* in4 = reinterpret_cast<const float4*>(input + base_pixel * 3);
        float4 d0 = __ldg(&in4[0]);
        float4 d1 = __ldg(&in4[1]);
        float4 d2 = __ldg(&in4[2]);

        float4 out;
        out.x = __fmaf_rn(wr, d0.x, __fmaf_rn(wg, d0.y, wb * d0.z));
        out.y = __fmaf_rn(wr, d0.w, __fmaf_rn(wg, d1.x, wb * d1.y));
        out.z = __fmaf_rn(wr, d1.z, __fmaf_rn(wg, d1.w, wb * d2.x));
        out.w = __fmaf_rn(wr, d2.y, __fmaf_rn(wg, d2.z, wb * d2.w));

        reinterpret_cast<float4*>(output + base_pixel)[0] = out;
    } else {
        for (int i = 0; i < 4 && base_pixel + i < n_pixels; i++) {
            int idx = (base_pixel + i) * 3;
            output[base_pixel + i] = wr * input[idx] + wg * input[idx+1] + wb * input[idx+2];
        }
    }
}

torch::Tensor grayscale_forward(torch::Tensor input, torch::Tensor output) {
    int n_pixels = input.numel() / 3;
    int threads = 256;
    int blocks = (n_pixels / 4 + threads - 1) / threads + 1;
    grayscale_v4_kernel<<<blocks, threads>>>(
        input.data_ptr<float>(),
        output.data_ptr<float>(),
        n_pixels
    );
    return output;
}
"""

cpp_source = """
torch::Tensor grayscale_forward(torch::Tensor input, torch::Tensor output);
"""

_module = None

def _get_module():
    global _module
    if _module is None:
        _module = load_inline(
            name="grayscale_v4_fma_v3",
            cpp_sources=cpp_source,
            cuda_sources=cuda_source,
            functions=["grayscale_forward"],
            extra_cuda_cflags=["-O3", "--use_fast_math"],
            verbose=False,
        )
    return _module


def custom_kernel(data: input_t) -> output_t:
    x, output = data
    mod = _get_module()
    mod.grayscale_forward(x.contiguous().view(-1), output.view(-1))
    return output
scrolls · 80 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON