Skip to content
KernelIndex
Search⌘K

submission 66506

aikitoria · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 90 lines, June 9 Researcher Reciprocity License v1.0.

greyscale_v2_simple.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-66506?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA B200
740.3µs
#67 of 84
2025-11-04

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:c73316b84e6820ae6933c2da98bc70b5e3cb93368224d251e6c22a447edcaeeb
license declaredunknown
license concludedunknown
authorsaikitoria
imported2026-08-15

Kernel source

greyscale_v2_simple.py90 lines
# submission.py
from task import input_t, output_t
import torch
from torch.utils.cpp_extension import load_inline

_cuda = r"""
#include <torch/extension.h>
#include <cuda.h>
#include <cuda_runtime.h>
#include <ATen/cuda/CUDAContext.h>

template <typename scalar_t>
__global__ void rgb2gray_kernel(const scalar_t* __restrict__ in,
                                scalar_t* __restrict__ out,
                                size_t n_pix,
                                const float w0,
                                const float w1,
                                const float w2) {
    size_t idx = blockIdx.x * blockDim.x + threadIdx.x;
    size_t stride = blockDim.x * gridDim.x;

    for (size_t i = idx; i < n_pix; i += stride) {
        float r = static_cast<float>(in[3*i + 0]);
        float g = static_cast<float>(in[3*i + 1]);
        float b = static_cast<float>(in[3*i + 2]);
        float y = w0 * r + w1 * g + w2 * b;
        out[i] = static_cast<scalar_t>(y);
    }
}

torch::Tensor grayscale_forward(torch::Tensor input,
                                torch::Tensor output,
                                double w0,
                                double w1,
                                double w2) {
    TORCH_CHECK(input.is_cuda() && output.is_cuda(), "tensors must be CUDA");
    TORCH_CHECK(input.is_contiguous() && output.is_contiguous(), "tensors must be contiguous");
    TORCH_CHECK(input.scalar_type() == output.scalar_type(), "dtype mismatch");
    TORCH_CHECK(input.size(-1) == 3, "last dim must be 3");
    const size_t n_pix = static_cast<size_t>(input.numel() / 3);
    TORCH_CHECK(static_cast<size_t>(output.numel()) == n_pix, "output.numel mismatch");

    const int block = 256;
    const int max_blocks = 32768;
    const int grid = std::min<int>((n_pix + block - 1) / block, max_blocks);

    AT_DISPATCH_FLOATING_TYPES_AND_HALF(input.scalar_type(), "rgb2gray_kernel", [&] {
        const scalar_t* in_ptr = input.data_ptr<scalar_t>();
        scalar_t* out_ptr = output.data_ptr<scalar_t>();
        rgb2gray_kernel<scalar_t><<<grid, block, 0, at::cuda::getCurrentCUDAStream()>>>(
            in_ptr, out_ptr, n_pix, static_cast<float>(w0), static_cast<float>(w1), static_cast<float>(w2));
    });
    C10_CUDA_KERNEL_LAUNCH_CHECK();
    return output;
}
"""

# Minimal declaration so the autogenerated main.cpp can call the CUDA function.
_cpp_decl = r"""
#include <torch/extension.h>
torch::Tensor grayscale_forward(torch::Tensor, torch::Tensor, double, double, double);
"""

mod = load_inline(
    name="rgb2gray_inline",
    cpp_sources=_cpp_decl,          # declaration TU
    cuda_sources=_cuda,             # definition TU
    functions=["grayscale_forward"],
    extra_cuda_cflags=["-O3", "-lineinfo", "--use_fast_math"],
    verbose=False,
)

_WEIGHTS = (0.2989, 0.5870, 0.1140)

def _ensure_contig(x: torch.Tensor) -> torch.Tensor:
    return x if x.is_contiguous() else x.contiguous()

@torch.inference_mode()
def custom_kernel(data: input_t) -> output_t:
    x, out = data
    assert x.is_cuda and out.is_cuda, "move tensors to CUDA"
    assert x.size(-1) == 3, "last dim must be 3"
    x = _ensure_contig(x)
    out = _ensure_contig(out)
    # Flatten views
    xv = x.view(-1, 3)
    ov = out.view(-1)
    mod.grayscale_forward(xv, ov, *_WEIGHTS)
    return out
scrolls · 90 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON