submission 66506
aikitoria · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 90 lines, June 9 Researcher Reciprocity License v1.0.
greyscale_v2_simple.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-66506?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:c73316b84e6820ae6933c2da98bc70b5e3cb93368224d251e6c22a447edcaeeb
license declaredunknown
license concludedunknown
authorsaikitoria
imported2026-08-15
Kernel source
greyscale_v2_simple.py90 lines
# submission.py
from task import input_t, output_t
import torch
from torch.utils.cpp_extension import load_inline
_cuda = r"""
#include <torch/extension.h>
#include <cuda.h>
#include <cuda_runtime.h>
#include <ATen/cuda/CUDAContext.h>
template <typename scalar_t>
__global__ void rgb2gray_kernel(const scalar_t* __restrict__ in,
scalar_t* __restrict__ out,
size_t n_pix,
const float w0,
const float w1,
const float w2) {
size_t idx = blockIdx.x * blockDim.x + threadIdx.x;
size_t stride = blockDim.x * gridDim.x;
for (size_t i = idx; i < n_pix; i += stride) {
float r = static_cast<float>(in[3*i + 0]);
float g = static_cast<float>(in[3*i + 1]);
float b = static_cast<float>(in[3*i + 2]);
float y = w0 * r + w1 * g + w2 * b;
out[i] = static_cast<scalar_t>(y);
}
}
torch::Tensor grayscale_forward(torch::Tensor input,
torch::Tensor output,
double w0,
double w1,
double w2) {
TORCH_CHECK(input.is_cuda() && output.is_cuda(), "tensors must be CUDA");
TORCH_CHECK(input.is_contiguous() && output.is_contiguous(), "tensors must be contiguous");
TORCH_CHECK(input.scalar_type() == output.scalar_type(), "dtype mismatch");
TORCH_CHECK(input.size(-1) == 3, "last dim must be 3");
const size_t n_pix = static_cast<size_t>(input.numel() / 3);
TORCH_CHECK(static_cast<size_t>(output.numel()) == n_pix, "output.numel mismatch");
const int block = 256;
const int max_blocks = 32768;
const int grid = std::min<int>((n_pix + block - 1) / block, max_blocks);
AT_DISPATCH_FLOATING_TYPES_AND_HALF(input.scalar_type(), "rgb2gray_kernel", [&] {
const scalar_t* in_ptr = input.data_ptr<scalar_t>();
scalar_t* out_ptr = output.data_ptr<scalar_t>();
rgb2gray_kernel<scalar_t><<<grid, block, 0, at::cuda::getCurrentCUDAStream()>>>(
in_ptr, out_ptr, n_pix, static_cast<float>(w0), static_cast<float>(w1), static_cast<float>(w2));
});
C10_CUDA_KERNEL_LAUNCH_CHECK();
return output;
}
"""
# Minimal declaration so the autogenerated main.cpp can call the CUDA function.
_cpp_decl = r"""
#include <torch/extension.h>
torch::Tensor grayscale_forward(torch::Tensor, torch::Tensor, double, double, double);
"""
mod = load_inline(
name="rgb2gray_inline",
cpp_sources=_cpp_decl, # declaration TU
cuda_sources=_cuda, # definition TU
functions=["grayscale_forward"],
extra_cuda_cflags=["-O3", "-lineinfo", "--use_fast_math"],
verbose=False,
)
_WEIGHTS = (0.2989, 0.5870, 0.1140)
def _ensure_contig(x: torch.Tensor) -> torch.Tensor:
return x if x.is_contiguous() else x.contiguous()
@torch.inference_mode()
def custom_kernel(data: input_t) -> output_t:
x, out = data
assert x.is_cuda and out.is_cuda, "move tensors to CUDA"
assert x.size(-1) == 3, "last dim must be 3"
x = _ensure_contig(x)
out = _ensure_contig(out)
# Flatten views
xv = x.view(-1, 3)
ov = out.view(-1)
mod.grayscale_forward(xv, ov, *_WEIGHTS)
return out
scrolls · 90 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON