submission 512879
JordanNanos · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 80 lines, June 9 Researcher Reciprocity License v1.0.
grayscale_py_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-512879?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:0f8947e0758b3304ec9dc543485453f450585e746555c8d003347b8d3431e99c
license declaredunknown
license concludedunknown
authorsJordanNanos
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
vector-width = float4
const float4* in4 = reinterpret_cast<const float4*>(input + base_pixel * 3);Kernel source
grayscale_py_submission.py80 lines
import torch
from task import input_t, output_t
from torch.utils.cpp_extension import load_inline
cuda_source = """
#include <cuda_runtime.h>
#include <torch/extension.h>
__global__ void grayscale_v4_kernel(
const float* __restrict__ input,
float* __restrict__ output,
int n_pixels
) {
int tid = blockIdx.x * blockDim.x + threadIdx.x;
int base_pixel = tid * 4;
const float wr = 0.2989f;
const float wg = 0.5870f;
const float wb = 0.1140f;
if (base_pixel + 3 < n_pixels) {
const float4* in4 = reinterpret_cast<const float4*>(input + base_pixel * 3);
float4 d0 = __ldg(&in4[0]);
float4 d1 = __ldg(&in4[1]);
float4 d2 = __ldg(&in4[2]);
float4 out;
out.x = __fmaf_rn(wr, d0.x, __fmaf_rn(wg, d0.y, wb * d0.z));
out.y = __fmaf_rn(wr, d0.w, __fmaf_rn(wg, d1.x, wb * d1.y));
out.z = __fmaf_rn(wr, d1.z, __fmaf_rn(wg, d1.w, wb * d2.x));
out.w = __fmaf_rn(wr, d2.y, __fmaf_rn(wg, d2.z, wb * d2.w));
reinterpret_cast<float4*>(output + base_pixel)[0] = out;
} else {
for (int i = 0; i < 4 && base_pixel + i < n_pixels; i++) {
int idx = (base_pixel + i) * 3;
output[base_pixel + i] = wr * input[idx] + wg * input[idx+1] + wb * input[idx+2];
}
}
}
torch::Tensor grayscale_forward(torch::Tensor input, torch::Tensor output) {
int n_pixels = input.numel() / 3;
int threads = 256;
int blocks = (n_pixels / 4 + threads - 1) / threads + 1;
grayscale_v4_kernel<<<blocks, threads>>>(
input.data_ptr<float>(),
output.data_ptr<float>(),
n_pixels
);
return output;
}
"""
cpp_source = """
torch::Tensor grayscale_forward(torch::Tensor input, torch::Tensor output);
"""
_module = None
def _get_module():
global _module
if _module is None:
_module = load_inline(
name="grayscale_v4_fma_v3",
cpp_sources=cpp_source,
cuda_sources=cuda_source,
functions=["grayscale_forward"],
extra_cuda_cflags=["-O3", "--use_fast_math"],
verbose=False,
)
return _module
def custom_kernel(data: input_t) -> output_t:
x, output = data
mod = _get_module()
mod.grayscale_forward(x.contiguous().view(-1), output.view(-1))
return output
scrolls · 80 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON