submission 688170
MatrixGod-max · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 71 lines, June 9 Researcher Reciprocity License v1.0.
grayscale_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-688170?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:a8a3174b387d6c5a2982249c35feb14608493d282a243d0731ab1625c55bb04e
license declaredunknown
license concludedunknown
authorsMatrixGod-max
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
vector-width = float4
const float4* p = reinterpret_cast<const float4*>(input + i * 12);Kernel source
grayscale_submission.py71 lines
from task import input_t, output_t
import torch
from torch.utils.cpp_extension import load_inline
cuda_source = r"""
#include <torch/extension.h>
__global__ void grayscale_kernel(
const float* __restrict__ input,
float* __restrict__ output,
const int total_pixels
) {
const float wr = 0.2989f;
const float wg = 0.5870f;
const float wb = 0.1140f;
int idx = blockIdx.x * blockDim.x + threadIdx.x;
int stride = blockDim.x * gridDim.x;
// Process 4 pixels per iteration: 3 x float4 loads, 1 x float4 store
int n4 = total_pixels >> 2;
for (int i = idx; i < n4; i += stride) {
const float4* p = reinterpret_cast<const float4*>(input + i * 12);
float4 a = __ldg(p); // R0 G0 B0 R1
float4 b = __ldg(p + 1); // G1 B1 R2 G2
float4 c = __ldg(p + 2); // B2 R3 G3 B3
float4 o;
o.x = __fmaf_rn(wr, a.x, __fmaf_rn(wg, a.y, wb * a.z));
o.y = __fmaf_rn(wr, a.w, __fmaf_rn(wg, b.x, wb * b.y));
o.z = __fmaf_rn(wr, b.z, __fmaf_rn(wg, b.w, wb * c.x));
o.w = __fmaf_rn(wr, c.y, __fmaf_rn(wg, c.z, wb * c.w));
reinterpret_cast<float4*>(output)[i] = o;
}
}
torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output) {
int total_pixels = input.size(0) * input.size(1);
int n4 = total_pixels >> 2;
const int threads = 256;
int blocks = min((n4 + threads - 1) / threads, 65535);
if (blocks < 1) blocks = 1;
grayscale_kernel<<<blocks, threads>>>(
input.data_ptr<float>(),
output.data_ptr<float>(),
total_pixels
);
return output;
}
"""
cpp_source = "torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output);"
module = load_inline(
name="grayscale_fast",
cpp_sources=cpp_source,
cuda_sources=cuda_source,
functions=["grayscale_cuda"],
verbose=False,
extra_cuda_cflags=["-O3", "--use_fast_math"],
)
def custom_kernel(data: input_t) -> output_t:
data, output = data
module.grayscale_cuda(data, output)
return output
scrolls · 71 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 687616.
from task import input_t, output_timport torch- import triton- import triton.language as tl+ from torch.utils.cpp_extension import load_inline+ cuda_source = r"""+ #include <torch/extension.h>- @triton.jit- def grayscale_kernel(- in_ptr, out_ptr,- total_pixels,- BLOCK_SIZE: tl.constexpr,- ):- wr = 0.2989- wg = 0.5870- wb = 0.1140+ __global__ void grayscale_kernel(+ const float* __restrict__ input,+ float* __restrict__ output,+ const int total_pixels+ ) {+ const float wr = 0.2989f;+ const float wg = 0.5870f;+ const float wb = 0.1140f;- pid = tl.program_id(0)- offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)- mask = offsets < total_pixels+ int idx = blockIdx.x * blockDim.x + threadIdx.x;+ int stride = blockDim.x * gridDim.x;- # Load R, G, B channels (input is [H, W, 3] contiguous)- base = offsets * 3- r = tl.load(in_ptr + base, mask=mask)- g = tl.load(in_ptr + base + 1, mask=mask)- b = tl.load(in_ptr + base + 2, mask=mask)+ // Process 4 pixels per iteration: 3 x float4 loads, 1 x float4 store+ int n4 = total_pixels >> 2;+ for (int i = idx; i < n4; i += stride) {+ const float4* p = reinterpret_cast<const float4*>(input + i * 12);+ float4 a = __ldg(p); // R0 G0 B0 R1+ float4 b = __ldg(p + 1); // G1 B1 R2 G2+ float4 c = __ldg(p + 2); // B2 R3 G3 B3- gray = r * wr + g * wg + b * wb+ float4 o;+ o.x = __fmaf_rn(wr, a.x, __fmaf_rn(wg, a.y, wb * a.z));+ o.y = __fmaf_rn(wr, a.w, __fmaf_rn(wg, b.x, wb * b.y));+ o.z = __fmaf_rn(wr, b.z, __fmaf_rn(wg, b.w, wb * c.x));+ o.w = __fmaf_rn(wr, c.y, __fmaf_rn(wg, c.z, wb * c.w));- tl.store(out_ptr + offsets, gray, mask=mask)+ reinterpret_cast<float4*>(output)[i] = o;+ }+ }+ torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output) {+ int total_pixels = input.size(0) * input.size(1);+ int n4 = total_pixels >> 2;- def custom_kernel(data: input_t) -> output_t:- data, output = data- total_pixels = data.shape[0] * data.shape[1]+ const int threads = 256;+ int blocks = min((n4 + threads - 1) / threads, 65535);+ if (blocks < 1) blocks = 1;- BLOCK_SIZE = 1024- grid = ((total_pixels + BLOCK_SIZE - 1) // BLOCK_SIZE,)+ grayscale_kernel<<<blocks, threads>>>(+ input.data_ptr<float>(),+ output.data_ptr<float>(),+ total_pixels+ );- grayscale_kernel[grid](- data, output,- total_pixels,- BLOCK_SIZE=BLOCK_SIZE,- )+ return output;+ }+ """+ cpp_source = "torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output);"++ module = load_inline(+ name="grayscale_fast",+ cpp_sources=cpp_source,+ cuda_sources=cuda_source,+ functions=["grayscale_cuda"],+ verbose=False,+ extra_cuda_cflags=["-O3", "--use_fast_math"],+ )++ def custom_kernel(data: input_t) -> output_t:+ data, output = data+ module.grayscale_cuda(data, output)return output
scrolls · 101 diff lines total
Best evidence level for this revision: reported
JSON