submission 687427
MatrixGod-max · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 89 lines, June 9 Researcher Reciprocity License v1.0.
grayscale_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-687427?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:dad63660e9a0cd4fbb097725963adcefd8f764bc2cbd247a0817fb4536321c9a
license declaredunknown
license concludedunknown
authorsMatrixGod-max
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
vector-width = float4
const float4* in4 = reinterpret_cast<const float4*>(input + base * 3);Kernel source
grayscale_submission.py89 lines
from task import input_t, output_t
import torch
from torch.utils.cpp_extension import load_inline
cuda_source = r"""
#include <torch/extension.h>
#include <cuda_runtime.h>
// Optimized grayscale kernel:
// - float4 vectorized loads (128-bit per load)
// - 4 pixels per thread per loop iteration
// - __ldg read-only cache
// - fused multiply-add (__fmaf_rn)
// - grid-stride loop for large images
__global__ void grayscale_kernel_v3(
const float* __restrict__ input,
float* __restrict__ output,
const int total_pixels
) {
constexpr float wr = 0.2989f;
constexpr float wg = 0.5870f;
constexpr float wb = 0.1140f;
int tid = blockIdx.x * blockDim.x + threadIdx.x;
int stride = blockDim.x * gridDim.x;
// Process 4 pixels at a time using float4 vectorized stores
int total_pixels_4 = total_pixels / 4 * 4;
for (int base = tid * 4; base < total_pixels_4; base += stride * 4) {
// Load 12 floats (4 pixels x 3 channels) via float4 (3 loads)
const float4* in4 = reinterpret_cast<const float4*>(input + base * 3);
float4 v0 = __ldg(in4); // R0 G0 B0 R1
float4 v1 = __ldg(in4 + 1); // G1 B1 R2 G2
float4 v2 = __ldg(in4 + 2); // B2 R3 G3 B3
float4 out4;
out4.x = __fmaf_rn(v0.x, wr, __fmaf_rn(v0.y, wg, v0.z * wb)); // pixel 0
out4.y = __fmaf_rn(v0.w, wr, __fmaf_rn(v1.x, wg, v1.y * wb)); // pixel 1
out4.z = __fmaf_rn(v1.z, wr, __fmaf_rn(v1.w, wg, v2.x * wb)); // pixel 2
out4.w = __fmaf_rn(v2.y, wr, __fmaf_rn(v2.z, wg, v2.w * wb)); // pixel 3
*reinterpret_cast<float4*>(output + base) = out4;
}
// Handle remaining pixels
for (int i = total_pixels_4 + tid; i < total_pixels; i += stride) {
int off = i * 3;
float r = __ldg(&input[off]);
float g = __ldg(&input[off + 1]);
float b = __ldg(&input[off + 2]);
output[i] = __fmaf_rn(r, wr, __fmaf_rn(g, wg, b * wb));
}
}
torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output) {
int total_pixels = input.size(0) * input.size(1);
const int threads = 256;
int blocks = min((total_pixels / 4 + threads - 1) / threads, 65535);
blocks = max(blocks, 1);
grayscale_kernel_v3<<<blocks, threads>>>(
input.data_ptr<float>(),
output.data_ptr<float>(),
total_pixels
);
return output;
}
"""
cpp_source = "torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output);"
module = load_inline(
name="grayscale_v3",
cpp_sources=cpp_source,
cuda_sources=cuda_source,
functions=["grayscale_cuda"],
verbose=False,
extra_cuda_cflags=["-O3", "--use_fast_math"],
)
def custom_kernel(data: input_t) -> output_t:
data, output = data
module.grayscale_cuda(data, output)
return output
scrolls · 89 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON