Skip to content
KernelIndex
Search⌘K

submission 688313

MatrixGod-max · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 70 lines, June 9 Researcher Reciprocity License v1.0.

grayscale_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-688313?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA A100
2.38ms
#5 of 137
2026-04-01

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:743428dcaa434c3b911153049f12c935bc9e8f0b88bbda2c947fecef35d2147c
license declaredunknown
license concludedunknown
authorsMatrixGod-max
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

vector-width = float4const float4* p = reinterpret_cast<const float4*>(input + i * 12);

Kernel source

grayscale_submission.py70 lines
from task import input_t, output_t
import torch
from torch.utils.cpp_extension import load_inline

cuda_source = r"""
#include <torch/extension.h>

__global__ void grayscale_kernel(
    const float* __restrict__ input,
    float* __restrict__ output,
    const int total_pixels
) {
    const float wr = 0.2989f;
    const float wg = 0.5870f;
    const float wb = 0.1140f;

    int idx = blockIdx.x * blockDim.x + threadIdx.x;
    int stride = blockDim.x * gridDim.x;

    int n4 = total_pixels >> 2;
    for (int i = idx; i < n4; i += stride) {
        const float4* p = reinterpret_cast<const float4*>(input + i * 12);
        float4 a = __ldg(p);
        float4 b = __ldg(p + 1);
        float4 c = __ldg(p + 2);

        float4 o;
        o.x = __fmaf_rn(wr, a.x, __fmaf_rn(wg, a.y, wb * a.z));
        o.y = __fmaf_rn(wr, a.w, __fmaf_rn(wg, b.x, wb * b.y));
        o.z = __fmaf_rn(wr, b.z, __fmaf_rn(wg, b.w, wb * c.x));
        o.w = __fmaf_rn(wr, c.y, __fmaf_rn(wg, c.z, wb * c.w));

        reinterpret_cast<float4*>(output)[i] = o;
    }
}

torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output) {
    int total_pixels = input.size(0) * input.size(1);
    int n4 = total_pixels >> 2;

    const int threads = 256;
    int blocks = min((n4 + threads - 1) / threads, 65535);
    if (blocks < 1) blocks = 1;

    grayscale_kernel<<<blocks, threads>>>(
        input.data_ptr<float>(),
        output.data_ptr<float>(),
        total_pixels
    );

    return output;
}
"""

cpp_source = "torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output);"

module = load_inline(
    name="grayscale_fast",
    cpp_sources=cpp_source,
    cuda_sources=cuda_source,
    functions=["grayscale_cuda"],
    verbose=False,
    extra_cuda_cflags=["-O3", "--use_fast_math"],
)

def custom_kernel(data: input_t) -> output_t:
    data, output = data
    module.grayscale_cuda(data, output)
    return output
scrolls · 70 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 688170.

⋯ 16 unchanged lines
int idx = blockIdx.x * blockDim.x + threadIdx.x;
int stride = blockDim.x * gridDim.x;
- // Process 4 pixels per iteration: 3 x float4 loads, 1 x float4 store
int n4 = total_pixels >> 2;
for (int i = idx; i < n4; i += stride) {
const float4* p = reinterpret_cast<const float4*>(input + i * 12);
- float4 a = __ldg(p); // R0 G0 B0 R1
- float4 b = __ldg(p + 1); // G1 B1 R2 G2
- float4 c = __ldg(p + 2); // B2 R3 G3 B3
+ float4 a = __ldg(p);
+ float4 b = __ldg(p + 1);
+ float4 c = __ldg(p + 2);
float4 o;
o.x = __fmaf_rn(wr, a.x, __fmaf_rn(wg, a.y, wb * a.z));

Best evidence level for this revision: reported

JSON