Skip to content
KernelIndex
Search⌘K

submission 687427

MatrixGod-max · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 89 lines, June 9 Researcher Reciprocity License v1.0.

grayscale_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-687427?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA A100
2.52ms
#22 of 137
2026-04-01

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:dad63660e9a0cd4fbb097725963adcefd8f764bc2cbd247a0817fb4536321c9a
license declaredunknown
license concludedunknown
authorsMatrixGod-max
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

vector-width = float4const float4* in4 = reinterpret_cast<const float4*>(input + base * 3);

Kernel source

grayscale_submission.py89 lines
from task import input_t, output_t
import torch
from torch.utils.cpp_extension import load_inline

cuda_source = r"""
#include <torch/extension.h>
#include <cuda_runtime.h>

// Optimized grayscale kernel:
// - float4 vectorized loads (128-bit per load)
// - 4 pixels per thread per loop iteration
// - __ldg read-only cache
// - fused multiply-add (__fmaf_rn)
// - grid-stride loop for large images

__global__ void grayscale_kernel_v3(
    const float* __restrict__ input,
    float* __restrict__ output,
    const int total_pixels
) {
    constexpr float wr = 0.2989f;
    constexpr float wg = 0.5870f;
    constexpr float wb = 0.1140f;

    int tid = blockIdx.x * blockDim.x + threadIdx.x;
    int stride = blockDim.x * gridDim.x;

    // Process 4 pixels at a time using float4 vectorized stores
    int total_pixels_4 = total_pixels / 4 * 4;

    for (int base = tid * 4; base < total_pixels_4; base += stride * 4) {
        // Load 12 floats (4 pixels x 3 channels) via float4 (3 loads)
        const float4* in4 = reinterpret_cast<const float4*>(input + base * 3);
        float4 v0 = __ldg(in4);      // R0 G0 B0 R1
        float4 v1 = __ldg(in4 + 1);  // G1 B1 R2 G2
        float4 v2 = __ldg(in4 + 2);  // B2 R3 G3 B3

        float4 out4;
        out4.x = __fmaf_rn(v0.x, wr, __fmaf_rn(v0.y, wg, v0.z * wb));  // pixel 0
        out4.y = __fmaf_rn(v0.w, wr, __fmaf_rn(v1.x, wg, v1.y * wb));  // pixel 1
        out4.z = __fmaf_rn(v1.z, wr, __fmaf_rn(v1.w, wg, v2.x * wb));  // pixel 2
        out4.w = __fmaf_rn(v2.y, wr, __fmaf_rn(v2.z, wg, v2.w * wb));  // pixel 3

        *reinterpret_cast<float4*>(output + base) = out4;
    }

    // Handle remaining pixels
    for (int i = total_pixels_4 + tid; i < total_pixels; i += stride) {
        int off = i * 3;
        float r = __ldg(&input[off]);
        float g = __ldg(&input[off + 1]);
        float b = __ldg(&input[off + 2]);
        output[i] = __fmaf_rn(r, wr, __fmaf_rn(g, wg, b * wb));
    }
}

torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output) {
    int total_pixels = input.size(0) * input.size(1);

    const int threads = 256;
    int blocks = min((total_pixels / 4 + threads - 1) / threads, 65535);
    blocks = max(blocks, 1);

    grayscale_kernel_v3<<<blocks, threads>>>(
        input.data_ptr<float>(),
        output.data_ptr<float>(),
        total_pixels
    );

    return output;
}
"""

cpp_source = "torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output);"

module = load_inline(
    name="grayscale_v3",
    cpp_sources=cpp_source,
    cuda_sources=cuda_source,
    functions=["grayscale_cuda"],
    verbose=False,
    extra_cuda_cflags=["-O3", "--use_fast_math"],
)

def custom_kernel(data: input_t) -> output_t:
    data, output = data
    module.grayscale_cuda(data, output)
    return output
scrolls · 89 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON