Skip to content
KernelIndex
Search⌘K

submission 688170

MatrixGod-max · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 71 lines, June 9 Researcher Reciprocity License v1.0.

grayscale_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-688170?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA A100
2.47ms
#13 of 137
2026-04-01

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:a8a3174b387d6c5a2982249c35feb14608493d282a243d0731ab1625c55bb04e
license declaredunknown
license concludedunknown
authorsMatrixGod-max
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

vector-width = float4const float4* p = reinterpret_cast<const float4*>(input + i * 12);

Kernel source

grayscale_submission.py71 lines
from task import input_t, output_t
import torch
from torch.utils.cpp_extension import load_inline

cuda_source = r"""
#include <torch/extension.h>

__global__ void grayscale_kernel(
    const float* __restrict__ input,
    float* __restrict__ output,
    const int total_pixels
) {
    const float wr = 0.2989f;
    const float wg = 0.5870f;
    const float wb = 0.1140f;

    int idx = blockIdx.x * blockDim.x + threadIdx.x;
    int stride = blockDim.x * gridDim.x;

    // Process 4 pixels per iteration: 3 x float4 loads, 1 x float4 store
    int n4 = total_pixels >> 2;
    for (int i = idx; i < n4; i += stride) {
        const float4* p = reinterpret_cast<const float4*>(input + i * 12);
        float4 a = __ldg(p);      // R0 G0 B0 R1
        float4 b = __ldg(p + 1);  // G1 B1 R2 G2
        float4 c = __ldg(p + 2);  // B2 R3 G3 B3

        float4 o;
        o.x = __fmaf_rn(wr, a.x, __fmaf_rn(wg, a.y, wb * a.z));
        o.y = __fmaf_rn(wr, a.w, __fmaf_rn(wg, b.x, wb * b.y));
        o.z = __fmaf_rn(wr, b.z, __fmaf_rn(wg, b.w, wb * c.x));
        o.w = __fmaf_rn(wr, c.y, __fmaf_rn(wg, c.z, wb * c.w));

        reinterpret_cast<float4*>(output)[i] = o;
    }
}

torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output) {
    int total_pixels = input.size(0) * input.size(1);
    int n4 = total_pixels >> 2;

    const int threads = 256;
    int blocks = min((n4 + threads - 1) / threads, 65535);
    if (blocks < 1) blocks = 1;

    grayscale_kernel<<<blocks, threads>>>(
        input.data_ptr<float>(),
        output.data_ptr<float>(),
        total_pixels
    );

    return output;
}
"""

cpp_source = "torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output);"

module = load_inline(
    name="grayscale_fast",
    cpp_sources=cpp_source,
    cuda_sources=cuda_source,
    functions=["grayscale_cuda"],
    verbose=False,
    extra_cuda_cflags=["-O3", "--use_fast_math"],
)

def custom_kernel(data: input_t) -> output_t:
    data, output = data
    module.grayscale_cuda(data, output)
    return output
scrolls · 71 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 687616.

from task import input_t, output_t
import torch
- import triton
- import triton.language as tl
+ from torch.utils.cpp_extension import load_inline
+ cuda_source = r"""
+ #include <torch/extension.h>
- @triton.jit
- def grayscale_kernel(
- in_ptr, out_ptr,
- total_pixels,
- BLOCK_SIZE: tl.constexpr,
- ):
- wr = 0.2989
- wg = 0.5870
- wb = 0.1140
+ __global__ void grayscale_kernel(
+ const float* __restrict__ input,
+ float* __restrict__ output,
+ const int total_pixels
+ ) {
+ const float wr = 0.2989f;
+ const float wg = 0.5870f;
+ const float wb = 0.1140f;
- pid = tl.program_id(0)
- offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
- mask = offsets < total_pixels
+ int idx = blockIdx.x * blockDim.x + threadIdx.x;
+ int stride = blockDim.x * gridDim.x;
- # Load R, G, B channels (input is [H, W, 3] contiguous)
- base = offsets * 3
- r = tl.load(in_ptr + base, mask=mask)
- g = tl.load(in_ptr + base + 1, mask=mask)
- b = tl.load(in_ptr + base + 2, mask=mask)
+ // Process 4 pixels per iteration: 3 x float4 loads, 1 x float4 store
+ int n4 = total_pixels >> 2;
+ for (int i = idx; i < n4; i += stride) {
+ const float4* p = reinterpret_cast<const float4*>(input + i * 12);
+ float4 a = __ldg(p); // R0 G0 B0 R1
+ float4 b = __ldg(p + 1); // G1 B1 R2 G2
+ float4 c = __ldg(p + 2); // B2 R3 G3 B3
- gray = r * wr + g * wg + b * wb
+ float4 o;
+ o.x = __fmaf_rn(wr, a.x, __fmaf_rn(wg, a.y, wb * a.z));
+ o.y = __fmaf_rn(wr, a.w, __fmaf_rn(wg, b.x, wb * b.y));
+ o.z = __fmaf_rn(wr, b.z, __fmaf_rn(wg, b.w, wb * c.x));
+ o.w = __fmaf_rn(wr, c.y, __fmaf_rn(wg, c.z, wb * c.w));
- tl.store(out_ptr + offsets, gray, mask=mask)
+ reinterpret_cast<float4*>(output)[i] = o;
+ }
+ }
+ torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output) {
+ int total_pixels = input.size(0) * input.size(1);
+ int n4 = total_pixels >> 2;
- def custom_kernel(data: input_t) -> output_t:
- data, output = data
- total_pixels = data.shape[0] * data.shape[1]
+ const int threads = 256;
+ int blocks = min((n4 + threads - 1) / threads, 65535);
+ if (blocks < 1) blocks = 1;
- BLOCK_SIZE = 1024
- grid = ((total_pixels + BLOCK_SIZE - 1) // BLOCK_SIZE,)
+ grayscale_kernel<<<blocks, threads>>>(
+ input.data_ptr<float>(),
+ output.data_ptr<float>(),
+ total_pixels
+ );
- grayscale_kernel[grid](
- data, output,
- total_pixels,
- BLOCK_SIZE=BLOCK_SIZE,
- )
+ return output;
+ }
+ """
+ cpp_source = "torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output);"
+
+ module = load_inline(
+ name="grayscale_fast",
+ cpp_sources=cpp_source,
+ cuda_sources=cuda_source,
+ functions=["grayscale_cuda"],
+ verbose=False,
+ extra_cuda_cflags=["-O3", "--use_fast_math"],
+ )
+
+ def custom_kernel(data: input_t) -> output_t:
+ data, output = data
+ module.grayscale_cuda(data, output)
return output
scrolls · 101 diff lines total

Best evidence level for this revision: reported

JSON