Skip to content
KernelIndex
Search⌘K

submission 687616

MatrixGod-max · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 46 lines, June 9 Researcher Reciprocity License v1.0.

grayscale_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-687616?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA A100
2.49ms
#18 of 137
2026-04-01

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:6baae42622e7f72cc5c0b26e22136fcebe30d97b5445005c2defc98a615ec760
license declaredunknown
license concludedunknown
authorsMatrixGod-max
imported2026-08-15

Kernel source

grayscale_submission.py46 lines
from task import input_t, output_t
import torch
import triton
import triton.language as tl


@triton.jit
def grayscale_kernel(
    in_ptr, out_ptr,
    total_pixels,
    BLOCK_SIZE: tl.constexpr,
):
    wr = 0.2989
    wg = 0.5870
    wb = 0.1140

    pid = tl.program_id(0)
    offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
    mask = offsets < total_pixels

    # Load R, G, B channels (input is [H, W, 3] contiguous)
    base = offsets * 3
    r = tl.load(in_ptr + base, mask=mask)
    g = tl.load(in_ptr + base + 1, mask=mask)
    b = tl.load(in_ptr + base + 2, mask=mask)

    gray = r * wr + g * wg + b * wb

    tl.store(out_ptr + offsets, gray, mask=mask)


def custom_kernel(data: input_t) -> output_t:
    data, output = data
    total_pixels = data.shape[0] * data.shape[1]

    BLOCK_SIZE = 1024
    grid = ((total_pixels + BLOCK_SIZE - 1) // BLOCK_SIZE,)

    grayscale_kernel[grid](
        data, output,
        total_pixels,
        BLOCK_SIZE=BLOCK_SIZE,
    )

    return output
scrolls · 46 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 687427.

from task import input_t, output_t
import torch
- from torch.utils.cpp_extension import load_inline
+ import triton
+ import triton.language as tl
- cuda_source = r"""
- #include <torch/extension.h>
- #include <cuda_runtime.h>
- // Optimized grayscale kernel:
- // - float4 vectorized loads (128-bit per load)
- // - 4 pixels per thread per loop iteration
- // - __ldg read-only cache
- // - fused multiply-add (__fmaf_rn)
- // - grid-stride loop for large images
+ @triton.jit
+ def grayscale_kernel(
+ in_ptr, out_ptr,
+ total_pixels,
+ BLOCK_SIZE: tl.constexpr,
+ ):
+ wr = 0.2989
+ wg = 0.5870
+ wb = 0.1140
- __global__ void grayscale_kernel_v3(
- const float* __restrict__ input,
- float* __restrict__ output,
- const int total_pixels
- ) {
- constexpr float wr = 0.2989f;
- constexpr float wg = 0.5870f;
- constexpr float wb = 0.1140f;
+ pid = tl.program_id(0)
+ offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
+ mask = offsets < total_pixels
- int tid = blockIdx.x * blockDim.x + threadIdx.x;
- int stride = blockDim.x * gridDim.x;
+ # Load R, G, B channels (input is [H, W, 3] contiguous)
+ base = offsets * 3
+ r = tl.load(in_ptr + base, mask=mask)
+ g = tl.load(in_ptr + base + 1, mask=mask)
+ b = tl.load(in_ptr + base + 2, mask=mask)
- // Process 4 pixels at a time using float4 vectorized stores
- int total_pixels_4 = total_pixels / 4 * 4;
+ gray = r * wr + g * wg + b * wb
- for (int base = tid * 4; base < total_pixels_4; base += stride * 4) {
- // Load 12 floats (4 pixels x 3 channels) via float4 (3 loads)
- const float4* in4 = reinterpret_cast<const float4*>(input + base * 3);
- float4 v0 = __ldg(in4); // R0 G0 B0 R1
- float4 v1 = __ldg(in4 + 1); // G1 B1 R2 G2
- float4 v2 = __ldg(in4 + 2); // B2 R3 G3 B3
+ tl.store(out_ptr + offsets, gray, mask=mask)
- float4 out4;
- out4.x = __fmaf_rn(v0.x, wr, __fmaf_rn(v0.y, wg, v0.z * wb)); // pixel 0
- out4.y = __fmaf_rn(v0.w, wr, __fmaf_rn(v1.x, wg, v1.y * wb)); // pixel 1
- out4.z = __fmaf_rn(v1.z, wr, __fmaf_rn(v1.w, wg, v2.x * wb)); // pixel 2
- out4.w = __fmaf_rn(v2.y, wr, __fmaf_rn(v2.z, wg, v2.w * wb)); // pixel 3
- *reinterpret_cast<float4*>(output + base) = out4;
- }
+ def custom_kernel(data: input_t) -> output_t:
+ data, output = data
+ total_pixels = data.shape[0] * data.shape[1]
- // Handle remaining pixels
- for (int i = total_pixels_4 + tid; i < total_pixels; i += stride) {
- int off = i * 3;
- float r = __ldg(&input[off]);
- float g = __ldg(&input[off + 1]);
- float b = __ldg(&input[off + 2]);
- output[i] = __fmaf_rn(r, wr, __fmaf_rn(g, wg, b * wb));
- }
- }
+ BLOCK_SIZE = 1024
+ grid = ((total_pixels + BLOCK_SIZE - 1) // BLOCK_SIZE,)
- torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output) {
- int total_pixels = input.size(0) * input.size(1);
+ grayscale_kernel[grid](
+ data, output,
+ total_pixels,
+ BLOCK_SIZE=BLOCK_SIZE,
+ )
- const int threads = 256;
- int blocks = min((total_pixels / 4 + threads - 1) / threads, 65535);
- blocks = max(blocks, 1);
-
- grayscale_kernel_v3<<<blocks, threads>>>(
- input.data_ptr<float>(),
- output.data_ptr<float>(),
- total_pixels
- );
-
- return output;
- }
- """
-
- cpp_source = "torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output);"
-
- module = load_inline(
- name="grayscale_v3",
- cpp_sources=cpp_source,
- cuda_sources=cuda_source,
- functions=["grayscale_cuda"],
- verbose=False,
- extra_cuda_cflags=["-O3", "--use_fast_math"],
- )
-
- def custom_kernel(data: input_t) -> output_t:
- data, output = data
- module.grayscale_cuda(data, output)
return output
scrolls · 119 diff lines total

Best evidence level for this revision: reported

JSON