Skip to content
KernelIndex
Search⌘K

submission 688493

MatrixGod-max · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 54 lines, June 9 Researcher Reciprocity License v1.0.

grayscale_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-688493?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA B200
666.7µs
#63 of 84
2026-04-01

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:a8f7f8994223d35033d464160b0188120b026cf9173fa24731b00d56d7a39ade
license declaredunknown
license concludedunknown
authorsMatrixGod-max
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

num-warps = 8num_warps=8,
stages = 4num_stages=4,

Kernel source

grayscale_submission.py54 lines
from task import input_t, output_t
import torch
import triton
import triton.language as tl


@triton.jit
def grayscale_kernel(
    input_ptr,
    output_ptr,
    total_pixels,
    BLOCK_SIZE: tl.constexpr,
):
    pid = tl.program_id(0)
    offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
    mask = offsets < total_pixels

    base = offsets * 3
    r = tl.load(input_ptr + base, mask=mask, other=0.0)
    g = tl.load(input_ptr + base + 1, mask=mask, other=0.0)
    b = tl.load(input_ptr + base + 2, mask=mask, other=0.0)

    gray = 0.2989 * r + 0.5870 * g + 0.1140 * b
    tl.store(output_ptr + offsets, gray, mask=mask)


def custom_kernel(data: input_t) -> output_t:
    data, output = data

    if not data.is_cuda or not output.is_cuda:
        raise RuntimeError("custom_kernel expects CUDA tensors")
    if data.dtype != torch.float32 or output.dtype != torch.float32:
        raise RuntimeError("custom_kernel expects float32 tensors")
    if not data.is_contiguous() or not output.is_contiguous():
        raise RuntimeError("custom_kernel expects contiguous tensors")

    total_pixels = output.numel()
    if total_pixels == 0:
        return output

    flat_input = data.view(-1)
    flat_output = output.view(-1)
    grid = (triton.cdiv(total_pixels, 1024),)

    grayscale_kernel[grid](
        flat_input,
        flat_output,
        total_pixels,
        BLOCK_SIZE=1024,
        num_warps=8,
        num_stages=4,
    )
    return output
scrolls · 54 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 688313.

from task import input_t, output_t
import torch
- from torch.utils.cpp_extension import load_inline
+ import triton
+ import triton.language as tl
- cuda_source = r"""
- #include <torch/extension.h>
- __global__ void grayscale_kernel(
- const float* __restrict__ input,
- float* __restrict__ output,
- const int total_pixels
- ) {
- const float wr = 0.2989f;
- const float wg = 0.5870f;
- const float wb = 0.1140f;
+ @triton.jit
+ def grayscale_kernel(
+ input_ptr,
+ output_ptr,
+ total_pixels,
+ BLOCK_SIZE: tl.constexpr,
+ ):
+ pid = tl.program_id(0)
+ offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
+ mask = offsets < total_pixels
- int idx = blockIdx.x * blockDim.x + threadIdx.x;
- int stride = blockDim.x * gridDim.x;
+ base = offsets * 3
+ r = tl.load(input_ptr + base, mask=mask, other=0.0)
+ g = tl.load(input_ptr + base + 1, mask=mask, other=0.0)
+ b = tl.load(input_ptr + base + 2, mask=mask, other=0.0)
- int n4 = total_pixels >> 2;
- for (int i = idx; i < n4; i += stride) {
- const float4* p = reinterpret_cast<const float4*>(input + i * 12);
- float4 a = __ldg(p);
- float4 b = __ldg(p + 1);
- float4 c = __ldg(p + 2);
+ gray = 0.2989 * r + 0.5870 * g + 0.1140 * b
+ tl.store(output_ptr + offsets, gray, mask=mask)
- float4 o;
- o.x = __fmaf_rn(wr, a.x, __fmaf_rn(wg, a.y, wb * a.z));
- o.y = __fmaf_rn(wr, a.w, __fmaf_rn(wg, b.x, wb * b.y));
- o.z = __fmaf_rn(wr, b.z, __fmaf_rn(wg, b.w, wb * c.x));
- o.w = __fmaf_rn(wr, c.y, __fmaf_rn(wg, c.z, wb * c.w));
- reinterpret_cast<float4*>(output)[i] = o;
- }
- }
+ def custom_kernel(data: input_t) -> output_t:
+ data, output = data
- torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output) {
- int total_pixels = input.size(0) * input.size(1);
- int n4 = total_pixels >> 2;
+ if not data.is_cuda or not output.is_cuda:
+ raise RuntimeError("custom_kernel expects CUDA tensors")
+ if data.dtype != torch.float32 or output.dtype != torch.float32:
+ raise RuntimeError("custom_kernel expects float32 tensors")
+ if not data.is_contiguous() or not output.is_contiguous():
+ raise RuntimeError("custom_kernel expects contiguous tensors")
- const int threads = 256;
- int blocks = min((n4 + threads - 1) / threads, 65535);
- if (blocks < 1) blocks = 1;
+ total_pixels = output.numel()
+ if total_pixels == 0:
+ return output
- grayscale_kernel<<<blocks, threads>>>(
- input.data_ptr<float>(),
- output.data_ptr<float>(),
- total_pixels
- );
+ flat_input = data.view(-1)
+ flat_output = output.view(-1)
+ grid = (triton.cdiv(total_pixels, 1024),)
- return output;
- }
- """
-
- cpp_source = "torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output);"
-
- module = load_inline(
- name="grayscale_fast",
- cpp_sources=cpp_source,
- cuda_sources=cuda_source,
- functions=["grayscale_cuda"],
- verbose=False,
- extra_cuda_cflags=["-O3", "--use_fast_math"],
- )
-
- def custom_kernel(data: input_t) -> output_t:
- data, output = data
- module.grayscale_cuda(data, output)
+ grayscale_kernel[grid](
+ flat_input,
+ flat_output,
+ total_pixels,
+ BLOCK_SIZE=1024,
+ num_warps=8,
+ num_stages=4,
+ )
return output
scrolls · 109 diff lines total

Best evidence level for this revision: reported

JSON