Skip to content
KernelIndex
Search⌘K

submission 67825

rex_cz · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 64 lines, June 9 Researcher Reciprocity License v1.0.

a.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-67825?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA B200
643.2µs
#59 of 84
2025-11-08

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:82b70a474bad691c109cb2620fcf6861b5345fd2fa07271c7681a9a30008c0ea
license declaredunknown
license concludedunknown
authorsrex_cz
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

num-warps = 4num_warps=4,
stages = 1num_stages=1,

Kernel source

a.py64 lines
#!POPCORN leaderboard grayscale_v2

from task import input_t, output_t
from utils import DeterministicContext
import triton
import triton.language as tl


@triton.jit
def _grayscale_kernel(
    rgb_ptr,
    output_ptr,
    width: tl.int32,
    stride_h: tl.int32,
    stride_w: tl.int32,
    stride_c: tl.int32,
    out_stride_h: tl.int32,
    out_stride_w: tl.int32,
    BLOCK_SIZE: tl.constexpr,
):
    pid = tl.program_id(0)

    offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)

    rows = offsets // width
    cols = offsets % width

    rgb_index = rows * stride_h + cols * stride_w

    r = tl.load(rgb_ptr + rgb_index + 0 * stride_c)
    g = tl.load(rgb_ptr + rgb_index + 1 * stride_c)
    b = tl.load(rgb_ptr + rgb_index + 2 * stride_c)

    grayscale = 0.2989 * r + 0.5870 * g + 0.1140 * b

    out_index = rows * out_stride_h + cols * out_stride_w
    tl.store(output_ptr + out_index, grayscale)


def custom_kernel(data: input_t) -> output_t:
    with DeterministicContext():
        rgb, output = data
        height, width, channels = rgb.shape
        if channels != 3:
            raise ValueError(f"Expected last dimension to be 3, got {channels}")

        BLOCK_SIZE = 256

        grid = (triton.cdiv(height * width, BLOCK_SIZE),)

        _grayscale_kernel[grid](
            rgb,
            output,
            width,
            rgb.stride(0),
            rgb.stride(1),
            rgb.stride(2),
            output.stride(0),
            output.stride(1),
            BLOCK_SIZE=BLOCK_SIZE,
            num_warps=4,
            num_stages=1,
        )
        return output
scrolls · 64 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 66984.

+ #!POPCORN leaderboard grayscale_v2
+
from task import input_t, output_t
- import torch
+ from utils import DeterministicContext
+ import triton
+ import triton.language as tl
+ @triton.jit
+ def _grayscale_kernel(
+ rgb_ptr,
+ output_ptr,
+ width: tl.int32,
+ stride_h: tl.int32,
+ stride_w: tl.int32,
+ stride_c: tl.int32,
+ out_stride_h: tl.int32,
+ out_stride_w: tl.int32,
+ BLOCK_SIZE: tl.constexpr,
+ ):
+ pid = tl.program_id(0)
+
+ offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
+
+ rows = offsets // width
+ cols = offsets % width
+
+ rgb_index = rows * stride_h + cols * stride_w
+
+ r = tl.load(rgb_ptr + rgb_index + 0 * stride_c)
+ g = tl.load(rgb_ptr + rgb_index + 1 * stride_c)
+ b = tl.load(rgb_ptr + rgb_index + 2 * stride_c)
+
+ grayscale = 0.2989 * r + 0.5870 * g + 0.1140 * b
+
+ out_index = rows * out_stride_h + cols * out_stride_w
+ tl.store(output_ptr + out_index, grayscale)
+
+
def custom_kernel(data: input_t) -> output_t:
- data, output = data
- weights = torch.tensor(
- [0.2989, 0.5870, 0.1140], device=data.device, dtype=data.dtype
- )
- output[...] = torch.sum(data * weights, dim=-1)
- return output
+ with DeterministicContext():
+ rgb, output = data
+ height, width, channels = rgb.shape
+ if channels != 3:
+ raise ValueError(f"Expected last dimension to be 3, got {channels}")
+
+ BLOCK_SIZE = 256
+
+ grid = (triton.cdiv(height * width, BLOCK_SIZE),)
+
+ _grayscale_kernel[grid](
+ rgb,
+ output,
+ width,
+ rgb.stride(0),
+ rgb.stride(1),
+ rgb.stride(2),
+ output.stride(0),
+ output.stride(1),
+ BLOCK_SIZE=BLOCK_SIZE,
+ num_warps=4,
+ num_stages=1,
+ )
+ return output
No newline at end of file
scrolls · 72 diff lines total

Best evidence level for this revision: reported

JSON