Skip to content
KernelIndex
Search⌘K

submission 779852

shivbhatia · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 42 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-779852?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA A100
2.48ms
#17 of 137
2026-04-23

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:16303ae4eefc030691b5e798d1b491269e07a741825a1d2ebd48211a7596db10
license declaredunknown
license concludedunknown
authorsshivbhatia
imported2026-08-15

Kernel source

submission.py42 lines
import triton
import triton.language as tl
from task import input_t, output_t


@triton.jit
def _grayscale_kernel(
    data_ptr,
    output_ptr,
    n_pixels,
    w0,
    w1,
    w2,
    BLOCK_SIZE: tl.constexpr,
):
    pid = tl.program_id(0)
    offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
    mask = offsets < n_pixels

    base = offsets * 3
    r = tl.load(data_ptr + base, mask=mask)
    g = tl.load(data_ptr + base + 1, mask=mask)
    b = tl.load(data_ptr + base + 2, mask=mask)

    tl.store(output_ptr + offsets, r * w0 + g * w1 + b * w2, mask=mask)


def custom_kernel(data: input_t) -> output_t:
    data, output = data
    n_pixels = output.numel()
    grid = (triton.cdiv(n_pixels, 1024),)
    _grayscale_kernel[grid](
        data.contiguous().view(-1),
        output.view(-1),
        n_pixels,
        0.2989,
        0.5870,
        0.1140,
        BLOCK_SIZE=1024,
    )
    return output
scrolls · 42 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 779820.

+ import triton
+ import triton.language as tl
from task import input_t, output_t
- import torch
- _weights: torch.Tensor | None = None
- @torch.compile
- def _grayscale(data: torch.Tensor, weights: torch.Tensor) -> torch.Tensor:
- return data @ weights
+ @triton.jit
+ def _grayscale_kernel(
+ data_ptr,
+ output_ptr,
+ n_pixels,
+ w0,
+ w1,
+ w2,
+ BLOCK_SIZE: tl.constexpr,
+ ):
+ pid = tl.program_id(0)
+ offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
+ mask = offsets < n_pixels
+ base = offsets * 3
+ r = tl.load(data_ptr + base, mask=mask)
+ g = tl.load(data_ptr + base + 1, mask=mask)
+ b = tl.load(data_ptr + base + 2, mask=mask)
+
+ tl.store(output_ptr + offsets, r * w0 + g * w1 + b * w2, mask=mask)
+
+
def custom_kernel(data: input_t) -> output_t:
- global _weights
data, output = data
- if _weights is None or _weights.device != data.device or _weights.dtype != data.dtype:
- _weights = torch.tensor([0.2989, 0.5870, 0.1140],
- device=data.device, dtype=data.dtype)
- output[...] = _grayscale(data, _weights)
+ n_pixels = output.numel()
+ grid = (triton.cdiv(n_pixels, 1024),)
+ _grayscale_kernel[grid](
+ data.contiguous().view(-1),
+ output.view(-1),
+ n_pixels,
+ 0.2989,
+ 0.5870,
+ 0.1140,
+ BLOCK_SIZE=1024,
+ )
return output
scrolls · 51 diff lines total

Best evidence level for this revision: reported

JSON