Skip to content
KernelIndex
Search⌘K

submission 67954

rex_cz · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 83 lines, June 9 Researcher Reciprocity License v1.0.

grayscale_v4.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-67954?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA A100
2.41ms
#10 of 137
2025-11-08

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:e6de58c1685f940792eed646b3e4f4f08f4998887acc62434d02cbd1cc62a135
license declaredunknown
license concludedunknown
authorsrex_cz
imported2026-08-15

Kernel source

grayscale_v4.py83 lines
# Reference: https://github.com/gpu-mode/reference-kernels/blob/main/problems/pmpp_v2/grayscale_py/reference.py
# B200: block dim 256:  64381.682μs
# B200: block dim 1024: 48362.844μs
import torch
from task import input_t, output_t
import cutlass.cute as cute
from cutlass.cute.runtime import from_dlpack


@cute.kernel
def grayscale_kernel(
    gA: cute.Tensor,
    gC: cute.Tensor,
):
    tidx, _, _ = cute.arch.thread_idx()
    bidx, _, _ = cute.arch.block_idx()
    bdimx, _, _ = cute.arch.block_dim()

    m, n, _ = gA.shape[1]
    thread_idx = bidx * bdimx + tidx
    mi = thread_idx // n
    ni = thread_idx % n

    # rgb.shape: ((1,4),3)
    rgb = gA[(None, (mi, ni, None))].load()

    r, g, b = rgb[(None, 0)], rgb[(None, 1)], rgb[(None, 2)]
    y = 0.2989 * r + 0.5870 * g + 0.1140 * b
    gC[(None, (mi, ni))].store(y)


@cute.jit
def grayscale(input):
    mA, mC = input
    num_threads_per_block = 1024
    # gA.shape: ((1,4),(128,32,3))
    gA = cute.zipped_divide(mA, (1, 4))
    gC = cute.zipped_divide(mC, (1, 4))
    grayscale_kernel(gA, gC).launch(
        grid=[cute.size(gA, mode=[1]) // num_threads_per_block // 3, 1, 1],
        block=[num_threads_per_block, 1, 1],
    )

def generate_input(size: int, seed: int) -> input_t:
    """
    Generates random RGB image tensor of specified size.
    Returns:
        Tensor of shape (size, size, 3) with values in [0, 1]
    """
    gen = torch.Generator(device="cuda")
    gen.manual_seed(seed)

    x = torch.rand(
        size, size, 3, device="cuda", dtype=torch.float32, generator=gen
    ).contiguous()

    y = torch.empty(size, size, device="cuda", dtype=torch.float32).contiguous()

    return x, y

sizes = [
    (1001, 128),
    (5531, 256),
    (9173, 512),
    (93246, 1024),
    (6256, 2048),
    (8841, 4096),
    (6252, 8192),
    (54352, 16384),
]
compiled_kernels = {}
for i in sizes:
    seed, size = i
    data, output = generate_input(size, seed)
    a_ = from_dlpack(data, assumed_align=16)
    c_ = from_dlpack(output, assumed_align=16)
    compiled_kernels[size] = cute.compile(grayscale, (a_, c_))

def custom_kernel(input):
    compiled_kernels[input[0].shape[0]](input)
    return input[1]

scrolls · 83 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 67845.

- #!POPCORN leaderboard grayscale_v2
-
+ # Reference: https://github.com/gpu-mode/reference-kernels/blob/main/problems/pmpp_v2/grayscale_py/reference.py
+ # B200: block dim 256: 64381.682μs
+ # B200: block dim 1024: 48362.844μs
+ import torch
from task import input_t, output_t
- from utils import DeterministicContext
- import triton
- import triton.language as tl
+ import cutlass.cute as cute
+ from cutlass.cute.runtime import from_dlpack
- @triton.jit
- def _grayscale_kernel(
- rgb_ptr,
- output_ptr,
- width: tl.int32,
- stride_h: tl.int32,
- stride_w: tl.int32,
- stride_c: tl.int32,
- out_stride_h: tl.int32,
- out_stride_w: tl.int32,
- BLOCK_SIZE: tl.constexpr,
+ @cute.kernel
+ def grayscale_kernel(
+ gA: cute.Tensor,
+ gC: cute.Tensor,
):
- pid = tl.program_id(0)
+ tidx, _, _ = cute.arch.thread_idx()
+ bidx, _, _ = cute.arch.block_idx()
+ bdimx, _, _ = cute.arch.block_dim()
- offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
+ m, n, _ = gA.shape[1]
+ thread_idx = bidx * bdimx + tidx
+ mi = thread_idx // n
+ ni = thread_idx % n
- rows = offsets // width
- cols = offsets % width
+ # rgb.shape: ((1,4),3)
+ rgb = gA[(None, (mi, ni, None))].load()
- rgb_index = rows * stride_h + cols * stride_w
+ r, g, b = rgb[(None, 0)], rgb[(None, 1)], rgb[(None, 2)]
+ y = 0.2989 * r + 0.5870 * g + 0.1140 * b
+ gC[(None, (mi, ni))].store(y)
- r = tl.load(rgb_ptr + rgb_index + 0 * stride_c)
- g = tl.load(rgb_ptr + rgb_index + 1 * stride_c)
- b = tl.load(rgb_ptr + rgb_index + 2 * stride_c)
- grayscale = 0.2989 * r + 0.5870 * g + 0.1140 * b
+ @cute.jit
+ def grayscale(input):
+ mA, mC = input
+ num_threads_per_block = 1024
+ # gA.shape: ((1,4),(128,32,3))
+ gA = cute.zipped_divide(mA, (1, 4))
+ gC = cute.zipped_divide(mC, (1, 4))
+ grayscale_kernel(gA, gC).launch(
+ grid=[cute.size(gA, mode=[1]) // num_threads_per_block // 3, 1, 1],
+ block=[num_threads_per_block, 1, 1],
+ )
- out_index = rows * out_stride_h + cols * out_stride_w
- tl.store(output_ptr + out_index, grayscale)
+ def generate_input(size: int, seed: int) -> input_t:
+ """
+ Generates random RGB image tensor of specified size.
+ Returns:
+ Tensor of shape (size, size, 3) with values in [0, 1]
+ """
+ gen = torch.Generator(device="cuda")
+ gen.manual_seed(seed)
+ x = torch.rand(
+ size, size, 3, device="cuda", dtype=torch.float32, generator=gen
+ ).contiguous()
- def custom_kernel(data: input_t) -> output_t:
- with DeterministicContext():
- rgb, output = data
- height, width, channels = rgb.shape
- if channels != 3:
- raise ValueError(f"Expected last dimension to be 3, got {channels}")
+ y = torch.empty(size, size, device="cuda", dtype=torch.float32).contiguous()
- BLOCK_SIZE = 1024
+ return x, y
- grid = (triton.cdiv(height * width, BLOCK_SIZE),)
+ sizes = [
+ (1001, 128),
+ (5531, 256),
+ (9173, 512),
+ (93246, 1024),
+ (6256, 2048),
+ (8841, 4096),
+ (6252, 8192),
+ (54352, 16384),
+ ]
+ compiled_kernels = {}
+ for i in sizes:
+ seed, size = i
+ data, output = generate_input(size, seed)
+ a_ = from_dlpack(data, assumed_align=16)
+ c_ = from_dlpack(output, assumed_align=16)
+ compiled_kernels[size] = cute.compile(grayscale, (a_, c_))
- _grayscale_kernel[grid](
- rgb,
- output,
- width,
- rgb.stride(0),
- rgb.stride(1),
- rgb.stride(2),
- output.stride(0),
- output.stride(1),
- BLOCK_SIZE=BLOCK_SIZE,
- num_warps=4,
- num_stages=1,
- )
- return output
No newline at end of file
+ def custom_kernel(input):
+ compiled_kernels[input[0].shape[0]](input)
+ return input[1]
+
scrolls · 132 diff lines total

Best evidence level for this revision: reported

JSON