submission 67954
rex_cz · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 83 lines, June 9 Researcher Reciprocity License v1.0.
grayscale_v4.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-67954?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:e6de58c1685f940792eed646b3e4f4f08f4998887acc62434d02cbd1cc62a135
license declaredunknown
license concludedunknown
authorsrex_cz
imported2026-08-15
Kernel source
grayscale_v4.py83 lines
# Reference: https://github.com/gpu-mode/reference-kernels/blob/main/problems/pmpp_v2/grayscale_py/reference.py
# B200: block dim 256: 64381.682μs
# B200: block dim 1024: 48362.844μs
import torch
from task import input_t, output_t
import cutlass.cute as cute
from cutlass.cute.runtime import from_dlpack
@cute.kernel
def grayscale_kernel(
gA: cute.Tensor,
gC: cute.Tensor,
):
tidx, _, _ = cute.arch.thread_idx()
bidx, _, _ = cute.arch.block_idx()
bdimx, _, _ = cute.arch.block_dim()
m, n, _ = gA.shape[1]
thread_idx = bidx * bdimx + tidx
mi = thread_idx // n
ni = thread_idx % n
# rgb.shape: ((1,4),3)
rgb = gA[(None, (mi, ni, None))].load()
r, g, b = rgb[(None, 0)], rgb[(None, 1)], rgb[(None, 2)]
y = 0.2989 * r + 0.5870 * g + 0.1140 * b
gC[(None, (mi, ni))].store(y)
@cute.jit
def grayscale(input):
mA, mC = input
num_threads_per_block = 1024
# gA.shape: ((1,4),(128,32,3))
gA = cute.zipped_divide(mA, (1, 4))
gC = cute.zipped_divide(mC, (1, 4))
grayscale_kernel(gA, gC).launch(
grid=[cute.size(gA, mode=[1]) // num_threads_per_block // 3, 1, 1],
block=[num_threads_per_block, 1, 1],
)
def generate_input(size: int, seed: int) -> input_t:
"""
Generates random RGB image tensor of specified size.
Returns:
Tensor of shape (size, size, 3) with values in [0, 1]
"""
gen = torch.Generator(device="cuda")
gen.manual_seed(seed)
x = torch.rand(
size, size, 3, device="cuda", dtype=torch.float32, generator=gen
).contiguous()
y = torch.empty(size, size, device="cuda", dtype=torch.float32).contiguous()
return x, y
sizes = [
(1001, 128),
(5531, 256),
(9173, 512),
(93246, 1024),
(6256, 2048),
(8841, 4096),
(6252, 8192),
(54352, 16384),
]
compiled_kernels = {}
for i in sizes:
seed, size = i
data, output = generate_input(size, seed)
a_ = from_dlpack(data, assumed_align=16)
c_ = from_dlpack(output, assumed_align=16)
compiled_kernels[size] = cute.compile(grayscale, (a_, c_))
def custom_kernel(input):
compiled_kernels[input[0].shape[0]](input)
return input[1]
scrolls · 83 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 67845.
- #!POPCORN leaderboard grayscale_v2-+ # Reference: https://github.com/gpu-mode/reference-kernels/blob/main/problems/pmpp_v2/grayscale_py/reference.py+ # B200: block dim 256: 64381.682μs+ # B200: block dim 1024: 48362.844μs+ import torchfrom task import input_t, output_t- from utils import DeterministicContext- import triton- import triton.language as tl+ import cutlass.cute as cute+ from cutlass.cute.runtime import from_dlpack- @triton.jit- def _grayscale_kernel(- rgb_ptr,- output_ptr,- width: tl.int32,- stride_h: tl.int32,- stride_w: tl.int32,- stride_c: tl.int32,- out_stride_h: tl.int32,- out_stride_w: tl.int32,- BLOCK_SIZE: tl.constexpr,+ @cute.kernel+ def grayscale_kernel(+ gA: cute.Tensor,+ gC: cute.Tensor,):- pid = tl.program_id(0)+ tidx, _, _ = cute.arch.thread_idx()+ bidx, _, _ = cute.arch.block_idx()+ bdimx, _, _ = cute.arch.block_dim()- offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)+ m, n, _ = gA.shape[1]+ thread_idx = bidx * bdimx + tidx+ mi = thread_idx // n+ ni = thread_idx % n- rows = offsets // width- cols = offsets % width+ # rgb.shape: ((1,4),3)+ rgb = gA[(None, (mi, ni, None))].load()- rgb_index = rows * stride_h + cols * stride_w+ r, g, b = rgb[(None, 0)], rgb[(None, 1)], rgb[(None, 2)]+ y = 0.2989 * r + 0.5870 * g + 0.1140 * b+ gC[(None, (mi, ni))].store(y)- r = tl.load(rgb_ptr + rgb_index + 0 * stride_c)- g = tl.load(rgb_ptr + rgb_index + 1 * stride_c)- b = tl.load(rgb_ptr + rgb_index + 2 * stride_c)- grayscale = 0.2989 * r + 0.5870 * g + 0.1140 * b+ @cute.jit+ def grayscale(input):+ mA, mC = input+ num_threads_per_block = 1024+ # gA.shape: ((1,4),(128,32,3))+ gA = cute.zipped_divide(mA, (1, 4))+ gC = cute.zipped_divide(mC, (1, 4))+ grayscale_kernel(gA, gC).launch(+ grid=[cute.size(gA, mode=[1]) // num_threads_per_block // 3, 1, 1],+ block=[num_threads_per_block, 1, 1],+ )- out_index = rows * out_stride_h + cols * out_stride_w- tl.store(output_ptr + out_index, grayscale)+ def generate_input(size: int, seed: int) -> input_t:+ """+ Generates random RGB image tensor of specified size.+ Returns:+ Tensor of shape (size, size, 3) with values in [0, 1]+ """+ gen = torch.Generator(device="cuda")+ gen.manual_seed(seed)+ x = torch.rand(+ size, size, 3, device="cuda", dtype=torch.float32, generator=gen+ ).contiguous()- def custom_kernel(data: input_t) -> output_t:- with DeterministicContext():- rgb, output = data- height, width, channels = rgb.shape- if channels != 3:- raise ValueError(f"Expected last dimension to be 3, got {channels}")+ y = torch.empty(size, size, device="cuda", dtype=torch.float32).contiguous()- BLOCK_SIZE = 1024+ return x, y- grid = (triton.cdiv(height * width, BLOCK_SIZE),)+ sizes = [+ (1001, 128),+ (5531, 256),+ (9173, 512),+ (93246, 1024),+ (6256, 2048),+ (8841, 4096),+ (6252, 8192),+ (54352, 16384),+ ]+ compiled_kernels = {}+ for i in sizes:+ seed, size = i+ data, output = generate_input(size, seed)+ a_ = from_dlpack(data, assumed_align=16)+ c_ = from_dlpack(output, assumed_align=16)+ compiled_kernels[size] = cute.compile(grayscale, (a_, c_))- _grayscale_kernel[grid](- rgb,- output,- width,- rgb.stride(0),- rgb.stride(1),- rgb.stride(2),- output.stride(0),- output.stride(1),- BLOCK_SIZE=BLOCK_SIZE,- num_warps=4,- num_stages=1,- )- return outputNo newline at end of file+ def custom_kernel(input):+ compiled_kernels[input[0].shape[0]](input)+ return input[1]+
scrolls · 132 diff lines total
Best evidence level for this revision: reported
JSON