submission 688493
MatrixGod-max · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 54 lines, June 9 Researcher Reciprocity License v1.0.
grayscale_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-688493?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:a8f7f8994223d35033d464160b0188120b026cf9173fa24731b00d56d7a39ade
license declaredunknown
license concludedunknown
authorsMatrixGod-max
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
num-warps = 8
num_warps=8,stages = 4
num_stages=4,Kernel source
grayscale_submission.py54 lines
from task import input_t, output_t
import torch
import triton
import triton.language as tl
@triton.jit
def grayscale_kernel(
input_ptr,
output_ptr,
total_pixels,
BLOCK_SIZE: tl.constexpr,
):
pid = tl.program_id(0)
offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
mask = offsets < total_pixels
base = offsets * 3
r = tl.load(input_ptr + base, mask=mask, other=0.0)
g = tl.load(input_ptr + base + 1, mask=mask, other=0.0)
b = tl.load(input_ptr + base + 2, mask=mask, other=0.0)
gray = 0.2989 * r + 0.5870 * g + 0.1140 * b
tl.store(output_ptr + offsets, gray, mask=mask)
def custom_kernel(data: input_t) -> output_t:
data, output = data
if not data.is_cuda or not output.is_cuda:
raise RuntimeError("custom_kernel expects CUDA tensors")
if data.dtype != torch.float32 or output.dtype != torch.float32:
raise RuntimeError("custom_kernel expects float32 tensors")
if not data.is_contiguous() or not output.is_contiguous():
raise RuntimeError("custom_kernel expects contiguous tensors")
total_pixels = output.numel()
if total_pixels == 0:
return output
flat_input = data.view(-1)
flat_output = output.view(-1)
grid = (triton.cdiv(total_pixels, 1024),)
grayscale_kernel[grid](
flat_input,
flat_output,
total_pixels,
BLOCK_SIZE=1024,
num_warps=8,
num_stages=4,
)
return output
scrolls · 54 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 688313.
from task import input_t, output_timport torch- from torch.utils.cpp_extension import load_inline+ import triton+ import triton.language as tl- cuda_source = r"""- #include <torch/extension.h>- __global__ void grayscale_kernel(- const float* __restrict__ input,- float* __restrict__ output,- const int total_pixels- ) {- const float wr = 0.2989f;- const float wg = 0.5870f;- const float wb = 0.1140f;+ @triton.jit+ def grayscale_kernel(+ input_ptr,+ output_ptr,+ total_pixels,+ BLOCK_SIZE: tl.constexpr,+ ):+ pid = tl.program_id(0)+ offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)+ mask = offsets < total_pixels- int idx = blockIdx.x * blockDim.x + threadIdx.x;- int stride = blockDim.x * gridDim.x;+ base = offsets * 3+ r = tl.load(input_ptr + base, mask=mask, other=0.0)+ g = tl.load(input_ptr + base + 1, mask=mask, other=0.0)+ b = tl.load(input_ptr + base + 2, mask=mask, other=0.0)- int n4 = total_pixels >> 2;- for (int i = idx; i < n4; i += stride) {- const float4* p = reinterpret_cast<const float4*>(input + i * 12);- float4 a = __ldg(p);- float4 b = __ldg(p + 1);- float4 c = __ldg(p + 2);+ gray = 0.2989 * r + 0.5870 * g + 0.1140 * b+ tl.store(output_ptr + offsets, gray, mask=mask)- float4 o;- o.x = __fmaf_rn(wr, a.x, __fmaf_rn(wg, a.y, wb * a.z));- o.y = __fmaf_rn(wr, a.w, __fmaf_rn(wg, b.x, wb * b.y));- o.z = __fmaf_rn(wr, b.z, __fmaf_rn(wg, b.w, wb * c.x));- o.w = __fmaf_rn(wr, c.y, __fmaf_rn(wg, c.z, wb * c.w));- reinterpret_cast<float4*>(output)[i] = o;- }- }+ def custom_kernel(data: input_t) -> output_t:+ data, output = data- torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output) {- int total_pixels = input.size(0) * input.size(1);- int n4 = total_pixels >> 2;+ if not data.is_cuda or not output.is_cuda:+ raise RuntimeError("custom_kernel expects CUDA tensors")+ if data.dtype != torch.float32 or output.dtype != torch.float32:+ raise RuntimeError("custom_kernel expects float32 tensors")+ if not data.is_contiguous() or not output.is_contiguous():+ raise RuntimeError("custom_kernel expects contiguous tensors")- const int threads = 256;- int blocks = min((n4 + threads - 1) / threads, 65535);- if (blocks < 1) blocks = 1;+ total_pixels = output.numel()+ if total_pixels == 0:+ return output- grayscale_kernel<<<blocks, threads>>>(- input.data_ptr<float>(),- output.data_ptr<float>(),- total_pixels- );+ flat_input = data.view(-1)+ flat_output = output.view(-1)+ grid = (triton.cdiv(total_pixels, 1024),)- return output;- }- """-- cpp_source = "torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output);"-- module = load_inline(- name="grayscale_fast",- cpp_sources=cpp_source,- cuda_sources=cuda_source,- functions=["grayscale_cuda"],- verbose=False,- extra_cuda_cflags=["-O3", "--use_fast_math"],- )-- def custom_kernel(data: input_t) -> output_t:- data, output = data- module.grayscale_cuda(data, output)+ grayscale_kernel[grid](+ flat_input,+ flat_output,+ total_pixels,+ BLOCK_SIZE=1024,+ num_warps=8,+ num_stages=4,+ )return output
scrolls · 109 diff lines total
Best evidence level for this revision: reported
JSON