submission 687616
MatrixGod-max · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 46 lines, June 9 Researcher Reciprocity License v1.0.
grayscale_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-687616?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:6baae42622e7f72cc5c0b26e22136fcebe30d97b5445005c2defc98a615ec760
license declaredunknown
license concludedunknown
authorsMatrixGod-max
imported2026-08-15
Kernel source
grayscale_submission.py46 lines
from task import input_t, output_t
import torch
import triton
import triton.language as tl
@triton.jit
def grayscale_kernel(
in_ptr, out_ptr,
total_pixels,
BLOCK_SIZE: tl.constexpr,
):
wr = 0.2989
wg = 0.5870
wb = 0.1140
pid = tl.program_id(0)
offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
mask = offsets < total_pixels
# Load R, G, B channels (input is [H, W, 3] contiguous)
base = offsets * 3
r = tl.load(in_ptr + base, mask=mask)
g = tl.load(in_ptr + base + 1, mask=mask)
b = tl.load(in_ptr + base + 2, mask=mask)
gray = r * wr + g * wg + b * wb
tl.store(out_ptr + offsets, gray, mask=mask)
def custom_kernel(data: input_t) -> output_t:
data, output = data
total_pixels = data.shape[0] * data.shape[1]
BLOCK_SIZE = 1024
grid = ((total_pixels + BLOCK_SIZE - 1) // BLOCK_SIZE,)
grayscale_kernel[grid](
data, output,
total_pixels,
BLOCK_SIZE=BLOCK_SIZE,
)
return output
scrolls · 46 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 687427.
from task import input_t, output_timport torch- from torch.utils.cpp_extension import load_inline+ import triton+ import triton.language as tl- cuda_source = r"""- #include <torch/extension.h>- #include <cuda_runtime.h>- // Optimized grayscale kernel:- // - float4 vectorized loads (128-bit per load)- // - 4 pixels per thread per loop iteration- // - __ldg read-only cache- // - fused multiply-add (__fmaf_rn)- // - grid-stride loop for large images+ @triton.jit+ def grayscale_kernel(+ in_ptr, out_ptr,+ total_pixels,+ BLOCK_SIZE: tl.constexpr,+ ):+ wr = 0.2989+ wg = 0.5870+ wb = 0.1140- __global__ void grayscale_kernel_v3(- const float* __restrict__ input,- float* __restrict__ output,- const int total_pixels- ) {- constexpr float wr = 0.2989f;- constexpr float wg = 0.5870f;- constexpr float wb = 0.1140f;+ pid = tl.program_id(0)+ offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)+ mask = offsets < total_pixels- int tid = blockIdx.x * blockDim.x + threadIdx.x;- int stride = blockDim.x * gridDim.x;+ # Load R, G, B channels (input is [H, W, 3] contiguous)+ base = offsets * 3+ r = tl.load(in_ptr + base, mask=mask)+ g = tl.load(in_ptr + base + 1, mask=mask)+ b = tl.load(in_ptr + base + 2, mask=mask)- // Process 4 pixels at a time using float4 vectorized stores- int total_pixels_4 = total_pixels / 4 * 4;+ gray = r * wr + g * wg + b * wb- for (int base = tid * 4; base < total_pixels_4; base += stride * 4) {- // Load 12 floats (4 pixels x 3 channels) via float4 (3 loads)- const float4* in4 = reinterpret_cast<const float4*>(input + base * 3);- float4 v0 = __ldg(in4); // R0 G0 B0 R1- float4 v1 = __ldg(in4 + 1); // G1 B1 R2 G2- float4 v2 = __ldg(in4 + 2); // B2 R3 G3 B3+ tl.store(out_ptr + offsets, gray, mask=mask)- float4 out4;- out4.x = __fmaf_rn(v0.x, wr, __fmaf_rn(v0.y, wg, v0.z * wb)); // pixel 0- out4.y = __fmaf_rn(v0.w, wr, __fmaf_rn(v1.x, wg, v1.y * wb)); // pixel 1- out4.z = __fmaf_rn(v1.z, wr, __fmaf_rn(v1.w, wg, v2.x * wb)); // pixel 2- out4.w = __fmaf_rn(v2.y, wr, __fmaf_rn(v2.z, wg, v2.w * wb)); // pixel 3- *reinterpret_cast<float4*>(output + base) = out4;- }+ def custom_kernel(data: input_t) -> output_t:+ data, output = data+ total_pixels = data.shape[0] * data.shape[1]- // Handle remaining pixels- for (int i = total_pixels_4 + tid; i < total_pixels; i += stride) {- int off = i * 3;- float r = __ldg(&input[off]);- float g = __ldg(&input[off + 1]);- float b = __ldg(&input[off + 2]);- output[i] = __fmaf_rn(r, wr, __fmaf_rn(g, wg, b * wb));- }- }+ BLOCK_SIZE = 1024+ grid = ((total_pixels + BLOCK_SIZE - 1) // BLOCK_SIZE,)- torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output) {- int total_pixels = input.size(0) * input.size(1);+ grayscale_kernel[grid](+ data, output,+ total_pixels,+ BLOCK_SIZE=BLOCK_SIZE,+ )- const int threads = 256;- int blocks = min((total_pixels / 4 + threads - 1) / threads, 65535);- blocks = max(blocks, 1);-- grayscale_kernel_v3<<<blocks, threads>>>(- input.data_ptr<float>(),- output.data_ptr<float>(),- total_pixels- );-- return output;- }- """-- cpp_source = "torch::Tensor grayscale_cuda(torch::Tensor input, torch::Tensor output);"-- module = load_inline(- name="grayscale_v3",- cpp_sources=cpp_source,- cuda_sources=cuda_source,- functions=["grayscale_cuda"],- verbose=False,- extra_cuda_cflags=["-O3", "--use_fast_math"],- )-- def custom_kernel(data: input_t) -> output_t:- data, output = data- module.grayscale_cuda(data, output)return output
scrolls · 119 diff lines total
Best evidence level for this revision: reported
JSON