submission 607935
dannywillowliu-uchi · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 62 lines, June 9 Researcher Reciprocity License v1.0.
submission_grayscale.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-607935?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:b5e514c06bab7eb5da5c4727f7e4b6920ae10e48a95324d40c6f28d6fe91925a
license declaredunknown
license concludedunknown
authorsdannywillowliu-uchi
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
vector-width = float4
const float4* __restrict__ in4 = reinterpret_cast<const float4*>(input + tid * 12);Kernel source
submission_grayscale.py62 lines
import torch
from torch.utils.cpp_extension import load_inline
from task import input_t, output_t
cuda_src = r'''
#include <torch/extension.h>
#include <cuda_runtime.h>
__global__ void __launch_bounds__(256, 4)
grayscale_kernel(const float* __restrict__ input,
float* __restrict__ output,
const int n_quads) {
const int tid = blockIdx.x * 256 + threadIdx.x;
if (tid < n_quads) {
const float4* __restrict__ in4 = reinterpret_cast<const float4*>(input + tid * 12);
const float4 v0 = __ldg(in4);
const float4 v1 = __ldg(in4 + 1);
const float4 v2 = __ldg(in4 + 2);
float4 out;
out.x = __fmaf_rn(0.2989f, v0.x, __fmaf_rn(0.5870f, v0.y, 0.1140f * v0.z));
out.y = __fmaf_rn(0.2989f, v0.w, __fmaf_rn(0.5870f, v1.x, 0.1140f * v1.y));
out.z = __fmaf_rn(0.2989f, v1.z, __fmaf_rn(0.5870f, v1.w, 0.1140f * v2.x));
out.w = __fmaf_rn(0.2989f, v2.y, __fmaf_rn(0.5870f, v2.z, 0.1140f * v2.w));
reinterpret_cast<float4*>(output)[tid] = out;
}
}
torch::Tensor launch_grayscale(torch::Tensor input, torch::Tensor output) {
const int n_pixels = input.size(0) * input.size(1);
const int n_quads = n_pixels >> 2;
const int blocks = (n_quads + 255) >> 8;
grayscale_kernel<<<blocks, 256>>>(
input.data_ptr<float>(),
output.data_ptr<float>(),
n_quads
);
return output;
}
'''
cpp_src = 'torch::Tensor launch_grayscale(torch::Tensor input, torch::Tensor output);'
module = load_inline(
name='grayscale_v4_final',
cpp_sources=cpp_src,
cuda_sources=cuda_src,
functions=['launch_grayscale'],
verbose=False,
extra_cuda_cflags=['-O3', '--use_fast_math'],
)
_launch = module.launch_grayscale
def custom_kernel(data: input_t) -> output_t:
x, output = data
return _launch(x, output)
scrolls · 62 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 607837.
⋯ 5 unchanged lines#include <torch/extension.h>#include <cuda_runtime.h>- // Each thread processes 2 groups of 4 pixels (8 pixels total)- // Grid-stride loop to handle any size- __global__ void __launch_bounds__(256)+ __global__ void __launch_bounds__(256, 4)grayscale_kernel(const float* __restrict__ input,float* __restrict__ output,const int n_quads) {- int tid = blockIdx.x * blockDim.x + threadIdx.x;-+ const int tid = blockIdx.x * 256 + threadIdx.x;+if (tid < n_quads) {- const float4* in4 = reinterpret_cast<const float4*>(input + tid * 12);- float4 v0 = __ldg(in4);- float4 v1 = __ldg(in4 + 1);- float4 v2 = __ldg(in4 + 2);+ const float4* __restrict__ in4 = reinterpret_cast<const float4*>(input + tid * 12);+ const float4 v0 = __ldg(in4);+ const float4 v1 = __ldg(in4 + 1);+ const float4 v2 = __ldg(in4 + 2);float4 out;out.x = __fmaf_rn(0.2989f, v0.x, __fmaf_rn(0.5870f, v0.y, 0.1140f * v0.z));⋯ 8 unchanged linestorch::Tensor launch_grayscale(torch::Tensor input, torch::Tensor output) {const int n_pixels = input.size(0) * input.size(1);const int n_quads = n_pixels >> 2;-- constexpr int threads = 256;const int blocks = (n_quads + 255) >> 8;-- grayscale_kernel<<<blocks, threads>>>(++ grayscale_kernel<<<blocks, 256>>>(input.data_ptr<float>(),output.data_ptr<float>(),n_quads);-+return output;}'''- cpp_src = '''- torch::Tensor launch_grayscale(torch::Tensor input, torch::Tensor output);- '''+ cpp_src = 'torch::Tensor launch_grayscale(torch::Tensor input, torch::Tensor output);'module = load_inline(- name='grayscale_best_final',+ name='grayscale_v4_final',cpp_sources=cpp_src,cuda_sources=cuda_src,functions=['launch_grayscale'],
scrolls · 59 diff lines total
Best evidence level for this revision: reported
JSON