Skip to content
KernelIndex
Search⌘K

submission 608073

dannywillowliu-uchi · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 57 lines, June 9 Researcher Reciprocity License v1.0.

submission_grayscale.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-608073?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA B200
599.0µs
#13 of 84
2026-03-22

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:f42591ae00733bc7732a9352aa5051d4898a16f5566733af0001c0329ddb60cc
license declaredunknown
license concludedunknown
authorsdannywillowliu-uchi
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

vector-width = float4const float4 v0 = __ldg((const float4*)(input + tid * 12));

Kernel source

submission_grayscale.py57 lines
import torch
from torch.utils.cpp_extension import load_inline
from task import input_t, output_t

cuda_src = r'''
#include <torch/extension.h>
#include <cuda_runtime.h>

__global__ void __launch_bounds__(1024, 1)
grayscale_kernel(const float* __restrict__ input,
                 float* __restrict__ output,
                 const int n_quads) {
    const int tid = blockIdx.x * 1024 + threadIdx.x;

    if (tid < n_quads) {
        const float4 v0 = __ldg((const float4*)(input + tid * 12));
        const float4 v1 = __ldg((const float4*)(input + tid * 12 + 4));
        const float4 v2 = __ldg((const float4*)(input + tid * 12 + 8));

        float4 out;
        out.x = __fmaf_rn(0.2989f, v0.x, __fmaf_rn(0.5870f, v0.y, 0.1140f * v0.z));
        out.y = __fmaf_rn(0.2989f, v0.w, __fmaf_rn(0.5870f, v1.x, 0.1140f * v1.y));
        out.z = __fmaf_rn(0.2989f, v1.z, __fmaf_rn(0.5870f, v1.w, 0.1140f * v2.x));
        out.w = __fmaf_rn(0.2989f, v2.y, __fmaf_rn(0.5870f, v2.z, 0.1140f * v2.w));

        ((float4*)output)[tid] = out;
    }
}

torch::Tensor launch_grayscale(torch::Tensor input, torch::Tensor output) {
    const int n_quads = (input.size(0) * input.size(1)) >> 2;
    grayscale_kernel<<<(n_quads + 1023) / 1024, 1024>>>(
        input.data_ptr<float>(),
        output.data_ptr<float>(),
        n_quads
    );
    return output;
}
'''

cpp_src = 'torch::Tensor launch_grayscale(torch::Tensor input, torch::Tensor output);'

module = load_inline(
    name='grayscale_v5_1024',
    cpp_sources=cpp_src,
    cuda_sources=cuda_src,
    functions=['launch_grayscale'],
    verbose=False,
    extra_cuda_cflags=['-O3', '--use_fast_math', '-maxrregcount=28'],
)

_launch = module.launch_grayscale

def custom_kernel(data: input_t) -> output_t:
    x, output = data
    return _launch(x, output)
scrolls · 57 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 607935.

⋯ 5 unchanged lines
#include <torch/extension.h>
#include <cuda_runtime.h>
- __global__ void __launch_bounds__(256, 4)
+ __global__ void __launch_bounds__(1024, 1)
grayscale_kernel(const float* __restrict__ input,
float* __restrict__ output,
const int n_quads) {
- const int tid = blockIdx.x * 256 + threadIdx.x;
+ const int tid = blockIdx.x * 1024 + threadIdx.x;
if (tid < n_quads) {
- const float4* __restrict__ in4 = reinterpret_cast<const float4*>(input + tid * 12);
- const float4 v0 = __ldg(in4);
- const float4 v1 = __ldg(in4 + 1);
- const float4 v2 = __ldg(in4 + 2);
+ const float4 v0 = __ldg((const float4*)(input + tid * 12));
+ const float4 v1 = __ldg((const float4*)(input + tid * 12 + 4));
+ const float4 v2 = __ldg((const float4*)(input + tid * 12 + 8));
float4 out;
out.x = __fmaf_rn(0.2989f, v0.x, __fmaf_rn(0.5870f, v0.y, 0.1140f * v0.z));
⋯ 1 unchanged lines
out.z = __fmaf_rn(0.2989f, v1.z, __fmaf_rn(0.5870f, v1.w, 0.1140f * v2.x));
out.w = __fmaf_rn(0.2989f, v2.y, __fmaf_rn(0.5870f, v2.z, 0.1140f * v2.w));
- reinterpret_cast<float4*>(output)[tid] = out;
+ ((float4*)output)[tid] = out;
}
}
torch::Tensor launch_grayscale(torch::Tensor input, torch::Tensor output) {
- const int n_pixels = input.size(0) * input.size(1);
- const int n_quads = n_pixels >> 2;
- const int blocks = (n_quads + 255) >> 8;
-
- grayscale_kernel<<<blocks, 256>>>(
+ const int n_quads = (input.size(0) * input.size(1)) >> 2;
+ grayscale_kernel<<<(n_quads + 1023) / 1024, 1024>>>(
input.data_ptr<float>(),
output.data_ptr<float>(),
n_quads
);
-
return output;
}
'''
⋯ 1 unchanged lines
cpp_src = 'torch::Tensor launch_grayscale(torch::Tensor input, torch::Tensor output);'
module = load_inline(
- name='grayscale_v4_final',
+ name='grayscale_v5_1024',
cpp_sources=cpp_src,
cuda_sources=cuda_src,
functions=['launch_grayscale'],
verbose=False,
- extra_cuda_cflags=['-O3', '--use_fast_math'],
+ extra_cuda_cflags=['-O3', '--use_fast_math', '-maxrregcount=28'],
)
_launch = module.launch_grayscale
scrolls · 63 diff lines total

Best evidence level for this revision: reported

JSON