Skip to content
KernelIndex
Search⌘K

submission 781386

thom.gg · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 95 lines, June 9 Researcher Reciprocity License v1.0.

grayscale_submission_v2.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-781386?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA A100
2.39ms
#8 of 137
2026-05-08

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:46098612a53ab6a45d0a13ec2282827262f2e1340e04dadab5e29188936754dc
license declaredunknown
license concludedunknown
authorsthom.gg
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

vector-width = float4const float4 * inputFloat4 = reinterpret_cast<const float4*>(rgb_image);

Kernel source

grayscale_submission_v2.py95 lines
import torch
from torch.utils.cpp_extension import load_inline
from typing import List
from task import input_t, output_t

convert_cuda_source = """
#define CEIL_DIV(a,b) (((a) + (b) - 1) / (b))
#define THREADS_PER_BLOCK 256

__global__ void grayscale(const float* __restrict__ rgb_image, float* __restrict__ grayscale_output) {
    int bX = blockIdx.x;
    int tX = threadIdx.x;

    int globalX = (bX * blockDim.x + tX);

    const float4 * inputFloat4 = reinterpret_cast<const float4*>(rgb_image);


    int baseIndex = globalX * 3; // 3 rgb values for each cell
    float4 l0 = inputFloat4[baseIndex + 0];
    float4 l1 = inputFloat4[baseIndex + 1];;
    float4 l2 = inputFloat4[baseIndex + 2];;


        
    float4 output;
    output.x = __fmaf_rn(0.299f, l0.x, __fmaf_rn(0.587f, l0.y, 0.114f * l0.z));
    output.y = __fmaf_rn(0.299f, l0.w, __fmaf_rn(0.587f, l1.x, 0.114f * l1.y));
    output.z = __fmaf_rn(0.299f, l1.z, __fmaf_rn(0.587f, l1.w, 0.114f * l2.x));
    output.w = __fmaf_rn(0.299f, l2.y, __fmaf_rn(0.587f, l2.z, 0.114f * l2.w));
        
    *reinterpret_cast<float4 *>(&grayscale_output[globalX*4]) = output;        
    
}


torch::Tensor convert_cuda(torch::Tensor input, torch::Tensor output) {
    TORCH_CHECK(input.device().is_cuda(), "Tensor input must be a CUDA tensor");
    TORCH_CHECK(output.device().is_cuda(), "Tensor output must be a CUDA tensor");
    
    auto sizes = input.sizes(); // retourne IntArrayRef
    int64_t height = sizes[0];
    int64_t width = sizes[1];

   int gDim = CEIL_DIV(height*width, THREADS_PER_BLOCK*4);

    int nbBlocks = gDim;
    grayscale<<<nbBlocks, THREADS_PER_BLOCK>>>(input.data_ptr<float>(), output.data_ptr<float>());


    

    cudaError_t err = cudaGetLastError();
    if (err != cudaSuccess) {
        throw std::runtime_error(cudaGetErrorString(err));
    }

    return output;
}
"""

convert_cpp_source = """
#include <torch/extension.h>

torch::Tensor convert_cuda(torch::Tensor input, torch::Tensor output);
"""

convert_module = load_inline(
    name='convert_cuda',
    cpp_sources=convert_cpp_source,
    cuda_sources=convert_cuda_source,
    functions=['convert_cuda'],
    verbose=True,
)

def convert(input,output):
    if not input.is_cuda or not output.is_cuda:
        raise RuntimeError("Tensor must be on GPU")
    return convert_module.convert_cuda(input,output)

def custom_kernel(data: input_t) -> output_t:
    """
    Custom implementation of vector sum reduction using CUDA.
    Args:
        inputs: List of pairs of tensors [A, B] to be added.
    Returns:
        Tensor containing element-wise sum.
    """
    input, output = data
    assert input.is_cuda, "Input tensor must be on GPU"

    # Simply reuse the existing add function we already defined
    # This avoids the compilation issues with the inline kernel
    res =  convert(input, output)
    return res
scrolls · 95 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 781373.

⋯ 3 unchanged lines
from task import input_t, output_t
convert_cuda_source = """
-
#define CEIL_DIV(a,b) (((a) + (b) - 1) / (b))
+ #define THREADS_PER_BLOCK 256
- __global__ void grayscale(const float* rgb_image, float* grayscale_output, size_t height, size_t width, int cellsPerThread) {
+ __global__ void grayscale(const float* __restrict__ rgb_image, float* __restrict__ grayscale_output) {
int bX = blockIdx.x;
- int bY = blockIdx.y;
int tX = threadIdx.x;
- int tY = threadIdx.y;
- int xCoord = bX * blockDim.x + tX;
- int yCoord = (bY * blockDim.y + tY) * cellsPerThread;
+ int globalX = (bX * blockDim.x + tX);
+ const float4 * inputFloat4 = reinterpret_cast<const float4*>(rgb_image);
- for (int cell = 0; cell<cellsPerThread; cell+=4) {
- float4 l0 = *reinterpret_cast<const float4 *>(&rgb_image[xCoord * width * 3 + yCoord * 3 + 0*4]);
- float4 l1 = *reinterpret_cast<const float4 *>(&rgb_image[xCoord * width * 3 + yCoord * 3 + 1*4]);
- float4 l2 = *reinterpret_cast<const float4 *>(&rgb_image[xCoord * width * 3 + yCoord * 3 + 2*4]);
- float r0 = l0.x; float g0 = l0.y; float b0 = l0.z;
- float r1 = l0.w; float g1 = l1.x; float b1 = l1.y;
- float r2 = l1.z; float g2 = l1.w; float b2 = l2.x;
- float r3 = l2.y; float g3 = l2.z; float b3 = l2.w;
-
- float4 output = make_float4(
- 0.299f * r0 + 0.587f * g0 + 0.114f * b0,
- 0.299f * r1 + 0.587f * g1 + 0.114f * b1,
- 0.299f * r2 + 0.587f * g2 + 0.114f * b2,
- 0.299f * r3 + 0.587f * g3 + 0.114f * b3
- );
+ int baseIndex = globalX * 3; // 3 rgb values for each cell
+ float4 l0 = inputFloat4[baseIndex + 0];
+ float4 l1 = inputFloat4[baseIndex + 1];;
+ float4 l2 = inputFloat4[baseIndex + 2];;
+
- *reinterpret_cast<float4 *>(&grayscale_output[xCoord * width + yCoord + cell]) = output;
- }
+ float4 output;
+ output.x = __fmaf_rn(0.299f, l0.x, __fmaf_rn(0.587f, l0.y, 0.114f * l0.z));
+ output.y = __fmaf_rn(0.299f, l0.w, __fmaf_rn(0.587f, l1.x, 0.114f * l1.y));
+ output.z = __fmaf_rn(0.299f, l1.z, __fmaf_rn(0.587f, l1.w, 0.114f * l2.x));
+ output.w = __fmaf_rn(0.299f, l2.y, __fmaf_rn(0.587f, l2.z, 0.114f * l2.w));
+
+ *reinterpret_cast<float4 *>(&grayscale_output[globalX*4]) = output;
+
}
⋯ 5 unchanged lines
int64_t height = sizes[0];
int64_t width = sizes[1];
- int blockSize = 8;
- int cellsPerThread = 4;
- int bDimX = CEIL_DIV(height, blockSize);
- int bDimY = CEIL_DIV(width, blockSize*cellsPerThread);
- dim3 threadsPerBlock(blockSize,blockSize);
- dim3 nbBlocks( bDimX, bDimY);
+ int gDim = CEIL_DIV(height*width, THREADS_PER_BLOCK*4);
- grayscale<<<nbBlocks, threadsPerBlock>>>(input.data_ptr<float>(), output.data_ptr<float>(), height, width, cellsPerThread);
+ int nbBlocks = gDim;
+ grayscale<<<nbBlocks, THREADS_PER_BLOCK>>>(input.data_ptr<float>(), output.data_ptr<float>());
scrolls · 75 diff lines total

Best evidence level for this revision: reported

JSON