Skip to content
KernelIndex
Search⌘K

submission 780739

thom.gg · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 130 lines, June 9 Researcher Reciprocity License v1.0.

grayscale_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-780739?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA A100
4.29ms
#43 of 137
2026-05-06

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:6632eae3b156b7b3297da328b29e3a6fc143b0557b79a220d276c5494dd9dfb7
license declaredunknown
license concludedunknown
authorsthom.gg
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

vector-width = float4float4 rgbVals, newVals;

Kernel source

grayscale_submission.py130 lines
import torch
from torch.utils.cpp_extension import load_inline
from typing import List
from task import input_t, output_t

convert_cuda_source = """

#define CEIL_DIV(a,b) (((a) + (b) - 1) / (b))

__global__ void grayscale(const float* rgb_image, float* grayscale_output, size_t height, size_t width, int cellsPerThread) {
    int bX = blockIdx.x;
    int bY = blockIdx.y;
    int tX = threadIdx.x;
    int tY = threadIdx.y;

    int xCoord = bX * blockDim.x + tX;
    int yCoord = (bY * blockDim.y + tY) * cellsPerThread;

    float4 rgbVals, newVals;
    float4 output;
    for (int cell = 0; cell<cellsPerThread; cell++) {
        float r,g,b;
        int modulo = cell % 4;
        int yOffset = modulo * 4;
        if (modulo < 3)
            newVals = *reinterpret_cast<const float4 *>(&rgb_image[xCoord * width * 3 + yCoord * 3 + yOffset ]);
        if (modulo == 0) {
            // nothing interesting from previous iterqtions
            r=newVals.x;
            g=newVals.y;
            b=newVals.z;
        }
        else if (modulo == 1) {
            r = rgbVals.w; // last val from previous read
            g = newVals.x;
            b = newVals.y;
        }
        else if (modulo == 2) {
            r = rgbVals.z;
            g = rgbVals.w;
            b = newVals.x;
        }
        else if (modulo == 3) {
            // taking everything from, previous read, there is no current read
            r = rgbVals.y;
            g = rgbVals.z;
            b = rgbVals.w;
        }

        rgbVals = newVals;


        // Computing gray scale
        float gray = 0.299f * r + 0.587f * g + 0.114f * b;

        if (modulo == 0) {output.x = gray;}
        else if (modulo == 1) {output.y = gray;}
        else if (modulo == 2) {output.z = gray;}
        else if (modulo == 3) {output.w = gray;
            *reinterpret_cast<float4 *>(&grayscale_output[xCoord * width + yCoord + cell / 4]) = output;
        }
        // if (xCoord < height && yCoord+cell < width)
        // grayscale_output[xCoord*width + yCoord + cell] = gray;
    }
}


torch::Tensor convert_cuda(torch::Tensor input, torch::Tensor output) {
    TORCH_CHECK(input.device().is_cuda(), "Tensor input must be a CUDA tensor");
    TORCH_CHECK(output.device().is_cuda(), "Tensor output must be a CUDA tensor");
    
    auto sizes = input.sizes(); // retourne IntArrayRef
    int64_t height = sizes[0];
    int64_t width = sizes[1];

    int blockSize = 32;
    int cellsPerThread = 4;
    int bDimX = CEIL_DIV(height, blockSize);
    int bDimY = CEIL_DIV(width, blockSize*cellsPerThread);
    dim3 threadsPerBlock(blockSize,blockSize);
    dim3 nbBlocks( bDimX, bDimY);

    grayscale<<<nbBlocks, threadsPerBlock>>>(input.data_ptr<float>(), output.data_ptr<float>(), height, width, cellsPerThread);


    

    cudaError_t err = cudaGetLastError();
    if (err != cudaSuccess) {
        throw std::runtime_error(cudaGetErrorString(err));
    }

    return output;
}
"""

convert_cpp_source = """
#include <torch/extension.h>

torch::Tensor convert_cuda(torch::Tensor input, torch::Tensor output);
"""

convert_module = load_inline(
    name='convert_cuda',
    cpp_sources=convert_cpp_source,
    cuda_sources=convert_cuda_source,
    functions=['convert_cuda'],
    verbose=True,
)

def convert(input,output):
    if not input.is_cuda or not output.is_cuda:
        raise RuntimeError("Tensor must be on GPU")
    return convert_module.convert_cuda(input,output)

def custom_kernel(data: input_t) -> output_t:
    """
    Custom implementation of vector sum reduction using CUDA.
    Args:
        inputs: List of pairs of tensors [A, B] to be added.
    Returns:
        Tensor containing element-wise sum.
    """
    input, output = data
    assert input.is_cuda, "Input tensor must be on GPU"

    # Simply reuse the existing add function we already defined
    # This avoids the compilation issues with the inline kernel
    res =  convert(input, output)
    return res
scrolls · 130 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 780738.

⋯ 6 unchanged lines
#define CEIL_DIV(a,b) (((a) + (b) - 1) / (b))
-
__global__ void grayscale(const float* rgb_image, float* grayscale_output, size_t height, size_t width, int cellsPerThread) {
int bX = blockIdx.x;
int bY = blockIdx.y;
⋯ 4 unchanged lines
int yCoord = (bY * blockDim.y + tY) * cellsPerThread;
float4 rgbVals, newVals;
+ float4 output;
for (int cell = 0; cell<cellsPerThread; cell++) {
float r,g,b;
int modulo = cell % 4;
⋯ 27 unchanged lines
// Computing gray scale
- float gray = 0.299 * r + 0.587 * g + 0.114 * b;
- if (xCoord < height && yCoord+cell < width)
- grayscale_output[xCoord*width + yCoord + cell] = gray;
+ float gray = 0.299f * r + 0.587f * g + 0.114f * b;
+
+ if (modulo == 0) {output.x = gray;}
+ else if (modulo == 1) {output.y = gray;}
+ else if (modulo == 2) {output.z = gray;}
+ else if (modulo == 3) {output.w = gray;
+ *reinterpret_cast<float4 *>(&grayscale_output[xCoord * width + yCoord + cell / 4]) = output;
+ }
+ // if (xCoord < height && yCoord+cell < width)
+ // grayscale_output[xCoord*width + yCoord + cell] = gray;
}
}
-
torch::Tensor convert_cuda(torch::Tensor input, torch::Tensor output) {
TORCH_CHECK(input.device().is_cuda(), "Tensor input must be a CUDA tensor");
TORCH_CHECK(output.device().is_cuda(), "Tensor output must be a CUDA tensor");
scrolls · 41 diff lines total

Best evidence level for this revision: reported

JSON