submission 781371
thom.gg · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 105 lines, June 9 Researcher Reciprocity License v1.0.
grayscale_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-781371?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:3d55487e5315ba2b89ac2455911a5a7c2362170f85cde50f7c95b43de4adaf85
license declaredunknown
license concludedunknown
authorsthom.gg
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
vector-width = float4
float4 l0 = *reinterpret_cast<const float4 *>(&rgb_image[xCoord * width * 3 + yCoord * 3 + 0*4]);Kernel source
grayscale_submission.py105 lines
import torch
from torch.utils.cpp_extension import load_inline
from typing import List
from task import input_t, output_t
convert_cuda_source = """
#define CEIL_DIV(a,b) (((a) + (b) - 1) / (b))
__global__ void grayscale(const float* rgb_image, float* grayscale_output, size_t height, size_t width, int cellsPerThread) {
int bX = blockIdx.x;
int bY = blockIdx.y;
int tX = threadIdx.x;
int tY = threadIdx.y;
int xCoord = bX * blockDim.x + tX;
int yCoord = (bY * blockDim.y + tY) * cellsPerThread;
for (int cell = 0; cell<cellsPerThread; cell+=4) {
float4 l0 = *reinterpret_cast<const float4 *>(&rgb_image[xCoord * width * 3 + yCoord * 3 + 0*4]);
float4 l1 = *reinterpret_cast<const float4 *>(&rgb_image[xCoord * width * 3 + yCoord * 3 + 1*4]);
float4 l2 = *reinterpret_cast<const float4 *>(&rgb_image[xCoord * width * 3 + yCoord * 3 + 2*4]);
float r0 = l0.x; float g0 = l0.y; float b0 = l0.z;
float r1 = l0.w; float g1 = l1.x; float b1 = l1.y;
float r2 = l1.z; float g2 = l1.w; float b2 = l2.x;
float r3 = l2.y; float g3 = l2.z; float b3 = l2.w;
float4 output = make_float4(
0.299f * r0 + 0.587f * g0 + 0.114f * b0,
0.299f * r1 + 0.587f * g1 + 0.114f * b1,
0.299f * r2 + 0.587f * g2 + 0.114f * b2,
0.299f * r3 + 0.587f * g3 + 0.114f * b3
);
*reinterpret_cast<float4 *>(&grayscale_output[xCoord * width + yCoord + cell]) = output;
}
}
torch::Tensor convert_cuda(torch::Tensor input, torch::Tensor output) {
TORCH_CHECK(input.device().is_cuda(), "Tensor input must be a CUDA tensor");
TORCH_CHECK(output.device().is_cuda(), "Tensor output must be a CUDA tensor");
auto sizes = input.sizes(); // retourne IntArrayRef
int64_t height = sizes[0];
int64_t width = sizes[1];
int blockSize = 16;
int cellsPerThread = 4;
int bDimX = CEIL_DIV(height, blockSize);
int bDimY = CEIL_DIV(width, blockSize*cellsPerThread);
dim3 threadsPerBlock(blockSize,blockSize);
dim3 nbBlocks( bDimX, bDimY);
grayscale<<<nbBlocks, threadsPerBlock>>>(input.data_ptr<float>(), output.data_ptr<float>(), height, width, cellsPerThread);
cudaError_t err = cudaGetLastError();
if (err != cudaSuccess) {
throw std::runtime_error(cudaGetErrorString(err));
}
return output;
}
"""
convert_cpp_source = """
#include <torch/extension.h>
torch::Tensor convert_cuda(torch::Tensor input, torch::Tensor output);
"""
convert_module = load_inline(
name='convert_cuda',
cpp_sources=convert_cpp_source,
cuda_sources=convert_cuda_source,
functions=['convert_cuda'],
verbose=True,
)
def convert(input,output):
if not input.is_cuda or not output.is_cuda:
raise RuntimeError("Tensor must be on GPU")
return convert_module.convert_cuda(input,output)
def custom_kernel(data: input_t) -> output_t:
"""
Custom implementation of vector sum reduction using CUDA.
Args:
inputs: List of pairs of tensors [A, B] to be added.
Returns:
Tensor containing element-wise sum.
"""
input, output = data
assert input.is_cuda, "Input tensor must be on GPU"
# Simply reuse the existing add function we already defined
# This avoids the compilation issues with the inline kernel
res = convert(input, output)
return resscrolls · 105 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 780739.
⋯ 15 unchanged linesint xCoord = bX * blockDim.x + tX;int yCoord = (bY * blockDim.y + tY) * cellsPerThread;- float4 rgbVals, newVals;- float4 output;- for (int cell = 0; cell<cellsPerThread; cell++) {- float r,g,b;- int modulo = cell % 4;- int yOffset = modulo * 4;- if (modulo < 3)- newVals = *reinterpret_cast<const float4 *>(&rgb_image[xCoord * width * 3 + yCoord * 3 + yOffset ]);- if (modulo == 0) {- // nothing interesting from previous iterqtions- r=newVals.x;- g=newVals.y;- b=newVals.z;- }- else if (modulo == 1) {- r = rgbVals.w; // last val from previous read- g = newVals.x;- b = newVals.y;- }- else if (modulo == 2) {- r = rgbVals.z;- g = rgbVals.w;- b = newVals.x;- }- else if (modulo == 3) {- // taking everything from, previous read, there is no current read- r = rgbVals.y;- g = rgbVals.z;- b = rgbVals.w;- }- rgbVals = newVals;+ for (int cell = 0; cell<cellsPerThread; cell+=4) {+ float4 l0 = *reinterpret_cast<const float4 *>(&rgb_image[xCoord * width * 3 + yCoord * 3 + 0*4]);+ float4 l1 = *reinterpret_cast<const float4 *>(&rgb_image[xCoord * width * 3 + yCoord * 3 + 1*4]);+ float4 l2 = *reinterpret_cast<const float4 *>(&rgb_image[xCoord * width * 3 + yCoord * 3 + 2*4]);+ float r0 = l0.x; float g0 = l0.y; float b0 = l0.z;+ float r1 = l0.w; float g1 = l1.x; float b1 = l1.y;+ float r2 = l1.z; float g2 = l1.w; float b2 = l2.x;+ float r3 = l2.y; float g3 = l2.z; float b3 = l2.w;++ float4 output = make_float4(+ 0.299f * r0 + 0.587f * g0 + 0.114f * b0,+ 0.299f * r1 + 0.587f * g1 + 0.114f * b1,+ 0.299f * r2 + 0.587f * g2 + 0.114f * b2,+ 0.299f * r3 + 0.587f * g3 + 0.114f * b3+ );- // Computing gray scale- float gray = 0.299f * r + 0.587f * g + 0.114f * b;-- if (modulo == 0) {output.x = gray;}- else if (modulo == 1) {output.y = gray;}- else if (modulo == 2) {output.z = gray;}- else if (modulo == 3) {output.w = gray;- *reinterpret_cast<float4 *>(&grayscale_output[xCoord * width + yCoord + cell / 4]) = output;- }- // if (xCoord < height && yCoord+cell < width)- // grayscale_output[xCoord*width + yCoord + cell] = gray;++ *reinterpret_cast<float4 *>(&grayscale_output[xCoord * width + yCoord + cell]) = output;}}⋯ 6 unchanged linesint64_t height = sizes[0];int64_t width = sizes[1];- int blockSize = 32;+ int blockSize = 16;int cellsPerThread = 4;int bDimX = CEIL_DIV(height, blockSize);int bDimY = CEIL_DIV(width, blockSize*cellsPerThread);
scrolls · 78 diff lines total
Best evidence level for this revision: reported
JSON