submission 66627
Joao · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 127 lines, June 9 Researcher Reciprocity License v1.0.
solution5.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-66627?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp16
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:d10390d2fafef5db83d3fe22f80a8a24aa991104484aa27131f6b19f07aa7fcb
license declaredunknown
license concludedunknown
authorsJoao
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
vector-width = float4
__global__ void vectorAdd_float4(const float4* __restrict__ a, const float4* __restrict__ b, float4* __restrict__ c) {Kernel source
solution5.py127 lines
#!POPCORN leaderboard vectoradd_v2
import torch
from torch.utils.cpp_extension import load_inline
from typing import List
from task import input_t, output_t
add_cuda_source = """
#include <cuda_fp16.h>
//__global__ void vectorAdd_float4(const float4* __restrict__ a, const float4* __restrict__ b, float4* __restrict__ c, long long n_float8) {
__global__ void vectorAdd_float4(const float4* __restrict__ a, const float4* __restrict__ b, float4* __restrict__ c) {
// Calculate the global thread ID using a grid-stride loop
//for (long long i = blockIdx.x * blockDim.x + threadIdx.x;
// i < n_float8;
// i += gridDim.x * blockDim.x)
//{
int i = blockIdx.x * blockDim.x + threadIdx.x;
float4 a_vec = a[i];
float4 b_vec = b[i];
const half2* a_h = reinterpret_cast<const half2*>(&a_vec);
const half2* b_h = reinterpret_cast<const half2*>(&b_vec);
half2 c_h[4];
c_h[0] = __hadd2(a_h[0], b_h[0]);
c_h[1] = __hadd2(a_h[1], b_h[1]);
c_h[2] = __hadd2(a_h[2], b_h[2]);
c_h[3] = __hadd2(a_h[3], b_h[3]);
c[i] = *reinterpret_cast<float4*>(c_h);
//}
}
torch::Tensor add_cuda(torch::Tensor A, torch::Tensor B, torch::Tensor C) {
int N = A.numel();
//// N = 268435456
//// n_float8 = N/8 = 33554432
//const int N = 268435456;
const long long n_float8 = N / 8;
const int threads = 256;
// blocks = 65536
const int blocks = (n_float8 + threads - 1) / threads;
vectorAdd_float4<<<blocks,threads>>>(
reinterpret_cast<float4*>(A.data_ptr<at::Half>()),
reinterpret_cast<float4*>(B.data_ptr<at::Half>()),
reinterpret_cast<float4*>(C.data_ptr<at::Half>()));
// n_float8);
return C;
}
"""
add_cpp_source = """
#include <torch/extension.h>
#include <cuda_fp16.h>
torch::Tensor add_cuda(torch::Tensor A, torch::Tensor B, torch::Tensor C);
"""
add_module = load_inline(
name='add_cuda',
cpp_sources=add_cpp_source,
cuda_sources=add_cuda_source,
functions=['add_cuda'],
verbose=False,
)
def print_gpu_properties():
import sys
"""
Prints key properties for all available CUDA-enabled GPUs.
"""
if not torch.cuda.is_available():
print("CUDA is not available on this system.", file=sys.stderr)
return
device_count = torch.cuda.device_count()
print(f"Found {device_count} CUDA-enabled device(s).")
print("-" * 70)
for i in range(device_count):
try:
# Get the device properties
props = torch.cuda.get_device_properties(i)
print(f"--- Device {i}: {props.name} ---")
# --- Core Architecture ---
print(" Architecture:")
print(f" CUDA Compute Capability: {props.major}.{props.minor}")
print(f" Streaming Multiprocessors (SMs): {props.multi_processor_count}")
# Total cores = SMs * (cores/SM). Varies by architecture
# (e.g., 128 for Ampere, 64 for Turing). This is a rough guide.
# --- Memory ---
print(" Memory:")
# Convert bytes to Gigabytes (GiB)
total_mem_gb = props.total_memory / (1024**3)
print(f" Total Global Memory: {total_mem_gb:.2f} GiB")
print(f" Shared Memory per SM: {props.shared_memory_per_multiprocessor / 1024:.0f} KiB")
print(f" Shared Memory per Block: {props.shared_memory_per_block / 1024:.0f} KiB")
print(f" L2 Cache Size: {props.l2_cache_size / (1024**2):.0f} MiB")
print(f" Memory Bus Width: {props.memory_bus_width}-bit")
print(f" Memory Clock Rate: {props.memory_clock_rate / 1000:.2f} GHz")
# --- Threading & Execution ---
print(" Execution & Threading:")
print(f" Max Threads per Block: {props.max_threads_per_block}")
print(f" Max Threads per SM: {props.max_threads_per_multiprocessor}")
print(" Max Block Dimensions (x,y,z): "
f"({props.max_grid_size[0]}, {props.max_grid_size[1]}, {props.max_grid_size[2]})")
print(" Max Thread Dimensions (x,y,z): "
f"({props.max_threads_dim[0]}, {props.max_threads_dim[1]}, {props.max_threads_dim[2]})")
print(f" Warp Size: {props.warp_size}")
print("-" * 70)
except Exception as e:
print(f"Error getting properties for device {i}: {e}", file=sys.stderr)
def custom_kernel(data: input_t) -> output_t:
#print_gpu_properties()
return add_module.add_cuda(data[0], data[1], data[2])
scrolls · 127 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 66622.
⋯ 32 unchanged linestorch::Tensor add_cuda(torch::Tensor A, torch::Tensor B, torch::Tensor C) {int N = A.numel();- // N = 268435456- // n_float8 = N/8 = 33554432+ //// N = 268435456+ //// n_float8 = N/8 = 33554432+ //const int N = 268435456;const long long n_float8 = N / 8;const int threads = 256;⋯ 22 unchanged linescpp_sources=add_cpp_source,cuda_sources=add_cuda_source,functions=['add_cuda'],- verbose=True,+ verbose=False,)
Best evidence level for this revision: reported
JSON