submission 66622
Joao · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 126 lines, June 9 Researcher Reciprocity License v1.0.
solution5.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-66622?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp16
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:aa714b4bbdd83fd956c26d22b709c1f5e5da569bf5811fea8b5e09d94935f8a5
license declaredunknown
license concludedunknown
authorsJoao
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
vector-width = float4
__global__ void vectorAdd_float4(const float4* __restrict__ a, const float4* __restrict__ b, float4* __restrict__ c) {Kernel source
solution5.py126 lines
#!POPCORN leaderboard vectoradd_v2
import torch
from torch.utils.cpp_extension import load_inline
from typing import List
from task import input_t, output_t
add_cuda_source = """
#include <cuda_fp16.h>
//__global__ void vectorAdd_float4(const float4* __restrict__ a, const float4* __restrict__ b, float4* __restrict__ c, long long n_float8) {
__global__ void vectorAdd_float4(const float4* __restrict__ a, const float4* __restrict__ b, float4* __restrict__ c) {
// Calculate the global thread ID using a grid-stride loop
//for (long long i = blockIdx.x * blockDim.x + threadIdx.x;
// i < n_float8;
// i += gridDim.x * blockDim.x)
//{
int i = blockIdx.x * blockDim.x + threadIdx.x;
float4 a_vec = a[i];
float4 b_vec = b[i];
const half2* a_h = reinterpret_cast<const half2*>(&a_vec);
const half2* b_h = reinterpret_cast<const half2*>(&b_vec);
half2 c_h[4];
c_h[0] = __hadd2(a_h[0], b_h[0]);
c_h[1] = __hadd2(a_h[1], b_h[1]);
c_h[2] = __hadd2(a_h[2], b_h[2]);
c_h[3] = __hadd2(a_h[3], b_h[3]);
c[i] = *reinterpret_cast<float4*>(c_h);
//}
}
torch::Tensor add_cuda(torch::Tensor A, torch::Tensor B, torch::Tensor C) {
int N = A.numel();
// N = 268435456
// n_float8 = N/8 = 33554432
const long long n_float8 = N / 8;
const int threads = 256;
// blocks = 65536
const int blocks = (n_float8 + threads - 1) / threads;
vectorAdd_float4<<<blocks,threads>>>(
reinterpret_cast<float4*>(A.data_ptr<at::Half>()),
reinterpret_cast<float4*>(B.data_ptr<at::Half>()),
reinterpret_cast<float4*>(C.data_ptr<at::Half>()));
// n_float8);
return C;
}
"""
add_cpp_source = """
#include <torch/extension.h>
#include <cuda_fp16.h>
torch::Tensor add_cuda(torch::Tensor A, torch::Tensor B, torch::Tensor C);
"""
add_module = load_inline(
name='add_cuda',
cpp_sources=add_cpp_source,
cuda_sources=add_cuda_source,
functions=['add_cuda'],
verbose=True,
)
def print_gpu_properties():
import sys
"""
Prints key properties for all available CUDA-enabled GPUs.
"""
if not torch.cuda.is_available():
print("CUDA is not available on this system.", file=sys.stderr)
return
device_count = torch.cuda.device_count()
print(f"Found {device_count} CUDA-enabled device(s).")
print("-" * 70)
for i in range(device_count):
try:
# Get the device properties
props = torch.cuda.get_device_properties(i)
print(f"--- Device {i}: {props.name} ---")
# --- Core Architecture ---
print(" Architecture:")
print(f" CUDA Compute Capability: {props.major}.{props.minor}")
print(f" Streaming Multiprocessors (SMs): {props.multi_processor_count}")
# Total cores = SMs * (cores/SM). Varies by architecture
# (e.g., 128 for Ampere, 64 for Turing). This is a rough guide.
# --- Memory ---
print(" Memory:")
# Convert bytes to Gigabytes (GiB)
total_mem_gb = props.total_memory / (1024**3)
print(f" Total Global Memory: {total_mem_gb:.2f} GiB")
print(f" Shared Memory per SM: {props.shared_memory_per_multiprocessor / 1024:.0f} KiB")
print(f" Shared Memory per Block: {props.shared_memory_per_block / 1024:.0f} KiB")
print(f" L2 Cache Size: {props.l2_cache_size / (1024**2):.0f} MiB")
print(f" Memory Bus Width: {props.memory_bus_width}-bit")
print(f" Memory Clock Rate: {props.memory_clock_rate / 1000:.2f} GHz")
# --- Threading & Execution ---
print(" Execution & Threading:")
print(f" Max Threads per Block: {props.max_threads_per_block}")
print(f" Max Threads per SM: {props.max_threads_per_multiprocessor}")
print(" Max Block Dimensions (x,y,z): "
f"({props.max_grid_size[0]}, {props.max_grid_size[1]}, {props.max_grid_size[2]})")
print(" Max Thread Dimensions (x,y,z): "
f"({props.max_threads_dim[0]}, {props.max_threads_dim[1]}, {props.max_threads_dim[2]})")
print(f" Warp Size: {props.warp_size}")
print("-" * 70)
except Exception as e:
print(f"Error getting properties for device {i}: {e}", file=sys.stderr)
def custom_kernel(data: input_t) -> output_t:
#print_gpu_properties()
return add_module.add_cuda(data[0], data[1], data[2])
scrolls · 126 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 66458.
⋯ 3 unchanged linesfrom torch.utils.cpp_extension import load_inlinefrom typing import Listfrom task import input_t, output_t- #import sysadd_cuda_source = """#include <cuda_fp16.h>- __global__ void vectorAdd_float4(const float4* __restrict__ a, const float4* __restrict__ b, float4* __restrict__ c, long long n_float8) {+ //__global__ void vectorAdd_float4(const float4* __restrict__ a, const float4* __restrict__ b, float4* __restrict__ c, long long n_float8) {+ __global__ void vectorAdd_float4(const float4* __restrict__ a, const float4* __restrict__ b, float4* __restrict__ c) {// Calculate the global thread ID using a grid-stride loop//for (long long i = blockIdx.x * blockDim.x + threadIdx.x;⋯ 1 unchanged lines// i += gridDim.x * blockDim.x)//{int i = blockIdx.x * blockDim.x + threadIdx.x;- float4 a_vec = a[i];- float4 b_vec = b[i];- const half2* a_h = reinterpret_cast<const half2*>(&a_vec);- const half2* b_h = reinterpret_cast<const half2*>(&b_vec);- half2 c_h[4];- c_h[0] = __hadd2(a_h[0], b_h[0]);- c_h[1] = __hadd2(a_h[1], b_h[1]);- c_h[2] = __hadd2(a_h[2], b_h[2]);- c_h[3] = __hadd2(a_h[3], b_h[3]);- c[i] = *reinterpret_cast<float4*>(c_h);+ float4 a_vec = a[i];+ float4 b_vec = b[i];+ const half2* a_h = reinterpret_cast<const half2*>(&a_vec);+ const half2* b_h = reinterpret_cast<const half2*>(&b_vec);+ half2 c_h[4];+ c_h[0] = __hadd2(a_h[0], b_h[0]);+ c_h[1] = __hadd2(a_h[1], b_h[1]);+ c_h[2] = __hadd2(a_h[2], b_h[2]);+ c_h[3] = __hadd2(a_h[3], b_h[3]);+ c[i] = *reinterpret_cast<float4*>(c_h);//}}⋯ 10 unchanged linesvectorAdd_float4<<<blocks,threads>>>(reinterpret_cast<float4*>(A.data_ptr<at::Half>()),reinterpret_cast<float4*>(B.data_ptr<at::Half>()),- reinterpret_cast<float4*>(C.data_ptr<at::Half>()),- n_float8);+ reinterpret_cast<float4*>(C.data_ptr<at::Half>()));+ // n_float8);return C;}⋯ 14 unchanged linesverbose=True,)++ def print_gpu_properties():+ import sys+ """+ Prints key properties for all available CUDA-enabled GPUs.+ """+ if not torch.cuda.is_available():+ print("CUDA is not available on this system.", file=sys.stderr)+ return++ device_count = torch.cuda.device_count()+ print(f"Found {device_count} CUDA-enabled device(s).")+ print("-" * 70)++ for i in range(device_count):+ try:+ # Get the device properties+ props = torch.cuda.get_device_properties(i)++ print(f"--- Device {i}: {props.name} ---")++ # --- Core Architecture ---+ print(" Architecture:")+ print(f" CUDA Compute Capability: {props.major}.{props.minor}")+ print(f" Streaming Multiprocessors (SMs): {props.multi_processor_count}")+ # Total cores = SMs * (cores/SM). Varies by architecture+ # (e.g., 128 for Ampere, 64 for Turing). This is a rough guide.++ # --- Memory ---+ print(" Memory:")+ # Convert bytes to Gigabytes (GiB)+ total_mem_gb = props.total_memory / (1024**3)+ print(f" Total Global Memory: {total_mem_gb:.2f} GiB")+ print(f" Shared Memory per SM: {props.shared_memory_per_multiprocessor / 1024:.0f} KiB")+ print(f" Shared Memory per Block: {props.shared_memory_per_block / 1024:.0f} KiB")+ print(f" L2 Cache Size: {props.l2_cache_size / (1024**2):.0f} MiB")+ print(f" Memory Bus Width: {props.memory_bus_width}-bit")+ print(f" Memory Clock Rate: {props.memory_clock_rate / 1000:.2f} GHz")++ # --- Threading & Execution ---+ print(" Execution & Threading:")+ print(f" Max Threads per Block: {props.max_threads_per_block}")+ print(f" Max Threads per SM: {props.max_threads_per_multiprocessor}")+ print(" Max Block Dimensions (x,y,z): "+ f"({props.max_grid_size[0]}, {props.max_grid_size[1]}, {props.max_grid_size[2]})")+ print(" Max Thread Dimensions (x,y,z): "+ f"({props.max_threads_dim[0]}, {props.max_threads_dim[1]}, {props.max_threads_dim[2]})")+ print(f" Warp Size: {props.warp_size}")++ print("-" * 70)++ except Exception as e:+ print(f"Error getting properties for device {i}: {e}", file=sys.stderr)+def custom_kernel(data: input_t) -> output_t:+ #print_gpu_properties()return add_module.add_cuda(data[0], data[1], data[2])
scrolls · 115 diff lines total
Best evidence level for this revision: reported
JSON