Skip to content
KernelIndex
Search⌘K

submission 66622

Joao · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 126 lines, June 9 Researcher Reciprocity License v1.0.

solution5.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-66622?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp16

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
FP16 vector additionsuite of 5 cases
NVIDIA A100
896.0µs
#11= of 87
2025-11-05

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:aa714b4bbdd83fd956c26d22b709c1f5e5da569bf5811fea8b5e09d94935f8a5
license declaredunknown
license concludedunknown
authorsJoao
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

vector-width = float4__global__ void vectorAdd_float4(const float4* __restrict__ a, const float4* __restrict__ b, float4* __restrict__ c) {

Kernel source

solution5.py126 lines
#!POPCORN leaderboard vectoradd_v2

import torch
from torch.utils.cpp_extension import load_inline
from typing import List
from task import input_t, output_t

add_cuda_source = """

#include <cuda_fp16.h>

//__global__ void vectorAdd_float4(const float4*  __restrict__ a, const float4* __restrict__ b, float4* __restrict__ c, long long n_float8) {
__global__ void vectorAdd_float4(const float4*  __restrict__ a, const float4* __restrict__ b, float4* __restrict__ c) {
    
    // Calculate the global thread ID using a grid-stride loop
    //for (long long i = blockIdx.x * blockDim.x + threadIdx.x; 
    //     i < n_float8; 
    //     i += gridDim.x * blockDim.x) 
    //{
    int i = blockIdx.x * blockDim.x + threadIdx.x; 
    float4 a_vec = a[i];
    float4 b_vec = b[i];
    const half2* a_h = reinterpret_cast<const half2*>(&a_vec);
    const half2* b_h = reinterpret_cast<const half2*>(&b_vec);
    half2 c_h[4];
    c_h[0] = __hadd2(a_h[0], b_h[0]);
    c_h[1] = __hadd2(a_h[1], b_h[1]);
    c_h[2] = __hadd2(a_h[2], b_h[2]);
    c_h[3] = __hadd2(a_h[3], b_h[3]);
    c[i] = *reinterpret_cast<float4*>(c_h);
    //}
}

torch::Tensor add_cuda(torch::Tensor A, torch::Tensor B, torch::Tensor C) {
    int N = A.numel();  
    // N = 268435456
    // n_float8 = N/8 = 33554432
    const long long n_float8 = N / 8;

    const int threads = 256; 
    // blocks = 65536
    const int blocks = (n_float8 + threads - 1) / threads;

    vectorAdd_float4<<<blocks,threads>>>(
        reinterpret_cast<float4*>(A.data_ptr<at::Half>()),
        reinterpret_cast<float4*>(B.data_ptr<at::Half>()),
        reinterpret_cast<float4*>(C.data_ptr<at::Half>()));
//        n_float8);

    return C;
}
"""

add_cpp_source = """
#include <torch/extension.h>
#include <cuda_fp16.h>

torch::Tensor add_cuda(torch::Tensor A, torch::Tensor B, torch::Tensor C);
"""

add_module = load_inline(
    name='add_cuda',
    cpp_sources=add_cpp_source,
    cuda_sources=add_cuda_source,
    functions=['add_cuda'],
    verbose=True,
)


def print_gpu_properties():
    import sys
    """
    Prints key properties for all available CUDA-enabled GPUs.
    """
    if not torch.cuda.is_available():
        print("CUDA is not available on this system.", file=sys.stderr)
        return

    device_count = torch.cuda.device_count()
    print(f"Found {device_count} CUDA-enabled device(s).")
    print("-" * 70)

    for i in range(device_count):
        try:
            # Get the device properties
            props = torch.cuda.get_device_properties(i)

            print(f"--- Device {i}: {props.name} ---")

            # --- Core Architecture ---
            print("  Architecture:")
            print(f"    CUDA Compute Capability:   {props.major}.{props.minor}")
            print(f"    Streaming Multiprocessors (SMs): {props.multi_processor_count}")
            # Total cores = SMs * (cores/SM). Varies by architecture
            # (e.g., 128 for Ampere, 64 for Turing). This is a rough guide.

            # --- Memory ---
            print("  Memory:")
            # Convert bytes to Gigabytes (GiB)
            total_mem_gb = props.total_memory / (1024**3)
            print(f"    Total Global Memory:     {total_mem_gb:.2f} GiB")
            print(f"    Shared Memory per SM:    {props.shared_memory_per_multiprocessor / 1024:.0f} KiB")
            print(f"    Shared Memory per Block: {props.shared_memory_per_block / 1024:.0f} KiB")
            print(f"    L2 Cache Size:           {props.l2_cache_size / (1024**2):.0f} MiB")
            print(f"    Memory Bus Width:        {props.memory_bus_width}-bit")
            print(f"    Memory Clock Rate:       {props.memory_clock_rate / 1000:.2f} GHz")

            # --- Threading & Execution ---
            print("  Execution & Threading:")
            print(f"    Max Threads per Block:   {props.max_threads_per_block}")
            print(f"    Max Threads per SM:      {props.max_threads_per_multiprocessor}")
            print("    Max Block Dimensions (x,y,z): "
                  f"({props.max_grid_size[0]}, {props.max_grid_size[1]}, {props.max_grid_size[2]})")
            print("    Max Thread Dimensions (x,y,z): "
                  f"({props.max_threads_dim[0]}, {props.max_threads_dim[1]}, {props.max_threads_dim[2]})")
            print(f"    Warp Size:               {props.warp_size}")

            print("-" * 70)

        except Exception as e:
            print(f"Error getting properties for device {i}: {e}", file=sys.stderr)

def custom_kernel(data: input_t) -> output_t:
    #print_gpu_properties()
    return add_module.add_cuda(data[0], data[1], data[2])
scrolls · 126 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 66458.

⋯ 3 unchanged lines
from torch.utils.cpp_extension import load_inline
from typing import List
from task import input_t, output_t
- #import sys
add_cuda_source = """
#include <cuda_fp16.h>
- __global__ void vectorAdd_float4(const float4* __restrict__ a, const float4* __restrict__ b, float4* __restrict__ c, long long n_float8) {
+ //__global__ void vectorAdd_float4(const float4* __restrict__ a, const float4* __restrict__ b, float4* __restrict__ c, long long n_float8) {
+ __global__ void vectorAdd_float4(const float4* __restrict__ a, const float4* __restrict__ b, float4* __restrict__ c) {
// Calculate the global thread ID using a grid-stride loop
//for (long long i = blockIdx.x * blockDim.x + threadIdx.x;
⋯ 1 unchanged lines
// i += gridDim.x * blockDim.x)
//{
int i = blockIdx.x * blockDim.x + threadIdx.x;
- float4 a_vec = a[i];
- float4 b_vec = b[i];
- const half2* a_h = reinterpret_cast<const half2*>(&a_vec);
- const half2* b_h = reinterpret_cast<const half2*>(&b_vec);
- half2 c_h[4];
- c_h[0] = __hadd2(a_h[0], b_h[0]);
- c_h[1] = __hadd2(a_h[1], b_h[1]);
- c_h[2] = __hadd2(a_h[2], b_h[2]);
- c_h[3] = __hadd2(a_h[3], b_h[3]);
- c[i] = *reinterpret_cast<float4*>(c_h);
+ float4 a_vec = a[i];
+ float4 b_vec = b[i];
+ const half2* a_h = reinterpret_cast<const half2*>(&a_vec);
+ const half2* b_h = reinterpret_cast<const half2*>(&b_vec);
+ half2 c_h[4];
+ c_h[0] = __hadd2(a_h[0], b_h[0]);
+ c_h[1] = __hadd2(a_h[1], b_h[1]);
+ c_h[2] = __hadd2(a_h[2], b_h[2]);
+ c_h[3] = __hadd2(a_h[3], b_h[3]);
+ c[i] = *reinterpret_cast<float4*>(c_h);
//}
}
⋯ 10 unchanged lines
vectorAdd_float4<<<blocks,threads>>>(
reinterpret_cast<float4*>(A.data_ptr<at::Half>()),
reinterpret_cast<float4*>(B.data_ptr<at::Half>()),
- reinterpret_cast<float4*>(C.data_ptr<at::Half>()),
- n_float8);
+ reinterpret_cast<float4*>(C.data_ptr<at::Half>()));
+ // n_float8);
return C;
}
⋯ 14 unchanged lines
verbose=True,
)
+
+ def print_gpu_properties():
+ import sys
+ """
+ Prints key properties for all available CUDA-enabled GPUs.
+ """
+ if not torch.cuda.is_available():
+ print("CUDA is not available on this system.", file=sys.stderr)
+ return
+
+ device_count = torch.cuda.device_count()
+ print(f"Found {device_count} CUDA-enabled device(s).")
+ print("-" * 70)
+
+ for i in range(device_count):
+ try:
+ # Get the device properties
+ props = torch.cuda.get_device_properties(i)
+
+ print(f"--- Device {i}: {props.name} ---")
+
+ # --- Core Architecture ---
+ print(" Architecture:")
+ print(f" CUDA Compute Capability: {props.major}.{props.minor}")
+ print(f" Streaming Multiprocessors (SMs): {props.multi_processor_count}")
+ # Total cores = SMs * (cores/SM). Varies by architecture
+ # (e.g., 128 for Ampere, 64 for Turing). This is a rough guide.
+
+ # --- Memory ---
+ print(" Memory:")
+ # Convert bytes to Gigabytes (GiB)
+ total_mem_gb = props.total_memory / (1024**3)
+ print(f" Total Global Memory: {total_mem_gb:.2f} GiB")
+ print(f" Shared Memory per SM: {props.shared_memory_per_multiprocessor / 1024:.0f} KiB")
+ print(f" Shared Memory per Block: {props.shared_memory_per_block / 1024:.0f} KiB")
+ print(f" L2 Cache Size: {props.l2_cache_size / (1024**2):.0f} MiB")
+ print(f" Memory Bus Width: {props.memory_bus_width}-bit")
+ print(f" Memory Clock Rate: {props.memory_clock_rate / 1000:.2f} GHz")
+
+ # --- Threading & Execution ---
+ print(" Execution & Threading:")
+ print(f" Max Threads per Block: {props.max_threads_per_block}")
+ print(f" Max Threads per SM: {props.max_threads_per_multiprocessor}")
+ print(" Max Block Dimensions (x,y,z): "
+ f"({props.max_grid_size[0]}, {props.max_grid_size[1]}, {props.max_grid_size[2]})")
+ print(" Max Thread Dimensions (x,y,z): "
+ f"({props.max_threads_dim[0]}, {props.max_threads_dim[1]}, {props.max_threads_dim[2]})")
+ print(f" Warp Size: {props.warp_size}")
+
+ print("-" * 70)
+
+ except Exception as e:
+ print(f"Error getting properties for device {i}: {e}", file=sys.stderr)
+
def custom_kernel(data: input_t) -> output_t:
+ #print_gpu_properties()
return add_module.add_cuda(data[0], data[1], data[2])
scrolls · 115 diff lines total

Best evidence level for this revision: reported

JSON