Skip to content
KernelIndex
Search⌘K

submission 66627

Joao · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 127 lines, June 9 Researcher Reciprocity License v1.0.

solution5.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-66627?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp16

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
FP16 vector additionsuite of 5 cases
NVIDIA B200
237.0µs
#31 of 66
2025-11-05

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:d10390d2fafef5db83d3fe22f80a8a24aa991104484aa27131f6b19f07aa7fcb
license declaredunknown
license concludedunknown
authorsJoao
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

vector-width = float4__global__ void vectorAdd_float4(const float4* __restrict__ a, const float4* __restrict__ b, float4* __restrict__ c) {

Kernel source

solution5.py127 lines
#!POPCORN leaderboard vectoradd_v2

import torch
from torch.utils.cpp_extension import load_inline
from typing import List
from task import input_t, output_t

add_cuda_source = """

#include <cuda_fp16.h>

//__global__ void vectorAdd_float4(const float4*  __restrict__ a, const float4* __restrict__ b, float4* __restrict__ c, long long n_float8) {
__global__ void vectorAdd_float4(const float4*  __restrict__ a, const float4* __restrict__ b, float4* __restrict__ c) {
    
    // Calculate the global thread ID using a grid-stride loop
    //for (long long i = blockIdx.x * blockDim.x + threadIdx.x; 
    //     i < n_float8; 
    //     i += gridDim.x * blockDim.x) 
    //{
    int i = blockIdx.x * blockDim.x + threadIdx.x; 
    float4 a_vec = a[i];
    float4 b_vec = b[i];
    const half2* a_h = reinterpret_cast<const half2*>(&a_vec);
    const half2* b_h = reinterpret_cast<const half2*>(&b_vec);
    half2 c_h[4];
    c_h[0] = __hadd2(a_h[0], b_h[0]);
    c_h[1] = __hadd2(a_h[1], b_h[1]);
    c_h[2] = __hadd2(a_h[2], b_h[2]);
    c_h[3] = __hadd2(a_h[3], b_h[3]);
    c[i] = *reinterpret_cast<float4*>(c_h);
    //}
}

torch::Tensor add_cuda(torch::Tensor A, torch::Tensor B, torch::Tensor C) {
    int N = A.numel();  
    //// N = 268435456
    //// n_float8 = N/8 = 33554432
    //const int N = 268435456;
    const long long n_float8 = N / 8;

    const int threads = 256; 
    // blocks = 65536
    const int blocks = (n_float8 + threads - 1) / threads;

    vectorAdd_float4<<<blocks,threads>>>(
        reinterpret_cast<float4*>(A.data_ptr<at::Half>()),
        reinterpret_cast<float4*>(B.data_ptr<at::Half>()),
        reinterpret_cast<float4*>(C.data_ptr<at::Half>()));
//        n_float8);

    return C;
}
"""

add_cpp_source = """
#include <torch/extension.h>
#include <cuda_fp16.h>

torch::Tensor add_cuda(torch::Tensor A, torch::Tensor B, torch::Tensor C);
"""

add_module = load_inline(
    name='add_cuda',
    cpp_sources=add_cpp_source,
    cuda_sources=add_cuda_source,
    functions=['add_cuda'],
    verbose=False,
)


def print_gpu_properties():
    import sys
    """
    Prints key properties for all available CUDA-enabled GPUs.
    """
    if not torch.cuda.is_available():
        print("CUDA is not available on this system.", file=sys.stderr)
        return

    device_count = torch.cuda.device_count()
    print(f"Found {device_count} CUDA-enabled device(s).")
    print("-" * 70)

    for i in range(device_count):
        try:
            # Get the device properties
            props = torch.cuda.get_device_properties(i)

            print(f"--- Device {i}: {props.name} ---")

            # --- Core Architecture ---
            print("  Architecture:")
            print(f"    CUDA Compute Capability:   {props.major}.{props.minor}")
            print(f"    Streaming Multiprocessors (SMs): {props.multi_processor_count}")
            # Total cores = SMs * (cores/SM). Varies by architecture
            # (e.g., 128 for Ampere, 64 for Turing). This is a rough guide.

            # --- Memory ---
            print("  Memory:")
            # Convert bytes to Gigabytes (GiB)
            total_mem_gb = props.total_memory / (1024**3)
            print(f"    Total Global Memory:     {total_mem_gb:.2f} GiB")
            print(f"    Shared Memory per SM:    {props.shared_memory_per_multiprocessor / 1024:.0f} KiB")
            print(f"    Shared Memory per Block: {props.shared_memory_per_block / 1024:.0f} KiB")
            print(f"    L2 Cache Size:           {props.l2_cache_size / (1024**2):.0f} MiB")
            print(f"    Memory Bus Width:        {props.memory_bus_width}-bit")
            print(f"    Memory Clock Rate:       {props.memory_clock_rate / 1000:.2f} GHz")

            # --- Threading & Execution ---
            print("  Execution & Threading:")
            print(f"    Max Threads per Block:   {props.max_threads_per_block}")
            print(f"    Max Threads per SM:      {props.max_threads_per_multiprocessor}")
            print("    Max Block Dimensions (x,y,z): "
                  f"({props.max_grid_size[0]}, {props.max_grid_size[1]}, {props.max_grid_size[2]})")
            print("    Max Thread Dimensions (x,y,z): "
                  f"({props.max_threads_dim[0]}, {props.max_threads_dim[1]}, {props.max_threads_dim[2]})")
            print(f"    Warp Size:               {props.warp_size}")

            print("-" * 70)

        except Exception as e:
            print(f"Error getting properties for device {i}: {e}", file=sys.stderr)

def custom_kernel(data: input_t) -> output_t:
    #print_gpu_properties()
    return add_module.add_cuda(data[0], data[1], data[2])
scrolls · 127 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 66622.

⋯ 32 unchanged lines
torch::Tensor add_cuda(torch::Tensor A, torch::Tensor B, torch::Tensor C) {
int N = A.numel();
- // N = 268435456
- // n_float8 = N/8 = 33554432
+ //// N = 268435456
+ //// n_float8 = N/8 = 33554432
+ //const int N = 268435456;
const long long n_float8 = N / 8;
const int threads = 256;
⋯ 22 unchanged lines
cpp_sources=add_cpp_source,
cuda_sources=add_cuda_source,
functions=['add_cuda'],
- verbose=True,
+ verbose=False,
)

Best evidence level for this revision: reported

JSON