Skip to content
KernelIndex
Search⌘K

submission 68349

rex_cz · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 113 lines, June 9 Researcher Reciprocity License v1.0.

vector_add_v1.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-68349?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp16

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
FP16 vector additionsuite of 5 cases
NVIDIA A100
897.0µs
#13 of 87
2025-11-08

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:0232b05079b6c5c314294c9c4bf085f01e44feca96e1b1f7351886ccb9becd3e
license declaredunknown
license concludedunknown
authorsrex_cz
imported2026-08-15

Kernel source

vector_add_v1.py113 lines
# Reference: https://github.com/gpu-mode/reference-kernels/blob/main/problems/pmpp_v2/vectoradd_py/reference.py
# B200: 76.2 ± 0.40 ms
import torch
from task import input_t, output_t
import cutlass.cute as cute
from cutlass.cute.runtime import from_dlpack


@cute.kernel
def vector_add_kernel(gA: cute.Tensor, gB: cute.Tensor, gC: cute.Tensor):
    tidx, _, _ = cute.arch.thread_idx()
    bidx, _, _ = cute.arch.block_idx()
    bdim, _, _ = cute.arch.block_dim()

    thread_idx = bidx * bdim + tidx

    m, n = gA.shape
    numel = m * n

    if thread_idx < numel:
        mi = thread_idx // n
        ni = thread_idx % n
        gC[mi, ni] = gA[mi, ni] + gB[mi, ni]


@cute.kernel
def vector_add_vectorized_kernel(
    gA: cute.Tensor,
    gB: cute.Tensor,
    gC: cute.Tensor,
):
    tidx, _, _ = cute.arch.thread_idx()
    bidx, _, _ = cute.arch.block_idx()
    bdim, _, _ = cute.arch.block_dim()

    thread_idx = bidx * bdim + tidx

    m, n = gA.shape[1]
    ni = thread_idx % n
    mi = thread_idx // n
    a_val = gA[(None, (mi, ni))].load()
    b_val = gB[(None, (mi, ni))].load()

    gC[(None, (mi, ni))] = a_val + b_val


@cute.jit
def vector_add(mA, mB, mC):
    threads_per_block = 256
    m, n = mA.shape
    vector_add_kernel(mA, mB, mC).launch(
        grid=((m * n) // threads_per_block + 1, 1, 1),
        block=(threads_per_block, 1, 1),
    )

@cute.jit
def vector_add_vectorized(mA, mB, mC):
    threads_per_block = 128
    gA = cute.zipped_divide(mA, (1, 8))
    gB = cute.zipped_divide(mB, (1, 8))
    gC = cute.zipped_divide(mC, (1, 8))
    vector_add_vectorized_kernel(gA, gB, gC).launch(
        grid=(cute.size(gC, mode=[1]) // threads_per_block, 1, 1),
        block=(threads_per_block, 1, 1),
    )

def generate_input(size: int, seed: int) -> input_t:
    """
    Generates random input tensors of specified shapes.
    Returns:
        Tuple of tensors [A, B] to be added.
    """
    gen = torch.Generator(device="cuda")
    gen.manual_seed(seed)
    A = torch.randn(
        size, size, device="cuda", dtype=torch.float16, generator=gen
    ).contiguous()
    B = torch.randn(
        size, size, device="cuda", dtype=torch.float16, generator=gen
    ).contiguous()
    C = torch.empty(size, size, device="cuda", dtype=torch.float16).contiguous()
    return A, B, C

sizes = [
  (127, 4242),
  (128, 5236),
  (129, 1001),
  (256, 5531),
  (512, 9173),
  (1024, 31232),
  (2048, 4052),
  (4096, 2146),
  (8192, 3129),
  (16384, 54352),
]
compiled_kernels = {}
for i in sizes:
    size, seed = i
    a, b, c = generate_input(size, seed)
    a_ = from_dlpack(a, assumed_align=16)
    b_ = from_dlpack(b, assumed_align=16)
    c_ = from_dlpack(c, assumed_align=16)
    kernel = vector_add
    if size % 4 == 0:
        kernel = vector_add_vectorized
    compiled_kernels[size] = cute.compile(kernel, a_, b_, c_)

def custom_kernel(input):
    a, b, c = input
    compiled_kernels[a.shape[0]](a, b, c)
    return c

scrolls · 113 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 68346.

⋯ 54 unchanged lines
@cute.jit
def vector_add_vectorized(mA, mB, mC):
- threads_per_block = 256
+ threads_per_block = 128
gA = cute.zipped_divide(mA, (1, 8))
gB = cute.zipped_divide(mB, (1, 8))
gC = cute.zipped_divide(mC, (1, 8))

Best evidence level for this revision: reported

JSON