submission 68346
rex_cz · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 113 lines, June 9 Researcher Reciprocity License v1.0.
vector_add_v1.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-68346?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp16
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:836bd2ce042ac247e74fa4610af26cd8b329d73941b71688314da97e8db6afbb
license declaredunknown
license concludedunknown
authorsrex_cz
imported2026-08-15
Kernel source
vector_add_v1.py113 lines
# Reference: https://github.com/gpu-mode/reference-kernels/blob/main/problems/pmpp_v2/vectoradd_py/reference.py
# B200: 76.2 ± 0.40 ms
import torch
from task import input_t, output_t
import cutlass.cute as cute
from cutlass.cute.runtime import from_dlpack
@cute.kernel
def vector_add_kernel(gA: cute.Tensor, gB: cute.Tensor, gC: cute.Tensor):
tidx, _, _ = cute.arch.thread_idx()
bidx, _, _ = cute.arch.block_idx()
bdim, _, _ = cute.arch.block_dim()
thread_idx = bidx * bdim + tidx
m, n = gA.shape
numel = m * n
if thread_idx < numel:
mi = thread_idx // n
ni = thread_idx % n
gC[mi, ni] = gA[mi, ni] + gB[mi, ni]
@cute.kernel
def vector_add_vectorized_kernel(
gA: cute.Tensor,
gB: cute.Tensor,
gC: cute.Tensor,
):
tidx, _, _ = cute.arch.thread_idx()
bidx, _, _ = cute.arch.block_idx()
bdim, _, _ = cute.arch.block_dim()
thread_idx = bidx * bdim + tidx
m, n = gA.shape[1]
ni = thread_idx % n
mi = thread_idx // n
a_val = gA[(None, (mi, ni))].load()
b_val = gB[(None, (mi, ni))].load()
gC[(None, (mi, ni))] = a_val + b_val
@cute.jit
def vector_add(mA, mB, mC):
threads_per_block = 256
m, n = mA.shape
vector_add_kernel(mA, mB, mC).launch(
grid=((m * n) // threads_per_block + 1, 1, 1),
block=(threads_per_block, 1, 1),
)
@cute.jit
def vector_add_vectorized(mA, mB, mC):
threads_per_block = 256
gA = cute.zipped_divide(mA, (1, 8))
gB = cute.zipped_divide(mB, (1, 8))
gC = cute.zipped_divide(mC, (1, 8))
vector_add_vectorized_kernel(gA, gB, gC).launch(
grid=(cute.size(gC, mode=[1]) // threads_per_block, 1, 1),
block=(threads_per_block, 1, 1),
)
def generate_input(size: int, seed: int) -> input_t:
"""
Generates random input tensors of specified shapes.
Returns:
Tuple of tensors [A, B] to be added.
"""
gen = torch.Generator(device="cuda")
gen.manual_seed(seed)
A = torch.randn(
size, size, device="cuda", dtype=torch.float16, generator=gen
).contiguous()
B = torch.randn(
size, size, device="cuda", dtype=torch.float16, generator=gen
).contiguous()
C = torch.empty(size, size, device="cuda", dtype=torch.float16).contiguous()
return A, B, C
sizes = [
(127, 4242),
(128, 5236),
(129, 1001),
(256, 5531),
(512, 9173),
(1024, 31232),
(2048, 4052),
(4096, 2146),
(8192, 3129),
(16384, 54352),
]
compiled_kernels = {}
for i in sizes:
size, seed = i
a, b, c = generate_input(size, seed)
a_ = from_dlpack(a, assumed_align=16)
b_ = from_dlpack(b, assumed_align=16)
c_ = from_dlpack(c, assumed_align=16)
kernel = vector_add
if size % 4 == 0:
kernel = vector_add_vectorized
compiled_kernels[size] = cute.compile(kernel, a_, b_, c_)
def custom_kernel(input):
a, b, c = input
compiled_kernels[a.shape[0]](a, b, c)
return c
scrolls · 113 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 68340.
⋯ 37 unchanged linesm, n = gA.shape[1]ni = thread_idx % nmi = thread_idx // n- if mi < m:- a_val = gA[(None, (mi, ni))].load()- b_val = gB[(None, (mi, ni))].load()+ a_val = gA[(None, (mi, ni))].load()+ b_val = gB[(None, (mi, ni))].load()- gC[(None, (mi, ni))] = a_val + b_val+ gC[(None, (mi, ni))] = a_val + b_val@cute.jit⋯ 8 unchanged lines@cute.jitdef vector_add_vectorized(mA, mB, mC):threads_per_block = 256- gA = cute.zipped_divide(mA, (1, 4))- gB = cute.zipped_divide(mB, (1, 4))- gC = cute.zipped_divide(mC, (1, 4))+ gA = cute.zipped_divide(mA, (1, 8))+ gB = cute.zipped_divide(mB, (1, 8))+ gC = cute.zipped_divide(mC, (1, 8))vector_add_vectorized_kernel(gA, gB, gC).launch(grid=(cute.size(gC, mode=[1]) // threads_per_block, 1, 1),block=(threads_per_block, 1, 1),
scrolls · 28 diff lines total
Best evidence level for this revision: reported
JSON