Skip to content
KernelIndex
Search⌘K

claude-opus-4-1-20250805 / tritona20c42

claude-opus-4-1-20250805_triton_a20c42 · claude-opus-4-1-20250805 · triton · Apache-2.0

Use it

Vendorable · source mirrored · Apache-2.0View source →

No package. Vendor the mirrored source: 136 lines, Apache-2.0, pinned at da91508.

main.py
curl "https://kernelindex.com/api/v1/implementations/flashinfer-claude-opus-4-1-20250805-triton-a20c42?include=source"
interfacetriton
revisionda915083d4c7
symbolrun
pathmain.py
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp16

Benchmark evidence

25 measurements across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
GEMM n128 k2048fp16 · [1, 2048]
NVIDIA B200
21.8µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [4, 2048]
NVIDIA B200
22.2µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [64, 2048]
NVIDIA B200
22.3µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [5, 2048]
NVIDIA B200
22.4µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [8, 2048]
NVIDIA B200
22.4µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [34, 2048]
NVIDIA B200
22.4µs
#3 of 7
2025-10-16
GEMM n128 k2048fp16 · [16, 2048]
NVIDIA B200
22.5µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [93, 2048]
NVIDIA B200
22.5µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [17, 2048]
NVIDIA B200
22.5µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [63, 2048]
NVIDIA B200
22.5µs
#4 of 7
2025-10-16
Show all 25 measurements ›
GEMM n128 k2048fp16 · [172, 2048]
NVIDIA B200
22.5µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [32, 2048]
NVIDIA B200
22.5µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [2, 2048]
NVIDIA B200
22.6µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [128, 2048]
NVIDIA B200
22.6µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [25, 2048]
NVIDIA B200
22.6µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [289, 2048]
NVIDIA B200
22.6µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [492, 2048]
NVIDIA B200
22.7µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [952, 2048]
NVIDIA B200
22.9µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [6, 2048]
NVIDIA B200
23.0µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [8828, 2048]
NVIDIA B200
26.7µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [11006, 2048]
NVIDIA B200
29.2µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [12251, 2048]
NVIDIA B200
31.0µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [14915, 2048]
NVIDIA B200
32.8µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [16294, 2048]
NVIDIA B200
33.7µs
#4 of 7
2025-10-16
GEMM n128 k2048fp16 · [12853, 2048]
NVIDIA B200
41.6µs
#4 of 7
2025-10-16

Reported · How evidence levels are derived →

Source and license

sourcehttps://huggingface.co/datasets/flashinfer-ai/flashinfer-trace
commitda915083d4c7c5e61aa3005e3d17ae488e0fc71c
revision digestsha256:64d7d9b4692bde283b5498504549c06575f5817b95aebb314a5650e7617fa974
license declaredApache-2.0
license concludedApache-2.0
authorsclaude-opus-4-1-20250805
imported2026-08-20

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

mmaacc += tl.dot(a_block, tl.trans(b_block))
tile-k = 64BLOCK_SIZE_K = 64 # Tile K dimension for better cache usage
tile-m = 128BLOCK_SIZE_M = 128
tile-n = 64BLOCK_SIZE_N = 64 # N=128, so we use 2 blocks

Kernel source

main.py136 lines
import torch
import triton
import triton.language as tl


@triton.jit
def gemm_n128_k2048_kernel(
    a_ptr, b_ptr, c_ptr,
    M, N, K,
    stride_am, stride_ak,
    stride_bn, stride_bk,
    stride_cm, stride_cn,
    BLOCK_SIZE_M: tl.constexpr,
    BLOCK_SIZE_N: tl.constexpr,
    BLOCK_SIZE_K: tl.constexpr,
):
    """Optimized GEMM kernel for N=128, K=2048 configuration."""
    # Program ID
    pid_m = tl.program_id(0)
    pid_n = tl.program_id(1)
    
    # Block indices
    block_start_m = pid_m * BLOCK_SIZE_M
    block_start_n = pid_n * BLOCK_SIZE_N
    
    # Thread block offsets
    offs_m = block_start_m + tl.arange(0, BLOCK_SIZE_M)
    offs_n = block_start_n + tl.arange(0, BLOCK_SIZE_N)
    offs_k = tl.arange(0, BLOCK_SIZE_K)
    
    # Pointers to first blocks of A and B
    a_ptrs = a_ptr + (offs_m[:, None] * stride_am + offs_k[None, :] * stride_ak)
    b_ptrs = b_ptr + (offs_n[:, None] * stride_bn + offs_k[None, :] * stride_bk)
    
    # Initialize accumulator
    acc = tl.zeros((BLOCK_SIZE_M, BLOCK_SIZE_N), dtype=tl.float32)
    
    # Main loop over K dimension
    for k in range(0, K, BLOCK_SIZE_K):
        # Load blocks from A and B with boundary checks
        mask_m = offs_m < M
        mask_n = offs_n < N
        mask_k = (k + offs_k) < K
        
        a_block = tl.load(a_ptrs, mask=mask_m[:, None] & mask_k[None, :], other=0.0)
        b_block = tl.load(b_ptrs, mask=mask_n[:, None] & mask_k[None, :], other=0.0)
        
        # Compute dot product for this K block
        # B is transposed in memory access pattern
        acc += tl.dot(a_block, tl.trans(b_block))
        
        # Advance pointers to next K block
        a_ptrs += BLOCK_SIZE_K * stride_ak
        b_ptrs += BLOCK_SIZE_K * stride_bk
    
    # Store result with boundary check
    c_ptrs = c_ptr + (offs_m[:, None] * stride_cm + offs_n[None, :] * stride_cn)
    mask = (offs_m[:, None] < M) & (offs_n[None, :] < N)
    tl.store(c_ptrs, acc.to(tl.float16), mask=mask)


def run(*args, **kwargs):
    """Entry point function that handles device management and kernel execution."""
    # Handle both args and kwargs
    if len(args) == 2:
        A, B = args
    elif 'A' in kwargs and 'B' in kwargs:
        A = kwargs['A']
        B = kwargs['B']
    else:
        raise ValueError("Expected either (A, B) as positional args or as keyword args")
    
    # Check input shapes and dtypes
    assert A.ndim == 2 and B.ndim == 2, "Input tensors must be 2D"
    M, K_a = A.shape
    N, K_b = B.shape
    assert K_a == 2048 and K_b == 2048, f"Expected K=2048, got K_a={K_a}, K_b={K_b}"
    assert N == 128, f"Expected N=128, got N={N}"
    
    # Store original devices
    device_a = A.device
    device_b = B.device
    
    # Move to GPU if needed
    if not torch.cuda.is_available():
        if A.is_cuda or B.is_cuda:
            raise RuntimeError("CUDA is not available but GPU tensors were provided")
        raise RuntimeError("CUDA is not available for GPU computation")
    
    # Move CPU tensors to GPU
    if not A.is_cuda:
        A = A.cuda()
    if not B.is_cuda:
        B = B.cuda()
    
    # Ensure correct dtype
    if A.dtype != torch.float16:
        A = A.to(torch.float16)
    if B.dtype != torch.float16:
        B = B.to(torch.float16)
    
    # Ensure tensors are on the same device
    if A.device != B.device:
        B = B.to(A.device)
    
    # Allocate output tensor
    C = torch.empty((M, N), dtype=torch.float16, device=A.device)
    
    # Configure kernel parameters optimized for B200
    # B200 has large shared memory and high compute throughput
    BLOCK_SIZE_M = 128
    BLOCK_SIZE_N = 64  # N=128, so we use 2 blocks
    BLOCK_SIZE_K = 64  # Tile K dimension for better cache usage
    
    # Compute grid dimensions
    grid = (triton.cdiv(M, BLOCK_SIZE_M), triton.cdiv(N, BLOCK_SIZE_N))
    
    # Launch kernel
    gemm_n128_k2048_kernel[grid](
        A, B, C,
        M, N, 2048,
        A.stride(0), A.stride(1),
        B.stride(0), B.stride(1),
        C.stride(0), C.stride(1),
        BLOCK_SIZE_M=BLOCK_SIZE_M,
        BLOCK_SIZE_N=BLOCK_SIZE_N,
        BLOCK_SIZE_K=BLOCK_SIZE_K,
    )
    
    # Move result back to original device if needed
    if device_a.type == 'cpu':
        C = C.cpu()
    elif device_a != C.device:
        C = C.to(device_a)
    
    return C
scrolls · 136 lines total

Source code from FlashInfer-Bench (flashinfer-ai/flashinfer-trace) · Apache-2.0

Best evidence level for this revision: reported

JSON