claude-opus-4-1 / triton79b898
claude-opus-4-1_triton_79b898 · claude-opus-4-1-20250805 · triton · Apache-2.0
Use it
Vendorable · source mirrored · Apache-2.0View source →
No package. Vendor the mirrored source: 122 lines, Apache-2.0, pinned at da91508.
main.py
curl "https://kernelindex.com/api/v1/implementations/flashinfer-claude-opus-4-1-triton-79b898?include=source"interfacetriton
revisionda915083d4c7
symbolrun
pathmain.py
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp16
Benchmark evidence
43 measurements across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Show all 43 measurements ›Showing all 43 measurements ⌄
Reported · How evidence levels are derived →
Source and license
sourcehttps://huggingface.co/datasets/flashinfer-ai/flashinfer-trace
commitda915083d4c7c5e61aa3005e3d17ae488e0fc71c
revision digestsha256:ab82c9081be76d46e03abcc9fa4792b47f347b4ef6fb336f64bff663c057398d
license declaredApache-2.0
license concludedApache-2.0
authorsclaude-opus-4-1-20250805
imported2026-08-20
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
mma
accumulator += tl.dot(a, b.T, allow_tf32=True)tile-k = 64
BLOCK_SIZE_K = 64tile-m = 128
BLOCK_SIZE_M = 128tile-n = 128
BLOCK_SIZE_N = 128Kernel source
main.py122 lines
import torch
import triton
import triton.language as tl
import math
@triton.jit
def gemm_kernel(
a_ptr, b_ptr, c_ptr,
M, N, K,
stride_am, stride_ak,
stride_bn, stride_bk,
stride_cm, stride_cn,
BLOCK_SIZE_M: tl.constexpr,
BLOCK_SIZE_N: tl.constexpr,
BLOCK_SIZE_K: tl.constexpr,
):
# Program ID
pid = tl.program_id(axis=0)
num_pid_m = tl.cdiv(M, BLOCK_SIZE_M)
num_pid_n = tl.cdiv(N, BLOCK_SIZE_N)
# 2D grid mapping
pid_m = pid // num_pid_n
pid_n = pid % num_pid_n
# Skip if out of bounds
if pid_m >= num_pid_m:
return
# Block indices
offs_am = pid_m * BLOCK_SIZE_M + tl.arange(0, BLOCK_SIZE_M)
offs_bn = pid_n * BLOCK_SIZE_N + tl.arange(0, BLOCK_SIZE_N)
offs_k = tl.arange(0, BLOCK_SIZE_K)
# Accumulator
accumulator = tl.zeros((BLOCK_SIZE_M, BLOCK_SIZE_N), dtype=tl.float32)
# Loop over K dimension
for k in range(0, K, BLOCK_SIZE_K):
# Compute current k offsets
curr_k = k + offs_k
# Load tiles with boundary checks
a_ptrs = a_ptr + (offs_am[:, None] * stride_am + curr_k[None, :] * stride_ak)
b_ptrs = b_ptr + (offs_bn[:, None] * stride_bn + curr_k[None, :] * stride_bk)
a_mask = (offs_am[:, None] < M) & (curr_k[None, :] < K)
b_mask = (offs_bn[:, None] < N) & (curr_k[None, :] < K)
a = tl.load(a_ptrs, mask=a_mask, other=0.0)
b = tl.load(b_ptrs, mask=b_mask, other=0.0)
# Matrix multiply and accumulate - b is already transposed in memory layout
accumulator += tl.dot(a, b.T, allow_tf32=True)
# Convert back to fp16 and store
c = accumulator.to(tl.float16)
# Store output with boundary checks
offs_cm = pid_m * BLOCK_SIZE_M + tl.arange(0, BLOCK_SIZE_M)
offs_cn = pid_n * BLOCK_SIZE_N + tl.arange(0, BLOCK_SIZE_N)
c_ptrs = c_ptr + stride_cm * offs_cm[:, None] + stride_cn * offs_cn[None, :]
c_mask = (offs_cm[:, None] < M) & (offs_cn[None, :] < N)
tl.store(c_ptrs, c, mask=c_mask)
def run(A, B):
# Handle device management
device_a = A.device
device_b = B.device
# Move to GPU if needed
if A.device.type == 'cpu':
if not torch.cuda.is_available():
raise RuntimeError("CUDA is not available but GPU tensors are required")
A = A.cuda()
if B.device.type == 'cpu':
if not torch.cuda.is_available():
raise RuntimeError("CUDA is not available but GPU tensors are required")
B = B.cuda()
# Ensure tensors are on the same GPU
if A.device != B.device:
B = B.to(A.device)
# Get dimensions
M = A.shape[0]
N = 28672 # constant
K = 4096 # constant
# Allocate output
C = torch.empty((M, N), device=A.device, dtype=torch.float16)
# Block sizes optimized for B200
BLOCK_SIZE_M = 128
BLOCK_SIZE_N = 128
BLOCK_SIZE_K = 64
# Grid configuration
def grid(META):
return (triton.cdiv(M, META['BLOCK_SIZE_M']) * triton.cdiv(N, META['BLOCK_SIZE_N']),)
# Launch kernel
gemm_kernel[grid](
A, B, C,
M, N, K,
A.stride(0), A.stride(1),
B.stride(0), B.stride(1),
C.stride(0), C.stride(1),
BLOCK_SIZE_M=BLOCK_SIZE_M,
BLOCK_SIZE_N=BLOCK_SIZE_N,
BLOCK_SIZE_K=BLOCK_SIZE_K,
)
# Move result back to original device if needed
if device_a.type == 'cpu':
C = C.cpu()
elif device_a != C.device:
C = C.to(device_a)
return Cscrolls · 122 lines total
Source code from FlashInfer-Bench (flashinfer-ai/flashinfer-trace) · Apache-2.0
Best evidence level for this revision: reported
JSON