Skip to content
KernelIndex
Search⌘K

submission 545300

rajesh0042 · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 39 lines, June 9 Researcher Reciprocity License v1.0.

histogram_v4.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-histogram-v2-545300?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesuint8

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
Histogramsuite of 6 cases
NVIDIA B200
148.8µs
#40 of 54
2026-03-13

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:0c0260a11de95cd714e4eea020bdc1f9eec8c60fbc34b542605150463a0bcd87
license declaredunknown
license concludedunknown
authorsrajesh0042
imported2026-08-15

Kernel source

histogram_v4.py39 lines
import os
os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"

import torch
import triton
import triton.language as tl
from task import input_t, output_t

# Per-block private histogram approach:
# Phase 1: Each block builds its own 256-bin histogram (low contention)
# Phase 2: Sum all per-block histograms

@triton.jit
def histogram_per_block_kernel(
    data_ptr, hist_ptr, n_elements, n_bins: tl.constexpr,
    BLOCK_SIZE: tl.constexpr,
):
    pid = tl.program_id(0)
    offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
    mask = offsets < n_elements
    vals = tl.load(data_ptr + offsets, mask=mask, other=0).to(tl.int32)
    # Each block writes to its own 256-bin region
    base = pid * n_bins
    tl.atomic_add(hist_ptr + base + vals, 1, mask=mask)

def custom_kernel(data: input_t) -> output_t:
    data, output = data
    n = data.numel()
    BLOCK_SIZE = 4096
    n_blocks = (n + BLOCK_SIZE - 1) // BLOCK_SIZE
    # Allocate per-block histograms
    per_block = torch.zeros(n_blocks, 256, device=data.device, dtype=torch.int32)
    histogram_per_block_kernel[(n_blocks,)](
        data, per_block, n, n_bins=256, BLOCK_SIZE=BLOCK_SIZE
    )
    # Reduce: sum across blocks
    output[...] = per_block.sum(dim=0)
    return output
scrolls · 39 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 545265.

⋯ 5 unchanged lines
import triton.language as tl
from task import input_t, output_t
+ # Per-block private histogram approach:
+ # Phase 1: Each block builds its own 256-bin histogram (low contention)
+ # Phase 2: Sum all per-block histograms
+
@triton.jit
- def histogram_kernel(
- data_ptr, output_ptr, n_elements,
+ def histogram_per_block_kernel(
+ data_ptr, hist_ptr, n_elements, n_bins: tl.constexpr,
BLOCK_SIZE: tl.constexpr,
):
pid = tl.program_id(0)
offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
mask = offsets < n_elements
- vals = tl.load(data_ptr + offsets, mask=mask, other=0)
- # Convert to int32 for indexing
- vals = vals.to(tl.int32)
- # Atomic add to global histogram
- for i in range(BLOCK_SIZE):
- if pid * BLOCK_SIZE + i < n_elements:
- bin_idx = tl.load(data_ptr + pid * BLOCK_SIZE + i).to(tl.int32)
- tl.atomic_add(output_ptr + bin_idx, 1)
+ vals = tl.load(data_ptr + offsets, mask=mask, other=0).to(tl.int32)
+ # Each block writes to its own 256-bin region
+ base = pid * n_bins
+ tl.atomic_add(hist_ptr + base + vals, 1, mask=mask)
def custom_kernel(data: input_t) -> output_t:
data, output = data
- # torch.bincount is already very fast, let's just use it
- output[...] = torch.bincount(data, minlength=256)
+ n = data.numel()
+ BLOCK_SIZE = 4096
+ n_blocks = (n + BLOCK_SIZE - 1) // BLOCK_SIZE
+ # Allocate per-block histograms
+ per_block = torch.zeros(n_blocks, 256, device=data.device, dtype=torch.int32)
+ histogram_per_block_kernel[(n_blocks,)](
+ data, per_block, n, n_bins=256, BLOCK_SIZE=BLOCK_SIZE
+ )
+ # Reduce: sum across blocks
+ output[...] = per_block.sum(dim=0)
return output
scrolls · 46 diff lines total

Best evidence level for this revision: reported

JSON