Team Ai
Datasetpublic

togethercomputer/ParallelKernelBench_Problems

ParallelKernelBench (benchmark) Reference problems for ParallelKernelBench: a benchmark for LLM-generated multi-GPU CUDA kernels. This dataset contains 87 reference implementations in reference/ and the input tensor specification in utils/input_output_tensors.py. Inputs are deterministic — reproduce them with create_input_tensor(rank, world_size, problem_id, base_shape, dtype, trial) from that file; you do not need stored .pt files. Files Path Description… See the full description on the dataset page: https://huggingface.co/datasets/togethercomputer/ParallelKernelBench_Problems.

sourceHugging Faceapache-2.0updated 4mo agoView on Hugging Face
0likes177downloads
71_hyena_conv1d_boundary_exchange.py95 linesDownload Raw Back to reference
1from typing import Optional, Tuple2 3import torch4import torch.distributed as dist5import torch.nn.functional as F6 7 8def _zigzag_get_overlapping_patches(9    data: torch.Tensor,10    seq_dim: int,11    overlap_size: int,12) -> Tuple[torch.Tensor, torch.Tensor]:13    shape = list(data.shape)14    shape[seq_dim : seq_dim + 1] = [2, data.shape[seq_dim] // 2]15    chunks = data.reshape(shape)16 17    order = list(range(chunks.dim()))18    order.insert(0, order.pop(seq_dim))19    chunks = chunks.permute(order)20 21    chunk_len = chunks.shape[seq_dim + 1]22    overlaps = chunks.narrow(seq_dim + 1, chunk_len - overlap_size, overlap_size)23    return overlaps[0], overlaps[1]24 25 26@torch.no_grad()27def solution(28    x: torch.Tensor,29    weight: torch.Tensor,30    group: Optional[dist.ProcessGroup] = None,31) -> torch.Tensor:32    group = group or dist.group.WORLD33    group_ranks = dist.get_process_group_ranks(group)34    group_rank = dist.get_rank(group)35    group_world_size = len(group_ranks)36 37    batch, hidden, local_seq = x.shape38    chunk_len = local_seq // 239    pad_size = weight.shape[-1] - 140    chunk_a, chunk_b = _zigzag_get_overlapping_patches(x, 2, pad_size)41 42    ops = []43    recv_prev_a = None44    recv_next_b = None45 46    if group_rank > 0:47        recv_prev_a = torch.empty_like(chunk_a)48        ops.append(49            dist.P2POp(dist.irecv, recv_prev_a, group_ranks[group_rank - 1], group)50        )51    if group_rank < group_world_size - 1:52        ops.append(53            dist.P2POp(54                dist.isend,55                chunk_a.contiguous(),56                group_ranks[group_rank + 1],57                group,58            )59        )60 61    if group_rank < group_world_size - 1:62        recv_next_b = torch.empty_like(chunk_b)63        ops.append(64            dist.P2POp(dist.irecv, recv_next_b, group_ranks[group_rank + 1], group)65        )66    if group_rank > 0:67        ops.append(68            dist.P2POp(69                dist.isend,70                chunk_b.contiguous(),71                group_ranks[group_rank - 1],72                group,73            )74        )75 76    for request in dist.batch_isend_irecv(ops):77        request.wait()78 79    if recv_prev_a is None:80        recv_prev_a = torch.zeros_like(chunk_a)81    if recv_next_b is None:82        recv_next_b = chunk_a.clone().contiguous()83 84    # Move the two zigzag chunks into batch so both use the same grouped conv.85    x_chunks = x.reshape(batch, hidden, 2, chunk_len).permute(2, 0, 1, 3)86    x_chunks = x_chunks.reshape(2 * batch, hidden, chunk_len)87    padding = torch.cat([recv_prev_a, recv_next_b], dim=0)88    x_padded = torch.cat([padding, x_chunks], dim=-1)89 90    y = F.conv1d(x_padded, weight, bias=None, stride=1, padding=0, groups=hidden)91    return (92        y.reshape(2, batch, hidden, chunk_len)93        .permute(1, 2, 0, 3)94        .reshape(batch, hidden, local_seq)95    )