Team Ai
Datasetpublic

willychan21/ParallelKernelBench_Problems

ParallelKernelBench (benchmark) Reference problems for ParallelKernelBench: a benchmark for LLM-generated multi-GPU CUDA kernels. This dataset contains 87 reference implementations in reference/ and the input tensor specification in utils/input_output_tensors.py. Files Path Description data/problems.parquet One row per problem (tabular access) reference/*.py Reference solution() implementations utils/input_output_tensors.py Input/output tensor… See the full description on the dataset page: https://huggingface.co/datasets/willychan21/ParallelKernelBench_Problems.

sourceHugging Faceapache-2.0updated 5mo agoView on Hugging Face
0likes123downloads
73_vocab_parallel_cross_entropy_loss.py42 linesDownload Raw Back to reference
1from typing import Optional, Tuple2 3import torch4import torch.distributed as dist5 6 7def _vocab_range(partition_vocab_size: int, rank: int) -> Tuple[int, int]:8    start = rank * partition_vocab_size9    return start, start + partition_vocab_size10 11 12@torch.no_grad()13def solution(14    vocab_parallel_logits: torch.Tensor,15    target: torch.Tensor,16    group: Optional[dist.ProcessGroup] = None,17) -> torch.Tensor:18    group = group or dist.group.WORLD19    rank = dist.get_rank(group=group)20 21    logits_max = torch.max(vocab_parallel_logits, dim=-1).values22    dist.all_reduce(logits_max, op=dist.ReduceOp.MAX, group=group)23    vocab_parallel_logits.sub_(logits_max.unsqueeze(dim=-1))24 25    partition_vocab_size = vocab_parallel_logits.shape[-1]26    vocab_start, vocab_end = _vocab_range(partition_vocab_size, rank)27    target_mask = (target < vocab_start) | (target >= vocab_end)28    masked_target = target - vocab_start29    masked_target = masked_target.masked_fill(target_mask, 0)30 31    logits_2d = vocab_parallel_logits.reshape(-1, partition_vocab_size)32    target_1d = masked_target.reshape(-1)33    row_ids = torch.arange(logits_2d.shape[0], device=logits_2d.device)34    predicted_logits = logits_2d[row_ids, target_1d].clone().reshape_as(target)35    predicted_logits = predicted_logits.masked_fill(target_mask, 0.0)36    dist.all_reduce(predicted_logits, op=dist.ReduceOp.SUM, group=group)37 38    exp_logits = torch.exp(vocab_parallel_logits)39    sum_exp_logits = exp_logits.sum(dim=-1)40    dist.all_reduce(sum_exp_logits, op=dist.ReduceOp.SUM, group=group)41 42    return torch.log(sum_exp_logits) - predicted_logits