Team Ai
Datasetpublic

willychan21/ParallelKernelBench_Problems

ParallelKernelBench (benchmark) Reference problems for ParallelKernelBench: a benchmark for LLM-generated multi-GPU CUDA kernels. This dataset contains 87 reference implementations in reference/ and the input tensor specification in utils/input_output_tensors.py. Files Path Description data/problems.parquet One row per problem (tabular access) reference/*.py Reference solution() implementations utils/input_output_tensors.py Input/output tensor… See the full description on the dataset page: https://huggingface.co/datasets/willychan21/ParallelKernelBench_Problems.

sourceHugging Faceapache-2.0updated 5mo agoView on Hugging Face
0likes123downloads
58_openclip_contrastive_loss.py94 linesDownload Raw Back to reference
1from typing import Optional2 3import torch4import torch.distributed as dist5import torch.nn.functional as F6 7 8def _siglip_loss(9    image_features: torch.Tensor,10    text_features: torch.Tensor,11    logit_scale: float,12    logit_bias: float,13    negative_only: bool,14) -> torch.Tensor:15    batch = image_features.size(0)16    logits = logit_scale * image_features @ text_features.T + logit_bias17 18    if negative_only:19        return -F.logsigmoid(-logits).sum() / batch20 21    labels = -torch.ones((batch, batch), device=logits.device, dtype=logits.dtype)22    labels.diagonal().fill_(1)23    return -F.logsigmoid(labels * logits).sum() / batch24 25 26def _exchange(27    group: dist.ProcessGroup,28    recv_from: int,29    send_to: int,30    tensor: torch.Tensor,31) -> torch.Tensor:32    out = torch.empty_like(tensor)33    ops = [34        dist.P2POp(dist.isend, tensor, dist.get_global_rank(group, send_to), group=group),35        dist.P2POp(dist.irecv, out, dist.get_global_rank(group, recv_from), group=group),36    ]37    for req in dist.batch_isend_irecv(ops):38        req.wait()39    return out40 41 42def _exchange_bidir(43    group: dist.ProcessGroup,44    left: int,45    right: int,46    tensor_to_left: torch.Tensor,47    tensor_to_right: torch.Tensor,48) -> tuple[torch.Tensor, torch.Tensor]:49    from_left = torch.empty_like(tensor_to_right)50    from_right = torch.empty_like(tensor_to_left)51    left_global = dist.get_global_rank(group, left)52    right_global = dist.get_global_rank(group, right)53    ops = [54        dist.P2POp(dist.isend, tensor_to_right, right_global, group=group),55        dist.P2POp(dist.isend, tensor_to_left, left_global, group=group),56        dist.P2POp(dist.irecv, from_right, right_global, group=group),57        dist.P2POp(dist.irecv, from_left, left_global, group=group),58    ]59    for req in dist.batch_isend_irecv(ops):60        req.wait()61    return from_right, from_left62 63 64@torch.no_grad()65def solution(66    image_features: torch.Tensor,67    text_features: torch.Tensor,68    logit_scale: float,69    logit_bias: float = 0.0,70    group: Optional[dist.ProcessGroup] = None,71) -> torch.Tensor:72    group = group or dist.group.WORLD73    rank = dist.get_rank(group)74    world_size = dist.get_world_size(group)75 76    loss = _siglip_loss(image_features, text_features, logit_scale, logit_bias, False)77 78    left = (rank - 1) % world_size79    right = (rank + 1) % world_size80    text_to_left = text_features81    text_to_right = text_features82    num_bidir, remainder = divmod(world_size - 1, 2)83 84    for _ in range(num_bidir):85        from_right, from_left = _exchange_bidir(group, left, right, text_to_left, text_to_right)86        loss = loss + _siglip_loss(image_features, from_right, logit_scale, logit_bias, True)87        loss = loss + _siglip_loss(image_features, from_left, logit_scale, logit_bias, True)88        text_to_left, text_to_right = from_right, from_left89 90    if remainder:91        text_recv = _exchange(group, left, right, text_to_right)92        loss = loss + _siglip_loss(image_features, text_recv, logit_scale, logit_bias, True)93 94    return loss