Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
utils.hpp200 linesDownload Raw Back to nvcomp
1/*
2 * SPDX-FileCopyrightText: Copyright (c) 2018-2025 NVIDIA CORPORATION & AFFILIATES.
3 * All rights reserved. SPDX-License-Identifier: LicenseRef-NvidiaProprietary
4 *
5 * NVIDIA CORPORATION, its affiliates and licensors retain all intellectual
6 * property and proprietary rights in and to this material, related
7 * documentation and any modifications thereto. Any use, reproduction,
8 * disclosure or distribution of this material and related documentation
9 * without an express license agreement from NVIDIA CORPORATION or
10 * its affiliates is strictly prohibited.
11*/
12
13#ifndef DOXYGEN_SHOULD_SKIP_THIS
14#pragma once
15
16#include <limits>
17#include <type_traits>
18#include <cassert>
19
20#ifndef __NVCC__
21#define NVCOMP_HOST_DEVICE_FUNCTION
22#else
23#define NVCOMP_HOST_DEVICE_FUNCTION __host__ __device__
24#endif // __NVCC__
25
26namespace nvcomp {
27
28/**
29 * @brief Return the ceiling of the ratio of input num and input chunk.
30 *
31 * @tparam U The type of the argument num.
32 * @tparam T The type of the argument chunk.
33 * @param[in] num The dividend.
34 * @param[in] chunk The divisor.
35 *
36 * @return The rounded quotient of the division.
37 */
38template <typename U, typename T>
39constexpr NVCOMP_HOST_DEVICE_FUNCTION U roundUpDiv(const U num, const T chunk) noexcept
40{
41  return (num + chunk - 1) / chunk;
42}
43
44/**
45 * @brief Round down the input num to an integer multiple of the input chunk.
46 *
47 * @tparam U The type of the argument num.
48 * @tparam T The type of the argument chunk.
49 * @param[in] num The original amount to be rounded down.
50 * @param[in] chunk The rounding multiple.
51 *
52 * @return The rounded-down input.
53 */
54template <typename U, typename T>
55constexpr NVCOMP_HOST_DEVICE_FUNCTION U roundDownTo(const U num, const T chunk) noexcept
56{
57  return (num / chunk) * chunk;
58}
59
60/**
61 * @brief Round up the input num to an integer multiple of the input chunk.
62 *
63 * @tparam U The type of the argument num.
64 * @tparam T The type of the argument chunk.
65 * @param[in] num The original amount to be rounded up.
66 * @param[in] chunk The rounding multiple.
67 *
68 * @return The rounded-up input.
69 */
70template <typename U, typename T>
71constexpr NVCOMP_HOST_DEVICE_FUNCTION U roundUpTo(const U num, const T chunk) noexcept
72{
73  return roundUpDiv(num, chunk) * chunk;
74}
75
76/**
77 * @brief Return the smallest power of two larger or equal to the input x.
78 *
79 * @tparam T The type of the argument x.
80 * @param[in] x The original amount to be rounded up.
81 *
82 * @return The rounded-up input.
83 */
84template<typename T>
85constexpr NVCOMP_HOST_DEVICE_FUNCTION T roundUpPow2(const T x) noexcept
86{
87  size_t res = 1;
88  while(res < x) {
89    res *= 2;
90  }
91  return res;
92}
93
94/**
95 * @brief Calculate the first aligned location after `ptr`.
96 *
97 * @tparam T Type such that the alignment requirement is satisfied.
98 * @param[in] ptr Input pointer.
99 *
100 * @return The first pointer after `ptr` that satisfies the alignment requirement.
101 */
102template <typename T>
103constexpr NVCOMP_HOST_DEVICE_FUNCTION T* roundUpToAlignment(void* ptr) noexcept
104{
105  constexpr auto alignment = alignof(T);
106  const auto address = reinterpret_cast<uintptr_t>(ptr);
107  return reinterpret_cast<T*>((address + alignment - 1) & ~(alignment - 1));
108}
109
110/**
111 * @brief Calculate the first aligned location after `ptr`.
112 *
113 * @tparam T Type such that the alignment requirement is satisfied.
114 * @param[in] ptr Input pointer pointing to constant data.
115 *
116 * @return The first pointer after `ptr` that satisfies the alignment requirement.
117 */
118template <typename T>
119constexpr NVCOMP_HOST_DEVICE_FUNCTION const T* roundUpToAlignment(const void* ptr) noexcept
120{
121  constexpr auto alignment = alignof(T);
122  const auto address = reinterpret_cast<uintptr_t>(ptr);
123  return reinterpret_cast<const T*>((address + alignment - 1) & ~(alignment - 1));
124}
125
126/**
127 * @brief Verifies whether a given cast from InputT type to OutputT type is valid.
128 *
129 * @tparam OutputT The output type we intend to cast to.
130 * @tparam InputT The input type we intend to cast from.
131 *
132 * @return Boolean indicating whether the cast is valid.
133 */
134template <typename OutputT, typename InputT>
135constexpr NVCOMP_HOST_DEVICE_FUNCTION bool is_cast_valid(const InputT i) noexcept
136{
137  static_assert(
138      std::numeric_limits<OutputT>::is_integer && std::numeric_limits<InputT>::is_integer,
139      "Types for is_cast_valid must both be integers");
140  if (std::is_unsigned<InputT>::value) {
141      // The minimum bound is always satisfied, so just check the maximum bound.
142      // Use larger type, breaking tie with InputT, which is already known unsigned.
143      using largerT = typename std::conditional<(sizeof(OutputT) > sizeof(InputT)), OutputT, InputT>::type;
144      return static_cast<largerT>(i) <= static_cast<largerT>((std::numeric_limits<OutputT>::max)());
145  }
146
147  // At this point, InputT is signed, but because this code will still be compiled
148  // for unsigned InputT, force InputT to be signed, to avoid warnings about signed
149  // vs. unsigned comparison.
150  using signedInputT = typename std::make_signed<InputT>::type;
151  using signedOutputT = typename std::make_signed<OutputT>::type;
152
153  // Check whether the input is less than the minimum value of OutputT.
154  // I.e. a negative signed integer is casting to an unsigned
155  // Note, if OutputT is unsigned, the minimum is zero, which is safe to cast to
156  // a signed type.
157  if (static_cast<signedInputT>(i)
158      < static_cast<signedOutputT>((std::numeric_limits<OutputT>::min)())) {
159    return false;
160  }
161
162  // Because we've already checked whether the inputT is "too negative", if it's
163  // negative at all this is valid
164  // InputT is signed and larger than the minimum value of OutputT.
165  if (static_cast<signedInputT>(i) <= static_cast<signedInputT>(0)) {
166    return true;
167  }
168
169  // InputT is signed, but larger than zero, so can be cast to unsigned.
170  using unsignedInputT = typename std::make_unsigned<InputT>::type;
171  using unsignedOutputT = typename std::make_unsigned<OutputT>::type;
172
173  return static_cast<unsignedInputT>(i)
174         <= static_cast<unsignedOutputT>((std::numeric_limits<OutputT>::max)());
175}
176
177/**
178 * @brief Cast to uint, with debug-only range check, for CUDA kernel launch grid
179 * or block dimensions.
180 *
181 * @tparam InputT The input type we intend to cast from.
182 * @param[in] i Input dimension to cast.
183 *
184 * @return The input casted to unsigned integer.
185 */
186template <typename InputT>
187constexpr unsigned int cuda_dim_cast(const InputT i) noexcept
188{
189  // On current architectures (7.5 to 12.0, both inclusive)
190  // Maximum x-dimension of a grid of thread blocks: 2^31-1
191  // Maximum y- or z-dimension of a grid of thread blocks: 65535
192  assert(is_cast_valid<unsigned int>(i) &&  i < (1u << 31));
193
194  return static_cast<unsigned int>(i);
195}
196
197} // namespace nvcomp
198
199#endif /* DOXYGEN_SHOULD_SKIP_THIS */
200 
codekingpro/portable-devtools · Team Ai