codekingpro/portable-devtools
114k
1/*
2* Copyright 1993-2024 NVIDIA Corporation. All rights reserved.
3*
4* NOTICE TO LICENSEE:
5*
6* This source code and/or documentation ("Licensed Deliverables") are
7* subject to NVIDIA intellectual property rights under U.S. and
8* international Copyright laws.
9*
10* These Licensed Deliverables contained herein is PROPRIETARY and
11* CONFIDENTIAL to NVIDIA and is being provided under the terms and
12* conditions of a form of NVIDIA software license agreement by and
13* between NVIDIA and Licensee ("License Agreement") or electronically
14* accepted by Licensee. Notwithstanding any terms or conditions to
15* the contrary in the License Agreement, reproduction or disclosure
16* of the Licensed Deliverables to any third party without the express
17* written consent of NVIDIA is prohibited.
18*
19* NOTWITHSTANDING ANY TERMS OR CONDITIONS TO THE CONTRARY IN THE
20* LICENSE AGREEMENT, NVIDIA MAKES NO REPRESENTATION ABOUT THE
21* SUITABILITY OF THESE LICENSED DELIVERABLES FOR ANY PURPOSE. IT IS
22* PROVIDED "AS IS" WITHOUT EXPRESS OR IMPLIED WARRANTY OF ANY KIND.
23* NVIDIA DISCLAIMS ALL WARRANTIES WITH REGARD TO THESE LICENSED
24* DELIVERABLES, INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY,
25* NONINFRINGEMENT, AND FITNESS FOR A PARTICULAR PURPOSE.
26* NOTWITHSTANDING ANY TERMS OR CONDITIONS TO THE CONTRARY IN THE
27* LICENSE AGREEMENT, IN NO EVENT SHALL NVIDIA BE LIABLE FOR ANY
28* SPECIAL, INDIRECT, INCIDENTAL, OR CONSEQUENTIAL DAMAGES, OR ANY
29* DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS,
30* WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS
31* ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR PERFORMANCE
32* OF THESE LICENSED DELIVERABLES.
33*
34* U.S. Government End Users. These Licensed Deliverables are a
35* "commercial item" as that term is defined at 48 C.F.R. 2.101 (OCT
36* 1995), consisting of "commercial computer software" and "commercial
37* computer software documentation" as such terms are used in 48
38* C.F.R. 12.212 (SEPT 1995) and is provided to the U.S. Government
39* only as a commercial end item. Consistent with 48 C.F.R.12.212 and
40* 48 C.F.R. 227.7202-1 through 227.7202-4 (JUNE 1995), all
41* U.S. Government End Users acquire the Licensed Deliverables with
42* only those rights set forth herein.
43*
44* Any use of the Licensed Deliverables in individual and commercial
45* software must include, in the user documentation and internal
46* comments to the code, the above Disclaimer and U.S. Government End
47* Users Notice.
48*/
49
50#if !defined(__CUDA_BF16_HPP__)
51#define __CUDA_BF16_HPP__
52
53#if !defined(__CUDA_BF16_H__)
54#error "Do not include this file directly. Instead, include cuda_bf16.h."
55#endif
56
57#if !defined(IF_DEVICE_OR_CUDACC)
58#if defined(__CUDACC__)
59 #define IF_DEVICE_OR_CUDACC(d, c, f) NV_IF_ELSE_TARGET(NV_IS_DEVICE, d, c)
60#else
61 #define IF_DEVICE_OR_CUDACC(d, c, f) NV_IF_ELSE_TARGET(NV_IS_DEVICE, d, f)
62#endif
63#endif
64
65/* All other definitions in this file are only visible to C++ compilers */
66#if defined(__cplusplus)
67/**
68 * \ingroup CUDA_MATH_INTRINSIC_BFLOAT16_CONSTANTS
69 * \brief Defines floating-point positive infinity value for the \p nv_bfloat16 data type
70 */
71#define CUDART_INF_BF16 __ushort_as_bfloat16((unsigned short)0x7F80U)
72/**
73 * \ingroup CUDA_MATH_INTRINSIC_BFLOAT16_CONSTANTS
74 * \brief Defines canonical NaN value for the \p nv_bfloat16 data type
75 */
76#define CUDART_NAN_BF16 __ushort_as_bfloat16((unsigned short)0x7FFFU)
77/**
78 * \ingroup CUDA_MATH_INTRINSIC_BFLOAT16_CONSTANTS
79 * \brief Defines a minimum representable (denormalized) value for the \p nv_bfloat16 data type
80 */
81#define CUDART_MIN_DENORM_BF16 __ushort_as_bfloat16((unsigned short)0x0001U)
82/**
83 * \ingroup CUDA_MATH_INTRINSIC_BFLOAT16_CONSTANTS
84 * \brief Defines a maximum representable value for the \p nv_bfloat16 data type
85 */
86#define CUDART_MAX_NORMAL_BF16 __ushort_as_bfloat16((unsigned short)0x7F7FU)
87/**
88 * \ingroup CUDA_MATH_INTRINSIC_BFLOAT16_CONSTANTS
89 * \brief Defines a negative zero value for the \p nv_bfloat16 data type
90 */
91#define CUDART_NEG_ZERO_BF16 __ushort_as_bfloat16((unsigned short)0x8000U)
92/**
93 * \ingroup CUDA_MATH_INTRINSIC_BFLOAT16_CONSTANTS
94 * \brief Defines a positive zero value for the \p nv_bfloat16 data type
95 */
96#define CUDART_ZERO_BF16 __ushort_as_bfloat16((unsigned short)0x0000U)
97/**
98 * \ingroup CUDA_MATH_INTRINSIC_BFLOAT16_CONSTANTS
99 * \brief Defines a value of 1.0 for the \p nv_bfloat16 data type
100 */
101#define CUDART_ONE_BF16 __ushort_as_bfloat16((unsigned short)0x3F80U)
102
103 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16 &__nv_bfloat16::operator=(const __nv_bfloat16_raw &hr) { __x = hr.x; return *this; }
104 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ volatile __nv_bfloat16 &__nv_bfloat16::operator=(const __nv_bfloat16_raw &hr) volatile { __x = hr.x; return *this; }
105 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ volatile __nv_bfloat16 &__nv_bfloat16::operator=(const volatile __nv_bfloat16_raw &hr) volatile { __x = hr.x; return *this; }
106 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator __nv_bfloat16_raw() const { __nv_bfloat16_raw ret; ret.x = __x; return ret; }
107 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator __nv_bfloat16_raw() const volatile { __nv_bfloat16_raw ret; ret.x = __x; return ret; }
108
109#if !defined(__CUDA_NO_BFLOAT16_CONVERSIONS__)
110 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator float() const { return __bfloat162float(*this); }
111 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16 &__nv_bfloat16::operator=(const float f) { __x = __float2bfloat16(f).__x; return *this; }
112 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16 &__nv_bfloat16::operator=(const double f) { __x = __double2bfloat16(f).__x; return *this; }
113
114/*
115 * Implicit type conversions to/from integer types were only available to nvcc compilation.
116 * Introducing them for all compilers is a potentially breaking change that may affect
117 * overloads resolution and will require users to update their code.
118 * Define __CUDA_BF16_DISABLE_IMPLICIT_INTEGER_CONVERTS_FOR_HOST_COMPILERS__ to opt-out.
119 */
120#if !(defined __CUDA_BF16_DISABLE_IMPLICIT_INTEGER_CONVERTS_FOR_HOST_COMPILERS__) || (defined __CUDACC__)
121 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator signed char() const { return __bfloat162char_rz(*this); }
122 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator unsigned char() const { return __bfloat162uchar_rz(*this); }
123 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator char() const {
124 char value;
125 /* Suppress VS warning: warning C4127: conditional expression is constant */
126#if defined(_MSC_VER) && !defined(__CUDA_ARCH__)
127#pragma warning (push)
128#pragma warning (disable: 4127)
129#endif /* _MSC_VER && !defined(__CUDA_ARCH__) */
130 if (((char)-1) < (char)0)
131#if defined(_MSC_VER) && !defined(__CUDA_ARCH__)
132#pragma warning (pop)
133#endif /* _MSC_VER && !defined(__CUDA_ARCH__) */
134 {
135 value = static_cast<char>(__bfloat162char_rz(*this));
136 }
137 else
138 {
139 value = static_cast<char>(__bfloat162uchar_rz(*this));
140 }
141 return value;
142 }
143 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator short() const { return __bfloat162short_rz(*this); }
144 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator unsigned short() const { return __bfloat162ushort_rz(*this); }
145 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator int() const { return __bfloat162int_rz(*this); }
146 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator unsigned int() const { return __bfloat162uint_rz(*this); }
147 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator long() const {
148 long retval;
149 /* Suppress VS warning: warning C4127: conditional expression is constant */
150#if defined(_MSC_VER) && !defined(__CUDA_ARCH__)
151#pragma warning (push)
152#pragma warning (disable: 4127)
153#endif /* _MSC_VER && !defined(__CUDA_ARCH__) */
154 if (sizeof(long) == sizeof(long long))
155#if defined(_MSC_VER) && !defined(__CUDA_ARCH__)
156#pragma warning (pop)
157#endif /* _MSC_VER && !defined(__CUDA_ARCH__) */
158 {
159 retval = static_cast<long>(__bfloat162ll_rz(*this));
160 }
161 else
162 {
163 retval = static_cast<long>(__bfloat162int_rz(*this));
164 }
165 return retval;
166 }
167 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator unsigned long() const {
168 unsigned long retval;
169 /* Suppress VS warning: warning C4127: conditional expression is constant */
170#if defined(_MSC_VER) && !defined(__CUDA_ARCH__)
171#pragma warning (push)
172#pragma warning (disable: 4127)
173#endif /* _MSC_VER && !defined(__CUDA_ARCH__) */
174 if (sizeof(unsigned long) == sizeof(unsigned long long))
175#if defined(_MSC_VER) && !defined(__CUDA_ARCH__)
176#pragma warning (pop)
177#endif /* _MSC_VER && !defined(__CUDA_ARCH__) */
178 {
179 retval = static_cast<unsigned long>(__bfloat162ull_rz(*this));
180 }
181 else
182 {
183 retval = static_cast<unsigned long>(__bfloat162uint_rz(*this));
184 }
185 return retval;
186 }
187 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator long long() const { return __bfloat162ll_rz(*this); }
188 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator unsigned long long() const { return __bfloat162ull_rz(*this); }
189 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16 &__nv_bfloat16::operator=(short val) { __x = __short2bfloat16_rn(val).__x; return *this; }
190 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16 &__nv_bfloat16::operator=(unsigned short val) { __x = __ushort2bfloat16_rn(val).__x; return *this; }
191 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16 &__nv_bfloat16::operator=(int val) { __x = __int2bfloat16_rn(val).__x; return *this; }
192 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16 &__nv_bfloat16::operator=(unsigned int val) { __x = __uint2bfloat16_rn(val).__x; return *this; }
193 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16 &__nv_bfloat16::operator=(long long val) { __x = __ll2bfloat16_rn(val).__x; return *this; }
194 __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16 &__nv_bfloat16::operator=(unsigned long long val) { __x = __ull2bfloat16_rn(val).__x; return *this; }
195#endif /* !(defined __CUDA_BF16_DISABLE_IMPLICIT_INTEGER_CONVERTS_FOR_HOST_COMPILERS__) || (defined __CUDACC__) */
196#endif /* !defined(__CUDA_NO_BFLOAT16_CONVERSIONS__) */
197
198
199#if !defined(__CUDA_NO_BFLOAT16_OPERATORS__)
200__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 operator+(const __nv_bfloat16 &lh, const __nv_bfloat16 &rh) { return __hadd(lh, rh); }
201__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 operator-(const __nv_bfloat16 &lh, const __nv_bfloat16 &rh) { return __hsub(lh, rh); }
202__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 operator*(const __nv_bfloat16 &lh, const __nv_bfloat16 &rh) { return __hmul(lh, rh); }
203__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 operator/(const __nv_bfloat16 &lh, const __nv_bfloat16 &rh) { return __hdiv(lh, rh); }
204__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 &operator+=(__nv_bfloat16 &lh, const __nv_bfloat16 &rh) { lh = __hadd(lh, rh); return lh; }
205__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 &operator-=(__nv_bfloat16 &lh, const __nv_bfloat16 &rh) { lh = __hsub(lh, rh); return lh; }
206__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 &operator*=(__nv_bfloat16 &lh, const __nv_bfloat16 &rh) { lh = __hmul(lh, rh); return lh; }
207__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 &operator/=(__nv_bfloat16 &lh, const __nv_bfloat16 &rh) { lh = __hdiv(lh, rh); return lh; }
208__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 &operator++(__nv_bfloat16 &h) { __nv_bfloat16_raw one; one.x = 0x3F80U; h += one; return h; }
209__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 &operator--(__nv_bfloat16 &h) { __nv_bfloat16_raw one; one.x = 0x3F80U; h -= one; return h; }
210__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 operator++(__nv_bfloat16 &h, const int ignored)
211{
212 // ignored on purpose. Parameter only needed to distinguish the function declaration from other types of operators.
213 static_cast<void>(ignored);
214
215 const __nv_bfloat16 ret = h;
216 __nv_bfloat16_raw one;
217 one.x = 0x3F80U;
218 h += one;
219 return ret;
220}
221__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 operator--(__nv_bfloat16 &h, const int ignored)
222{
223 // ignored on purpose. Parameter only needed to distinguish the function declaration from other types of operators.
224 static_cast<void>(ignored);
225
226 const __nv_bfloat16 ret = h;
227 __nv_bfloat16_raw one;
228 one.x = 0x3F80U;
229 h -= one;
230 return ret;
231}
232__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 operator+(const __nv_bfloat16 &h) { return h; }
233__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 operator-(const __nv_bfloat16 &h) { return __hneg(h); }
234__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator==(const __nv_bfloat16 &lh, const __nv_bfloat16 &rh) { return __heq(lh, rh); }
235__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator!=(const __nv_bfloat16 &lh, const __nv_bfloat16 &rh) { return __hneu(lh, rh); }
236__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator> (const __nv_bfloat16 &lh, const __nv_bfloat16 &rh) { return __hgt(lh, rh); }
237__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator< (const __nv_bfloat16 &lh, const __nv_bfloat16 &rh) { return __hlt(lh, rh); }
238__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator>=(const __nv_bfloat16 &lh, const __nv_bfloat16 &rh) { return __hge(lh, rh); }
239__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator<=(const __nv_bfloat16 &lh, const __nv_bfloat16 &rh) { return __hle(lh, rh); }
240#endif /* !defined(__CUDA_NO_BFLOAT16_OPERATORS__) */
241
242#if defined(__CPP_VERSION_AT_LEAST_11_BF16)
243__CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat162::__nv_bfloat162(__nv_bfloat162 &&src) {
244NV_IF_ELSE_TARGET(NV_IS_DEVICE,
245 __BFLOAT162_TO_UI(*this) = std::move(__BFLOAT162_TO_CUI(src));
246,
247 this->x = src.x;
248 this->y = src.y;
249)
250}
251__CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat162 &__nv_bfloat162::operator=(__nv_bfloat162 &&src) {
252NV_IF_ELSE_TARGET(NV_IS_DEVICE,
253 __BFLOAT162_TO_UI(*this) = std::move(__BFLOAT162_TO_CUI(src));
254,
255 this->x = src.x;
256 this->y = src.y;
257)
258 return *this;
259}
260#else
261__CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat162::__nv_bfloat162() { }
262#endif /* defined(__CPP_VERSION_AT_LEAST_11_BF16) */
263__CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat162::__nv_bfloat162(const __nv_bfloat162 &src) {
264NV_IF_ELSE_TARGET(NV_IS_DEVICE,
265 __BFLOAT162_TO_UI(*this) = __BFLOAT162_TO_CUI(src);
266,
267 this->x = src.x;
268 this->y = src.y;
269)
270}
271__CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat162 &__nv_bfloat162::operator=(const __nv_bfloat162 &src) {
272NV_IF_ELSE_TARGET(NV_IS_DEVICE,
273 __BFLOAT162_TO_UI(*this) = __BFLOAT162_TO_CUI(src);
274,
275 this->x = src.x;
276 this->y = src.y;
277)
278 return *this;
279}
280__CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat162::__nv_bfloat162(const __nv_bfloat162_raw &h2r ) {
281NV_IF_ELSE_TARGET(NV_IS_DEVICE,
282 __BFLOAT162_TO_UI(*this) = __BFLOAT162_TO_CUI(h2r);
283,
284 __nv_bfloat16_raw tr;
285 tr.x = h2r.x;
286 this->x = static_cast<__nv_bfloat16>(tr);
287 tr.x = h2r.y;
288 this->y = static_cast<__nv_bfloat16>(tr);
289)
290}
291__CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat162 &__nv_bfloat162::operator=(const __nv_bfloat162_raw &h2r) {
292NV_IF_ELSE_TARGET(NV_IS_DEVICE,
293 __BFLOAT162_TO_UI(*this) = __BFLOAT162_TO_CUI(h2r);
294,
295 __nv_bfloat16_raw tr;
296 tr.x = h2r.x;
297 this->x = static_cast<__nv_bfloat16>(tr);
298 tr.x = h2r.y;
299 this->y = static_cast<__nv_bfloat16>(tr);
300)
301 return *this;
302}
303__CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat162::operator __nv_bfloat162_raw() const {
304 __nv_bfloat162_raw ret;
305NV_IF_ELSE_TARGET(NV_IS_DEVICE,
306 ret.x = 0U;
307 ret.y = 0U;
308 __BFLOAT162_TO_UI(ret) = __BFLOAT162_TO_CUI(*this);
309,
310 ret.x = static_cast<__nv_bfloat16_raw>(this->x).x;
311 ret.y = static_cast<__nv_bfloat16_raw>(this->y).x;
312)
313 return ret;
314}
315
316#if !defined(__CUDA_NO_BFLOAT162_OPERATORS__)
317__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162 operator+(const __nv_bfloat162 &lh, const __nv_bfloat162 &rh) { return __hadd2(lh, rh); }
318__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162 operator-(const __nv_bfloat162 &lh, const __nv_bfloat162 &rh) { return __hsub2(lh, rh); }
319__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162 operator*(const __nv_bfloat162 &lh, const __nv_bfloat162 &rh) { return __hmul2(lh, rh); }
320__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162 operator/(const __nv_bfloat162 &lh, const __nv_bfloat162 &rh) { return __h2div(lh, rh); }
321__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162& operator+=(__nv_bfloat162 &lh, const __nv_bfloat162 &rh) { lh = __hadd2(lh, rh); return lh; }
322__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162& operator-=(__nv_bfloat162 &lh, const __nv_bfloat162 &rh) { lh = __hsub2(lh, rh); return lh; }
323__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162& operator*=(__nv_bfloat162 &lh, const __nv_bfloat162 &rh) { lh = __hmul2(lh, rh); return lh; }
324__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162& operator/=(__nv_bfloat162 &lh, const __nv_bfloat162 &rh) { lh = __h2div(lh, rh); return lh; }
325__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162 &operator++(__nv_bfloat162 &h) { __nv_bfloat162_raw one; one.x = 0x3F80U; one.y = 0x3F80U; h = __hadd2(h, one); return h; }
326__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162 &operator--(__nv_bfloat162 &h) { __nv_bfloat162_raw one; one.x = 0x3F80U; one.y = 0x3F80U; h = __hsub2(h, one); return h; }
327__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162 operator++(__nv_bfloat162 &h, const int ignored)
328{
329 // ignored on purpose. Parameter only needed to distinguish the function declaration from other types of operators.
330 static_cast<void>(ignored);
331
332 const __nv_bfloat162 ret = h;
333 __nv_bfloat162_raw one;
334 one.x = 0x3F80U;
335 one.y = 0x3F80U;
336 h = __hadd2(h, one);
337 return ret;
338}
339__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162 operator--(__nv_bfloat162 &h, const int ignored)
340{
341 // ignored on purpose. Parameter only needed to distinguish the function declaration from other types of operators.
342 static_cast<void>(ignored);
343
344 const __nv_bfloat162 ret = h;
345 __nv_bfloat162_raw one;
346 one.x = 0x3F80U;
347 one.y = 0x3F80U;
348 h = __hsub2(h, one);
349 return ret;
350}
351__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162 operator+(const __nv_bfloat162 &h) { return h; }
352__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162 operator-(const __nv_bfloat162 &h) { return __hneg2(h); }
353__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator==(const __nv_bfloat162 &lh, const __nv_bfloat162 &rh) { return __hbeq2(lh, rh); }
354__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator!=(const __nv_bfloat162 &lh, const __nv_bfloat162 &rh) { return __hbneu2(lh, rh); }
355__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator>(const __nv_bfloat162 &lh, const __nv_bfloat162 &rh) { return __hbgt2(lh, rh); }
356__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator<(const __nv_bfloat162 &lh, const __nv_bfloat162 &rh) { return __hblt2(lh, rh); }
357__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator>=(const __nv_bfloat162 &lh, const __nv_bfloat162 &rh) { return __hbge2(lh, rh); }
358__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator<=(const __nv_bfloat162 &lh, const __nv_bfloat162 &rh) { return __hble2(lh, rh); }
359#endif /* !defined(__CUDA_NO_BFLOAT162_OPERATORS__) */
360
361/* Restore warning for multiple assignment operators */
362#if defined(_MSC_VER) && _MSC_VER >= 1500
363#pragma warning( pop )
364#endif /* defined(_MSC_VER) && _MSC_VER >= 1500 */
365
366/* Restore -Weffc++ warnings from here on */
367#if defined(__GNUC__)
368#if __GNUC__ > 4 || (__GNUC__ == 4 && __GNUC_MINOR__ >= 6)
369#pragma GCC diagnostic pop
370#endif /* __GNUC__ > 4 || (__GNUC__ == 4 && __GNUC_MINOR__ >= 6) */
371#endif /* defined(__GNUC__) */
372
373#undef __CUDA_HOSTDEVICE__
374#undef __CUDA_ALIGN__
375
376__CUDA_HOSTDEVICE_BF16_DECL__ unsigned int __internal_float_as_uint(const float f)
377{
378 unsigned int u;
379IF_DEVICE_OR_CUDACC(
380 u = __float_as_uint(f);
381,
382 memcpy(&u, &f, sizeof(f));
383,
384 std::memcpy(&u, &f, sizeof(f));
385)
386 return u;
387}
388
389__CUDA_HOSTDEVICE_BF16_DECL__ float __internal_uint_as_float(const unsigned int u)
390{
391 float f;
392IF_DEVICE_OR_CUDACC(
393 f = __uint_as_float(u);
394,
395 memcpy(&f, &u, sizeof(u));
396,
397 std::memcpy(&f, &u, sizeof(u));
398)
399 return f;
400}
401
402__CUDA_HOSTDEVICE_BF16_DECL__ unsigned short __internal_float2bfloat16(const float f, unsigned int &sign, unsigned int &remainder)
403{
404 unsigned int x;
405
406 x = __internal_float_as_uint(f);
407
408 if ((x & 0x7fffffffU) > 0x7f800000U) {
409 sign = 0U;
410 remainder = 0U;
411 return static_cast<unsigned short>(0x7fffU);
412 }
413 sign = x >> 31U;
414 remainder = x << 16U;
415 return static_cast<unsigned short>(x >> 16U);
416}
417
418__CUDA_HOSTDEVICE_BF16_DECL__ float __internal_double2float_rn(const double x)
419{
420 float r;
421NV_IF_ELSE_TARGET(NV_IS_DEVICE,
422 asm("cvt.rn.f32.f64 %0, %1;" : "=f"(r) : "d"(x));
423,
424 r = static_cast<float>(x);
425)
426 return r;
427}
428__CUDA_HOSTDEVICE_BF16_DECL__ double __internal_float2double(const float x)
429{
430 double r;
431NV_IF_ELSE_TARGET(NV_IS_DEVICE,
432 asm("cvt.f64.f32 %0, %1;" : "=d"(r) : "f"(x));
433,
434 r = static_cast<double>(x);
435)
436 return r;
437}
438
439__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat16 __double2bfloat16(const double x)
440{
441NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
442 __nv_bfloat16 val;
443 asm("{ cvt.rn.bf16.f64 %0, %1;}\n" : "=h"(__BFLOAT16_TO_US(val)) : "d"(x));
444 return val;
445,
446 float f = __internal_double2float_rn(x);
447 const double d = __internal_float2double(f);
448 unsigned int u = __internal_float_as_uint(f);
449
450 bool x_is_not_nan = ((u << (unsigned)1U) <= (unsigned)0xFF000000U);
451
452
453 if ((x > 0.0) && (d > x)) {
454 u--;
455 }
456 if ((x < 0.0) && (d < x)) {
457 u--;
458 }
459 if ((d != x) && x_is_not_nan) {
460 u |= 1U;
461 }
462
463 f = __internal_uint_as_float(u);
464
465 return __float2bfloat16(f);
466)
467}
468
469__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat16 __float2bfloat16(const float a)
470{
471 __nv_bfloat16 val;
472NV_IF_ELSE_TARGET(NV_PROVIDES_SM_80,
473 asm("{ cvt.rn.bf16.f32 %0, %1;}\n" : "=h"(__BFLOAT16_TO_US(val)) : "f"(a));
474,
475 __nv_bfloat16_raw r;
476 unsigned int sign = 0U;
477 unsigned int remainder = 0U;
478 r.x = __internal_float2bfloat16(a, sign, remainder);
479 if ((remainder > 0x80000000U) || ((remainder == 0x80000000U) && ((r.x & 0x1U) != 0U))) {
480 r.x++;
481 }
482 val = r;
483)
484 return val;
485}
486__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat16 __float2bfloat16_rn(const float a)
487{
488 __nv_bfloat16 val;
489NV_IF_ELSE_TARGET(NV_PROVIDES_SM_80,
490 asm("{ cvt.rn.bf16.f32 %0, %1;}\n" : "=h"(__BFLOAT16_TO_US(val)) : "f"(a));
491,
492 __nv_bfloat16_raw r;
493 unsigned int sign = 0U;
494 unsigned int remainder = 0U;
495 r.x = __internal_float2bfloat16(a, sign, remainder);
496 if ((remainder > 0x80000000U) || ((remainder == 0x80000000U) && ((r.x & 0x1U) != 0U))) {
497 r.x++;
498 }
499 val = r;
500)
501 return val;
502}
503__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat16 __float2bfloat16_rz(const float a)
504{
505 __nv_bfloat16 val;
506NV_IF_ELSE_TARGET(NV_PROVIDES_SM_80,
507 asm("{ cvt.rz.bf16.f32 %0, %1;}\n" : "=h"(__BFLOAT16_TO_US(val)) : "f"(a));
508,
509 __nv_bfloat16_raw r;
510 unsigned int sign = 0U;
511 unsigned int remainder = 0U;
512 r.x = __internal_float2bfloat16(a, sign, remainder);
513 val = r;
514)
515 return val;
516}
517__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat16 __float2bfloat16_rd(const float a)
518{
519NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
520 __nv_bfloat16 val;
521 asm("{ cvt.rm.bf16.f32 %0, %1;}\n" : "=h"(__BFLOAT16_TO_US(val)) : "f"(a));
522 return val;
523,
524 __nv_bfloat16 val;
525 __nv_bfloat16_raw r;
526 unsigned int sign = 0U;
527 unsigned int remainder = 0U;
528 r.x = __internal_float2bfloat16(a, sign, remainder);
529 if ((remainder != 0U) && (sign != 0U)) {
530 r.x++;
531 }
532 val = r;
533 return val;
534)
535}
536__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat16 __float2bfloat16_ru(const float a)
537{
538NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
539 __nv_bfloat16 val;
540 asm("{ cvt.rp.bf16.f32 %0, %1;}\n" : "=h"(__BFLOAT16_TO_US(val)) : "f"(a));
541 return val;
542,
543 __nv_bfloat16 val;
544 __nv_bfloat16_raw r;
545 unsigned int sign = 0U;
546 unsigned int remainder = 0U;
547 r.x = __internal_float2bfloat16(a, sign, remainder);
548 if ((remainder != 0U) && (sign == 0U)) {
549 r.x++;
550 }
551 val = r;
552 return val;
553)
554}
555__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat162 __float2bfloat162_rn(const float a)
556{
557 __nv_bfloat162 val;
558NV_IF_ELSE_TARGET(NV_PROVIDES_SM_80,
559 asm("{.reg .b16 low;\n"
560 " cvt.rn.bf16.f32 low, %1;\n"
561 " mov.b32 %0, {low,low};}\n" : "=r"(__BFLOAT162_TO_UI(val)) : "f"(a));
562,
563 val = __nv_bfloat162(__float2bfloat16_rn(a), __float2bfloat16_rn(a));
564)
565 return val;
566}
567__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat162 __floats2bfloat162_rn(const float a, const float b)
568{
569 __nv_bfloat162 val;
570NV_IF_ELSE_TARGET(NV_PROVIDES_SM_80,
571 asm("{ cvt.rn.bf16x2.f32 %0, %2, %1;}\n"
572 : "=r"(__BFLOAT162_TO_UI(val)) : "f"(a), "f"(b));
573,
574 val = __nv_bfloat162(__float2bfloat16_rn(a), __float2bfloat16_rn(b));
575)
576 return val;
577}
578
579#if defined(__CUDACC__) || defined(_NVHPC_CUDA)
580__CUDA_BF16_DECL__ float __internal_device_bfloat162float(const unsigned short h)
581{
582 float f;
583NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
584 asm("{ cvt.f32.bf16 %0, %1;}\n" : "=f"(f) : "h"(h));
585,
586 asm("{ mov.b32 %0, {0,%1};}\n" : "=f"(f) : "h"(h));
587)
588 return f;
589}
590#endif /* defined(__CUDACC__) || defined(_NVHPC_CUDA) */
591
592__CUDA_HOSTDEVICE_BF16_DECL__ float __internal_bfloat162float(const unsigned short h)
593{
594 float f;
595NV_IF_ELSE_TARGET(NV_IS_DEVICE,
596 f = __internal_device_bfloat162float(h);
597,
598 unsigned int u = static_cast<unsigned int>(h) << 16;
599 f = __internal_uint_as_float(u);
600)
601 return f;
602}
603
604__CUDA_HOSTDEVICE_BF16_DECL__ float __bfloat162float(const __nv_bfloat16 a)
605{
606 return __internal_bfloat162float(static_cast<__nv_bfloat16_raw>(a).x);
607}
608__CUDA_HOSTDEVICE_BF16_DECL__ float __low2float(const __nv_bfloat162 a)
609{
610 return __internal_bfloat162float(static_cast<__nv_bfloat162_raw>(a).x);
611}
612
613__CUDA_HOSTDEVICE_BF16_DECL__ float __high2float(const __nv_bfloat162 a)
614{
615 return __internal_bfloat162float(static_cast<__nv_bfloat162_raw>(a).y);
616}
617
618/* CUDA vector-types compatible vector creation function (note returns __nv_bfloat162, not nv_bfloat162) */
619__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat162 make_bfloat162(const __nv_bfloat16 x, const __nv_bfloat16 y)
620{
621 __nv_bfloat162 t; t.x = x; t.y = y; return t;
622}
623
624/* Definitions of intrinsics */
625__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat162 __float22bfloat162_rn(const float2 a)
626{
627 __nv_bfloat162 val = __floats2bfloat162_rn(a.x, a.y);
628 return val;
629}
630__CUDA_HOSTDEVICE_BF16_DECL__ float2 __bfloat1622float2(const __nv_bfloat162 a)
631{
632 float hi_float;
633 float lo_float;
634 lo_float = __internal_bfloat162float(((__nv_bfloat162_raw)a).x);
635 hi_float = __internal_bfloat162float(((__nv_bfloat162_raw)a).y);
636 return make_float2(lo_float, hi_float);
637}
638#if defined(__CUDACC__) || defined(_NVHPC_CUDA)
639__CUDA_BF16_DECL__ int __bfloat162int_rn(const __nv_bfloat16 h)
640{
641NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
642 int val;
643 asm("{ cvt.rni.s32.bf16 %0, %1;}\n" : "=r"(val) : "h"(__BFLOAT16_TO_CUS(h)));
644 return val;
645,
646 return __float2int_rn(__bfloat162float(h));
647)
648}
649#endif /* defined(__CUDACC__) || defined(_NVHPC_CUDA) */
650
651__CUDA_HOSTDEVICE_BF16_DECL__ int __internal_bfloat162int_rz(const __nv_bfloat16 h)
652{
653 const float f = __bfloat162float(h);
654 int i;
655NV_IF_ELSE_TARGET(NV_IS_DEVICE,
656 i = __float2int_rz(f);
657,
658 const int max_val = (int)0x7fffffffU;
659 const int min_val = (int)0x80000000U;
660 const unsigned short bits = static_cast<unsigned short>(static_cast<__nv_bfloat16_raw>(h).x << 1U);
661 // saturation fixup
662 if (bits > (unsigned short)0xFF00U) {
663 // NaN
664 i = 0;
665 } else if (f >= static_cast<float>(max_val)) {
666 // saturate maximum
667 i = max_val;
668 } else if (f < static_cast<float>(min_val)) {
669 // saturate minimum
670 i = min_val;
671 } else {
672 i = static_cast<int>(f);
673 }
674)
675 return i;
676}
677
678__CUDA_HOSTDEVICE_BF16_DECL__ int __bfloat162int_rz(const __nv_bfloat16 h)
679{
680NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
681 int val;
682 asm("{ cvt.rzi.s32.bf16 %0, %1;}\n" : "=r"(val) : "h"(__BFLOAT16_TO_CUS(h)));
683 return val;
684,
685 return __internal_bfloat162int_rz(h);
686)
687}
688#if defined(__CUDACC__) || defined(_NVHPC_CUDA)
689__CUDA_BF16_DECL__ int __bfloat162int_rd(const __nv_bfloat16 h)
690{
691 int val;
692NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
693 asm("{ cvt.rmi.s32.bf16 %0, %1;}\n" : "=r"(val) : "h"(__BFLOAT16_TO_CUS(h)));
694,
695 const float f = __bfloat162float(h);
696 asm("cvt.rmi.s32.f32 %0, %1;" : "=r"(val) : "f"(f));
697)
698 return val;
699}
700__CUDA_BF16_DECL__ int __bfloat162int_ru(const __nv_bfloat16 h)
701{
702 int val;
703NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
704 asm("{ cvt.rpi.s32.bf16 %0, %1;}\n" : "=r"(val) : "h"(__BFLOAT16_TO_CUS(h)));
705,
706 const float f = __bfloat162float(h);
707 asm("cvt.rpi.s32.f32 %0, %1;" : "=r"(val) : "f"(f));
708)
709 return val;
710}
711
712__CUDA_BF16_DECL__ __nv_bfloat16 __internal_device_int2bfloat16_rn(const int i)
713{
714NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
715 __nv_bfloat16 val;
716 asm("cvt.rn.bf16.s32 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "r"(i));
717 return val;
718,
719 const float ru = __int2float_ru(i);
720 const float rd = __int2float_rd(i);
721 float rz = __int2float_rz(i);
722 if (ru != rd) {
723 rz = __uint_as_float(__float_as_uint(rz) | 1U);
724 }
725 return __float2bfloat16_rn(rz);
726)
727}
728#endif /* defined(__CUDACC__) || defined(_NVHPC_CUDA) */
729__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat16 __int2bfloat16_rn(const int i)
730{
731NV_IF_ELSE_TARGET(NV_IS_DEVICE,
732 return __internal_device_int2bfloat16_rn(i);
733,
734 const double d = static_cast<double>(i);
735 return __double2bfloat16(d);
736)
737}
738__CUDA_HOSTDEVICE_BF16_DECL__ signed char __bfloat162char_rz(const __nv_bfloat16 h)
739{
740 signed char i;
741NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
742 unsigned short tmp = 0;
743 asm("{ .reg.b8 myreg;\n"
744 " cvt.rzi.s8.bf16 myreg, %1;\n"
745 " mov.b16 %0, {myreg, 0};\n}"
746 :"=h"(tmp) : "h"(__BFLOAT16_TO_CUS(h)));
747 const unsigned char u = static_cast<unsigned char>(tmp);
748 i = static_cast<signed char>(u);
749,
750 const float f = __bfloat162float(h);
751 const signed char max_val = (signed char)0x7fU;
752 const signed char min_val = (signed char)0x80U;
753 const unsigned short bits = static_cast<unsigned short>(static_cast<__nv_bfloat16_raw>(h).x << 1U);
754 // saturation fixup
755 if (bits > (unsigned short)0xFF00U) {
756 // NaN
757 i = 0;
758 } else if (f > static_cast<float>(max_val)) {
759 // saturate maximum
760 i = max_val;
761 } else if (f < static_cast<float>(min_val)) {
762 // saturate minimum
763 i = min_val;
764 } else {
765 // normal value, conversion is well-defined
766 i = static_cast<signed char>(f);
767 }
768)
769 return i;
770}
771
772__CUDA_HOSTDEVICE_BF16_DECL__ unsigned char __bfloat162uchar_rz(const __nv_bfloat16 h)
773{
774 unsigned char i;
775NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
776 unsigned short tmp = 0;
777 asm("{ .reg.b8 myreg;\n"
778 " cvt.rzi.u8.bf16 myreg, %1;\n"
779 " mov.b16 %0, {myreg, 0};\n}"
780 :"=h"(tmp) : "h"(__BFLOAT16_TO_CUS(h)));
781 i = static_cast<unsigned char>(tmp);
782,
783 const float f = __bfloat162float(h);
784 const unsigned char max_val = 0xffU;
785 const unsigned char min_val = 0U;
786 const unsigned short bits = static_cast<unsigned short>(static_cast<__nv_bfloat16_raw>(h).x << 1U);
787 // saturation fixup
788 if (bits > (unsigned short)0xFF00U) {
789 // NaN
790 i = 0U;
791 } else if (f > static_cast<float>(max_val)) {
792 // saturate maximum
793 i = max_val;
794 } else if (f < static_cast<float>(min_val)) {
795 // saturate minimum
796 i = min_val;
797 } else {
798 // normal value, conversion is well-defined
799 i = static_cast<unsigned char>(f);
800 }
801)
802 return i;
803}
804#if defined(__CUDACC__) || defined(_NVHPC_CUDA)
805__CUDA_BF16_DECL__ __nv_bfloat16 __int2bfloat16_rz(const int i)
806{
807NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
808 __nv_bfloat16 val;
809 asm("cvt.rz.bf16.s32 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "r"(i));
810 return val;
811,
812 return __float2bfloat16_rz(__int2float_rz(i));
813)
814}
815__CUDA_BF16_DECL__ __nv_bfloat16 __int2bfloat16_rd(const int i)
816{
817NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
818 __nv_bfloat16 val;
819 asm("cvt.rm.bf16.s32 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "r"(i));
820 return val;
821,
822 return __float2bfloat16_rd(__int2float_rd(i));
823)
824}
825
826__CUDA_BF16_DECL__ __nv_bfloat16 __int2bfloat16_ru(const int i)
827{
828NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
829 __nv_bfloat16 val;
830 asm("cvt.rp.bf16.s32 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "r"(i));
831 return val;
832,
833 return __float2bfloat16_ru(__int2float_ru(i));
834)
835}
836
837__CUDA_BF16_DECL__ short int __bfloat162short_rn(const __nv_bfloat16 h)
838{
839 short int val;
840NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
841 asm("cvt.rni.s16.bf16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
842,
843 asm("{ .reg.f32 f;\n"
844 " mov.b32 f, {0,%1};\n"
845 " cvt.rni.s16.f32 %0,f;\n}"
846 :"=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
847)
848 return val;
849}
850
851__CUDA_BF16_DECL__ short int __internal_device_bfloat162short_rz(const __nv_bfloat16 h)
852{
853 short int val;
854NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
855 asm("cvt.rzi.s16.bf16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
856,
857 asm("{ .reg.f32 f;\n"
858 " mov.b32 f, {0,%1};\n"
859 " cvt.rzi.s16.f32 %0,f;\n}"
860 :"=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
861)
862 return val;
863}
864#endif /* defined(__CUDACC__) || defined(_NVHPC_CUDA) */
865__CUDA_HOSTDEVICE_BF16_DECL__ short int __bfloat162short_rz(const __nv_bfloat16 h)
866{
867 short int val;
868NV_IF_ELSE_TARGET(NV_IS_DEVICE,
869 val = __internal_device_bfloat162short_rz(h);
870,
871 const float f = __bfloat162float(h);
872 const short int max_val = (short int)0x7fffU;
873 const short int min_val = (short int)0x8000U;
874 const unsigned short bits = static_cast<unsigned short>(static_cast<__nv_bfloat16_raw>(h).x << 1U);
875 // saturation fixup
876 if (bits > (unsigned short)0xFF00U) {
877 // NaN
878 val = 0;
879 } else if (f > static_cast<float>(max_val)) {
880 // saturate maximum
881 val = max_val;
882 } else if (f < static_cast<float>(min_val)) {
883 // saturate minimum
884 val = min_val;
885 } else {
886 val = static_cast<short int>(f);
887 }
888)
889 return val;
890}
891#if defined(__CUDACC__) || defined(_NVHPC_CUDA)
892__CUDA_BF16_DECL__ short int __bfloat162short_rd(const __nv_bfloat16 h)
893{
894 short int val;
895NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
896 asm("cvt.rmi.s16.bf16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
897,
898 asm("{ .reg.f32 f;\n"
899 " mov.b32 f, {0,%1};\n"
900 " cvt.rmi.s16.f32 %0,f;\n}"
901 :"=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
902)
903 return val;
904}
905__CUDA_BF16_DECL__ short int __bfloat162short_ru(const __nv_bfloat16 h)
906{
907 short int val;
908NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
909 asm("cvt.rpi.s16.bf16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
910,
911 asm("{ .reg.f32 f;\n"
912 " mov.b32 f, {0,%1};\n"
913 " cvt.rpi.s16.f32 %0,f;\n}"
914 :"=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
915)
916 return val;
917}
918#endif /* defined(__CUDACC__) || defined(_NVHPC_CUDA) */
919__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat16 __short2bfloat16_rn(const short int i)
920{
921NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
922 __nv_bfloat16 val;
923 asm("cvt.rn.bf16.s16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(i));
924 return val;
925,
926 const float f = static_cast<float>(i);
927 return __float2bfloat16_rn(f);
928)
929}
930#if defined(__CUDACC__) || defined(_NVHPC_CUDA)
931__CUDA_BF16_DECL__ __nv_bfloat16 __short2bfloat16_rz(const short int i)
932{
933NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
934 __nv_bfloat16 val;
935 asm("cvt.rz.bf16.s16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(i));
936 return val;
937,
938 return __float2bfloat16_rz(__int2float_rz(static_cast<int>(i)));
939)
940}
941__CUDA_BF16_DECL__ __nv_bfloat16 __short2bfloat16_rd(const short int i)
942{
943NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
944 __nv_bfloat16 val;
945 asm("cvt.rm.bf16.s16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(i));
946 return val;
947,
948 return __float2bfloat16_rd(__int2float_rd(static_cast<int>(i)));
949)
950}
951__CUDA_BF16_DECL__ __nv_bfloat16 __short2bfloat16_ru(const short int i)
952{
953NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
954 __nv_bfloat16 val;
955 asm("cvt.rp.bf16.s16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(i));
956 return val;
957,
958 return __float2bfloat16_ru(__int2float_ru(static_cast<int>(i)));
959)
960}
961
962__CUDA_BF16_DECL__ unsigned int __bfloat162uint_rn(const __nv_bfloat16 h)
963{
964NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
965 unsigned int val;
966 asm("{ cvt.rni.u32.bf16 %0, %1;}\n" : "=r"(val) : "h"(__BFLOAT16_TO_CUS(h)));
967 return val;
968,
969 return __float2uint_rn(__bfloat162float(h));
970)
971}
972#endif /* defined(__CUDACC__) || defined(_NVHPC_CUDA) */
973__CUDA_HOSTDEVICE_BF16_DECL__ unsigned int __internal_bfloat162uint_rz(const __nv_bfloat16 h)
974{
975 const float f = __bfloat162float(h);
976 unsigned int i;
977NV_IF_ELSE_TARGET(NV_IS_DEVICE,
978 i = __float2uint_rz(f);
979,
980 const unsigned int max_val = 0xffffffffU;
981 const unsigned int min_val = 0U;
982 const unsigned short bits = static_cast<unsigned short>(static_cast<__nv_bfloat16_raw>(h).x << 1U);
983 // saturation fixup
984 if (bits > (unsigned short)0xFF00U) {
985 // NaN
986 i = 0U;
987 } else if (f >= static_cast<float>(max_val)) {
988 // saturate maximum
989 i = max_val;
990 } else if (f < static_cast<float>(min_val)) {
991 // saturate minimum
992 i = min_val;
993 } else {
994 i = static_cast<unsigned int>(f);
995 }
996)
997 return i;
998}
999
1000__CUDA_HOSTDEVICE_BF16_DECL__ unsigned int __bfloat162uint_rz(const __nv_bfloat16 h)
1001{
1002NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1003 unsigned int val;
1004 asm("{ cvt.rzi.u32.bf16 %0, %1;}\n" : "=r"(val) : "h"(__BFLOAT16_TO_CUS(h)));
1005 return val;
1006,
1007 return __internal_bfloat162uint_rz(h);
1008)
1009}
1010#if defined(__CUDACC__) || defined(_NVHPC_CUDA)
1011__CUDA_BF16_DECL__ unsigned int __bfloat162uint_rd(const __nv_bfloat16 h)
1012{
1013NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1014 unsigned int val;
1015 asm("{ cvt.rmi.u32.bf16 %0, %1;}\n" : "=r"(val) : "h"(__BFLOAT16_TO_CUS(h)));
1016 return val;
1017,
1018 return __float2uint_rd(__bfloat162float(h));
1019)
1020}
1021__CUDA_BF16_DECL__ unsigned int __bfloat162uint_ru(const __nv_bfloat16 h)
1022{
1023 unsigned int val;
1024NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1025 asm("{ cvt.rpi.u32.bf16 %0, %1;}\n" : "=r"(val) : "h"(__BFLOAT16_TO_CUS(h)));
1026,
1027 const float f = __bfloat162float(h);
1028 asm("cvt.rpi.u32.f32 %0, %1;" : "=r"(val) : "f"(f));
1029)
1030 return val;
1031}
1032
1033__CUDA_BF16_DECL__ __nv_bfloat16 __internal_device_uint2bfloat16_rn(const unsigned int i)
1034{
1035NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1036 __nv_bfloat16 val;
1037 asm("cvt.rn.bf16.u32 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "r"(i));
1038 return val;
1039,
1040 const float ru = __uint2float_ru(i);
1041 const float rd = __uint2float_rd(i);
1042 float rz = __uint2float_rz(i);
1043 if (ru != rd) {
1044 rz = __uint_as_float(__float_as_uint(rz) | 1U);
1045 }
1046 return __float2bfloat16_rn(rz);
1047)
1048}
1049#endif /* defined(__CUDACC__) || defined(_NVHPC_CUDA) */
1050__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat16 __uint2bfloat16_rn(const unsigned int i)
1051{
1052NV_IF_ELSE_TARGET(NV_IS_DEVICE,
1053 return __internal_device_uint2bfloat16_rn(i);
1054,
1055 const double d = static_cast<double>(i);
1056 return __double2bfloat16(d);
1057)
1058}
1059#if defined(__CUDACC__) || defined(_NVHPC_CUDA)
1060__CUDA_BF16_DECL__ __nv_bfloat16 __uint2bfloat16_rz(const unsigned int i)
1061{
1062NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1063 __nv_bfloat16 val;
1064 asm("cvt.rz.bf16.u32 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "r"(i));
1065 return val;
1066,
1067 return __float2bfloat16_rz(__uint2float_rz(i));
1068)
1069}
1070__CUDA_BF16_DECL__ __nv_bfloat16 __uint2bfloat16_rd(const unsigned int i)
1071{
1072NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1073 __nv_bfloat16 val;
1074 asm("cvt.rm.bf16.u32 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "r"(i));
1075 return val;
1076,
1077 return __float2bfloat16_rd(__uint2float_rd(i));
1078)
1079}
1080__CUDA_BF16_DECL__ __nv_bfloat16 __uint2bfloat16_ru(const unsigned int i)
1081{
1082NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1083 __nv_bfloat16 val;
1084 asm("cvt.rp.bf16.u32 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "r"(i));
1085 return val;
1086,
1087 return __float2bfloat16_ru(__uint2float_ru(i));
1088)
1089}
1090
1091__CUDA_BF16_DECL__ unsigned short int __bfloat162ushort_rn(const __nv_bfloat16 h)
1092{
1093 unsigned short int val;
1094NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1095 asm("cvt.rni.u16.bf16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
1096,
1097 asm("{ .reg.f32 f;\n"
1098 " mov.b32 f, {0,%1};\n"
1099 " cvt.rni.u16.f32 %0,f;\n}"
1100 :"=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
1101)
1102 return val;
1103}
1104
1105__CUDA_BF16_DECL__ unsigned short int __internal_device_bfloat162ushort_rz(const __nv_bfloat16 h)
1106{
1107 unsigned short int val;
1108NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1109 asm("cvt.rzi.u16.bf16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
1110,
1111 asm("{ .reg.f32 f;\n"
1112 " mov.b32 f, {0,%1};\n"
1113 " cvt.rzi.u16.f32 %0,f;\n}"
1114 :"=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
1115)
1116 return val;
1117}
1118#endif /* defined(__CUDACC__) || defined(_NVHPC_CUDA) */
1119__CUDA_HOSTDEVICE_BF16_DECL__ unsigned short int __bfloat162ushort_rz(const __nv_bfloat16 h)
1120{
1121 unsigned short int val;
1122NV_IF_ELSE_TARGET(NV_IS_DEVICE,
1123 val = __internal_device_bfloat162ushort_rz(h);
1124,
1125 const float f = __bfloat162float(h);
1126 const unsigned short int max_val = 0xffffU;
1127 const unsigned short int min_val = 0U;
1128 const unsigned short bits = static_cast<unsigned short>(static_cast<__nv_bfloat16_raw>(h).x << 1U);
1129 // saturation fixup
1130 if (bits > (unsigned short)0xFF00U) {
1131 // NaN
1132 val = 0U;
1133 } else if (f > static_cast<float>(max_val)) {
1134 // saturate maximum
1135 val = max_val;
1136 } else if (f < static_cast<float>(min_val)) {
1137 // saturate minimum
1138 val = min_val;
1139 } else {
1140 val = static_cast<unsigned short int>(f);
1141 }
1142)
1143 return val;
1144}
1145#if defined(__CUDACC__) || defined(_NVHPC_CUDA)
1146__CUDA_BF16_DECL__ unsigned short int __bfloat162ushort_rd(const __nv_bfloat16 h)
1147{
1148 unsigned short int val;
1149NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1150 asm("cvt.rmi.u16.bf16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
1151,
1152 asm("{ .reg.f32 f;\n"
1153 " mov.b32 f, {0,%1};\n"
1154 " cvt.rmi.u16.f32 %0,f;\n}"
1155 :"=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
1156)
1157 return val;
1158}
1159__CUDA_BF16_DECL__ unsigned short int __bfloat162ushort_ru(const __nv_bfloat16 h)
1160{
1161 unsigned short int val;
1162NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1163 asm("cvt.rpi.u16.bf16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
1164,
1165 asm("{ .reg.f32 f;\n"
1166 " mov.b32 f, {0,%1};\n"
1167 " cvt.rpi.u16.f32 %0,f;\n}"
1168 :"=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
1169)
1170 return val;
1171}
1172#endif /* defined(__CUDACC__) || defined(_NVHPC_CUDA) */
1173__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat16 __ushort2bfloat16_rn(const unsigned short int i)
1174{
1175NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1176 __nv_bfloat16 val;
1177 asm("cvt.rn.bf16.u16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(i));
1178 return val;
1179,
1180 const float f = static_cast<float>(i);
1181 return __float2bfloat16_rn(f);
1182)
1183}
1184#if defined(__CUDACC__) || defined(_NVHPC_CUDA)
1185__CUDA_BF16_DECL__ __nv_bfloat16 __ushort2bfloat16_rz(const unsigned short int i)
1186{
1187NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1188 __nv_bfloat16 val;
1189 asm("cvt.rz.bf16.u16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(i));
1190 return val;
1191,
1192 return __float2bfloat16_rz(__uint2float_rz(static_cast<unsigned int>(i)));
1193)
1194}
1195__CUDA_BF16_DECL__ __nv_bfloat16 __ushort2bfloat16_rd(const unsigned short int i)
1196{
1197NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1198 __nv_bfloat16 val;
1199 asm("cvt.rm.bf16.u16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(i));
1200 return val;
