Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
cuda_bf16.hpp3792 linesDownload Raw Back to include
1/*
2* Copyright 1993-2024 NVIDIA Corporation.  All rights reserved.
3*
4* NOTICE TO LICENSEE:
5*
6* This source code and/or documentation ("Licensed Deliverables") are
7* subject to NVIDIA intellectual property rights under U.S. and
8* international Copyright laws.
9*
10* These Licensed Deliverables contained herein is PROPRIETARY and
11* CONFIDENTIAL to NVIDIA and is being provided under the terms and
12* conditions of a form of NVIDIA software license agreement by and
13* between NVIDIA and Licensee ("License Agreement") or electronically
14* accepted by Licensee.  Notwithstanding any terms or conditions to
15* the contrary in the License Agreement, reproduction or disclosure
16* of the Licensed Deliverables to any third party without the express
17* written consent of NVIDIA is prohibited.
18*
19* NOTWITHSTANDING ANY TERMS OR CONDITIONS TO THE CONTRARY IN THE
20* LICENSE AGREEMENT, NVIDIA MAKES NO REPRESENTATION ABOUT THE
21* SUITABILITY OF THESE LICENSED DELIVERABLES FOR ANY PURPOSE.  IT IS
22* PROVIDED "AS IS" WITHOUT EXPRESS OR IMPLIED WARRANTY OF ANY KIND.
23* NVIDIA DISCLAIMS ALL WARRANTIES WITH REGARD TO THESE LICENSED
24* DELIVERABLES, INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY,
25* NONINFRINGEMENT, AND FITNESS FOR A PARTICULAR PURPOSE.
26* NOTWITHSTANDING ANY TERMS OR CONDITIONS TO THE CONTRARY IN THE
27* LICENSE AGREEMENT, IN NO EVENT SHALL NVIDIA BE LIABLE FOR ANY
28* SPECIAL, INDIRECT, INCIDENTAL, OR CONSEQUENTIAL DAMAGES, OR ANY
29* DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS,
30* WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS
31* ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR PERFORMANCE
32* OF THESE LICENSED DELIVERABLES.
33*
34* U.S. Government End Users.  These Licensed Deliverables are a
35* "commercial item" as that term is defined at 48 C.F.R. 2.101 (OCT
36* 1995), consisting of "commercial computer software" and "commercial
37* computer software documentation" as such terms are used in 48
38* C.F.R. 12.212 (SEPT 1995) and is provided to the U.S. Government
39* only as a commercial end item.  Consistent with 48 C.F.R.12.212 and
40* 48 C.F.R. 227.7202-1 through 227.7202-4 (JUNE 1995), all
41* U.S. Government End Users acquire the Licensed Deliverables with
42* only those rights set forth herein.
43*
44* Any use of the Licensed Deliverables in individual and commercial
45* software must include, in the user documentation and internal
46* comments to the code, the above Disclaimer and U.S. Government End
47* Users Notice.
48*/
49
50#if !defined(__CUDA_BF16_HPP__)
51#define __CUDA_BF16_HPP__
52
53#if !defined(__CUDA_BF16_H__)
54#error "Do not include this file directly. Instead, include cuda_bf16.h."
55#endif
56
57#if !defined(IF_DEVICE_OR_CUDACC)
58#if defined(__CUDACC__)
59    #define IF_DEVICE_OR_CUDACC(d, c, f) NV_IF_ELSE_TARGET(NV_IS_DEVICE, d, c)
60#else
61    #define IF_DEVICE_OR_CUDACC(d, c, f) NV_IF_ELSE_TARGET(NV_IS_DEVICE, d, f)
62#endif
63#endif
64
65/* All other definitions in this file are only visible to C++ compilers */
66#if defined(__cplusplus)
67/**
68 * \ingroup CUDA_MATH_INTRINSIC_BFLOAT16_CONSTANTS
69 * \brief Defines floating-point positive infinity value for the \p nv_bfloat16 data type
70 */
71#define CUDART_INF_BF16            __ushort_as_bfloat16((unsigned short)0x7F80U)
72/**
73 * \ingroup CUDA_MATH_INTRINSIC_BFLOAT16_CONSTANTS
74 * \brief Defines canonical NaN value for the \p nv_bfloat16 data type
75 */
76#define CUDART_NAN_BF16            __ushort_as_bfloat16((unsigned short)0x7FFFU)
77/**
78 * \ingroup CUDA_MATH_INTRINSIC_BFLOAT16_CONSTANTS
79 * \brief Defines a minimum representable (denormalized) value for the \p nv_bfloat16 data type
80 */
81#define CUDART_MIN_DENORM_BF16     __ushort_as_bfloat16((unsigned short)0x0001U)
82/**
83 * \ingroup CUDA_MATH_INTRINSIC_BFLOAT16_CONSTANTS
84 * \brief Defines a maximum representable value for the \p nv_bfloat16 data type
85 */
86#define CUDART_MAX_NORMAL_BF16     __ushort_as_bfloat16((unsigned short)0x7F7FU)
87/**
88 * \ingroup CUDA_MATH_INTRINSIC_BFLOAT16_CONSTANTS
89 * \brief Defines a negative zero value for the \p nv_bfloat16 data type
90 */
91#define CUDART_NEG_ZERO_BF16       __ushort_as_bfloat16((unsigned short)0x8000U)
92/**
93 * \ingroup CUDA_MATH_INTRINSIC_BFLOAT16_CONSTANTS
94 * \brief Defines a positive zero value for the \p nv_bfloat16 data type
95 */
96#define CUDART_ZERO_BF16           __ushort_as_bfloat16((unsigned short)0x0000U)
97/**
98 * \ingroup CUDA_MATH_INTRINSIC_BFLOAT16_CONSTANTS
99 * \brief Defines a value of 1.0 for the \p nv_bfloat16 data type
100 */
101#define CUDART_ONE_BF16            __ushort_as_bfloat16((unsigned short)0x3F80U)
102
103    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16 &__nv_bfloat16::operator=(const __nv_bfloat16_raw &hr) { __x = hr.x; return *this; }
104    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ volatile __nv_bfloat16 &__nv_bfloat16::operator=(const __nv_bfloat16_raw &hr) volatile { __x = hr.x; return *this; }
105    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ volatile __nv_bfloat16 &__nv_bfloat16::operator=(const volatile __nv_bfloat16_raw &hr) volatile { __x = hr.x; return *this; }
106    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator __nv_bfloat16_raw() const { __nv_bfloat16_raw ret; ret.x = __x; return ret; }
107    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator __nv_bfloat16_raw() const volatile { __nv_bfloat16_raw ret; ret.x = __x; return ret; }
108
109#if !defined(__CUDA_NO_BFLOAT16_CONVERSIONS__)
110    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator float() const { return __bfloat162float(*this); }
111    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16 &__nv_bfloat16::operator=(const float f) { __x = __float2bfloat16(f).__x; return *this; }
112    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16 &__nv_bfloat16::operator=(const double f) { __x = __double2bfloat16(f).__x; return *this; }
113
114/*
115 * Implicit type conversions to/from integer types were only available to nvcc compilation.
116 * Introducing them for all compilers is a potentially breaking change that may affect
117 * overloads resolution and will require users to update their code.
118 * Define __CUDA_BF16_DISABLE_IMPLICIT_INTEGER_CONVERTS_FOR_HOST_COMPILERS__ to opt-out.
119 */
120#if !(defined __CUDA_BF16_DISABLE_IMPLICIT_INTEGER_CONVERTS_FOR_HOST_COMPILERS__) || (defined __CUDACC__)
121    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator signed char() const { return __bfloat162char_rz(*this); }
122    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator unsigned char() const { return __bfloat162uchar_rz(*this); }
123    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator char() const {
124        char value;
125        /* Suppress VS warning: warning C4127: conditional expression is constant */
126#if defined(_MSC_VER) && !defined(__CUDA_ARCH__)
127#pragma warning (push)
128#pragma warning (disable: 4127)
129#endif /* _MSC_VER && !defined(__CUDA_ARCH__) */
130        if (((char)-1) < (char)0)
131#if defined(_MSC_VER) && !defined(__CUDA_ARCH__)
132#pragma warning (pop)
133#endif /* _MSC_VER && !defined(__CUDA_ARCH__) */
134        {
135            value = static_cast<char>(__bfloat162char_rz(*this));
136        }
137        else
138        {
139            value = static_cast<char>(__bfloat162uchar_rz(*this));
140        }
141        return value;
142    }
143    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator short() const { return __bfloat162short_rz(*this); }
144    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator unsigned short() const { return __bfloat162ushort_rz(*this); }
145    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator int() const { return __bfloat162int_rz(*this); }
146    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator unsigned int() const { return __bfloat162uint_rz(*this); }
147    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator long() const {
148        long retval;
149        /* Suppress VS warning: warning C4127: conditional expression is constant */
150#if defined(_MSC_VER) && !defined(__CUDA_ARCH__)
151#pragma warning (push)
152#pragma warning (disable: 4127)
153#endif /* _MSC_VER && !defined(__CUDA_ARCH__) */
154        if (sizeof(long) == sizeof(long long))
155#if defined(_MSC_VER) && !defined(__CUDA_ARCH__)
156#pragma warning (pop)
157#endif /* _MSC_VER && !defined(__CUDA_ARCH__) */
158        {
159            retval = static_cast<long>(__bfloat162ll_rz(*this));
160        }
161        else
162        {
163            retval = static_cast<long>(__bfloat162int_rz(*this));
164        }
165        return retval;
166    }
167    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator unsigned long() const {
168        unsigned long retval;
169        /* Suppress VS warning: warning C4127: conditional expression is constant */
170#if defined(_MSC_VER) && !defined(__CUDA_ARCH__)
171#pragma warning (push)
172#pragma warning (disable: 4127)
173#endif /* _MSC_VER && !defined(__CUDA_ARCH__) */
174        if (sizeof(unsigned long) == sizeof(unsigned long long))
175#if defined(_MSC_VER) && !defined(__CUDA_ARCH__)
176#pragma warning (pop)
177#endif /* _MSC_VER && !defined(__CUDA_ARCH__) */
178        {
179            retval = static_cast<unsigned long>(__bfloat162ull_rz(*this));
180        }
181        else
182        {
183            retval = static_cast<unsigned long>(__bfloat162uint_rz(*this));
184        }
185        return retval;
186    }
187    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator long long() const { return __bfloat162ll_rz(*this); }
188    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16::operator unsigned long long() const { return __bfloat162ull_rz(*this); }
189    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16 &__nv_bfloat16::operator=(short val) { __x = __short2bfloat16_rn(val).__x; return *this; }
190    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16 &__nv_bfloat16::operator=(unsigned short val) { __x = __ushort2bfloat16_rn(val).__x; return *this; }
191    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16 &__nv_bfloat16::operator=(int val) { __x = __int2bfloat16_rn(val).__x; return *this; }
192    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16 &__nv_bfloat16::operator=(unsigned int val) { __x = __uint2bfloat16_rn(val).__x; return *this; }
193    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16 &__nv_bfloat16::operator=(long long val) { __x = __ll2bfloat16_rn(val).__x; return *this; }
194    __CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat16 &__nv_bfloat16::operator=(unsigned long long val) { __x = __ull2bfloat16_rn(val).__x; return *this; }
195#endif /* !(defined __CUDA_BF16_DISABLE_IMPLICIT_INTEGER_CONVERTS_FOR_HOST_COMPILERS__) || (defined __CUDACC__) */
196#endif /* !defined(__CUDA_NO_BFLOAT16_CONVERSIONS__) */
197
198
199#if !defined(__CUDA_NO_BFLOAT16_OPERATORS__)
200__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 operator+(const __nv_bfloat16 &lh, const __nv_bfloat16 &rh) { return __hadd(lh, rh); }
201__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 operator-(const __nv_bfloat16 &lh, const __nv_bfloat16 &rh) { return __hsub(lh, rh); }
202__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 operator*(const __nv_bfloat16 &lh, const __nv_bfloat16 &rh) { return __hmul(lh, rh); }
203__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 operator/(const __nv_bfloat16 &lh, const __nv_bfloat16 &rh) { return __hdiv(lh, rh); }
204__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 &operator+=(__nv_bfloat16 &lh, const __nv_bfloat16 &rh) { lh = __hadd(lh, rh); return lh; }
205__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 &operator-=(__nv_bfloat16 &lh, const __nv_bfloat16 &rh) { lh = __hsub(lh, rh); return lh; }
206__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 &operator*=(__nv_bfloat16 &lh, const __nv_bfloat16 &rh) { lh = __hmul(lh, rh); return lh; }
207__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 &operator/=(__nv_bfloat16 &lh, const __nv_bfloat16 &rh) { lh = __hdiv(lh, rh); return lh; }
208__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 &operator++(__nv_bfloat16 &h)      { __nv_bfloat16_raw one; one.x = 0x3F80U; h += one; return h; }
209__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 &operator--(__nv_bfloat16 &h)      { __nv_bfloat16_raw one; one.x = 0x3F80U; h -= one; return h; }
210__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16  operator++(__nv_bfloat16 &h, const int ignored)
211{
212    // ignored on purpose. Parameter only needed to distinguish the function declaration from other types of operators.
213    static_cast<void>(ignored);
214
215    const __nv_bfloat16 ret = h;
216    __nv_bfloat16_raw one;
217    one.x = 0x3F80U;
218    h += one;
219    return ret;
220}
221__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16  operator--(__nv_bfloat16 &h, const int ignored)
222{
223    // ignored on purpose. Parameter only needed to distinguish the function declaration from other types of operators.
224    static_cast<void>(ignored);
225
226    const __nv_bfloat16 ret = h;
227    __nv_bfloat16_raw one;
228    one.x = 0x3F80U;
229    h -= one;
230    return ret;
231}
232__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 operator+(const __nv_bfloat16 &h) { return h; }
233__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat16 operator-(const __nv_bfloat16 &h) { return __hneg(h); }
234__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator==(const __nv_bfloat16 &lh, const __nv_bfloat16 &rh) { return __heq(lh, rh); }
235__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator!=(const __nv_bfloat16 &lh, const __nv_bfloat16 &rh) { return __hneu(lh, rh); }
236__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator> (const __nv_bfloat16 &lh, const __nv_bfloat16 &rh) { return __hgt(lh, rh); }
237__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator< (const __nv_bfloat16 &lh, const __nv_bfloat16 &rh) { return __hlt(lh, rh); }
238__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator>=(const __nv_bfloat16 &lh, const __nv_bfloat16 &rh) { return __hge(lh, rh); }
239__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator<=(const __nv_bfloat16 &lh, const __nv_bfloat16 &rh) { return __hle(lh, rh); }
240#endif /* !defined(__CUDA_NO_BFLOAT16_OPERATORS__) */
241
242#if defined(__CPP_VERSION_AT_LEAST_11_BF16)
243__CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat162::__nv_bfloat162(__nv_bfloat162 &&src) {
244NV_IF_ELSE_TARGET(NV_IS_DEVICE,
245    __BFLOAT162_TO_UI(*this) = std::move(__BFLOAT162_TO_CUI(src));
246,
247    this->x = src.x;
248    this->y = src.y;
249)
250}
251__CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat162 &__nv_bfloat162::operator=(__nv_bfloat162 &&src) {
252NV_IF_ELSE_TARGET(NV_IS_DEVICE,
253    __BFLOAT162_TO_UI(*this) = std::move(__BFLOAT162_TO_CUI(src));
254,
255    this->x = src.x;
256    this->y = src.y;
257)
258    return *this;
259}
260#else
261__CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat162::__nv_bfloat162() { }
262#endif /* defined(__CPP_VERSION_AT_LEAST_11_BF16) */
263__CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat162::__nv_bfloat162(const __nv_bfloat162 &src) {
264NV_IF_ELSE_TARGET(NV_IS_DEVICE,
265   __BFLOAT162_TO_UI(*this) = __BFLOAT162_TO_CUI(src);
266,
267    this->x = src.x;
268    this->y = src.y;
269)
270}
271__CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat162 &__nv_bfloat162::operator=(const __nv_bfloat162 &src) {
272NV_IF_ELSE_TARGET(NV_IS_DEVICE,
273   __BFLOAT162_TO_UI(*this) = __BFLOAT162_TO_CUI(src);
274,
275    this->x = src.x;
276    this->y = src.y;
277)
278    return *this;
279}
280__CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat162::__nv_bfloat162(const __nv_bfloat162_raw &h2r ) {
281NV_IF_ELSE_TARGET(NV_IS_DEVICE,
282    __BFLOAT162_TO_UI(*this) = __BFLOAT162_TO_CUI(h2r);
283,
284    __nv_bfloat16_raw tr;
285    tr.x = h2r.x;
286    this->x = static_cast<__nv_bfloat16>(tr);
287    tr.x = h2r.y;
288    this->y = static_cast<__nv_bfloat16>(tr);
289)
290}
291__CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat162 &__nv_bfloat162::operator=(const __nv_bfloat162_raw &h2r) {
292NV_IF_ELSE_TARGET(NV_IS_DEVICE,
293    __BFLOAT162_TO_UI(*this) = __BFLOAT162_TO_CUI(h2r);
294,
295    __nv_bfloat16_raw tr;
296    tr.x = h2r.x;
297    this->x = static_cast<__nv_bfloat16>(tr);
298    tr.x = h2r.y;
299    this->y = static_cast<__nv_bfloat16>(tr);
300)
301    return *this;
302}
303__CUDA_HOSTDEVICE__ __CUDA_BF16_INLINE__ __nv_bfloat162::operator __nv_bfloat162_raw() const {
304    __nv_bfloat162_raw ret;
305NV_IF_ELSE_TARGET(NV_IS_DEVICE,
306    ret.x = 0U;
307    ret.y = 0U;
308    __BFLOAT162_TO_UI(ret) = __BFLOAT162_TO_CUI(*this);
309,
310    ret.x = static_cast<__nv_bfloat16_raw>(this->x).x;
311    ret.y = static_cast<__nv_bfloat16_raw>(this->y).x;
312)
313    return ret;
314}
315
316#if !defined(__CUDA_NO_BFLOAT162_OPERATORS__)
317__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162 operator+(const __nv_bfloat162 &lh, const __nv_bfloat162 &rh) { return __hadd2(lh, rh); }
318__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162 operator-(const __nv_bfloat162 &lh, const __nv_bfloat162 &rh) { return __hsub2(lh, rh); }
319__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162 operator*(const __nv_bfloat162 &lh, const __nv_bfloat162 &rh) { return __hmul2(lh, rh); }
320__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162 operator/(const __nv_bfloat162 &lh, const __nv_bfloat162 &rh) { return __h2div(lh, rh); }
321__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162& operator+=(__nv_bfloat162 &lh, const __nv_bfloat162 &rh) { lh = __hadd2(lh, rh); return lh; }
322__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162& operator-=(__nv_bfloat162 &lh, const __nv_bfloat162 &rh) { lh = __hsub2(lh, rh); return lh; }
323__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162& operator*=(__nv_bfloat162 &lh, const __nv_bfloat162 &rh) { lh = __hmul2(lh, rh); return lh; }
324__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162& operator/=(__nv_bfloat162 &lh, const __nv_bfloat162 &rh) { lh = __h2div(lh, rh); return lh; }
325__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162 &operator++(__nv_bfloat162 &h)      { __nv_bfloat162_raw one; one.x = 0x3F80U; one.y = 0x3F80U; h = __hadd2(h, one); return h; }
326__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162 &operator--(__nv_bfloat162 &h)      { __nv_bfloat162_raw one; one.x = 0x3F80U; one.y = 0x3F80U; h = __hsub2(h, one); return h; }
327__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162  operator++(__nv_bfloat162 &h, const int ignored)
328{
329    // ignored on purpose. Parameter only needed to distinguish the function declaration from other types of operators.
330    static_cast<void>(ignored);
331
332    const __nv_bfloat162 ret = h;
333    __nv_bfloat162_raw one;
334    one.x = 0x3F80U;
335    one.y = 0x3F80U;
336    h = __hadd2(h, one);
337    return ret;
338}
339__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162  operator--(__nv_bfloat162 &h, const int ignored)
340{
341    // ignored on purpose. Parameter only needed to distinguish the function declaration from other types of operators.
342    static_cast<void>(ignored);
343
344    const __nv_bfloat162 ret = h;
345    __nv_bfloat162_raw one;
346    one.x = 0x3F80U;
347    one.y = 0x3F80U;
348    h = __hsub2(h, one);
349    return ret;
350}
351__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162 operator+(const __nv_bfloat162 &h) { return h; }
352__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ __nv_bfloat162 operator-(const __nv_bfloat162 &h) { return __hneg2(h); }
353__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator==(const __nv_bfloat162 &lh, const __nv_bfloat162 &rh) { return __hbeq2(lh, rh); }
354__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator!=(const __nv_bfloat162 &lh, const __nv_bfloat162 &rh) { return __hbneu2(lh, rh); }
355__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator>(const __nv_bfloat162 &lh, const __nv_bfloat162 &rh) { return __hbgt2(lh, rh); }
356__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator<(const __nv_bfloat162 &lh, const __nv_bfloat162 &rh) { return __hblt2(lh, rh); }
357__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator>=(const __nv_bfloat162 &lh, const __nv_bfloat162 &rh) { return __hbge2(lh, rh); }
358__CUDA_HOSTDEVICE__ __CUDA_BF16_FORCEINLINE__ bool operator<=(const __nv_bfloat162 &lh, const __nv_bfloat162 &rh) { return __hble2(lh, rh); }
359#endif /* !defined(__CUDA_NO_BFLOAT162_OPERATORS__) */
360
361/* Restore warning for multiple assignment operators */
362#if defined(_MSC_VER) && _MSC_VER >= 1500
363#pragma warning( pop )
364#endif /* defined(_MSC_VER) && _MSC_VER >= 1500 */
365
366/* Restore -Weffc++ warnings from here on */
367#if defined(__GNUC__)
368#if __GNUC__ > 4 || (__GNUC__ == 4 && __GNUC_MINOR__ >= 6)
369#pragma GCC diagnostic pop
370#endif /* __GNUC__ > 4 || (__GNUC__ == 4 && __GNUC_MINOR__ >= 6) */
371#endif /* defined(__GNUC__) */
372
373#undef __CUDA_HOSTDEVICE__
374#undef __CUDA_ALIGN__
375
376__CUDA_HOSTDEVICE_BF16_DECL__ unsigned int __internal_float_as_uint(const float f)
377{
378    unsigned int u;
379IF_DEVICE_OR_CUDACC(
380    u = __float_as_uint(f);
381,
382    memcpy(&u, &f, sizeof(f));
383,
384    std::memcpy(&u, &f, sizeof(f));
385)
386    return u;
387}
388
389__CUDA_HOSTDEVICE_BF16_DECL__ float __internal_uint_as_float(const unsigned int u)
390{
391    float f;
392IF_DEVICE_OR_CUDACC(
393    f = __uint_as_float(u);
394,
395    memcpy(&f, &u, sizeof(u));
396,
397    std::memcpy(&f, &u, sizeof(u));
398)
399    return f;
400}
401
402__CUDA_HOSTDEVICE_BF16_DECL__ unsigned short __internal_float2bfloat16(const float f, unsigned int &sign, unsigned int &remainder)
403{
404    unsigned int x;
405
406    x = __internal_float_as_uint(f);
407
408    if ((x & 0x7fffffffU) > 0x7f800000U) {
409        sign = 0U;
410        remainder = 0U;
411        return static_cast<unsigned short>(0x7fffU);
412    }
413    sign = x >> 31U;
414    remainder = x << 16U;
415    return static_cast<unsigned short>(x >> 16U);
416}
417
418__CUDA_HOSTDEVICE_BF16_DECL__ float __internal_double2float_rn(const double x)
419{
420    float r;
421NV_IF_ELSE_TARGET(NV_IS_DEVICE,
422    asm("cvt.rn.f32.f64 %0, %1;" : "=f"(r) : "d"(x));
423,
424    r = static_cast<float>(x);
425)
426    return r;
427}
428__CUDA_HOSTDEVICE_BF16_DECL__ double __internal_float2double(const float x)
429{
430    double r;
431NV_IF_ELSE_TARGET(NV_IS_DEVICE,
432    asm("cvt.f64.f32 %0, %1;" : "=d"(r) : "f"(x));
433,
434    r = static_cast<double>(x);
435)
436    return r;
437}
438
439__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat16 __double2bfloat16(const double x)
440{
441NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
442    __nv_bfloat16 val;
443    asm("{  cvt.rn.bf16.f64 %0, %1;}\n" : "=h"(__BFLOAT16_TO_US(val)) : "d"(x));
444    return val;
445,
446    float f = __internal_double2float_rn(x);
447    const double d = __internal_float2double(f);
448    unsigned int u = __internal_float_as_uint(f);
449
450    bool x_is_not_nan = ((u << (unsigned)1U) <= (unsigned)0xFF000000U);
451
452
453    if ((x > 0.0) && (d > x)) {
454        u--;
455    }
456    if ((x < 0.0) && (d < x)) {
457        u--;
458    }
459    if ((d != x) && x_is_not_nan) {
460        u |= 1U;
461    }
462
463    f = __internal_uint_as_float(u);
464
465    return __float2bfloat16(f);
466)
467}
468
469__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat16 __float2bfloat16(const float a)
470{
471    __nv_bfloat16 val;
472NV_IF_ELSE_TARGET(NV_PROVIDES_SM_80,
473    asm("{  cvt.rn.bf16.f32 %0, %1;}\n" : "=h"(__BFLOAT16_TO_US(val)) : "f"(a));
474,
475    __nv_bfloat16_raw r;
476    unsigned int sign = 0U;
477    unsigned int remainder = 0U;
478    r.x = __internal_float2bfloat16(a, sign, remainder);
479    if ((remainder > 0x80000000U) || ((remainder == 0x80000000U) && ((r.x & 0x1U) != 0U))) {
480        r.x++;
481    }
482    val = r;
483)
484    return val;
485}
486__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat16 __float2bfloat16_rn(const float a)
487{
488    __nv_bfloat16 val;
489NV_IF_ELSE_TARGET(NV_PROVIDES_SM_80,
490    asm("{  cvt.rn.bf16.f32 %0, %1;}\n" : "=h"(__BFLOAT16_TO_US(val)) : "f"(a));
491,
492    __nv_bfloat16_raw r;
493    unsigned int sign = 0U;
494    unsigned int remainder = 0U;
495    r.x = __internal_float2bfloat16(a, sign, remainder);
496    if ((remainder > 0x80000000U) || ((remainder == 0x80000000U) && ((r.x & 0x1U) != 0U))) {
497        r.x++;
498    }
499    val = r;
500)
501    return val;
502}
503__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat16 __float2bfloat16_rz(const float a)
504{
505    __nv_bfloat16 val;
506NV_IF_ELSE_TARGET(NV_PROVIDES_SM_80,
507    asm("{  cvt.rz.bf16.f32 %0, %1;}\n" : "=h"(__BFLOAT16_TO_US(val)) : "f"(a));
508,
509    __nv_bfloat16_raw r;
510    unsigned int sign = 0U;
511    unsigned int remainder = 0U;
512    r.x = __internal_float2bfloat16(a, sign, remainder);
513    val = r;
514)
515    return val;
516}
517__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat16 __float2bfloat16_rd(const float a)
518{
519NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
520    __nv_bfloat16 val;
521    asm("{  cvt.rm.bf16.f32 %0, %1;}\n" : "=h"(__BFLOAT16_TO_US(val)) : "f"(a));
522    return val;
523,
524    __nv_bfloat16 val;
525    __nv_bfloat16_raw r;
526    unsigned int sign = 0U;
527    unsigned int remainder = 0U;
528    r.x = __internal_float2bfloat16(a, sign, remainder);
529    if ((remainder != 0U) && (sign != 0U)) {
530        r.x++;
531    }
532    val = r;
533    return val;
534)
535}
536__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat16 __float2bfloat16_ru(const float a)
537{
538NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
539    __nv_bfloat16 val;
540    asm("{  cvt.rp.bf16.f32 %0, %1;}\n" : "=h"(__BFLOAT16_TO_US(val)) : "f"(a));
541    return val;
542,
543    __nv_bfloat16 val;
544    __nv_bfloat16_raw r;
545    unsigned int sign = 0U;
546    unsigned int remainder = 0U;
547    r.x = __internal_float2bfloat16(a, sign, remainder);
548    if ((remainder != 0U) && (sign == 0U)) {
549        r.x++;
550    }
551    val = r;
552    return val;
553)
554}
555__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat162 __float2bfloat162_rn(const float a)
556{
557    __nv_bfloat162 val;
558NV_IF_ELSE_TARGET(NV_PROVIDES_SM_80,
559    asm("{.reg .b16 low;\n"
560        "  cvt.rn.bf16.f32 low, %1;\n"
561        "  mov.b32 %0, {low,low};}\n" : "=r"(__BFLOAT162_TO_UI(val)) : "f"(a));
562,
563    val = __nv_bfloat162(__float2bfloat16_rn(a), __float2bfloat16_rn(a));
564)
565    return val;
566}
567__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat162 __floats2bfloat162_rn(const float a, const float b)
568{
569    __nv_bfloat162 val;
570NV_IF_ELSE_TARGET(NV_PROVIDES_SM_80,
571    asm("{ cvt.rn.bf16x2.f32 %0, %2, %1;}\n"
572        : "=r"(__BFLOAT162_TO_UI(val)) : "f"(a), "f"(b));
573,
574    val = __nv_bfloat162(__float2bfloat16_rn(a), __float2bfloat16_rn(b));
575)
576    return val;
577}
578
579#if defined(__CUDACC__) || defined(_NVHPC_CUDA)
580__CUDA_BF16_DECL__ float __internal_device_bfloat162float(const unsigned short h)
581{
582    float f;
583NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
584    asm("{ cvt.f32.bf16 %0, %1;}\n" : "=f"(f) : "h"(h));
585,
586    asm("{ mov.b32 %0, {0,%1};}\n" : "=f"(f) : "h"(h));
587)
588    return f;
589}
590#endif /* defined(__CUDACC__) || defined(_NVHPC_CUDA) */
591
592__CUDA_HOSTDEVICE_BF16_DECL__ float __internal_bfloat162float(const unsigned short h)
593{
594    float f;
595NV_IF_ELSE_TARGET(NV_IS_DEVICE,
596    f = __internal_device_bfloat162float(h);
597,
598    unsigned int u = static_cast<unsigned int>(h) << 16;
599    f = __internal_uint_as_float(u);
600)
601    return f;
602}
603
604__CUDA_HOSTDEVICE_BF16_DECL__ float __bfloat162float(const __nv_bfloat16 a)
605{
606    return __internal_bfloat162float(static_cast<__nv_bfloat16_raw>(a).x);
607}
608__CUDA_HOSTDEVICE_BF16_DECL__ float __low2float(const __nv_bfloat162 a)
609{
610    return __internal_bfloat162float(static_cast<__nv_bfloat162_raw>(a).x);
611}
612
613__CUDA_HOSTDEVICE_BF16_DECL__ float __high2float(const __nv_bfloat162 a)
614{
615    return __internal_bfloat162float(static_cast<__nv_bfloat162_raw>(a).y);
616}
617
618/* CUDA vector-types compatible vector creation function (note returns __nv_bfloat162, not nv_bfloat162) */
619__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat162 make_bfloat162(const __nv_bfloat16 x, const __nv_bfloat16 y)
620{
621    __nv_bfloat162 t; t.x = x; t.y = y; return t;
622}
623
624/* Definitions of intrinsics */
625__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat162 __float22bfloat162_rn(const float2 a)
626{
627    __nv_bfloat162 val = __floats2bfloat162_rn(a.x, a.y);
628    return val;
629}
630__CUDA_HOSTDEVICE_BF16_DECL__ float2 __bfloat1622float2(const __nv_bfloat162 a)
631{
632    float hi_float;
633    float lo_float;
634    lo_float = __internal_bfloat162float(((__nv_bfloat162_raw)a).x);
635    hi_float = __internal_bfloat162float(((__nv_bfloat162_raw)a).y);
636    return make_float2(lo_float, hi_float);
637}
638#if defined(__CUDACC__) || defined(_NVHPC_CUDA)
639__CUDA_BF16_DECL__ int __bfloat162int_rn(const __nv_bfloat16 h)
640{
641NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
642    int val;
643    asm("{  cvt.rni.s32.bf16 %0, %1;}\n" : "=r"(val) : "h"(__BFLOAT16_TO_CUS(h)));
644    return val;
645,
646    return __float2int_rn(__bfloat162float(h));
647)
648}
649#endif /* defined(__CUDACC__) || defined(_NVHPC_CUDA) */
650
651__CUDA_HOSTDEVICE_BF16_DECL__ int __internal_bfloat162int_rz(const __nv_bfloat16 h)
652{
653    const float f = __bfloat162float(h);
654    int   i;
655NV_IF_ELSE_TARGET(NV_IS_DEVICE,
656    i = __float2int_rz(f);
657,
658    const int max_val = (int)0x7fffffffU;
659    const int min_val = (int)0x80000000U;
660    const unsigned short bits = static_cast<unsigned short>(static_cast<__nv_bfloat16_raw>(h).x << 1U);
661    // saturation fixup
662    if (bits > (unsigned short)0xFF00U) {
663        // NaN
664        i = 0;
665    } else if (f >= static_cast<float>(max_val)) {
666        // saturate maximum
667        i = max_val;
668    } else if (f < static_cast<float>(min_val)) {
669        // saturate minimum
670        i = min_val;
671    } else {
672        i = static_cast<int>(f);
673    }
674)
675    return i;
676}
677
678__CUDA_HOSTDEVICE_BF16_DECL__ int __bfloat162int_rz(const __nv_bfloat16 h)
679{
680NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
681    int val;
682    asm("{  cvt.rzi.s32.bf16 %0, %1;}\n" : "=r"(val) : "h"(__BFLOAT16_TO_CUS(h)));
683    return val;
684,
685    return __internal_bfloat162int_rz(h);
686)
687}
688#if defined(__CUDACC__) || defined(_NVHPC_CUDA)
689__CUDA_BF16_DECL__ int __bfloat162int_rd(const __nv_bfloat16 h)
690{
691    int val;
692NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
693    asm("{  cvt.rmi.s32.bf16 %0, %1;}\n" : "=r"(val) : "h"(__BFLOAT16_TO_CUS(h)));
694,
695    const float f = __bfloat162float(h);
696    asm("cvt.rmi.s32.f32 %0, %1;" : "=r"(val) : "f"(f));
697)
698    return val;
699}
700__CUDA_BF16_DECL__ int __bfloat162int_ru(const __nv_bfloat16 h)
701{
702    int val;
703NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
704    asm("{  cvt.rpi.s32.bf16 %0, %1;}\n" : "=r"(val) : "h"(__BFLOAT16_TO_CUS(h)));
705,
706    const float f = __bfloat162float(h);
707    asm("cvt.rpi.s32.f32 %0, %1;" : "=r"(val) : "f"(f));
708)
709    return val;
710}
711
712__CUDA_BF16_DECL__ __nv_bfloat16 __internal_device_int2bfloat16_rn(const int i)
713{
714NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
715        __nv_bfloat16 val;
716       asm("cvt.rn.bf16.s32 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "r"(i));
717       return val;
718,
719        const float ru = __int2float_ru(i);
720        const float rd = __int2float_rd(i);
721        float rz = __int2float_rz(i);
722        if (ru != rd) {
723            rz = __uint_as_float(__float_as_uint(rz) | 1U);
724        }
725        return __float2bfloat16_rn(rz);
726)
727}
728#endif /* defined(__CUDACC__) || defined(_NVHPC_CUDA) */
729__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat16 __int2bfloat16_rn(const int i)
730{
731NV_IF_ELSE_TARGET(NV_IS_DEVICE,
732    return __internal_device_int2bfloat16_rn(i);
733,
734    const double d = static_cast<double>(i);
735    return __double2bfloat16(d);
736)
737}
738__CUDA_HOSTDEVICE_BF16_DECL__ signed char __bfloat162char_rz(const __nv_bfloat16 h)
739{
740    signed char i;
741NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
742    unsigned short tmp = 0;
743    asm("{ .reg.b8 myreg;\n"
744        "  cvt.rzi.s8.bf16 myreg, %1;\n"
745        "  mov.b16 %0, {myreg, 0};\n}"
746         :"=h"(tmp) : "h"(__BFLOAT16_TO_CUS(h)));
747    const unsigned char u = static_cast<unsigned char>(tmp);
748    i = static_cast<signed char>(u);
749,
750    const float f = __bfloat162float(h);
751    const signed char max_val = (signed char)0x7fU;
752    const signed char min_val = (signed char)0x80U;
753    const unsigned short bits = static_cast<unsigned short>(static_cast<__nv_bfloat16_raw>(h).x << 1U);
754    // saturation fixup
755    if (bits > (unsigned short)0xFF00U) {
756        // NaN
757        i = 0;
758    } else if (f > static_cast<float>(max_val)) {
759        // saturate maximum
760        i = max_val;
761    } else if (f < static_cast<float>(min_val)) {
762        // saturate minimum
763        i = min_val;
764    } else {
765        // normal value, conversion is well-defined
766        i = static_cast<signed char>(f);
767    }
768)
769    return i;
770}
771
772__CUDA_HOSTDEVICE_BF16_DECL__ unsigned char __bfloat162uchar_rz(const __nv_bfloat16 h)
773{
774    unsigned char i;
775NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
776    unsigned short tmp = 0;
777    asm("{ .reg.b8 myreg;\n"
778        "  cvt.rzi.u8.bf16 myreg, %1;\n"
779        "  mov.b16 %0, {myreg, 0};\n}"
780         :"=h"(tmp) : "h"(__BFLOAT16_TO_CUS(h)));
781    i = static_cast<unsigned char>(tmp);
782,
783    const float f = __bfloat162float(h);
784    const unsigned char max_val = 0xffU;
785    const unsigned char min_val = 0U;
786    const unsigned short bits = static_cast<unsigned short>(static_cast<__nv_bfloat16_raw>(h).x << 1U);
787    // saturation fixup
788    if (bits > (unsigned short)0xFF00U) {
789        // NaN
790        i = 0U;
791    } else if (f > static_cast<float>(max_val)) {
792        // saturate maximum
793        i = max_val;
794    } else if (f < static_cast<float>(min_val)) {
795        // saturate minimum
796        i = min_val;
797    } else {
798        // normal value, conversion is well-defined
799        i = static_cast<unsigned char>(f);
800    }
801)
802    return i;
803}
804#if defined(__CUDACC__) || defined(_NVHPC_CUDA)
805__CUDA_BF16_DECL__ __nv_bfloat16 __int2bfloat16_rz(const int i)
806{
807NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
808     __nv_bfloat16 val;
809    asm("cvt.rz.bf16.s32 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "r"(i));
810    return val;
811,
812    return __float2bfloat16_rz(__int2float_rz(i));
813)
814}
815__CUDA_BF16_DECL__ __nv_bfloat16 __int2bfloat16_rd(const int i)
816{
817NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
818     __nv_bfloat16 val;
819    asm("cvt.rm.bf16.s32 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "r"(i));
820    return val;
821,
822    return __float2bfloat16_rd(__int2float_rd(i));
823)
824}
825
826__CUDA_BF16_DECL__ __nv_bfloat16 __int2bfloat16_ru(const int i)
827{
828NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
829     __nv_bfloat16 val;
830    asm("cvt.rp.bf16.s32 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "r"(i));
831    return val;
832,
833    return __float2bfloat16_ru(__int2float_ru(i));
834)
835}
836
837__CUDA_BF16_DECL__ short int __bfloat162short_rn(const __nv_bfloat16 h)
838{
839   short int val;
840NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
841   asm("cvt.rni.s16.bf16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
842,
843   asm("{ .reg.f32 f;\n"
844       "  mov.b32 f, {0,%1};\n"
845       "  cvt.rni.s16.f32 %0,f;\n}"
846        :"=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
847)
848   return val;
849}
850
851__CUDA_BF16_DECL__ short int __internal_device_bfloat162short_rz(const __nv_bfloat16 h)
852{
853    short int val;
854NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
855    asm("cvt.rzi.s16.bf16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
856,
857    asm("{ .reg.f32 f;\n"
858        "  mov.b32 f, {0,%1};\n"
859        "  cvt.rzi.s16.f32 %0,f;\n}"
860         :"=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
861)
862    return val;
863}
864#endif /* defined(__CUDACC__) || defined(_NVHPC_CUDA) */
865__CUDA_HOSTDEVICE_BF16_DECL__ short int __bfloat162short_rz(const __nv_bfloat16 h)
866{
867    short int val;
868NV_IF_ELSE_TARGET(NV_IS_DEVICE,
869    val = __internal_device_bfloat162short_rz(h);
870,
871    const float f = __bfloat162float(h);
872    const short int max_val = (short int)0x7fffU;
873    const short int min_val = (short int)0x8000U;
874    const unsigned short bits = static_cast<unsigned short>(static_cast<__nv_bfloat16_raw>(h).x << 1U);
875    // saturation fixup
876    if (bits > (unsigned short)0xFF00U) {
877        // NaN
878        val = 0;
879    } else if (f > static_cast<float>(max_val)) {
880        // saturate maximum
881        val = max_val;
882    } else if (f < static_cast<float>(min_val)) {
883        // saturate minimum
884        val = min_val;
885    } else {
886        val = static_cast<short int>(f);
887    }
888)
889   return val;
890}
891#if defined(__CUDACC__) || defined(_NVHPC_CUDA)
892__CUDA_BF16_DECL__ short int __bfloat162short_rd(const __nv_bfloat16 h)
893{
894   short int val;
895NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
896   asm("cvt.rmi.s16.bf16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
897,
898   asm("{ .reg.f32 f;\n"
899       "  mov.b32 f, {0,%1};\n"
900       "  cvt.rmi.s16.f32 %0,f;\n}"
901        :"=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
902)
903   return val;
904}
905__CUDA_BF16_DECL__ short int __bfloat162short_ru(const __nv_bfloat16 h)
906{
907   short int val;
908NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
909   asm("cvt.rpi.s16.bf16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
910,
911   asm("{ .reg.f32 f;\n"
912       "  mov.b32 f, {0,%1};\n"
913       "  cvt.rpi.s16.f32 %0,f;\n}"
914        :"=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
915)
916   return val;
917}
918#endif /* defined(__CUDACC__) || defined(_NVHPC_CUDA) */
919__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat16 __short2bfloat16_rn(const short int i)
920{
921NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
922    __nv_bfloat16 val;
923    asm("cvt.rn.bf16.s16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(i));
924    return val;
925,
926    const float f = static_cast<float>(i);
927    return __float2bfloat16_rn(f);
928)
929}
930#if defined(__CUDACC__) || defined(_NVHPC_CUDA)
931__CUDA_BF16_DECL__ __nv_bfloat16 __short2bfloat16_rz(const short int i)
932{
933NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
934    __nv_bfloat16 val;
935    asm("cvt.rz.bf16.s16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(i));
936    return val;
937,
938    return __float2bfloat16_rz(__int2float_rz(static_cast<int>(i)));
939)
940}
941__CUDA_BF16_DECL__ __nv_bfloat16 __short2bfloat16_rd(const short int i)
942{
943NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
944    __nv_bfloat16 val;
945    asm("cvt.rm.bf16.s16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(i));
946    return val;
947,
948    return __float2bfloat16_rd(__int2float_rd(static_cast<int>(i)));
949)
950}
951__CUDA_BF16_DECL__ __nv_bfloat16 __short2bfloat16_ru(const short int i)
952{
953NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
954    __nv_bfloat16 val;
955    asm("cvt.rp.bf16.s16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(i));
956    return val;
957,
958    return __float2bfloat16_ru(__int2float_ru(static_cast<int>(i)));
959)
960}
961
962__CUDA_BF16_DECL__ unsigned int __bfloat162uint_rn(const __nv_bfloat16 h)
963{
964NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
965    unsigned int val;
966    asm("{  cvt.rni.u32.bf16 %0, %1;}\n" : "=r"(val) : "h"(__BFLOAT16_TO_CUS(h)));
967    return val;
968,
969    return __float2uint_rn(__bfloat162float(h));
970)
971}
972#endif /* defined(__CUDACC__) || defined(_NVHPC_CUDA) */
973__CUDA_HOSTDEVICE_BF16_DECL__ unsigned int __internal_bfloat162uint_rz(const __nv_bfloat16 h)
974{
975    const float f = __bfloat162float(h);
976    unsigned int i;
977NV_IF_ELSE_TARGET(NV_IS_DEVICE,
978    i = __float2uint_rz(f);
979,
980    const unsigned int max_val = 0xffffffffU;
981    const unsigned int min_val = 0U;
982    const unsigned short bits = static_cast<unsigned short>(static_cast<__nv_bfloat16_raw>(h).x << 1U);
983    // saturation fixup
984    if (bits > (unsigned short)0xFF00U) {
985        // NaN
986        i = 0U;
987    } else if (f >= static_cast<float>(max_val)) {
988        // saturate maximum
989        i = max_val;
990    } else if (f < static_cast<float>(min_val)) {
991        // saturate minimum
992        i = min_val;
993    } else {
994        i = static_cast<unsigned int>(f);
995    }
996)
997    return i;
998}
999
1000__CUDA_HOSTDEVICE_BF16_DECL__ unsigned int __bfloat162uint_rz(const __nv_bfloat16 h)
1001{
1002NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1003    unsigned int val;
1004    asm("{  cvt.rzi.u32.bf16 %0, %1;}\n" : "=r"(val) : "h"(__BFLOAT16_TO_CUS(h)));
1005    return val;
1006,
1007    return __internal_bfloat162uint_rz(h);
1008)
1009}
1010#if defined(__CUDACC__) || defined(_NVHPC_CUDA)
1011__CUDA_BF16_DECL__ unsigned int __bfloat162uint_rd(const __nv_bfloat16 h)
1012{
1013NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1014    unsigned int val;
1015    asm("{  cvt.rmi.u32.bf16 %0, %1;}\n" : "=r"(val) : "h"(__BFLOAT16_TO_CUS(h)));
1016    return val;
1017,
1018    return __float2uint_rd(__bfloat162float(h));
1019)
1020}
1021__CUDA_BF16_DECL__ unsigned int __bfloat162uint_ru(const __nv_bfloat16 h)
1022{
1023    unsigned int val;
1024NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1025    asm("{  cvt.rpi.u32.bf16 %0, %1;}\n" : "=r"(val) : "h"(__BFLOAT16_TO_CUS(h)));
1026,
1027    const float f = __bfloat162float(h);
1028    asm("cvt.rpi.u32.f32 %0, %1;" : "=r"(val) : "f"(f));
1029)
1030    return val;
1031}
1032
1033__CUDA_BF16_DECL__ __nv_bfloat16 __internal_device_uint2bfloat16_rn(const unsigned int i)
1034{
1035NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1036    __nv_bfloat16 val;
1037    asm("cvt.rn.bf16.u32 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "r"(i));
1038    return val;
1039,
1040    const float ru = __uint2float_ru(i);
1041    const float rd = __uint2float_rd(i);
1042    float rz = __uint2float_rz(i);
1043    if (ru != rd) {
1044        rz = __uint_as_float(__float_as_uint(rz) | 1U);
1045    }
1046    return __float2bfloat16_rn(rz);
1047)
1048}
1049#endif /* defined(__CUDACC__) || defined(_NVHPC_CUDA) */
1050__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat16 __uint2bfloat16_rn(const unsigned int i)
1051{
1052NV_IF_ELSE_TARGET(NV_IS_DEVICE,
1053    return __internal_device_uint2bfloat16_rn(i);
1054,
1055    const double d = static_cast<double>(i);
1056    return __double2bfloat16(d);
1057)
1058}
1059#if defined(__CUDACC__) || defined(_NVHPC_CUDA)
1060__CUDA_BF16_DECL__ __nv_bfloat16 __uint2bfloat16_rz(const unsigned int i)
1061{
1062NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1063     __nv_bfloat16 val;
1064    asm("cvt.rz.bf16.u32 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "r"(i));
1065    return val;
1066,
1067    return __float2bfloat16_rz(__uint2float_rz(i));
1068)
1069}
1070__CUDA_BF16_DECL__ __nv_bfloat16 __uint2bfloat16_rd(const unsigned int i)
1071{
1072NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1073     __nv_bfloat16 val;
1074    asm("cvt.rm.bf16.u32 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "r"(i));
1075    return val;
1076,
1077    return __float2bfloat16_rd(__uint2float_rd(i));
1078)
1079}
1080__CUDA_BF16_DECL__ __nv_bfloat16 __uint2bfloat16_ru(const unsigned int i)
1081{
1082NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1083     __nv_bfloat16 val;
1084    asm("cvt.rp.bf16.u32 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "r"(i));
1085    return val;
1086,
1087    return __float2bfloat16_ru(__uint2float_ru(i));
1088)
1089}
1090
1091__CUDA_BF16_DECL__ unsigned short int __bfloat162ushort_rn(const __nv_bfloat16 h)
1092{
1093   unsigned short int val;
1094NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1095   asm("cvt.rni.u16.bf16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
1096,
1097   asm("{ .reg.f32 f;\n"
1098       "  mov.b32 f, {0,%1};\n"
1099       "  cvt.rni.u16.f32 %0,f;\n}"
1100        :"=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
1101)
1102   return val;
1103}
1104
1105__CUDA_BF16_DECL__ unsigned short int __internal_device_bfloat162ushort_rz(const __nv_bfloat16 h)
1106{
1107   unsigned short int val;
1108NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1109   asm("cvt.rzi.u16.bf16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
1110,
1111   asm("{ .reg.f32 f;\n"
1112       "  mov.b32 f, {0,%1};\n"
1113       "  cvt.rzi.u16.f32 %0,f;\n}"
1114        :"=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
1115)
1116   return val;
1117}
1118#endif /* defined(__CUDACC__) || defined(_NVHPC_CUDA) */
1119__CUDA_HOSTDEVICE_BF16_DECL__ unsigned short int __bfloat162ushort_rz(const __nv_bfloat16 h)
1120{
1121   unsigned short int val;
1122NV_IF_ELSE_TARGET(NV_IS_DEVICE,
1123   val = __internal_device_bfloat162ushort_rz(h);
1124,
1125    const float f = __bfloat162float(h);
1126    const unsigned short int max_val = 0xffffU;
1127    const unsigned short int min_val = 0U;
1128    const unsigned short bits = static_cast<unsigned short>(static_cast<__nv_bfloat16_raw>(h).x << 1U);
1129    // saturation fixup
1130    if (bits > (unsigned short)0xFF00U) {
1131        // NaN
1132        val = 0U;
1133    } else if (f > static_cast<float>(max_val)) {
1134        // saturate maximum
1135        val = max_val;
1136    } else if (f < static_cast<float>(min_val)) {
1137        // saturate minimum
1138        val = min_val;
1139    } else {
1140        val = static_cast<unsigned short int>(f);
1141    }
1142)
1143   return val;
1144}
1145#if defined(__CUDACC__) || defined(_NVHPC_CUDA)
1146__CUDA_BF16_DECL__ unsigned short int __bfloat162ushort_rd(const __nv_bfloat16 h)
1147{
1148   unsigned short int val;
1149NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1150   asm("cvt.rmi.u16.bf16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
1151,
1152   asm("{ .reg.f32 f;\n"
1153       "  mov.b32 f, {0,%1};\n"
1154       "  cvt.rmi.u16.f32 %0,f;\n}"
1155        :"=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
1156)
1157   return val;
1158}
1159__CUDA_BF16_DECL__ unsigned short int __bfloat162ushort_ru(const __nv_bfloat16 h)
1160{
1161   unsigned short int val;
1162NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1163   asm("cvt.rpi.u16.bf16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
1164,
1165   asm("{ .reg.f32 f;\n"
1166       "  mov.b32 f, {0,%1};\n"
1167       "  cvt.rpi.u16.f32 %0,f;\n}"
1168        :"=h"(__BFLOAT16_TO_US(val)) : "h"(__BFLOAT16_TO_CUS(h)));
1169)
1170   return val;
1171}
1172#endif /* defined(__CUDACC__) || defined(_NVHPC_CUDA) */
1173__CUDA_HOSTDEVICE_BF16_DECL__ __nv_bfloat16 __ushort2bfloat16_rn(const unsigned short int i)
1174{
1175NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1176    __nv_bfloat16 val;
1177    asm("cvt.rn.bf16.u16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(i));
1178    return val;
1179,
1180    const float f = static_cast<float>(i);
1181    return __float2bfloat16_rn(f);
1182)
1183}
1184#if defined(__CUDACC__) || defined(_NVHPC_CUDA)
1185__CUDA_BF16_DECL__ __nv_bfloat16 __ushort2bfloat16_rz(const unsigned short int i)
1186{
1187NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1188    __nv_bfloat16 val;
1189    asm("cvt.rz.bf16.u16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(i));
1190    return val;
1191,
1192    return __float2bfloat16_rz(__uint2float_rz(static_cast<unsigned int>(i)));
1193)
1194}
1195__CUDA_BF16_DECL__ __nv_bfloat16 __ushort2bfloat16_rd(const unsigned short int i)
1196{
1197NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
1198    __nv_bfloat16 val;
1199    asm("cvt.rm.bf16.u16 %0, %1;" : "=h"(__BFLOAT16_TO_US(val)) : "h"(i));
1200    return val;

Showing the first 1,200 of 3792 lines. Download the file for the rest.

codekingpro/portable-devtools · Team Ai