codekingpro/portable-devtools
114k
1/*
2 * Copyright 1993-2019 NVIDIA Corporation. All rights reserved.
3 *
4 * NOTICE TO LICENSEE:
5 *
6 * This source code and/or documentation ("Licensed Deliverables") are
7 * subject to NVIDIA intellectual property rights under U.S. and
8 * international Copyright laws.
9 *
10 * These Licensed Deliverables contained herein is PROPRIETARY and
11 * CONFIDENTIAL to NVIDIA and is being provided under the terms and
12 * conditions of a form of NVIDIA software license agreement by and
13 * between NVIDIA and Licensee ("License Agreement") or electronically
14 * accepted by Licensee. Notwithstanding any terms or conditions to
15 * the contrary in the License Agreement, reproduction or disclosure
16 * of the Licensed Deliverables to any third party without the express
17 * written consent of NVIDIA is prohibited.
18 *
19 * NOTWITHSTANDING ANY TERMS OR CONDITIONS TO THE CONTRARY IN THE
20 * LICENSE AGREEMENT, NVIDIA MAKES NO REPRESENTATION ABOUT THE
21 * SUITABILITY OF THESE LICENSED DELIVERABLES FOR ANY PURPOSE. IT IS
22 * PROVIDED "AS IS" WITHOUT EXPRESS OR IMPLIED WARRANTY OF ANY KIND.
23 * NVIDIA DISCLAIMS ALL WARRANTIES WITH REGARD TO THESE LICENSED
24 * DELIVERABLES, INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY,
25 * NONINFRINGEMENT, AND FITNESS FOR A PARTICULAR PURPOSE.
26 * NOTWITHSTANDING ANY TERMS OR CONDITIONS TO THE CONTRARY IN THE
27 * LICENSE AGREEMENT, IN NO EVENT SHALL NVIDIA BE LIABLE FOR ANY
28 * SPECIAL, INDIRECT, INCIDENTAL, OR CONSEQUENTIAL DAMAGES, OR ANY
29 * DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS,
30 * WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS
31 * ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR PERFORMANCE
32 * OF THESE LICENSED DELIVERABLES.
33 *
34 * U.S. Government End Users. These Licensed Deliverables are a
35 * "commercial item" as that term is defined at 48 C.F.R. 2.101 (OCT
36 * 1995), consisting of "commercial computer software" and "commercial
37 * computer software documentation" as such terms are used in 48
38 * C.F.R. 12.212 (SEPT 1995) and is provided to the U.S. Government
39 * only as a commercial end item. Consistent with 48 C.F.R.12.212 and
40 * 48 C.F.R. 227.7202-1 through 227.7202-4 (JUNE 1995), all
41 * U.S. Government End Users acquire the Licensed Deliverables with
42 * only those rights set forth herein.
43 *
44 * Any use of the Licensed Deliverables in individual and commercial
45 * software must include, in the user documentation and internal
46 * comments to the code, the above Disclaimer and U.S. Government End
47 * Users Notice.
48 */
49
50#ifndef _CUDA_PIPELINE_H_
51# define _CUDA_PIPELINE_H_
52
53# include "cuda_pipeline_primitives.h"
54
55# if !defined(_CUDA_PIPELINE_CPLUSPLUS_11_OR_LATER)
56# error This file requires compiler support for the ISO C++ 2011 standard. This support must be enabled with the \
57 -std=c++11 compiler option.
58# endif
59
60# if defined(_CUDA_PIPELINE_ARCH_700_OR_LATER)
61# include "cuda_awbarrier.h"
62# endif
63
64// Integration with libcu++'s cuda::barrier<cuda::thread_scope_block>.
65
66# if defined(_CUDA_PIPELINE_ARCH_700_OR_LATER)
67# if defined(_LIBCUDACXX_CUDA_ABI_VERSION)
68# define _LIBCUDACXX_PIPELINE_ASSUMED_ABI_VERSION _LIBCUDACXX_CUDA_ABI_VERSION
69# else
70# define _LIBCUDACXX_PIPELINE_ASSUMED_ABI_VERSION 4
71# endif
72
73# define _LIBCUDACXX_PIPELINE_CONCAT(X, Y) X ## Y
74# define _LIBCUDACXX_PIPELINE_CONCAT2(X, Y) _LIBCUDACXX_PIPELINE_CONCAT(X, Y)
75# define _LIBCUDACXX_PIPELINE_INLINE_NAMESPACE _LIBCUDACXX_PIPELINE_CONCAT2(__, _LIBCUDACXX_PIPELINE_ASSUMED_ABI_VERSION)
76
77namespace cuda { inline namespace _LIBCUDACXX_PIPELINE_INLINE_NAMESPACE {
78 struct __block_scope_barrier_base;
79}}
80
81# endif
82
83_CUDA_PIPELINE_BEGIN_NAMESPACE
84
85template<size_t N, typename T>
86_CUDA_PIPELINE_QUALIFIER
87auto segment(T* ptr) -> T(*)[N];
88
89class pipeline {
90public:
91 pipeline(const pipeline&) = delete;
92 pipeline(pipeline&&) = delete;
93 pipeline& operator=(const pipeline&) = delete;
94 pipeline& operator=(pipeline&&) = delete;
95
96 _CUDA_PIPELINE_QUALIFIER pipeline();
97 _CUDA_PIPELINE_QUALIFIER size_t commit();
98 _CUDA_PIPELINE_QUALIFIER void commit_and_wait();
99 _CUDA_PIPELINE_QUALIFIER void wait(size_t batch);
100 template<unsigned N>
101 _CUDA_PIPELINE_QUALIFIER void wait_prior();
102
103# if defined(_CUDA_PIPELINE_ARCH_700_OR_LATER)
104 _CUDA_PIPELINE_QUALIFIER void arrive_on(awbarrier& barrier);
105 _CUDA_PIPELINE_QUALIFIER void arrive_on(cuda::__block_scope_barrier_base& barrier);
106# endif
107
108private:
109 size_t current_batch;
110};
111
112template<class T>
113_CUDA_PIPELINE_QUALIFIER
114void memcpy_async(T& dst, const T& src, pipeline& pipe);
115
116template<class T, size_t DstN, size_t SrcN>
117_CUDA_PIPELINE_QUALIFIER
118void memcpy_async(T(*dst)[DstN], const T(*src)[SrcN], pipeline& pipe);
119
120template<size_t N, typename T>
121_CUDA_PIPELINE_QUALIFIER
122auto segment(T* ptr) -> T(*)[N]
123{
124 return (T(*)[N])ptr;
125}
126
127_CUDA_PIPELINE_QUALIFIER
128pipeline::pipeline()
129 : current_batch(0)
130{
131}
132
133_CUDA_PIPELINE_QUALIFIER
134size_t pipeline::commit()
135{
136 _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_commit();
137 return this->current_batch++;
138}
139
140_CUDA_PIPELINE_QUALIFIER
141void pipeline::commit_and_wait()
142{
143 (void)pipeline::commit();
144 pipeline::wait_prior<0>();
145}
146
147_CUDA_PIPELINE_QUALIFIER
148void pipeline::wait(size_t batch)
149{
150 const size_t prior = this->current_batch > batch ? this->current_batch - batch : 0;
151
152 switch (prior) {
153 case 0 : _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_wait_prior<0>(); break;
154 case 1 : _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_wait_prior<1>(); break;
155 case 2 : _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_wait_prior<2>(); break;
156 case 3 : _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_wait_prior<3>(); break;
157 case 4 : _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_wait_prior<4>(); break;
158 case 5 : _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_wait_prior<5>(); break;
159 case 6 : _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_wait_prior<6>(); break;
160 case 7 : _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_wait_prior<7>(); break;
161 default : _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_wait_prior<8>(); break;
162 }
163}
164
165template<unsigned N>
166_CUDA_PIPELINE_QUALIFIER
167void pipeline::wait_prior()
168{
169 _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_wait_prior<N>();
170}
171
172# if defined(_CUDA_PIPELINE_ARCH_700_OR_LATER)
173_CUDA_PIPELINE_QUALIFIER
174void pipeline::arrive_on(awbarrier& barrier)
175{
176 _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_arrive_on(&barrier.barrier);
177}
178
179_CUDA_PIPELINE_QUALIFIER
180void pipeline::arrive_on(cuda::__block_scope_barrier_base & barrier)
181{
182 _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_arrive_on(reinterpret_cast<uint64_t *>(&barrier));
183}
184# endif
185
186template<class T>
187_CUDA_PIPELINE_QUALIFIER
188void memcpy_async(T& dst, const T& src, pipeline& pipe)
189{
190 _CUDA_PIPELINE_ASSERT(!(reinterpret_cast<uintptr_t>(&src) & (alignof(T) - 1)));
191 _CUDA_PIPELINE_ASSERT(!(reinterpret_cast<uintptr_t>(&dst) & (alignof(T) - 1)));
192
193 if (__is_trivially_copyable(T)) {
194 _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_copy_relaxed<sizeof(T), alignof(T)>(
195 reinterpret_cast<void*>(&dst), reinterpret_cast<const void*>(&src));
196 } else {
197 dst = src;
198 }
199}
200
201template<class T, size_t DstN, size_t SrcN>
202_CUDA_PIPELINE_QUALIFIER
203void memcpy_async(T(*dst)[DstN], const T(*src)[SrcN], pipeline& pipe)
204{
205 constexpr size_t dst_size = sizeof(*dst);
206 constexpr size_t src_size = sizeof(*src);
207 static_assert(dst_size == 4 || dst_size == 8 || dst_size == 16, "Unsupported copy size.");
208 static_assert(src_size <= dst_size, "Source size must be less than or equal to destination size.");
209 _CUDA_PIPELINE_ASSERT(!(reinterpret_cast<uintptr_t>(src) & (dst_size - 1)));
210 _CUDA_PIPELINE_ASSERT(!(reinterpret_cast<uintptr_t>(dst) & (dst_size - 1)));
211
212 if (__is_trivially_copyable(T)) {
213 _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_copy_strict<sizeof(*dst), sizeof(*src)>(
214 reinterpret_cast<void*>(*dst), reinterpret_cast<const void*>(*src));
215 } else {
216 for (size_t i = 0; i < DstN; ++i) {
217 (*dst)[i] = (i < SrcN) ? (*src)[i] : T();
218 }
219 }
220}
221
222_CUDA_PIPELINE_END_NAMESPACE
223
224#endif /* !_CUDA_PIPELINE_H_ */
225 