Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
cuda_pipeline.h225 linesDownload Raw Back to include
1/*
2 * Copyright 1993-2019 NVIDIA Corporation.  All rights reserved.
3 *
4 * NOTICE TO LICENSEE:
5 *
6 * This source code and/or documentation ("Licensed Deliverables") are
7 * subject to NVIDIA intellectual property rights under U.S. and
8 * international Copyright laws.
9 *
10 * These Licensed Deliverables contained herein is PROPRIETARY and
11 * CONFIDENTIAL to NVIDIA and is being provided under the terms and
12 * conditions of a form of NVIDIA software license agreement by and
13 * between NVIDIA and Licensee ("License Agreement") or electronically
14 * accepted by Licensee.  Notwithstanding any terms or conditions to
15 * the contrary in the License Agreement, reproduction or disclosure
16 * of the Licensed Deliverables to any third party without the express
17 * written consent of NVIDIA is prohibited.
18 *
19 * NOTWITHSTANDING ANY TERMS OR CONDITIONS TO THE CONTRARY IN THE
20 * LICENSE AGREEMENT, NVIDIA MAKES NO REPRESENTATION ABOUT THE
21 * SUITABILITY OF THESE LICENSED DELIVERABLES FOR ANY PURPOSE.  IT IS
22 * PROVIDED "AS IS" WITHOUT EXPRESS OR IMPLIED WARRANTY OF ANY KIND.
23 * NVIDIA DISCLAIMS ALL WARRANTIES WITH REGARD TO THESE LICENSED
24 * DELIVERABLES, INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY,
25 * NONINFRINGEMENT, AND FITNESS FOR A PARTICULAR PURPOSE.
26 * NOTWITHSTANDING ANY TERMS OR CONDITIONS TO THE CONTRARY IN THE
27 * LICENSE AGREEMENT, IN NO EVENT SHALL NVIDIA BE LIABLE FOR ANY
28 * SPECIAL, INDIRECT, INCIDENTAL, OR CONSEQUENTIAL DAMAGES, OR ANY
29 * DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS,
30 * WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS
31 * ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR PERFORMANCE
32 * OF THESE LICENSED DELIVERABLES.
33 *
34 * U.S. Government End Users.  These Licensed Deliverables are a
35 * "commercial item" as that term is defined at 48 C.F.R. 2.101 (OCT
36 * 1995), consisting of "commercial computer software" and "commercial
37 * computer software documentation" as such terms are used in 48
38 * C.F.R. 12.212 (SEPT 1995) and is provided to the U.S. Government
39 * only as a commercial end item.  Consistent with 48 C.F.R.12.212 and
40 * 48 C.F.R. 227.7202-1 through 227.7202-4 (JUNE 1995), all
41 * U.S. Government End Users acquire the Licensed Deliverables with
42 * only those rights set forth herein.
43 *
44 * Any use of the Licensed Deliverables in individual and commercial
45 * software must include, in the user documentation and internal
46 * comments to the code, the above Disclaimer and U.S. Government End
47 * Users Notice.
48 */
49
50#ifndef _CUDA_PIPELINE_H_
51# define _CUDA_PIPELINE_H_
52
53# include "cuda_pipeline_primitives.h"
54
55# if !defined(_CUDA_PIPELINE_CPLUSPLUS_11_OR_LATER)
56#  error This file requires compiler support for the ISO C++ 2011 standard. This support must be enabled with the \
57         -std=c++11 compiler option.
58# endif
59
60# if defined(_CUDA_PIPELINE_ARCH_700_OR_LATER)
61#  include "cuda_awbarrier.h"
62# endif
63
64// Integration with libcu++'s cuda::barrier<cuda::thread_scope_block>.
65
66# if defined(_CUDA_PIPELINE_ARCH_700_OR_LATER)
67#  if defined(_LIBCUDACXX_CUDA_ABI_VERSION)
68#   define _LIBCUDACXX_PIPELINE_ASSUMED_ABI_VERSION _LIBCUDACXX_CUDA_ABI_VERSION
69#  else
70#   define _LIBCUDACXX_PIPELINE_ASSUMED_ABI_VERSION 4
71#  endif
72
73#  define _LIBCUDACXX_PIPELINE_CONCAT(X, Y) X ## Y
74#  define _LIBCUDACXX_PIPELINE_CONCAT2(X, Y) _LIBCUDACXX_PIPELINE_CONCAT(X, Y)
75#  define _LIBCUDACXX_PIPELINE_INLINE_NAMESPACE _LIBCUDACXX_PIPELINE_CONCAT2(__, _LIBCUDACXX_PIPELINE_ASSUMED_ABI_VERSION)
76
77namespace cuda { inline namespace _LIBCUDACXX_PIPELINE_INLINE_NAMESPACE {
78    struct __block_scope_barrier_base;
79}}
80
81# endif
82
83_CUDA_PIPELINE_BEGIN_NAMESPACE
84
85template<size_t N, typename T>
86_CUDA_PIPELINE_QUALIFIER
87auto segment(T* ptr) -> T(*)[N];
88
89class pipeline {
90public:
91    pipeline(const pipeline&) = delete;
92    pipeline(pipeline&&) = delete;
93    pipeline& operator=(const pipeline&) = delete;
94    pipeline& operator=(pipeline&&) = delete;
95
96    _CUDA_PIPELINE_QUALIFIER pipeline();
97    _CUDA_PIPELINE_QUALIFIER size_t commit();
98    _CUDA_PIPELINE_QUALIFIER void commit_and_wait();
99    _CUDA_PIPELINE_QUALIFIER void wait(size_t batch);
100    template<unsigned N>
101    _CUDA_PIPELINE_QUALIFIER void wait_prior();
102
103# if defined(_CUDA_PIPELINE_ARCH_700_OR_LATER)
104    _CUDA_PIPELINE_QUALIFIER void arrive_on(awbarrier& barrier);
105    _CUDA_PIPELINE_QUALIFIER void arrive_on(cuda::__block_scope_barrier_base& barrier);
106# endif
107
108private:
109    size_t current_batch;
110};
111
112template<class T>
113_CUDA_PIPELINE_QUALIFIER
114void memcpy_async(T& dst, const T& src, pipeline& pipe);
115
116template<class T, size_t DstN, size_t SrcN>
117_CUDA_PIPELINE_QUALIFIER
118void memcpy_async(T(*dst)[DstN], const T(*src)[SrcN], pipeline& pipe);
119
120template<size_t N, typename T>
121_CUDA_PIPELINE_QUALIFIER
122auto segment(T* ptr) -> T(*)[N]
123{
124    return (T(*)[N])ptr;
125}
126
127_CUDA_PIPELINE_QUALIFIER
128pipeline::pipeline()
129    : current_batch(0)
130{
131}
132
133_CUDA_PIPELINE_QUALIFIER
134size_t pipeline::commit()
135{
136    _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_commit();
137    return this->current_batch++;
138}
139
140_CUDA_PIPELINE_QUALIFIER
141void pipeline::commit_and_wait()
142{
143    (void)pipeline::commit();
144    pipeline::wait_prior<0>();
145}
146
147_CUDA_PIPELINE_QUALIFIER
148void pipeline::wait(size_t batch)
149{
150    const size_t prior = this->current_batch > batch ? this->current_batch - batch : 0;
151
152    switch (prior) {
153    case  0 : _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_wait_prior<0>(); break;
154    case  1 : _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_wait_prior<1>(); break;
155    case  2 : _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_wait_prior<2>(); break;
156    case  3 : _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_wait_prior<3>(); break;
157    case  4 : _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_wait_prior<4>(); break;
158    case  5 : _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_wait_prior<5>(); break;
159    case  6 : _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_wait_prior<6>(); break;
160    case  7 : _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_wait_prior<7>(); break;
161    default : _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_wait_prior<8>(); break;
162    }
163}
164
165template<unsigned N>
166_CUDA_PIPELINE_QUALIFIER
167void pipeline::wait_prior()
168{
169    _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_wait_prior<N>();
170}
171
172# if defined(_CUDA_PIPELINE_ARCH_700_OR_LATER)
173_CUDA_PIPELINE_QUALIFIER
174void pipeline::arrive_on(awbarrier& barrier)
175{
176    _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_arrive_on(&barrier.barrier);
177}
178
179_CUDA_PIPELINE_QUALIFIER
180void pipeline::arrive_on(cuda::__block_scope_barrier_base & barrier)
181{
182    _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_arrive_on(reinterpret_cast<uint64_t *>(&barrier));
183}
184# endif
185
186template<class T>
187_CUDA_PIPELINE_QUALIFIER
188void memcpy_async(T& dst, const T& src, pipeline& pipe)
189{
190    _CUDA_PIPELINE_ASSERT(!(reinterpret_cast<uintptr_t>(&src) & (alignof(T) - 1)));
191    _CUDA_PIPELINE_ASSERT(!(reinterpret_cast<uintptr_t>(&dst) & (alignof(T) - 1)));
192
193    if (__is_trivially_copyable(T)) {
194        _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_copy_relaxed<sizeof(T), alignof(T)>(
195                reinterpret_cast<void*>(&dst), reinterpret_cast<const void*>(&src));
196    } else {
197        dst = src;
198    }
199}
200
201template<class T, size_t DstN, size_t SrcN>
202_CUDA_PIPELINE_QUALIFIER
203void memcpy_async(T(*dst)[DstN], const T(*src)[SrcN], pipeline& pipe)
204{
205    constexpr size_t dst_size = sizeof(*dst);
206    constexpr size_t src_size = sizeof(*src);
207    static_assert(dst_size == 4 || dst_size == 8 || dst_size == 16, "Unsupported copy size.");
208    static_assert(src_size <= dst_size, "Source size must be less than or equal to destination size.");
209    _CUDA_PIPELINE_ASSERT(!(reinterpret_cast<uintptr_t>(src) & (dst_size - 1)));
210    _CUDA_PIPELINE_ASSERT(!(reinterpret_cast<uintptr_t>(dst) & (dst_size - 1)));
211
212    if (__is_trivially_copyable(T)) {
213        _CUDA_PIPELINE_INTERNAL_NAMESPACE::pipeline_copy_strict<sizeof(*dst), sizeof(*src)>(
214                reinterpret_cast<void*>(*dst), reinterpret_cast<const void*>(*src));
215    } else {
216        for (size_t i = 0; i < DstN; ++i) {
217            (*dst)[i] = (i < SrcN) ? (*src)[i] : T();
218        }
219    }
220}
221
222_CUDA_PIPELINE_END_NAMESPACE
223
224#endif /* !_CUDA_PIPELINE_H_ */
225 
codekingpro/portable-devtools · Team Ai