Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
cufftXt.h260 linesDownload Raw Back to include
1
2 /* Copyright 2005-2021 NVIDIA Corporation.  All rights reserved.
3  *
4  * NOTICE TO LICENSEE:
5  *
6  * The source code and/or documentation ("Licensed Deliverables") are
7  * subject to NVIDIA intellectual property rights under U.S. and
8  * international Copyright laws.
9  *
10  * The Licensed Deliverables contained herein are PROPRIETARY and
11  * CONFIDENTIAL to NVIDIA and are being provided under the terms and
12  * conditions of a form of NVIDIA software license agreement by and
13  * between NVIDIA and Licensee ("License Agreement") or electronically
14  * accepted by Licensee.  Notwithstanding any terms or conditions to
15  * the contrary in the License Agreement, reproduction or disclosure
16  * of the Licensed Deliverables to any third party without the express
17  * written consent of NVIDIA is prohibited.
18  *
19  * NOTWITHSTANDING ANY TERMS OR CONDITIONS TO THE CONTRARY IN THE
20  * LICENSE AGREEMENT, NVIDIA MAKES NO REPRESENTATION ABOUT THE
21  * SUITABILITY OF THESE LICENSED DELIVERABLES FOR ANY PURPOSE.  THEY ARE
22  * PROVIDED "AS IS" WITHOUT EXPRESS OR IMPLIED WARRANTY OF ANY KIND.
23  * NVIDIA DISCLAIMS ALL WARRANTIES WITH REGARD TO THESE LICENSED
24  * DELIVERABLES, INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY,
25  * NONINFRINGEMENT, AND FITNESS FOR A PARTICULAR PURPOSE.
26  * NOTWITHSTANDING ANY TERMS OR CONDITIONS TO THE CONTRARY IN THE
27  * LICENSE AGREEMENT, IN NO EVENT SHALL NVIDIA BE LIABLE FOR ANY
28  * SPECIAL, INDIRECT, INCIDENTAL, OR CONSEQUENTIAL DAMAGES, OR ANY
29  * DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS,
30  * WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS
31  * ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR PERFORMANCE
32  * OF THESE LICENSED DELIVERABLES.
33  *
34  * U.S. Government End Users.  These Licensed Deliverables are a
35  * "commercial item" as that term is defined at 48 C.F.R. 2.101 (OCT
36  * 1995), consisting of "commercial computer software" and "commercial
37  * computer software documentation" as such terms are used in 48
38  * C.F.R. 12.212 (SEPT 1995) and are provided to the U.S. Government
39  * only as a commercial end item.  Consistent with 48 C.F.R.12.212 and
40  * 48 C.F.R. 227.7202-1 through 227.7202-4 (JUNE 1995), all
41  * U.S. Government End Users acquire the Licensed Deliverables with
42  * only those rights set forth herein.
43  *
44  * Any use of the Licensed Deliverables in individual and commercial
45  * software must include, in the user documentation and internal
46  * comments to the code, the above Disclaimer and U.S. Government End
47  * Users Notice.
48  */
49
50/*!
51* \file cufftXt.h
52* \brief Public header file for the NVIDIA CUDA FFT library (CUFFT)
53*/
54
55#ifndef _CUFFTXT_H_
56#define _CUFFTXT_H_
57#include "cudalibxt.h"
58#include "cufft.h"
59
60
61#ifndef CUFFTAPI
62#ifdef _WIN32
63#define CUFFTAPI __stdcall
64#else
65#define CUFFTAPI
66#endif
67#endif
68
69#ifdef __cplusplus
70extern "C" {
71#endif
72
73//
74// cufftXtSubFormat identifies the data layout of
75// a memory descriptor owned by cufft.
76// note that multi GPU cufft does not yet support out-of-place transforms
77//
78
79typedef enum cufftXtSubFormat_t {
80    CUFFT_XT_FORMAT_INPUT = 0x00,              //by default input is in linear order across GPUs
81    CUFFT_XT_FORMAT_OUTPUT = 0x01,             //by default output is in scrambled order depending on transform
82    CUFFT_XT_FORMAT_INPLACE = 0x02,            //by default inplace is input order, which is linear across GPUs
83    CUFFT_XT_FORMAT_INPLACE_SHUFFLED = 0x03,   //shuffled output order after execution of the transform
84    CUFFT_XT_FORMAT_1D_INPUT_SHUFFLED = 0x04,  //shuffled input order prior to execution of 1D transforms
85    CUFFT_XT_FORMAT_DISTRIBUTED_INPUT = 0x05,
86    CUFFT_XT_FORMAT_DISTRIBUTED_OUTPUT = 0x06,
87    CUFFT_FORMAT_UNDEFINED = 0x07
88} cufftXtSubFormat;
89
90//
91// cufftXtCopyType specifies the type of copy for cufftXtMemcpy
92//
93typedef enum cufftXtCopyType_t {
94    CUFFT_COPY_HOST_TO_DEVICE = 0x00,
95    CUFFT_COPY_DEVICE_TO_HOST = 0x01,
96    CUFFT_COPY_DEVICE_TO_DEVICE = 0x02,
97    CUFFT_COPY_UNDEFINED = 0x03
98} cufftXtCopyType;
99
100//
101// cufftXtQueryType specifies the type of query for cufftXtQueryPlan
102//
103typedef enum cufftXtQueryType_t {
104    CUFFT_QUERY_1D_FACTORS = 0x00,
105    CUFFT_QUERY_UNDEFINED = 0x01
106} cufftXtQueryType;
107
108typedef struct cufftXt1dFactors_t {
109    long long int size;
110    long long int stringCount;
111    long long int stringLength;
112    long long int substringLength;
113    long long int factor1;
114    long long int factor2;
115    long long int stringMask;
116    long long int substringMask;
117    long long int factor1Mask;
118    long long int factor2Mask;
119    int stringShift;
120    int substringShift;
121    int factor1Shift;
122    int factor2Shift;
123} cufftXt1dFactors;
124
125//
126// cufftXtWorkAreaPolicy specifies policy for cufftXtSetWorkAreaPolicy
127//
128typedef enum cufftXtWorkAreaPolicy_t {
129    CUFFT_WORKAREA_MINIMAL = 0, /* maximum reduction */
130    CUFFT_WORKAREA_USER = 1, /* use workSize parameter as limit */
131    CUFFT_WORKAREA_PERFORMANCE = 2, /* default - 1x overhead or more, maximum performance */
132} cufftXtWorkAreaPolicy;
133
134// multi-GPU routines
135cufftResult CUFFTAPI cufftXtSetGPUs(cufftHandle handle, int nGPUs, int *whichGPUs);
136
137cufftResult CUFFTAPI cufftXtMalloc(cufftHandle plan,
138                                   cudaLibXtDesc ** descriptor,
139                                   cufftXtSubFormat format);
140
141cufftResult CUFFTAPI cufftXtMemcpy(cufftHandle plan,
142                                   void *dstPointer,
143                                   void *srcPointer,
144                                   cufftXtCopyType type);
145
146cufftResult CUFFTAPI cufftXtFree(cudaLibXtDesc *descriptor);
147
148cufftResult CUFFTAPI cufftXtSetWorkArea(cufftHandle plan, void **workArea);
149
150cufftResult CUFFTAPI cufftXtExecDescriptorC2C(cufftHandle plan,
151                                              cudaLibXtDesc *input,
152                                              cudaLibXtDesc *output,
153                                              int direction);
154
155cufftResult CUFFTAPI cufftXtExecDescriptorR2C(cufftHandle plan,
156                                              cudaLibXtDesc *input,
157                                              cudaLibXtDesc *output);
158
159cufftResult CUFFTAPI cufftXtExecDescriptorC2R(cufftHandle plan,
160                                              cudaLibXtDesc *input,
161                                              cudaLibXtDesc *output);
162
163cufftResult CUFFTAPI cufftXtExecDescriptorZ2Z(cufftHandle plan,
164                                              cudaLibXtDesc *input,
165                                              cudaLibXtDesc *output,
166                                              int direction);
167
168cufftResult CUFFTAPI cufftXtExecDescriptorD2Z(cufftHandle plan,
169                                              cudaLibXtDesc *input,
170                                              cudaLibXtDesc *output);
171
172cufftResult CUFFTAPI cufftXtExecDescriptorZ2D(cufftHandle plan,
173                                              cudaLibXtDesc *input,
174                                              cudaLibXtDesc *output);
175
176// Utility functions
177
178cufftResult CUFFTAPI cufftXtQueryPlan(cufftHandle plan, void *queryStruct, cufftXtQueryType queryType);
179
180
181// callbacks
182
183
184typedef enum cufftXtCallbackType_t {
185    CUFFT_CB_LD_COMPLEX = 0x0,
186    CUFFT_CB_LD_COMPLEX_DOUBLE = 0x1,
187    CUFFT_CB_LD_REAL = 0x2,
188    CUFFT_CB_LD_REAL_DOUBLE = 0x3,
189    CUFFT_CB_ST_COMPLEX = 0x4,
190    CUFFT_CB_ST_COMPLEX_DOUBLE = 0x5,
191    CUFFT_CB_ST_REAL = 0x6,
192    CUFFT_CB_ST_REAL_DOUBLE = 0x7,
193    CUFFT_CB_UNDEFINED = 0x8
194
195} cufftXtCallbackType;
196
197typedef cufftComplex (*cufftCallbackLoadC)(void *dataIn, size_t offset, void *callerInfo, void *sharedPointer);
198typedef cufftDoubleComplex (*cufftCallbackLoadZ)(void *dataIn, size_t offset, void *callerInfo, void *sharedPointer);
199typedef cufftReal (*cufftCallbackLoadR)(void *dataIn, size_t offset, void *callerInfo, void *sharedPointer);
200typedef cufftDoubleReal(*cufftCallbackLoadD)(void *dataIn, size_t offset, void *callerInfo, void *sharedPointer);
201
202typedef void (*cufftCallbackStoreC)(void *dataOut, size_t offset, cufftComplex element, void *callerInfo, void *sharedPointer);
203typedef void (*cufftCallbackStoreZ)(void *dataOut, size_t offset, cufftDoubleComplex element, void *callerInfo, void *sharedPointer);
204typedef void (*cufftCallbackStoreR)(void *dataOut, size_t offset, cufftReal element, void *callerInfo, void *sharedPointer);
205typedef void (*cufftCallbackStoreD)(void *dataOut, size_t offset, cufftDoubleReal element, void *callerInfo, void *sharedPointer);
206
207
208cufftResult CUFFTAPI cufftXtSetCallback(cufftHandle plan, void **callback_routine, cufftXtCallbackType cbType, void **caller_info);
209cufftResult CUFFTAPI cufftXtClearCallback(cufftHandle plan, cufftXtCallbackType cbType);
210cufftResult CUFFTAPI cufftXtSetCallbackSharedSize(cufftHandle plan, cufftXtCallbackType cbType, size_t sharedSize);
211
212cufftResult CUFFTAPI cufftXtMakePlanMany(cufftHandle plan,
213                                         int rank,
214                                         long long int *n,
215                                         long long int *inembed,
216                                         long long int istride,
217                                         long long int idist,
218                                         cudaDataType inputtype,
219                                         long long int *onembed,
220                                         long long int ostride,
221                                         long long int odist,
222                                         cudaDataType outputtype,
223                                         long long int batch,
224                                         size_t *workSize,
225                                       	 cudaDataType executiontype);
226
227cufftResult CUFFTAPI cufftXtGetSizeMany(cufftHandle plan,
228                                        int rank,
229                                        long long int *n,
230                                        long long int *inembed,
231                                        long long int istride,
232                                        long long int idist,
233                                        cudaDataType inputtype,
234                                        long long int *onembed,
235                                        long long int ostride,
236                                        long long int odist,
237                                        cudaDataType outputtype,
238                                        long long int batch,
239                                        size_t *workSize,
240                                        cudaDataType executiontype);
241
242
243cufftResult CUFFTAPI cufftXtExec(cufftHandle plan,
244                                 void *input,
245                                 void *output,
246                                 int direction);
247
248cufftResult CUFFTAPI cufftXtExecDescriptor(cufftHandle plan,
249                                           cudaLibXtDesc *input,
250                                           cudaLibXtDesc *output,
251                                           int direction);
252
253cufftResult CUFFTAPI cufftXtSetWorkAreaPolicy(cufftHandle plan, cufftXtWorkAreaPolicy policy, size_t *workSize);
254
255#ifdef __cplusplus
256}
257#endif
258
259#endif
260 
codekingpro/portable-devtools · Team Ai