codekingpro/portable-devtools
114k
1
2 /* Copyright 2005-2021 NVIDIA Corporation. All rights reserved.
3 *
4 * NOTICE TO LICENSEE:
5 *
6 * The source code and/or documentation ("Licensed Deliverables") are
7 * subject to NVIDIA intellectual property rights under U.S. and
8 * international Copyright laws.
9 *
10 * The Licensed Deliverables contained herein are PROPRIETARY and
11 * CONFIDENTIAL to NVIDIA and are being provided under the terms and
12 * conditions of a form of NVIDIA software license agreement by and
13 * between NVIDIA and Licensee ("License Agreement") or electronically
14 * accepted by Licensee. Notwithstanding any terms or conditions to
15 * the contrary in the License Agreement, reproduction or disclosure
16 * of the Licensed Deliverables to any third party without the express
17 * written consent of NVIDIA is prohibited.
18 *
19 * NOTWITHSTANDING ANY TERMS OR CONDITIONS TO THE CONTRARY IN THE
20 * LICENSE AGREEMENT, NVIDIA MAKES NO REPRESENTATION ABOUT THE
21 * SUITABILITY OF THESE LICENSED DELIVERABLES FOR ANY PURPOSE. THEY ARE
22 * PROVIDED "AS IS" WITHOUT EXPRESS OR IMPLIED WARRANTY OF ANY KIND.
23 * NVIDIA DISCLAIMS ALL WARRANTIES WITH REGARD TO THESE LICENSED
24 * DELIVERABLES, INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY,
25 * NONINFRINGEMENT, AND FITNESS FOR A PARTICULAR PURPOSE.
26 * NOTWITHSTANDING ANY TERMS OR CONDITIONS TO THE CONTRARY IN THE
27 * LICENSE AGREEMENT, IN NO EVENT SHALL NVIDIA BE LIABLE FOR ANY
28 * SPECIAL, INDIRECT, INCIDENTAL, OR CONSEQUENTIAL DAMAGES, OR ANY
29 * DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS,
30 * WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS
31 * ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR PERFORMANCE
32 * OF THESE LICENSED DELIVERABLES.
33 *
34 * U.S. Government End Users. These Licensed Deliverables are a
35 * "commercial item" as that term is defined at 48 C.F.R. 2.101 (OCT
36 * 1995), consisting of "commercial computer software" and "commercial
37 * computer software documentation" as such terms are used in 48
38 * C.F.R. 12.212 (SEPT 1995) and are provided to the U.S. Government
39 * only as a commercial end item. Consistent with 48 C.F.R.12.212 and
40 * 48 C.F.R. 227.7202-1 through 227.7202-4 (JUNE 1995), all
41 * U.S. Government End Users acquire the Licensed Deliverables with
42 * only those rights set forth herein.
43 *
44 * Any use of the Licensed Deliverables in individual and commercial
45 * software must include, in the user documentation and internal
46 * comments to the code, the above Disclaimer and U.S. Government End
47 * Users Notice.
48 */
49
50/*!
51* \file cufftXt.h
52* \brief Public header file for the NVIDIA CUDA FFT library (CUFFT)
53*/
54
55#ifndef _CUFFTXT_H_
56#define _CUFFTXT_H_
57#include "cudalibxt.h"
58#include "cufft.h"
59
60
61#ifndef CUFFTAPI
62#ifdef _WIN32
63#define CUFFTAPI __stdcall
64#else
65#define CUFFTAPI
66#endif
67#endif
68
69#ifdef __cplusplus
70extern "C" {
71#endif
72
73//
74// cufftXtSubFormat identifies the data layout of
75// a memory descriptor owned by cufft.
76// note that multi GPU cufft does not yet support out-of-place transforms
77//
78
79typedef enum cufftXtSubFormat_t {
80 CUFFT_XT_FORMAT_INPUT = 0x00, //by default input is in linear order across GPUs
81 CUFFT_XT_FORMAT_OUTPUT = 0x01, //by default output is in scrambled order depending on transform
82 CUFFT_XT_FORMAT_INPLACE = 0x02, //by default inplace is input order, which is linear across GPUs
83 CUFFT_XT_FORMAT_INPLACE_SHUFFLED = 0x03, //shuffled output order after execution of the transform
84 CUFFT_XT_FORMAT_1D_INPUT_SHUFFLED = 0x04, //shuffled input order prior to execution of 1D transforms
85 CUFFT_XT_FORMAT_DISTRIBUTED_INPUT = 0x05,
86 CUFFT_XT_FORMAT_DISTRIBUTED_OUTPUT = 0x06,
87 CUFFT_FORMAT_UNDEFINED = 0x07
88} cufftXtSubFormat;
89
90//
91// cufftXtCopyType specifies the type of copy for cufftXtMemcpy
92//
93typedef enum cufftXtCopyType_t {
94 CUFFT_COPY_HOST_TO_DEVICE = 0x00,
95 CUFFT_COPY_DEVICE_TO_HOST = 0x01,
96 CUFFT_COPY_DEVICE_TO_DEVICE = 0x02,
97 CUFFT_COPY_UNDEFINED = 0x03
98} cufftXtCopyType;
99
100//
101// cufftXtQueryType specifies the type of query for cufftXtQueryPlan
102//
103typedef enum cufftXtQueryType_t {
104 CUFFT_QUERY_1D_FACTORS = 0x00,
105 CUFFT_QUERY_UNDEFINED = 0x01
106} cufftXtQueryType;
107
108typedef struct cufftXt1dFactors_t {
109 long long int size;
110 long long int stringCount;
111 long long int stringLength;
112 long long int substringLength;
113 long long int factor1;
114 long long int factor2;
115 long long int stringMask;
116 long long int substringMask;
117 long long int factor1Mask;
118 long long int factor2Mask;
119 int stringShift;
120 int substringShift;
121 int factor1Shift;
122 int factor2Shift;
123} cufftXt1dFactors;
124
125//
126// cufftXtWorkAreaPolicy specifies policy for cufftXtSetWorkAreaPolicy
127//
128typedef enum cufftXtWorkAreaPolicy_t {
129 CUFFT_WORKAREA_MINIMAL = 0, /* maximum reduction */
130 CUFFT_WORKAREA_USER = 1, /* use workSize parameter as limit */
131 CUFFT_WORKAREA_PERFORMANCE = 2, /* default - 1x overhead or more, maximum performance */
132} cufftXtWorkAreaPolicy;
133
134// multi-GPU routines
135cufftResult CUFFTAPI cufftXtSetGPUs(cufftHandle handle, int nGPUs, int *whichGPUs);
136
137cufftResult CUFFTAPI cufftXtMalloc(cufftHandle plan,
138 cudaLibXtDesc ** descriptor,
139 cufftXtSubFormat format);
140
141cufftResult CUFFTAPI cufftXtMemcpy(cufftHandle plan,
142 void *dstPointer,
143 void *srcPointer,
144 cufftXtCopyType type);
145
146cufftResult CUFFTAPI cufftXtFree(cudaLibXtDesc *descriptor);
147
148cufftResult CUFFTAPI cufftXtSetWorkArea(cufftHandle plan, void **workArea);
149
150cufftResult CUFFTAPI cufftXtExecDescriptorC2C(cufftHandle plan,
151 cudaLibXtDesc *input,
152 cudaLibXtDesc *output,
153 int direction);
154
155cufftResult CUFFTAPI cufftXtExecDescriptorR2C(cufftHandle plan,
156 cudaLibXtDesc *input,
157 cudaLibXtDesc *output);
158
159cufftResult CUFFTAPI cufftXtExecDescriptorC2R(cufftHandle plan,
160 cudaLibXtDesc *input,
161 cudaLibXtDesc *output);
162
163cufftResult CUFFTAPI cufftXtExecDescriptorZ2Z(cufftHandle plan,
164 cudaLibXtDesc *input,
165 cudaLibXtDesc *output,
166 int direction);
167
168cufftResult CUFFTAPI cufftXtExecDescriptorD2Z(cufftHandle plan,
169 cudaLibXtDesc *input,
170 cudaLibXtDesc *output);
171
172cufftResult CUFFTAPI cufftXtExecDescriptorZ2D(cufftHandle plan,
173 cudaLibXtDesc *input,
174 cudaLibXtDesc *output);
175
176// Utility functions
177
178cufftResult CUFFTAPI cufftXtQueryPlan(cufftHandle plan, void *queryStruct, cufftXtQueryType queryType);
179
180
181// callbacks
182
183
184typedef enum cufftXtCallbackType_t {
185 CUFFT_CB_LD_COMPLEX = 0x0,
186 CUFFT_CB_LD_COMPLEX_DOUBLE = 0x1,
187 CUFFT_CB_LD_REAL = 0x2,
188 CUFFT_CB_LD_REAL_DOUBLE = 0x3,
189 CUFFT_CB_ST_COMPLEX = 0x4,
190 CUFFT_CB_ST_COMPLEX_DOUBLE = 0x5,
191 CUFFT_CB_ST_REAL = 0x6,
192 CUFFT_CB_ST_REAL_DOUBLE = 0x7,
193 CUFFT_CB_UNDEFINED = 0x8
194
195} cufftXtCallbackType;
196
197typedef cufftComplex (*cufftCallbackLoadC)(void *dataIn, size_t offset, void *callerInfo, void *sharedPointer);
198typedef cufftDoubleComplex (*cufftCallbackLoadZ)(void *dataIn, size_t offset, void *callerInfo, void *sharedPointer);
199typedef cufftReal (*cufftCallbackLoadR)(void *dataIn, size_t offset, void *callerInfo, void *sharedPointer);
200typedef cufftDoubleReal(*cufftCallbackLoadD)(void *dataIn, size_t offset, void *callerInfo, void *sharedPointer);
201
202typedef void (*cufftCallbackStoreC)(void *dataOut, size_t offset, cufftComplex element, void *callerInfo, void *sharedPointer);
203typedef void (*cufftCallbackStoreZ)(void *dataOut, size_t offset, cufftDoubleComplex element, void *callerInfo, void *sharedPointer);
204typedef void (*cufftCallbackStoreR)(void *dataOut, size_t offset, cufftReal element, void *callerInfo, void *sharedPointer);
205typedef void (*cufftCallbackStoreD)(void *dataOut, size_t offset, cufftDoubleReal element, void *callerInfo, void *sharedPointer);
206
207
208cufftResult CUFFTAPI cufftXtSetCallback(cufftHandle plan, void **callback_routine, cufftXtCallbackType cbType, void **caller_info);
209cufftResult CUFFTAPI cufftXtClearCallback(cufftHandle plan, cufftXtCallbackType cbType);
210cufftResult CUFFTAPI cufftXtSetCallbackSharedSize(cufftHandle plan, cufftXtCallbackType cbType, size_t sharedSize);
211
212cufftResult CUFFTAPI cufftXtMakePlanMany(cufftHandle plan,
213 int rank,
214 long long int *n,
215 long long int *inembed,
216 long long int istride,
217 long long int idist,
218 cudaDataType inputtype,
219 long long int *onembed,
220 long long int ostride,
221 long long int odist,
222 cudaDataType outputtype,
223 long long int batch,
224 size_t *workSize,
225 cudaDataType executiontype);
226
227cufftResult CUFFTAPI cufftXtGetSizeMany(cufftHandle plan,
228 int rank,
229 long long int *n,
230 long long int *inembed,
231 long long int istride,
232 long long int idist,
233 cudaDataType inputtype,
234 long long int *onembed,
235 long long int ostride,
236 long long int odist,
237 cudaDataType outputtype,
238 long long int batch,
239 size_t *workSize,
240 cudaDataType executiontype);
241
242
243cufftResult CUFFTAPI cufftXtExec(cufftHandle plan,
244 void *input,
245 void *output,
246 int direction);
247
248cufftResult CUFFTAPI cufftXtExecDescriptor(cufftHandle plan,
249 cudaLibXtDesc *input,
250 cudaLibXtDesc *output,
251 int direction);
252
253cufftResult CUFFTAPI cufftXtSetWorkAreaPolicy(cufftHandle plan, cufftXtWorkAreaPolicy policy, size_t *workSize);
254
255#ifdef __cplusplus
256}
257#endif
258
259#endif
260 