codekingpro/portable-devtools
114k
1
2 /* Copyright 2005-2014 NVIDIA Corporation. All rights reserved.
3 *
4 * NOTICE TO LICENSEE:
5 *
6 * The source code and/or documentation ("Licensed Deliverables") are
7 * subject to NVIDIA intellectual property rights under U.S. and
8 * international Copyright laws.
9 *
10 * The Licensed Deliverables contained herein are PROPRIETARY and
11 * CONFIDENTIAL to NVIDIA and are being provided under the terms and
12 * conditions of a form of NVIDIA software license agreement by and
13 * between NVIDIA and Licensee ("License Agreement") or electronically
14 * accepted by Licensee. Notwithstanding any terms or conditions to
15 * the contrary in the License Agreement, reproduction or disclosure
16 * of the Licensed Deliverables to any third party without the express
17 * written consent of NVIDIA is prohibited.
18 *
19 * NOTWITHSTANDING ANY TERMS OR CONDITIONS TO THE CONTRARY IN THE
20 * LICENSE AGREEMENT, NVIDIA MAKES NO REPRESENTATION ABOUT THE
21 * SUITABILITY OF THESE LICENSED DELIVERABLES FOR ANY PURPOSE. THEY ARE
22 * PROVIDED "AS IS" WITHOUT EXPRESS OR IMPLIED WARRANTY OF ANY KIND.
23 * NVIDIA DISCLAIMS ALL WARRANTIES WITH REGARD TO THESE LICENSED
24 * DELIVERABLES, INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY,
25 * NONINFRINGEMENT, AND FITNESS FOR A PARTICULAR PURPOSE.
26 * NOTWITHSTANDING ANY TERMS OR CONDITIONS TO THE CONTRARY IN THE
27 * LICENSE AGREEMENT, IN NO EVENT SHALL NVIDIA BE LIABLE FOR ANY
28 * SPECIAL, INDIRECT, INCIDENTAL, OR CONSEQUENTIAL DAMAGES, OR ANY
29 * DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS,
30 * WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS
31 * ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR PERFORMANCE
32 * OF THESE LICENSED DELIVERABLES.
33 *
34 * U.S. Government End Users. These Licensed Deliverables are a
35 * "commercial item" as that term is defined at 48 C.F.R. 2.101 (OCT
36 * 1995), consisting of "commercial computer software" and "commercial
37 * computer software documentation" as such terms are used in 48
38 * C.F.R. 12.212 (SEPT 1995) and are provided to the U.S. Government
39 * only as a commercial end item. Consistent with 48 C.F.R.12.212 and
40 * 48 C.F.R. 227.7202-1 through 227.7202-4 (JUNE 1995), all
41 * U.S. Government End Users acquire the Licensed Deliverables with
42 * only those rights set forth herein.
43 *
44 * Any use of the Licensed Deliverables in individual and commercial
45 * software must include, in the user documentation and internal
46 * comments to the code, the above Disclaimer and U.S. Government End
47 * Users Notice.
48 */
49
50/*!
51* \file cufftw.h
52* \brief Public header file for the NVIDIA CUDA FFTW library (CUFFTW)
53*/
54
55#ifndef _CUFFTW_H_
56#define _CUFFTW_H_
57
58
59#include <stdio.h>
60#include "cufft.h"
61
62#ifdef __cplusplus
63extern "C" {
64#endif
65
66// Transform direction
67#define FFTW_FORWARD -1
68#define FFTW_INVERSE 1
69#define FFTW_BACKWARD 1
70
71// Planner flags
72#define FFTW_ESTIMATE 0x01
73#define FFTW_MEASURE 0x02
74#define FFTW_PATIENT 0x03
75#define FFTW_EXHAUSTIVE 0x04
76#define FFTW_WISDOM_ONLY 0x05
77
78// Algorithm restriction flags
79#define FFTW_DESTROY_INPUT 0x08
80#define FFTW_PRESERVE_INPUT 0x0C
81#define FFTW_UNALIGNED 0x10
82
83// CUFFTW defines and supports the following data types
84
85// note if complex.h has been included we use the C99 complex types
86#if !defined(FFTW_NO_Complex) && defined(_Complex_I) && defined (complex)
87 typedef double _Complex fftw_complex;
88 typedef float _Complex fftwf_complex;
89#else
90 typedef double fftw_complex[2];
91 typedef float fftwf_complex[2];
92#endif
93
94typedef void *fftw_plan;
95
96typedef void *fftwf_plan;
97
98typedef struct {
99 int n;
100 int is;
101 int os;
102} fftw_iodim;
103
104typedef fftw_iodim fftwf_iodim;
105
106typedef struct {
107 ptrdiff_t n;
108 ptrdiff_t is;
109 ptrdiff_t os;
110} fftw_iodim64;
111
112typedef fftw_iodim64 fftwf_iodim64;
113
114// CUFFTW defines and supports the following double precision APIs
115
116fftw_plan CUFFTAPI fftw_plan_dft_1d(int n,
117 fftw_complex *in,
118 fftw_complex *out,
119 int sign,
120 unsigned flags);
121
122fftw_plan CUFFTAPI fftw_plan_dft_2d(int n0,
123 int n1,
124 fftw_complex *in,
125 fftw_complex *out,
126 int sign,
127 unsigned flags);
128
129fftw_plan CUFFTAPI fftw_plan_dft_3d(int n0,
130 int n1,
131 int n2,
132 fftw_complex *in,
133 fftw_complex *out,
134 int sign,
135 unsigned flags);
136
137fftw_plan CUFFTAPI fftw_plan_dft(int rank,
138 const int *n,
139 fftw_complex *in,
140 fftw_complex *out,
141 int sign,
142 unsigned flags);
143
144fftw_plan CUFFTAPI fftw_plan_dft_r2c_1d(int n,
145 double *in,
146 fftw_complex *out,
147 unsigned flags);
148
149fftw_plan CUFFTAPI fftw_plan_dft_r2c_2d(int n0,
150 int n1,
151 double *in,
152 fftw_complex *out,
153 unsigned flags);
154
155fftw_plan CUFFTAPI fftw_plan_dft_r2c_3d(int n0,
156 int n1,
157 int n2,
158 double *in,
159 fftw_complex *out,
160 unsigned flags);
161
162fftw_plan CUFFTAPI fftw_plan_dft_r2c(int rank,
163 const int *n,
164 double *in,
165 fftw_complex *out,
166 unsigned flags);
167
168fftw_plan CUFFTAPI fftw_plan_dft_c2r_1d(int n,
169 fftw_complex *in,
170 double *out,
171 unsigned flags);
172
173fftw_plan CUFFTAPI fftw_plan_dft_c2r_2d(int n0,
174 int n1,
175 fftw_complex *in,
176 double *out,
177 unsigned flags);
178
179fftw_plan CUFFTAPI fftw_plan_dft_c2r_3d(int n0,
180 int n1,
181 int n2,
182 fftw_complex *in,
183 double *out,
184 unsigned flags);
185
186fftw_plan CUFFTAPI fftw_plan_dft_c2r(int rank,
187 const int *n,
188 fftw_complex *in,
189 double *out,
190 unsigned flags);
191
192
193fftw_plan CUFFTAPI fftw_plan_many_dft(int rank,
194 const int *n,
195 int batch,
196 fftw_complex *in,
197 const int *inembed, int istride, int idist,
198 fftw_complex *out,
199 const int *onembed, int ostride, int odist,
200 int sign, unsigned flags);
201
202fftw_plan CUFFTAPI fftw_plan_many_dft_r2c(int rank,
203 const int *n,
204 int batch,
205 double *in,
206 const int *inembed, int istride, int idist,
207 fftw_complex *out,
208 const int *onembed, int ostride, int odist,
209 unsigned flags);
210
211fftw_plan CUFFTAPI fftw_plan_many_dft_c2r(int rank,
212 const int *n,
213 int batch,
214 fftw_complex *in,
215 const int *inembed, int istride, int idist,
216 double *out,
217 const int *onembed, int ostride, int odist,
218 unsigned flags);
219
220fftw_plan CUFFTAPI fftw_plan_guru_dft(int rank, const fftw_iodim *dims,
221 int batch_rank, const fftw_iodim *batch_dims,
222 fftw_complex *in, fftw_complex *out,
223 int sign, unsigned flags);
224
225fftw_plan CUFFTAPI fftw_plan_guru_dft_r2c(int rank, const fftw_iodim *dims,
226 int batch_rank, const fftw_iodim *batch_dims,
227 double *in, fftw_complex *out,
228 unsigned flags);
229
230fftw_plan CUFFTAPI fftw_plan_guru_dft_c2r(int rank, const fftw_iodim *dims,
231 int batch_rank, const fftw_iodim *batch_dims,
232 fftw_complex *in, double *out,
233 unsigned flags);
234
235fftw_plan CUFFTAPI fftw_plan_guru64_dft(int rank, const fftw_iodim64* dims,
236 int batch_rank, const fftw_iodim64* batch_dims,
237 fftw_complex* in, fftw_complex* out,
238 int sign, unsigned flags);
239
240fftw_plan CUFFTAPI fftw_plan_guru64_dft_r2c(int rank, const fftw_iodim64* dims,
241 int batch_rank, const fftw_iodim64* batch_dims,
242 double* in, fftw_complex* out,
243 unsigned flags);
244
245fftw_plan CUFFTAPI fftw_plan_guru64_dft_c2r(int rank, const fftw_iodim64* dims,
246 int batch_rank, const fftw_iodim64* batch_dims,
247 fftw_complex* in, double* out,
248 unsigned flags);
249
250void CUFFTAPI fftw_execute(const fftw_plan plan);
251
252void CUFFTAPI fftw_execute_dft(const fftw_plan plan,
253 fftw_complex *idata,
254 fftw_complex *odata);
255
256void CUFFTAPI fftw_execute_dft_r2c(const fftw_plan plan,
257 double *idata,
258 fftw_complex *odata);
259
260void CUFFTAPI fftw_execute_dft_c2r(const fftw_plan plan,
261 fftw_complex *idata,
262 double *odata);
263
264// CUFFTW defines and supports the following single precision APIs
265
266fftwf_plan CUFFTAPI fftwf_plan_dft_1d(int n,
267 fftwf_complex *in,
268 fftwf_complex *out,
269 int sign,
270 unsigned flags);
271
272fftwf_plan CUFFTAPI fftwf_plan_dft_2d(int n0,
273 int n1,
274 fftwf_complex *in,
275 fftwf_complex *out,
276 int sign,
277 unsigned flags);
278
279fftwf_plan CUFFTAPI fftwf_plan_dft_3d(int n0,
280 int n1,
281 int n2,
282 fftwf_complex *in,
283 fftwf_complex *out,
284 int sign,
285 unsigned flags);
286
287fftwf_plan CUFFTAPI fftwf_plan_dft(int rank,
288 const int *n,
289 fftwf_complex *in,
290 fftwf_complex *out,
291 int sign,
292 unsigned flags);
293
294fftwf_plan CUFFTAPI fftwf_plan_dft_r2c_1d(int n,
295 float *in,
296 fftwf_complex *out,
297 unsigned flags);
298
299fftwf_plan CUFFTAPI fftwf_plan_dft_r2c_2d(int n0,
300 int n1,
301 float *in,
302 fftwf_complex *out,
303 unsigned flags);
304
305fftwf_plan CUFFTAPI fftwf_plan_dft_r2c_3d(int n0,
306 int n1,
307 int n2,
308 float *in,
309 fftwf_complex *out,
310 unsigned flags);
311
312fftwf_plan CUFFTAPI fftwf_plan_dft_r2c(int rank,
313 const int *n,
314 float *in,
315 fftwf_complex *out,
316 unsigned flags);
317
318fftwf_plan CUFFTAPI fftwf_plan_dft_c2r_1d(int n,
319 fftwf_complex *in,
320 float *out,
321 unsigned flags);
322
323fftwf_plan CUFFTAPI fftwf_plan_dft_c2r_2d(int n0,
324 int n1,
325 fftwf_complex *in,
326 float *out,
327 unsigned flags);
328
329fftwf_plan CUFFTAPI fftwf_plan_dft_c2r_3d(int n0,
330 int n1,
331 int n2,
332 fftwf_complex *in,
333 float *out,
334 unsigned flags);
335
336fftwf_plan CUFFTAPI fftwf_plan_dft_c2r(int rank,
337 const int *n,
338 fftwf_complex *in,
339 float *out,
340 unsigned flags);
341
342fftwf_plan CUFFTAPI fftwf_plan_many_dft(int rank,
343 const int *n,
344 int batch,
345 fftwf_complex *in,
346 const int *inembed, int istride, int idist,
347 fftwf_complex *out,
348 const int *onembed, int ostride, int odist,
349 int sign, unsigned flags);
350
351fftwf_plan CUFFTAPI fftwf_plan_many_dft_r2c(int rank,
352 const int *n,
353 int batch,
354 float *in,
355 const int *inembed, int istride, int idist,
356 fftwf_complex *out,
357 const int *onembed, int ostride, int odist,
358 unsigned flags);
359
360fftwf_plan CUFFTAPI fftwf_plan_many_dft_c2r(int rank,
361 const int *n,
362 int batch,
363 fftwf_complex *in,
364 const int *inembed, int istride, int idist,
365 float *out,
366 const int *onembed, int ostride, int odist,
367 unsigned flags);
368
369fftwf_plan CUFFTAPI fftwf_plan_guru_dft(int rank, const fftwf_iodim *dims,
370 int batch_rank, const fftwf_iodim *batch_dims,
371 fftwf_complex *in, fftwf_complex *out,
372 int sign, unsigned flags);
373
374fftwf_plan CUFFTAPI fftwf_plan_guru_dft_r2c(int rank, const fftwf_iodim *dims,
375 int batch_rank, const fftwf_iodim *batch_dims,
376 float *in, fftwf_complex *out,
377 unsigned flags);
378
379fftwf_plan CUFFTAPI fftwf_plan_guru_dft_c2r(int rank, const fftwf_iodim *dims,
380 int batch_rank, const fftwf_iodim *batch_dims,
381 fftwf_complex *in, float *out,
382 unsigned flags);
383
384fftwf_plan CUFFTAPI fftwf_plan_guru64_dft(int rank, const fftwf_iodim64* dims,
385 int batch_rank, const fftwf_iodim64* batch_dims,
386 fftwf_complex* in, fftwf_complex* out,
387 int sign, unsigned flags);
388
389fftwf_plan CUFFTAPI fftwf_plan_guru64_dft_r2c(int rank, const fftwf_iodim64* dims,
390 int batch_rank, const fftwf_iodim64* batch_dims,
391 float* in, fftwf_complex* out,
392 unsigned flags);
393
394fftwf_plan CUFFTAPI fftwf_plan_guru64_dft_c2r(int rank, const fftwf_iodim64* dims,
395 int batch_rank, const fftwf_iodim64* batch_dims,
396 fftwf_complex* in, float* out,
397 unsigned flags);
398
399void CUFFTAPI fftwf_execute(const fftw_plan plan);
400
401void CUFFTAPI fftwf_execute_dft(const fftwf_plan plan,
402 fftwf_complex *idata,
403 fftwf_complex *odata);
404
405void CUFFTAPI fftwf_execute_dft_r2c(const fftwf_plan plan,
406 float *idata,
407 fftwf_complex *odata);
408
409void CUFFTAPI fftwf_execute_dft_c2r(const fftwf_plan plan,
410 fftwf_complex *idata,
411 float *odata);
412
413#ifdef _WIN32
414#define _CUFFTAPI(T) T CUFFTAPI
415#else
416#define _CUFFTAPI(T) CUFFTAPI T
417#endif
418
419// CUFFTW defines and supports the following support APIs
420
421_CUFFTAPI(void *) fftw_malloc(size_t n);
422
423_CUFFTAPI(void *) fftwf_malloc(size_t n);
424
425void CUFFTAPI fftw_free(void *pointer);
426
427void CUFFTAPI fftwf_free(void *pointer);
428
429void CUFFTAPI fftw_export_wisdom_to_file(FILE * output_file);
430
431void CUFFTAPI fftwf_export_wisdom_to_file(FILE * output_file);
432
433int CUFFTAPI fftw_import_wisdom_from_file(FILE * input_file);
434
435int CUFFTAPI fftwf_import_wisdom_from_file(FILE * input_file);
436
437void CUFFTAPI fftw_print_plan(const fftw_plan plan);
438
439void CUFFTAPI fftwf_print_plan(const fftwf_plan plan);
440
441void CUFFTAPI fftw_set_timelimit(double seconds);
442
443void CUFFTAPI fftwf_set_timelimit(double seconds);
444
445double CUFFTAPI fftw_cost(const fftw_plan plan);
446
447double CUFFTAPI fftwf_cost(const fftw_plan plan);
448
449void CUFFTAPI fftw_flops(const fftw_plan plan, double *add, double *mul, double *fma);
450
451void CUFFTAPI fftwf_flops(const fftw_plan plan, double *add, double *mul, double *fma);
452
453void CUFFTAPI fftw_destroy_plan(fftw_plan plan);
454
455void CUFFTAPI fftwf_destroy_plan(fftwf_plan plan);
456
457void CUFFTAPI fftw_cleanup(void);
458
459void CUFFTAPI fftwf_cleanup(void);
460
461#ifdef __cplusplus
462}
463#endif
464
465#endif /* _CUFFTW_H_ */
466 