codekingpro/portable-devtools
114k
1//
2// NVIDIA_COPYRIGHT_BEGIN
3//
4// Copyright (c) 2014-2023, NVIDIA CORPORATION. All rights reserved.
5//
6// NVIDIA CORPORATION and its licensors retain all intellectual property
7// and proprietary rights in and to this software, related documentation
8// and any modifications thereto. Any use, reproduction, disclosure or
9// distribution of this software and related documentation without an express
10// license agreement from NVIDIA CORPORATION is strictly prohibited.
11//
12// NVIDIA_COPYRIGHT_END
13//
14
15#ifndef __NVRTC_H__
16#define __NVRTC_H__
17
18#ifdef __cplusplus
19extern "C" {
20#endif /* __cplusplus */
21
22#include <stdlib.h>
23
24
25/*************************************************************************//**
26 *
27 * \defgroup error Error Handling
28 *
29 * NVRTC defines the following enumeration type and function for API call
30 * error handling.
31 *
32 ****************************************************************************/
33
34
35/**
36 * \ingroup error
37 * \brief The enumerated type nvrtcResult defines API call result codes.
38 * NVRTC API functions return nvrtcResult to indicate the call
39 * result.
40 */
41typedef enum {
42 NVRTC_SUCCESS = 0,
43 NVRTC_ERROR_OUT_OF_MEMORY = 1,
44 NVRTC_ERROR_PROGRAM_CREATION_FAILURE = 2,
45 NVRTC_ERROR_INVALID_INPUT = 3,
46 NVRTC_ERROR_INVALID_PROGRAM = 4,
47 NVRTC_ERROR_INVALID_OPTION = 5,
48 NVRTC_ERROR_COMPILATION = 6,
49 NVRTC_ERROR_BUILTIN_OPERATION_FAILURE = 7,
50 NVRTC_ERROR_NO_NAME_EXPRESSIONS_AFTER_COMPILATION = 8,
51 NVRTC_ERROR_NO_LOWERED_NAMES_BEFORE_COMPILATION = 9,
52 NVRTC_ERROR_NAME_EXPRESSION_NOT_VALID = 10,
53 NVRTC_ERROR_INTERNAL_ERROR = 11,
54 NVRTC_ERROR_TIME_FILE_WRITE_FAILED = 12
55} nvrtcResult;
56
57
58/**
59 * \ingroup error
60 * \brief nvrtcGetErrorString is a helper function that returns a string
61 * describing the given nvrtcResult code, e.g., NVRTC_SUCCESS to
62 * \c "NVRTC_SUCCESS".
63 * For unrecognized enumeration values, it returns
64 * \c "NVRTC_ERROR unknown".
65 *
66 * \param [in] result CUDA Runtime Compilation API result code.
67 * \return Message string for the given #nvrtcResult code.
68 */
69const char *nvrtcGetErrorString(nvrtcResult result);
70
71
72/*************************************************************************//**
73 *
74 * \defgroup query General Information Query
75 *
76 * NVRTC defines the following function for general information query.
77 *
78 ****************************************************************************/
79
80
81/**
82 * \ingroup query
83 * \brief nvrtcVersion sets the output parameters \p major and \p minor
84 * with the CUDA Runtime Compilation version number.
85 *
86 * \param [out] major CUDA Runtime Compilation major version number.
87 * \param [out] minor CUDA Runtime Compilation minor version number.
88 * \return
89 * - \link #nvrtcResult NVRTC_SUCCESS \endlink
90 * - \link #nvrtcResult NVRTC_ERROR_INVALID_INPUT \endlink
91 *
92 */
93nvrtcResult nvrtcVersion(int *major, int *minor);
94
95
96/**
97 * \ingroup query
98 * \brief nvrtcGetNumSupportedArchs sets the output parameter \p numArchs
99 * with the number of architectures supported by NVRTC. This can
100 * then be used to pass an array to ::nvrtcGetSupportedArchs to
101 * get the supported architectures.
102 *
103 * \param [out] numArchs number of supported architectures.
104 * \return
105 * - \link #nvrtcResult NVRTC_SUCCESS \endlink
106 * - \link #nvrtcResult NVRTC_ERROR_INVALID_INPUT \endlink
107 *
108 * see ::nvrtcGetSupportedArchs
109 */
110nvrtcResult nvrtcGetNumSupportedArchs(int* numArchs);
111
112
113/**
114 * \ingroup query
115 * \brief nvrtcGetSupportedArchs populates the array passed via the output parameter
116 * \p supportedArchs with the architectures supported by NVRTC. The array is
117 * sorted in the ascending order. The size of the array to be passed can be
118 * determined using ::nvrtcGetNumSupportedArchs.
119 *
120 * \param [out] supportedArchs sorted array of supported architectures.
121 * \return
122 * - \link #nvrtcResult NVRTC_SUCCESS \endlink
123 * - \link #nvrtcResult NVRTC_ERROR_INVALID_INPUT \endlink
124 *
125 * see ::nvrtcGetNumSupportedArchs
126 */
127nvrtcResult nvrtcGetSupportedArchs(int* supportedArchs);
128
129
130/*************************************************************************//**
131 *
132 * \defgroup compilation Compilation
133 *
134 * NVRTC defines the following type and functions for actual compilation.
135 *
136 ****************************************************************************/
137
138
139/**
140 * \ingroup compilation
141 * \brief nvrtcProgram is the unit of compilation, and an opaque handle for
142 * a program.
143 *
144 * To compile a CUDA program string, an instance of nvrtcProgram must be
145 * created first with ::nvrtcCreateProgram, then compiled with
146 * ::nvrtcCompileProgram.
147 */
148typedef struct _nvrtcProgram *nvrtcProgram;
149
150
151/**
152 * \ingroup compilation
153 * \brief nvrtcCreateProgram creates an instance of nvrtcProgram with the
154 * given input parameters, and sets the output parameter \p prog with
155 * it.
156 *
157 * \param [out] prog CUDA Runtime Compilation program.
158 * \param [in] src CUDA program source.
159 * \param [in] name CUDA program name.\n
160 * \p name can be \c NULL; \c "default_program" is
161 * used when \p name is \c NULL or "".
162 * \param [in] numHeaders Number of headers used.\n
163 * \p numHeaders must be greater than or equal to 0.
164 * \param [in] headers Sources of the headers.\n
165 * \p headers can be \c NULL when \p numHeaders is
166 * 0.
167 * \param [in] includeNames Name of each header by which they can be
168 * included in the CUDA program source.\n
169 * \p includeNames can be \c NULL when \p numHeaders
170 * is 0. These headers must be included with the exact
171 * names specified here.
172 * \return
173 * - \link #nvrtcResult NVRTC_SUCCESS \endlink
174 * - \link #nvrtcResult NVRTC_ERROR_OUT_OF_MEMORY \endlink
175 * - \link #nvrtcResult NVRTC_ERROR_PROGRAM_CREATION_FAILURE \endlink
176 * - \link #nvrtcResult NVRTC_ERROR_INVALID_INPUT \endlink
177 * - \link #nvrtcResult NVRTC_ERROR_INVALID_PROGRAM \endlink
178 *
179 * \see ::nvrtcDestroyProgram
180 */
181nvrtcResult nvrtcCreateProgram(nvrtcProgram *prog,
182 const char *src,
183 const char *name,
184 int numHeaders,
185 const char * const *headers,
186 const char * const *includeNames);
187
188
189/**
190 * \ingroup compilation
191 * \brief nvrtcDestroyProgram destroys the given program.
192 *
193 * \param [in] prog CUDA Runtime Compilation program.
194 * \return
195 * - \link #nvrtcResult NVRTC_SUCCESS \endlink
196 * - \link #nvrtcResult NVRTC_ERROR_INVALID_PROGRAM \endlink
197 *
198 * \see ::nvrtcCreateProgram
199 */
200nvrtcResult nvrtcDestroyProgram(nvrtcProgram *prog);
201
202
203/**
204 * \ingroup compilation
205 * \brief nvrtcCompileProgram compiles the given program.
206 *
207 * \param [in] prog CUDA Runtime Compilation program.
208 * \param [in] numOptions Number of compiler options passed.
209 * \param [in] options Compiler options in the form of C string array.\n
210 * \p options can be \c NULL when \p numOptions is 0.
211 *
212 * \return
213 * - \link #nvrtcResult NVRTC_SUCCESS \endlink
214 * - \link #nvrtcResult NVRTC_ERROR_OUT_OF_MEMORY \endlink
215 * - \link #nvrtcResult NVRTC_ERROR_INVALID_INPUT \endlink
216 * - \link #nvrtcResult NVRTC_ERROR_INVALID_PROGRAM \endlink
217 * - \link #nvrtcResult NVRTC_ERROR_INVALID_OPTION \endlink
218 * - \link #nvrtcResult NVRTC_ERROR_COMPILATION \endlink
219 * - \link #nvrtcResult NVRTC_ERROR_BUILTIN_OPERATION_FAILURE \endlink
220 * - \link #nvrtcResult NVRTC_ERROR_TIME_FILE_WRITE_FAILED \endlink
221 *
222 * It supports compile options listed in \ref options.
223 */
224nvrtcResult nvrtcCompileProgram(nvrtcProgram prog,
225 int numOptions, const char * const *options);
226
227
228/**
229 * \ingroup compilation
230 * \brief nvrtcGetPTXSize sets the value of \p ptxSizeRet with the size of the PTX
231 * generated by the previous compilation of \p prog (including the
232 * trailing \c NULL).
233 *
234 * \param [in] prog CUDA Runtime Compilation program.
235 * \param [out] ptxSizeRet Size of the generated PTX (including the trailing
236 * \c NULL).
237 * \return
238 * - \link #nvrtcResult NVRTC_SUCCESS \endlink
239 * - \link #nvrtcResult NVRTC_ERROR_INVALID_INPUT \endlink
240 * - \link #nvrtcResult NVRTC_ERROR_INVALID_PROGRAM \endlink
241 *
242 * \see ::nvrtcGetPTX
243 */
244nvrtcResult nvrtcGetPTXSize(nvrtcProgram prog, size_t *ptxSizeRet);
245
246
247/**
248 * \ingroup compilation
249 * \brief nvrtcGetPTX stores the PTX generated by the previous compilation
250 * of \p prog in the memory pointed by \p ptx.
251 *
252 * \param [in] prog CUDA Runtime Compilation program.
253 * \param [out] ptx Compiled result.
254 * \return
255 * - \link #nvrtcResult NVRTC_SUCCESS \endlink
256 * - \link #nvrtcResult NVRTC_ERROR_INVALID_INPUT \endlink
257 * - \link #nvrtcResult NVRTC_ERROR_INVALID_PROGRAM \endlink
258 *
259 * \see ::nvrtcGetPTXSize
260 */
261nvrtcResult nvrtcGetPTX(nvrtcProgram prog, char *ptx);
262
263
264/**
265 * \ingroup compilation
266 * \brief nvrtcGetCUBINSize sets the value of \p cubinSizeRet with the size of the cubin
267 * generated by the previous compilation of \p prog. The value of
268 * cubinSizeRet is set to 0 if the value specified to \c -arch is a
269 * virtual architecture instead of an actual architecture.
270 *
271 * \param [in] prog CUDA Runtime Compilation program.
272 * \param [out] cubinSizeRet Size of the generated cubin.
273 * \return
274 * - \link #nvrtcResult NVRTC_SUCCESS \endlink
275 * - \link #nvrtcResult NVRTC_ERROR_INVALID_INPUT \endlink
276 * - \link #nvrtcResult NVRTC_ERROR_INVALID_PROGRAM \endlink
277 *
278 * \see ::nvrtcGetCUBIN
279 */
280nvrtcResult nvrtcGetCUBINSize(nvrtcProgram prog, size_t *cubinSizeRet);
281
282
283/**
284 * \ingroup compilation
285 * \brief nvrtcGetCUBIN stores the cubin generated by the previous compilation
286 * of \p prog in the memory pointed by \p cubin. No cubin is available
287 * if the value specified to \c -arch is a virtual architecture instead
288 * of an actual architecture.
289 *
290 * \param [in] prog CUDA Runtime Compilation program.
291 * \param [out] cubin Compiled and assembled result.
292 * \return
293 * - \link #nvrtcResult NVRTC_SUCCESS \endlink
294 * - \link #nvrtcResult NVRTC_ERROR_INVALID_INPUT \endlink
295 * - \link #nvrtcResult NVRTC_ERROR_INVALID_PROGRAM \endlink
296 *
297 * \see ::nvrtcGetCUBINSize
298 */
299nvrtcResult nvrtcGetCUBIN(nvrtcProgram prog, char *cubin);
300
301
302#if defined(_WIN32)
303# define __DEPRECATED__(msg) __declspec(deprecated(msg))
304#elif (defined(__GNUC__) && (__GNUC__ < 4 || (__GNUC__ == 4 && __GNUC_MINOR__ < 5 && !defined(__clang__))))
305# define __DEPRECATED__(msg) __attribute__((deprecated))
306#elif (defined(__GNUC__))
307# define __DEPRECATED__(msg) __attribute__((deprecated(msg)))
308#else
309# define __DEPRECATED__(msg)
310#endif
311
312/**
313 * \ingroup compilation
314 * \brief
315 * DEPRECATION NOTICE: This function will be removed in a future release. Please use
316 * nvrtcGetLTOIRSize (and nvrtcGetLTOIR) instead.
317 */
318__DEPRECATED__("This function will be removed in a future release. Please use nvrtcGetLTOIRSize instead")
319nvrtcResult nvrtcGetNVVMSize(nvrtcProgram prog, size_t *nvvmSizeRet);
320
321/**
322 * \ingroup compilation
323 * \brief
324 * DEPRECATION NOTICE: This function will be removed in a future release. Please use
325 * nvrtcGetLTOIR (and nvrtcGetLTOIRSize) instead.
326 */
327__DEPRECATED__("This function will be removed in a future release. Please use nvrtcGetLTOIR instead")
328nvrtcResult nvrtcGetNVVM(nvrtcProgram prog, char *nvvm);
329
330#undef __DEPRECATED__
331
332/**
333 * \ingroup compilation
334 * \brief nvrtcGetLTOIRSize sets the value of \p LTOIRSizeRet with the size of the LTO IR
335 * generated by the previous compilation of \p prog. The value of
336 * LTOIRSizeRet is set to 0 if the program was not compiled with
337 * \c -dlto.
338 *
339 * \param [in] prog CUDA Runtime Compilation program.
340 * \param [out] LTOIRSizeRet Size of the generated LTO IR.
341 * \return
342 * - \link #nvrtcResult NVRTC_SUCCESS \endlink
343 * - \link #nvrtcResult NVRTC_ERROR_INVALID_INPUT \endlink
344 * - \link #nvrtcResult NVRTC_ERROR_INVALID_PROGRAM \endlink
345 *
346 * \see ::nvrtcGetLTOIR
347 */
348nvrtcResult nvrtcGetLTOIRSize(nvrtcProgram prog, size_t *LTOIRSizeRet);
349
350
351/**
352 * \ingroup compilation
353 * \brief nvrtcGetLTOIR stores the LTO IR generated by the previous compilation
354 * of \p prog in the memory pointed by \p LTOIR. No LTO IR is available
355 * if the program was compiled without \c -dlto.
356 *
357 * \param [in] prog CUDA Runtime Compilation program.
358 * \param [out] LTOIR Compiled result.
359 * \return
360 * - \link #nvrtcResult NVRTC_SUCCESS \endlink
361 * - \link #nvrtcResult NVRTC_ERROR_INVALID_INPUT \endlink
362 * - \link #nvrtcResult NVRTC_ERROR_INVALID_PROGRAM \endlink
363 *
364 * \see ::nvrtcGetLTOIRSize
365 */
366nvrtcResult nvrtcGetLTOIR(nvrtcProgram prog, char *LTOIR);
367
368
369/**
370 * \ingroup compilation
371 * \brief nvrtcGetOptiXIRSize sets the value of \p optixirSizeRet with the size of the OptiX IR
372 * generated by the previous compilation of \p prog. The value of
373 * nvrtcGetOptiXIRSize is set to 0 if the program was compiled with
374 * options incompatible with OptiX IR generation.
375 *
376 * \param [in] prog CUDA Runtime Compilation program.
377 * \param [out] optixirSizeRet Size of the generated LTO IR.
378 * \return
379 * - \link #nvrtcResult NVRTC_SUCCESS \endlink
380 * - \link #nvrtcResult NVRTC_ERROR_INVALID_INPUT \endlink
381 * - \link #nvrtcResult NVRTC_ERROR_INVALID_PROGRAM \endlink
382 *
383 * \see ::nvrtcGetOptiXIR
384 */
385nvrtcResult nvrtcGetOptiXIRSize(nvrtcProgram prog, size_t *optixirSizeRet);
386
387
388/**
389 * \ingroup compilation
390 * \brief nvrtcGetOptiXIR stores the OptiX IR generated by the previous compilation
391 * of \p prog in the memory pointed by \p optixir. No OptiX IR is available
392 * if the program was compiled with options incompatible with OptiX IR generation.
393 *
394 * \param [in] prog CUDA Runtime Compilation program.
395 * \param [out] Optix IR Compiled result.
396 * \return
397 * - \link #nvrtcResult NVRTC_SUCCESS \endlink
398 * - \link #nvrtcResult NVRTC_ERROR_INVALID_INPUT \endlink
399 * - \link #nvrtcResult NVRTC_ERROR_INVALID_PROGRAM \endlink
400 *
401 * \see ::nvrtcGetOptiXIRSize
402 */
403nvrtcResult nvrtcGetOptiXIR(nvrtcProgram prog, char *optixir);
404
405/**
406 * \ingroup compilation
407 * \brief nvrtcGetProgramLogSize sets \p logSizeRet with the size of the
408 * log generated by the previous compilation of \p prog (including the
409 * trailing \c NULL).
410 *
411 * Note that compilation log may be generated with warnings and informative
412 * messages, even when the compilation of \p prog succeeds.
413 *
414 * \param [in] prog CUDA Runtime Compilation program.
415 * \param [out] logSizeRet Size of the compilation log
416 * (including the trailing \c NULL).
417 * \return
418 * - \link #nvrtcResult NVRTC_SUCCESS \endlink
419 * - \link #nvrtcResult NVRTC_ERROR_INVALID_INPUT \endlink
420 * - \link #nvrtcResult NVRTC_ERROR_INVALID_PROGRAM \endlink
421 *
422 * \see ::nvrtcGetProgramLog
423 */
424nvrtcResult nvrtcGetProgramLogSize(nvrtcProgram prog, size_t *logSizeRet);
425
426
427/**
428 * \ingroup compilation
429 * \brief nvrtcGetProgramLog stores the log generated by the previous
430 * compilation of \p prog in the memory pointed by \p log.
431 *
432 * \param [in] prog CUDA Runtime Compilation program.
433 * \param [out] log Compilation log.
434 * \return
435 * - \link #nvrtcResult NVRTC_SUCCESS \endlink
436 * - \link #nvrtcResult NVRTC_ERROR_INVALID_INPUT \endlink
437 * - \link #nvrtcResult NVRTC_ERROR_INVALID_PROGRAM \endlink
438 *
439 * \see ::nvrtcGetProgramLogSize
440 */
441nvrtcResult nvrtcGetProgramLog(nvrtcProgram prog, char *log);
442
443
444/**
445 * \ingroup compilation
446 * \brief nvrtcAddNameExpression notes the given name expression
447 * denoting the address of a __global__ function
448 * or __device__/__constant__ variable.
449 *
450 * The identical name expression string must be provided on a subsequent
451 * call to nvrtcGetLoweredName to extract the lowered name.
452 * \param [in] prog CUDA Runtime Compilation program.
453 * \param [in] name_expression constant expression denoting the address of
454 * a __global__ function or __device__/__constant__ variable.
455 * \return
456 * - \link #nvrtcResult NVRTC_SUCCESS \endlink
457 * - \link #nvrtcResult NVRTC_ERROR_INVALID_PROGRAM \endlink
458 * - \link #nvrtcResult NVRTC_ERROR_INVALID_INPUT \endlink
459 * - \link #nvrtcResult NVRTC_ERROR_NO_NAME_EXPRESSIONS_AFTER_COMPILATION \endlink
460 *
461 * \see ::nvrtcGetLoweredName
462 */
463nvrtcResult nvrtcAddNameExpression(nvrtcProgram prog,
464 const char * const name_expression);
465
466/**
467 * \ingroup compilation
468 * \brief nvrtcGetLoweredName extracts the lowered (mangled) name
469 * for a __global__ function or __device__/__constant__ variable,
470 * and updates *lowered_name to point to it. The memory containing
471 * the name is released when the NVRTC program is destroyed by
472 * nvrtcDestroyProgram.
473 * The identical name expression must have been previously
474 * provided to nvrtcAddNameExpression.
475 *
476 * \param [in] prog CUDA Runtime Compilation program.
477 * \param [in] name_expression constant expression denoting the address of
478 * a __global__ function or __device__/__constant__ variable.
479 * \param [out] lowered_name initialized by the function to point to a
480 * C string containing the lowered (mangled)
481 * name corresponding to the provided name expression.
482 * \return
483 * - \link #nvrtcResult NVRTC_SUCCESS \endlink
484 * - \link #nvrtcResult NVRTC_ERROR_NO_LOWERED_NAMES_BEFORE_COMPILATION \endlink
485 * - \link #nvrtcResult NVRTC_ERROR_INVALID_PROGRAM \endlink
486 * - \link #nvrtcResult NVRTC_ERROR_INVALID_INPUT \endlink
487 * - \link #nvrtcResult NVRTC_ERROR_NAME_EXPRESSION_NOT_VALID \endlink
488 *
489 * \see ::nvrtcAddNameExpression
490 */
491nvrtcResult nvrtcGetLoweredName(nvrtcProgram prog,
492 const char *const name_expression,
493 const char** lowered_name);
494
495
496/**
497 * \defgroup options Supported Compile Options
498 *
499 * NVRTC supports the compile options below.
500 * Option names with two preceding dashs (\c --) are long option names and
501 * option names with one preceding dash (\c -) are short option names.
502 * Short option names can be used instead of long option names.
503 * When a compile option takes an argument, an assignment operator (\c =)
504 * is used to separate the compile option argument from the compile option
505 * name, e.g., \c "--gpu-architecture=compute_60".
506 * Alternatively, the compile option name and the argument can be specified in
507 * separate strings without an assignment operator, .e.g,
508 * \c "--gpu-architecture" \c "compute_60".
509 * Single-character short option names, such as \c -D, \c -U, and \c -I, do
510 * not require an assignment operator, and the compile option name and the
511 * argument can be present in the same string with or without spaces between
512 * them.
513 * For instance, \c "-D=<def>", \c "-D<def>", and \c "-D <def>" are all
514 * supported.
515 *
516 * The valid compiler options are:
517 *
518 * - Compilation targets
519 * - \c --gpu-architecture=\<arch\> (\c -arch)\n
520 * Specify the name of the class of GPU architectures for which the
521 * input must be compiled.\n
522 * - Valid <c>\<arch\></c>s:
523 * - \c compute_50
524 * - \c compute_52
525 * - \c compute_53
526 * - \c compute_60
527 * - \c compute_61
528 * - \c compute_62
529 * - \c compute_70
530 * - \c compute_72
531 * - \c compute_75
532 * - \c compute_80
533 * - \c compute_87
534 * - \c compute_89
535 * - \c compute_90
536 * - \c compute_90a
537 * - \c sm_50
538 * - \c sm_52
539 * - \c sm_53
540 * - \c sm_60
541 * - \c sm_61
542 * - \c sm_62
543 * - \c sm_70
544 * - \c sm_72
545 * - \c sm_75
546 * - \c sm_80
547 * - \c sm_87
548 * - \c sm_89
549 * - \c sm_90
550 * - \c sm_90a
551 * - Default: \c compute_52
552 * - Separate compilation / whole-program compilation
553 * - \c --device-c (\c -dc)\n
554 * Generate relocatable code that can be linked with other relocatable
555 * device code. It is equivalent to --relocatable-device-code=true.
556 * - \c --device-w (\c -dw)\n
557 * Generate non-relocatable code. It is equivalent to
558 * \c --relocatable-device-code=false.
559 * - \c --relocatable-device-code={true|false} (\c -rdc)\n
560 * Enable (disable) the generation of relocatable device code.
561 * - Default: \c false
562 * - \c --extensible-whole-program (\c -ewp)\n
563 * Do extensible whole program compilation of device code.
564 * - Default: \c false
565 * - Debugging support
566 * - \c --device-debug (\c -G)\n
567 * Generate debug information. If --dopt is not specified,
568 * then turns off all optimizations.
569 * - \c --generate-line-info (\c -lineinfo)\n
570 * Generate line-number information.
571 * - Code generation
572 * - \c --dopt on (\c -dopt)\n
573 * - \c --dopt=on \n
574 * Enable device code optimization. When specified along with '-G', enables
575 * limited debug information generation for optimized device code (currently,
576 * only line number information).
577 * When '-G' is not specified, '-dopt=on' is implicit.
578 * - \c --ptxas-options \<options\> (\c -Xptxas)\n
579 * - \c --ptxas-options=\<options\> \n
580 * Specify options directly to ptxas, the PTX optimizing assembler.
581 * - \c --maxrregcount=\<N\> (\c -maxrregcount)\n
582 * Specify the maximum amount of registers that GPU functions can use.
583 * Until a function-specific limit, a higher value will generally
584 * increase the performance of individual GPU threads that execute this
585 * function. However, because thread registers are allocated from a
586 * global register pool on each GPU, a higher value of this option will
587 * also reduce the maximum thread block size, thereby reducing the amount
588 * of thread parallelism. Hence, a good maxrregcount value is the result
589 * of a trade-off. If this option is not specified, then no maximum is
590 * assumed. Value less than the minimum registers required by ABI will
591 * be bumped up by the compiler to ABI minimum limit.
592 * - \c --ftz={true|false} (\c -ftz)\n
593 * When performing single-precision floating-point operations, flush
594 * denormal values to zero or preserve denormal values.
595 * \c --use_fast_math implies \c --ftz=true.
596 * - Default: \c false
597 * - \c --prec-sqrt={true|false} (\c -prec-sqrt)\n
598 * For single-precision floating-point square root, use IEEE
599 * round-to-nearest mode or use a faster approximation.
600 * \c --use_fast_math implies \c --prec-sqrt=false.
601 * - Default: \c true
602 * - \c --prec-div={true|false} (\c -prec-div)\n
603 * For single-precision floating-point division and reciprocals, use IEEE
604 * round-to-nearest mode or use a faster approximation.
605 * \c --use_fast_math implies \c --prec-div=false.
606 * - Default: \c true
607 * - \c --fmad={true|false} (\c -fmad)\n
608 * Enables (disables) the contraction of floating-point multiplies and
609 * adds/subtracts into floating-point multiply-add operations (FMAD,
610 * FFMA, or DFMA). \c --use_fast_math implies \c --fmad=true.
611 * - Default: \c true
612 * - \c --use_fast_math (\c -use_fast_math)\n
613 * Make use of fast math operations.
614 * \c --use_fast_math implies \c --ftz=true \c --prec-div=false
615 * \c --prec-sqrt=false \c --fmad=true.
616 * - \c --extra-device-vectorization (\c -extra-device-vectorization)\n
617 * Enables more aggressive device code vectorization in the NVVM optimizer.
618 * - \c --modify-stack-limit={true|false} (\c -modify-stack-limit)\n
619 * On Linux, during compilation, use \c setrlimit() to increase stack size
620 * to maximum allowed. The limit is reset to the previous value at the
621 * end of compilation.
622 * Note: \c setrlimit() changes the value for the entire process.
623 * - Default: \c true
624 * - \c --dlink-time-opt (\c -dlto)\n
625 * Generate intermediate code for later link-time optimization.
626 * It implies \c -rdc=true.
627 * Note: when this option is used the nvrtcGetLTOIR API should be used,
628 * as PTX or Cubin will not be generated.
629 * - \c --gen-opt-lto (\c -gen-opt-lto)\n
630 * Run the optimizer passes before generating the LTO IR.
631 * - \c --optix-ir (\c -optix-ir)\n
632 * Generate OptiX IR. The Optix IR is only intended for consumption by OptiX
633 * through appropriate APIs. This feature is not supported with
634 * link-time-optimization (\c -dlto)\n.
635 * Note: when this option is used the nvrtcGetOptiX API should be used,
636 * as PTX or Cubin will not be generated.
637 * - \c --jump-table-density=[0-101] (\c -jtd)\n
638 * Specify the case density percentage in switch statements, and use it as
639 * a minimal threshold to determine whether jump table(brx.idx instruction)
640 * will be used to implement a switch statement. Default value is 101. The
641 * percentage ranges from 0 to 101 inclusively.
642 * - Preprocessing
643 * - \c --define-macro=\<def\> (\c -D)\n
644 * \c \<def\> can be either \c \<name\> or \c \<name=definitions\>.
645 * - \c \<name\> \n
646 * Predefine \c \<name\> as a macro with definition \c 1.
647 * - \c \<name\>=\<definition\> \n
648 * The contents of \c \<definition\> are tokenized and preprocessed
649 * as if they appeared during translation phase three in a \c \#define
650 * directive. In particular, the definition will be truncated by
651 * embedded new line characters.
652 * - \c --undefine-macro=\<def\> (\c -U)\n
653 * Cancel any previous definition of \c \<def\>.
654 * - \c --include-path=\<dir\> (\c -I)\n
655 * Add the directory \c \<dir\> to the list of directories to be
656 * searched for headers. These paths are searched after the list of
657 * headers given to ::nvrtcCreateProgram.
658 * - \c --pre-include=\<header\> (\c -include)\n
659 * Preinclude \c \<header\> during preprocessing.
660 * - \c --no-source-include (\c -no-source-include)
661 * The preprocessor by default adds the directory of each input sources
662 * to the include path. This option disables this feature and only
663 * considers the path specified explicitly.
664 * - Language Dialect
665 * - \c --std={c++03|c++11|c++14|c++17|c++20}
666 * (\c -std={c++11|c++14|c++17|c++20})\n
667 * Set language dialect to C++03, C++11, C++14, C++17 or C++20
668 * - Default: \c c++17
669 * - \c --builtin-move-forward={true|false} (\c -builtin-move-forward)\n
670 * Provide builtin definitions of \c std::move and \c std::forward,
671 * when C++11 or later language dialect is selected.
672 * - Default: \c true
673 * - \c --builtin-initializer-list={true|false}
674 * (\c -builtin-initializer-list)\n
675 * Provide builtin definitions of \c std::initializer_list class and
676 * member functions when C++11 or later language dialect is selected.
677 * - Default: \c true
678 * - Misc.
679 * - \c --disable-warnings (\c -w)\n
680 * Inhibit all warning messages.
681 * - \c --restrict (\c -restrict)\n
682 * Programmer assertion that all kernel pointer parameters are restrict
683 * pointers.
684 * - \c --device-as-default-execution-space
685 * (\c -default-device)\n
686 * Treat entities with no execution space annotation as \c __device__
687 * entities.
688 * - \c --device-int128 (\c -device-int128)\n
689 * Allow the \c __int128 type in device code. Also causes the macro \c __CUDACC_RTC_INT128__
690 * to be defined.
691 * - \c --optimization-info=\<kind\> (\c -opt-info)\n
692 * Provide optimization reports for the specified kind of optimization.
693 * The following kind tags are supported:
694 * - \c inline : emit a remark when a function is inlined.
695 * - \c --display-error-number (\c -err-no)\n
696 * Display diagnostic number for warning messages. (Default)
697 * - \c --no-display-error-number (\c -no-err-no)\n
698 * Disables the display of a diagnostic number for warning messages.
699 * - \c --diag-error=<error-number>,... (\c -diag-error)\n
700 * Emit error for specified diagnostic message number(s). Message numbers can be separated by comma.
701 * - \c --diag-suppress=<error-number>,... (\c -diag-suppress)\n
702 * Suppress specified diagnostic message number(s). Message numbers can be separated by comma.
703 * - \c --diag-warn=<error-number>,... (\c -diag-warn)\n
704 * Emit warning for specified diagnostic message number(s). Message numbers can be separated by comma.
705 * - \c --brief-diagnostics={true|false} (\c -brief-diag)\n
706 * This option disables or enables showing source line and column info
707 * in a diagnostic.
708 * The --brief-diagnostics=true will not show the source line and column info.
709 * - Default: \c false
710 * - \c --time=<file-name> (\c -time)\n
711 * Generate a comma separated value table with the time taken by each compilation
712 * phase, and append it at the end of the file given as the option argument.
713 * If the file does not exist, the column headings are generated in the first row
714 * of the table. If the file name is '-', the timing data is written to the compilation log.
715 * - \c --split-compile=<number of threads> (\c -split-compile=<number of threads>)\n
716 * Perform compiler optimizations in parallel.
717 * Split compilation attempts to reduce compile time by enabling the compiler to run certain
718 * optimization passes concurrently. This option accepts a numerical value that specifies the
719 * maximum number of threads the compiler can use. One can also allow the compiler to use the maximum
720 * threads available on the system by setting --split-compile=0.
721 * Setting --split-compile=1 will cause this option to be ignored.
722 * - \c --fdevice-syntax-only (\c -fdevice-syntax-only)\n
723 * Ends device compilation after front-end syntax checking. This option does not generate valid
724 * device code.
725 * - \c --minimal (\c -minimal)\n
726 * Omit certain language features to reduce compile time for small programs.
727 * In particular, the following are omitted:
728 * - Texture and surface functions and associated types, e.g., \c cudaTextureObject_t.
729 * - CUDA Runtime Functions that are provided by the cudadevrt device code library,
730 * typically named with prefix "cuda", e.g., \c cudaMalloc.
731 * - Kernel launch from device code.
732 * - Types and macros associated with CUDA Runtime and Driver APIs,
733 * provided by cuda/tools/cudart/driver_types.h, typically named with prefix "cuda", e.g., \c cudaError_t.
734 *
735 */
736
737#ifdef __cplusplus
738}
739#endif /* __cplusplus */
740
741
742/* The utility function 'nvrtcGetTypeName' is not available by default. Define
743 the macro 'NVRTC_GET_TYPE_NAME' to a non-zero value to make it available.
744*/
745
746#if NVRTC_GET_TYPE_NAME || __DOXYGEN_ONLY__
747
748#if NVRTC_USE_CXXABI || __clang__ || __GNUC__ || __DOXYGEN_ONLY__
749#include <cxxabi.h>
750#include <cstdlib>
751
752#elif defined(_WIN32)
753#include <Windows.h>
754#include <DbgHelp.h>
755#endif /* NVRTC_USE_CXXABI || __clang__ || __GNUC__ */
756
757
758#include <string>
759#include <typeinfo>
760
761template <typename T> struct __nvrtcGetTypeName_helper_t { };
762
763/*************************************************************************//**
764 *
765 * \defgroup hosthelper Host Helper
766 *
767 * NVRTC defines the following functions for easier interaction with host code.
768 *
769 ****************************************************************************/
770
771/**
772 * \ingroup hosthelper
773 * \brief nvrtcGetTypeName stores the source level name of a type in the given
774 * std::string location.
775 *
776 * This function is only provided when the macro NVRTC_GET_TYPE_NAME is
777 * defined with a non-zero value. It uses abi::__cxa_demangle or UnDecorateSymbolName
778 * function calls to extract the type name, when using gcc/clang or cl.exe compilers,
779 * respectively. If the name extraction fails, it will return NVRTC_INTERNAL_ERROR,
780 * otherwise *result is initialized with the extracted name.
781 *
782 * Windows-specific notes:
783 * - nvrtcGetTypeName() is not multi-thread safe because it calls UnDecorateSymbolName(),
784 * which is not multi-thread safe.
785 * - The returned string may contain Microsoft-specific keywords such as __ptr64 and __cdecl.
786 *
787 * \param [in] tinfo: reference to object of type std::type_info for a given type.
788 * \param [in] result: pointer to std::string in which to store the type name.
789 * \return
790 * - \link #nvrtcResult NVRTC_SUCCESS \endlink
791 * - \link #nvrtcResult NVRTC_ERROR_INTERNAL_ERROR \endlink
792 *
793 */
794inline nvrtcResult nvrtcGetTypeName(const std::type_info &tinfo, std::string *result)
795{
796#if USE_CXXABI || __clang__ || __GNUC__
797 const char *name = tinfo.name();
798 int status;
799 char *undecorated_name = abi::__cxa_demangle(name, 0, 0, &status);
800 if (status == 0) {
801 *result = undecorated_name;
802 free(undecorated_name);
803 return NVRTC_SUCCESS;
804 }
805#elif defined(_WIN32)
806 const char *name = tinfo.raw_name();
807 if (!name || *name != '.') {
808 return NVRTC_ERROR_INTERNAL_ERROR;
809 }
810 char undecorated_name[4096];
811 //name+1 skips over the '.' prefix
812 if(UnDecorateSymbolName(name+1, undecorated_name,
813 sizeof(undecorated_name) / sizeof(*undecorated_name),
814 //note: doesn't seem to work correctly without UNDNAME_NO_ARGUMENTS.
815 UNDNAME_NO_ARGUMENTS | UNDNAME_NAME_ONLY ) ) {
816 *result = undecorated_name;
817 return NVRTC_SUCCESS;
818 }
819#endif /* USE_CXXABI || __clang__ || __GNUC__ */
820
821 return NVRTC_ERROR_INTERNAL_ERROR;
822}
823
824/**
825 * \ingroup hosthelper
826 * \brief nvrtcGetTypeName stores the source level name of the template type argument
827 * T in the given std::string location.
828 *
829 * This function is only provided when the macro NVRTC_GET_TYPE_NAME is
830 * defined with a non-zero value. It uses abi::__cxa_demangle or UnDecorateSymbolName
831 * function calls to extract the type name, when using gcc/clang or cl.exe compilers,
832 * respectively. If the name extraction fails, it will return NVRTC_INTERNAL_ERROR,
833 * otherwise *result is initialized with the extracted name.
834 *
835 * Windows-specific notes:
836 * - nvrtcGetTypeName() is not multi-thread safe because it calls UnDecorateSymbolName(),
837 * which is not multi-thread safe.
838 * - The returned string may contain Microsoft-specific keywords such as __ptr64 and __cdecl.
839 *
840 * \param [in] result: pointer to std::string in which to store the type name.
841 * \return
842 * - \link #nvrtcResult NVRTC_SUCCESS \endlink
843 * - \link #nvrtcResult NVRTC_ERROR_INTERNAL_ERROR \endlink
844 *
845 */
846
847template <typename T>
848nvrtcResult nvrtcGetTypeName(std::string *result)
849{
850 nvrtcResult res = nvrtcGetTypeName(typeid(__nvrtcGetTypeName_helper_t<T>),
851 result);
852 if (res != NVRTC_SUCCESS)
853 return res;
854
855 std::string repr = *result;
856 std::size_t idx = repr.find("__nvrtcGetTypeName_helper_t");
857 idx = (idx != std::string::npos) ? repr.find("<", idx) : idx;
858 std::size_t last_idx = repr.find_last_of('>');
859 if (idx == std::string::npos || last_idx == std::string::npos) {
860 return NVRTC_ERROR_INTERNAL_ERROR;
861 }
862 ++idx;
863 *result = repr.substr(idx, last_idx - idx);
864 return NVRTC_SUCCESS;
865}
866
867#endif /* NVRTC_GET_TYPE_NAME */
868
869#endif /* __NVRTC_H__ */
870 