codekingpro/portable-devtools
114k
1 /* Copyright 2009-2022 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
2 *
3 * NOTICE TO LICENSEE:
4 *
5 * The source code and/or documentation ("Licensed Deliverables") are
6 * subject to NVIDIA intellectual property rights under U.S. and
7 * international Copyright laws.
8 *
9 * The Licensed Deliverables contained herein are PROPRIETARY and
10 * CONFIDENTIAL to NVIDIA and are being provided under the terms and
11 * conditions of a form of NVIDIA software license agreement by and
12 * between NVIDIA and Licensee ("License Agreement") or electronically
13 * accepted by Licensee. Notwithstanding any terms or conditions to
14 * the contrary in the License Agreement, reproduction or disclosure
15 * of the Licensed Deliverables to any third party without the express
16 * written consent of NVIDIA is prohibited.
17 *
18 * NOTWITHSTANDING ANY TERMS OR CONDITIONS TO THE CONTRARY IN THE
19 * LICENSE AGREEMENT, NVIDIA MAKES NO REPRESENTATION ABOUT THE
20 * SUITABILITY OF THESE LICENSED DELIVERABLES FOR ANY PURPOSE. THEY ARE
21 * PROVIDED "AS IS" WITHOUT EXPRESS OR IMPLIED WARRANTY OF ANY KIND.
22 * NVIDIA DISCLAIMS ALL WARRANTIES WITH REGARD TO THESE LICENSED
23 * DELIVERABLES, INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY,
24 * NONINFRINGEMENT, AND FITNESS FOR A PARTICULAR PURPOSE.
25 * NOTWITHSTANDING ANY TERMS OR CONDITIONS TO THE CONTRARY IN THE
26 * LICENSE AGREEMENT, IN NO EVENT SHALL NVIDIA BE LIABLE FOR ANY
27 * SPECIAL, INDIRECT, INCIDENTAL, OR CONSEQUENTIAL DAMAGES, OR ANY
28 * DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS,
29 * WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS
30 * ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR PERFORMANCE
31 * OF THESE LICENSED DELIVERABLES.
32 *
33 * U.S. Government End Users. These Licensed Deliverables are a
34 * "commercial item" as that term is defined at 48 C.F.R. 2.101 (OCT
35 * 1995), consisting of "commercial computer software" and "commercial
36 * computer software documentation" as such terms are used in 48
37 * C.F.R. 12.212 (SEPT 1995) and are provided to the U.S. Government
38 * only as a commercial end item. Consistent with 48 C.F.R.12.212 and
39 * 48 C.F.R. 227.7202-1 through 227.7202-4 (JUNE 1995), all
40 * U.S. Government End Users acquire the Licensed Deliverables with
41 * only those rights set forth herein.
42 *
43 * Any use of the Licensed Deliverables in individual and commercial
44 * software must include, in the user documentation and internal
45 * comments to the code, the above Disclaimer and U.S. Government End
46 * Users Notice.
47 */
48#ifndef NV_NPPCORE_H
49#define NV_NPPCORE_H
50
51#include <cuda_runtime_api.h>
52
53/**
54 * \file nppcore.h
55 * Basic NPP functionality.
56 * This file contains functions to query the NPP version as well as
57 * info about the CUDA compute capabilities on a given computer.
58 */
59
60#include "nppdefs.h"
61
62#ifdef __cplusplus
63#ifdef NPP_PLUS
64using namespace nppPlusV;
65#else
66extern "C" {
67#endif
68#endif
69
70/**
71 * \page core_npp NPP Core
72 * @defgroup core_npp NPP Core
73 * Basic functions for library management, in particular library version
74 * and device property query functions.
75 * @{
76 */
77
78/**
79 * Get the NPP library version.
80 *
81 * \return A struct containing separate values for major and minor revision
82 * and build number.
83 */
84const NppLibraryVersion *
85nppGetLibVersion(void);
86
87/**
88 * Get the number of Streaming Multiprocessors (SM) on the active CUDA device.
89 *
90 * \return Number of SMs of the default CUDA device.
91 */
92int
93nppGetGpuNumSMs(void);
94
95/**
96 * Get the maximum number of threads per block on the active CUDA device.
97 *
98 * \return Maximum number of threads per block on the active CUDA device.
99 */
100int
101nppGetMaxThreadsPerBlock(void);
102
103/**
104 * Get the maximum number of threads per SM for the active GPU
105 *
106 * \return Maximum number of threads per SM for the active GPU
107 */
108int
109nppGetMaxThreadsPerSM(void);
110
111/**
112 * Get the maximum number of threads per SM, maximum threads per block, and number of SMs for the active GPU
113 *
114 * \return cudaSuccess for success, -1 for failure
115 */
116int
117nppGetGpuDeviceProperties(int * pMaxThreadsPerSM, int * pMaxThreadsPerBlock, int * pNumberOfSMs);
118
119/**
120 * Get the name of the active CUDA device.
121 *
122 * \return Name string of the active graphics-card/compute device in a system.
123 */
124const char *
125nppGetGpuName(void);
126
127/**
128 * Get the NPP CUDA stream.
129 * NPP enables concurrent device tasks via a global stream state varible.
130 * The NPP stream by default is set to stream 0, i.e. non-concurrent mode.
131 * A user can set the NPP stream to any valid CUDA stream. All CUDA commands
132 * issued by NPP (e.g. kernels launched by the NPP library) are then
133 * issed to that NPP stream.
134 */
135cudaStream_t
136nppGetStream(void);
137
138/**
139 * Get the current NPP managed CUDA stream context as set by calls to nppSetStream().
140 * NPP enables concurrent device tasks via an NPP maintained global stream state context.
141 * The NPP stream by default is set to stream 0, i.e. non-concurrent mode.
142 * A user can set the NPP stream to any valid CUDA stream which will update the current NPP managed stream state context
143 * or supply application initialized stream contexts to NPP calls. All CUDA commands
144 * issued by NPP (e.g. kernels launched by the NPP library) are then
145 * issed to the current NPP managed stream or to application supplied stream contexts depending on whether
146 * the stream context is passed to the NPP function or not. NPP managed stream context calls (those without stream context parameters)
147 * can be intermixed with application managed stream context calls but any NPP managed stream context calls will always use the most recent
148 * stream set by nppSetStream() or the NULL stream if nppSetStream() has never been called.
149 */
150NppStatus
151nppGetStreamContext(NppStreamContext * pNppStreamContext);
152
153/**
154 * Get the number of SMs on the device associated with the current NPP CUDA stream.
155 * NPP enables concurrent device tasks via a global stream state varible.
156 * The NPP stream by default is set to stream 0, i.e. non-concurrent mode.
157 * A user can set the NPP stream to any valid CUDA stream. All CUDA commands
158 * issued by NPP (e.g. kernels launched by the NPP library) are then
159 * issed to that NPP stream. This call avoids a cudaGetDeviceProperties() call.
160 */
161unsigned int
162nppGetStreamNumSMs(void);
163
164/**
165 * Get the maximum number of threads per SM on the device associated with the current NPP CUDA stream.
166 * NPP enables concurrent device tasks via a global stream state varible.
167 * The NPP stream by default is set to stream 0, i.e. non-concurrent mode.
168 * A user can set the NPP stream to any valid CUDA stream. All CUDA commands
169 * issued by NPP (e.g. kernels launched by the NPP library) are then
170 * issed to that NPP stream. This call avoids a cudaGetDeviceProperties() call.
171 */
172unsigned int
173nppGetStreamMaxThreadsPerSM(void);
174
175/**
176 * Set the NPP CUDA stream. This function now returns an error if a problem occurs with Cuda stream management.
177 * This function should only be called if a call to nppGetStream() returns a stream number which is different from
178 * the desired stream since unnecessarily flushing the current stream can significantly affect performance.
179 * \see nppGetStream()
180 */
181NppStatus
182nppSetStream(cudaStream_t hStream);
183
184/** @} core_npp */
185
186#ifdef __cplusplus
187#ifndef NPP_PLUS
188} /* extern "C" */
189#endif
190#endif
191
192#endif /* NV_NPPCORE_H */
193 