Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
nppcore.h193 linesDownload Raw Back to include
1 /* Copyright 2009-2022 NVIDIA CORPORATION & AFFILIATES.  All rights reserved. 
2  * 
3  * NOTICE TO LICENSEE: 
4  * 
5  * The source code and/or documentation ("Licensed Deliverables") are 
6  * subject to NVIDIA intellectual property rights under U.S. and 
7  * international Copyright laws. 
8  * 
9  * The Licensed Deliverables contained herein are PROPRIETARY and 
10  * CONFIDENTIAL to NVIDIA and are being provided under the terms and 
11  * conditions of a form of NVIDIA software license agreement by and 
12  * between NVIDIA and Licensee ("License Agreement") or electronically 
13  * accepted by Licensee.  Notwithstanding any terms or conditions to 
14  * the contrary in the License Agreement, reproduction or disclosure 
15  * of the Licensed Deliverables to any third party without the express 
16  * written consent of NVIDIA is prohibited. 
17  * 
18  * NOTWITHSTANDING ANY TERMS OR CONDITIONS TO THE CONTRARY IN THE 
19  * LICENSE AGREEMENT, NVIDIA MAKES NO REPRESENTATION ABOUT THE 
20  * SUITABILITY OF THESE LICENSED DELIVERABLES FOR ANY PURPOSE.  THEY ARE 
21  * PROVIDED "AS IS" WITHOUT EXPRESS OR IMPLIED WARRANTY OF ANY KIND. 
22  * NVIDIA DISCLAIMS ALL WARRANTIES WITH REGARD TO THESE LICENSED 
23  * DELIVERABLES, INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY, 
24  * NONINFRINGEMENT, AND FITNESS FOR A PARTICULAR PURPOSE. 
25  * NOTWITHSTANDING ANY TERMS OR CONDITIONS TO THE CONTRARY IN THE 
26  * LICENSE AGREEMENT, IN NO EVENT SHALL NVIDIA BE LIABLE FOR ANY 
27  * SPECIAL, INDIRECT, INCIDENTAL, OR CONSEQUENTIAL DAMAGES, OR ANY 
28  * DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, 
29  * WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS 
30  * ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR PERFORMANCE 
31  * OF THESE LICENSED DELIVERABLES. 
32  * 
33  * U.S. Government End Users.  These Licensed Deliverables are a 
34  * "commercial item" as that term is defined at 48 C.F.R. 2.101 (OCT 
35  * 1995), consisting of "commercial computer software" and "commercial 
36  * computer software documentation" as such terms are used in 48 
37  * C.F.R. 12.212 (SEPT 1995) and are provided to the U.S. Government 
38  * only as a commercial end item.  Consistent with 48 C.F.R.12.212 and 
39  * 48 C.F.R. 227.7202-1 through 227.7202-4 (JUNE 1995), all 
40  * U.S. Government End Users acquire the Licensed Deliverables with 
41  * only those rights set forth herein. 
42  * 
43  * Any use of the Licensed Deliverables in individual and commercial 
44  * software must include, in the user documentation and internal 
45  * comments to the code, the above Disclaimer and U.S. Government End 
46  * Users Notice. 
47  */ 
48#ifndef NV_NPPCORE_H
49#define NV_NPPCORE_H
50
51#include <cuda_runtime_api.h>
52
53/**
54 * \file nppcore.h
55 * Basic NPP functionality. 
56 *  This file contains functions to query the NPP version as well as 
57 *  info about the CUDA compute capabilities on a given computer.
58 */
59 
60#include "nppdefs.h"
61
62#ifdef __cplusplus
63#ifdef NPP_PLUS
64using namespace nppPlusV;
65#else
66extern "C" {
67#endif
68#endif
69 
70/** 
71 * \page core_npp NPP Core
72 * @defgroup core_npp NPP Core
73 * Basic functions for library management, in particular library version
74 * and device property query functions.
75 * @{
76 */
77
78/**
79 * Get the NPP library version.
80 *
81 * \return A struct containing separate values for major and minor revision 
82 *      and build number.
83 */
84const NppLibraryVersion * 
85nppGetLibVersion(void);
86
87/**
88 * Get the number of Streaming Multiprocessors (SM) on the active CUDA device.
89 *
90 * \return Number of SMs of the default CUDA device.
91 */
92int 
93nppGetGpuNumSMs(void);
94
95/**
96 * Get the maximum number of threads per block on the active CUDA device.
97 *
98 * \return Maximum number of threads per block on the active CUDA device.
99 */
100int 
101nppGetMaxThreadsPerBlock(void);
102
103/**
104 * Get the maximum number of threads per SM for the active GPU
105 *
106 * \return Maximum number of threads per SM for the active GPU
107 */
108int 
109nppGetMaxThreadsPerSM(void);
110
111/**
112 * Get the maximum number of threads per SM, maximum threads per block, and number of SMs for the active GPU
113 *
114 * \return cudaSuccess for success, -1 for failure
115 */
116int 
117nppGetGpuDeviceProperties(int * pMaxThreadsPerSM, int * pMaxThreadsPerBlock, int * pNumberOfSMs);
118
119/** 
120 * Get the name of the active CUDA device.
121 *
122 * \return Name string of the active graphics-card/compute device in a system.
123 */
124const char * 
125nppGetGpuName(void);
126
127/**
128 * Get the NPP CUDA stream.
129 * NPP enables concurrent device tasks via a global stream state varible.
130 * The NPP stream by default is set to stream 0, i.e. non-concurrent mode.
131 * A user can set the NPP stream to any valid CUDA stream. All CUDA commands
132 * issued by NPP (e.g. kernels launched by the NPP library) are then
133 * issed to that NPP stream.
134 */
135cudaStream_t
136nppGetStream(void);
137
138/**
139 * Get the current NPP managed CUDA stream context as set by calls to nppSetStream().
140 * NPP enables concurrent device tasks via an NPP maintained global stream state context.
141 * The NPP stream by default is set to stream 0, i.e. non-concurrent mode.
142 * A user can set the NPP stream to any valid CUDA stream which will update the current NPP managed stream state context 
143 * or supply application initialized stream contexts to NPP calls. All CUDA commands
144 * issued by NPP (e.g. kernels launched by the NPP library) are then
145 * issed to the current NPP managed stream or to application supplied stream contexts depending on whether 
146 * the stream context is passed to the NPP function or not.  NPP managed stream context calls (those without stream context parameters) 
147 * can be intermixed with application managed stream context calls but any NPP managed stream context calls will always use the most recent 
148 * stream set by nppSetStream() or the NULL stream if nppSetStream() has never been called. 
149 */
150NppStatus
151nppGetStreamContext(NppStreamContext * pNppStreamContext);
152
153/**
154 * Get the number of SMs on the device associated with the current NPP CUDA stream.
155 * NPP enables concurrent device tasks via a global stream state varible.
156 * The NPP stream by default is set to stream 0, i.e. non-concurrent mode.
157 * A user can set the NPP stream to any valid CUDA stream. All CUDA commands
158 * issued by NPP (e.g. kernels launched by the NPP library) are then
159 * issed to that NPP stream.  This call avoids a cudaGetDeviceProperties() call.
160 */
161unsigned int
162nppGetStreamNumSMs(void);
163
164/**
165 * Get the maximum number of threads per SM on the device associated with the current NPP CUDA stream.
166 * NPP enables concurrent device tasks via a global stream state varible.
167 * The NPP stream by default is set to stream 0, i.e. non-concurrent mode.
168 * A user can set the NPP stream to any valid CUDA stream. All CUDA commands
169 * issued by NPP (e.g. kernels launched by the NPP library) are then
170 * issed to that NPP stream.  This call avoids a cudaGetDeviceProperties() call.
171 */
172unsigned int
173nppGetStreamMaxThreadsPerSM(void);
174
175/**
176 * Set the NPP CUDA stream.  This function now returns an error if a problem occurs with Cuda stream management. 
177 *   This function should only be called if a call to nppGetStream() returns a stream number which is different from
178 *   the desired stream since unnecessarily flushing the current stream can significantly affect performance.
179 * \see nppGetStream()
180 */
181NppStatus
182nppSetStream(cudaStream_t hStream);
183
184/** @} core_npp */ 
185
186#ifdef __cplusplus
187#ifndef NPP_PLUS
188} /* extern "C" */
189#endif
190#endif
191
192#endif /* NV_NPPCORE_H */
193 
codekingpro/portable-devtools · Team Ai