codekingpro/portable-devtools
114k
1/*
2 * SPDX-FileCopyrightText: Copyright (c) 2017-2025 NVIDIA CORPORATION & AFFILIATES.
3 * All rights reserved. SPDX-License-Identifier: LicenseRef-NvidiaProprietary
4 *
5 * NVIDIA CORPORATION, its affiliates and licensors retain all intellectual
6 * property and proprietary rights in and to this material, related
7 * documentation and any modifications thereto. Any use, reproduction,
8 * disclosure or distribution of this material and related documentation
9 * without an express license agreement from NVIDIA CORPORATION or
10 * its affiliates is strictly prohibited.
11*/
12
13#ifndef NVCOMP_BITCOMP_H
14#define NVCOMP_BITCOMP_H
15
16#include "nvcomp.h"
17
18#ifdef __cplusplus
19extern "C" {
20#endif
21
22/**
23 * @brief Bitcomp compression options for the low-level API
24 */
25typedef struct
26{
27 /**
28 * @brief Bitcomp algorithm options.
29 *
30 * - 0 : Default algorithm, usually gives the best compression ratios
31 * - 1 : "Sparse" algorithm, works well on sparse data (with lots of zeroes)
32 * and is usually faster than the default algorithm.
33 */
34 int algorithm;
35 /**
36 * @brief One of nvcomp's possible data types
37 */
38 nvcompType_t data_type;
39 /**
40 * @brief These bytes are unused and must be zeroed. This ensures
41 * compatibility if additional fields are added in the future.
42 */
43 char reserved[56];
44} nvcompBatchedBitcompCompressOpts_t;
45
46/**
47 * @brief Bitcomp decompression options for the low-level API
48 */
49typedef struct {
50 /**
51 * @brief Decompression backend to use.
52 */
53 nvcompDecompressBackend_t backend;
54 /**
55 * @brief These bytes are unused and must be zeroed. This ensures
56 * compatibility if additional fields are added in the future.
57 */
58 char reserved[60];
59} nvcompBatchedBitcompDecompressOpts_t;
60
61/**
62 * @brief Default Bitcomp compression options
63 */
64static const nvcompBatchedBitcompCompressOpts_t nvcompBatchedBitcompCompressDefaultOpts =
65 {0, NVCOMP_TYPE_UCHAR, {0}};
66
67/**
68 * @brief Default Bitcomp decompression options
69 */
70static const nvcompBatchedBitcompDecompressOpts_t nvcompBatchedBitcompDecompressDefaultOpts =
71 {NVCOMP_DECOMPRESS_BACKEND_DEFAULT, {0}};
72
73/**
74 * @brief The maximum supported uncompressed chunk size in bytes for the Bitcomp compressor.
75 */
76static const size_t nvcompBitcompCompressionMaxAllowedChunkSize = 1 << 24;
77
78/**
79 * @brief The maximum supported compressed and decompressed chunk size in bytes for the Bitcomp decompressor.
80 * @note To maximize decompression performance, users are encouraged to compress in smaller chunks, for example 64KiB.
81 */
82static const size_t nvcompBitcompDecompressionMaxAllowedChunkSize = 1ull << 25;
83
84/**
85 * @brief The most restrictive of the minimum alignment requirements for void-type CUDA memory buffers
86 * used for input, output, or temporary memory, passed to compression functions.
87 *
88 * @note In all cases, typed memory buffers must still be aligned to their type's size,
89 * e.g., 4 bytes for `int`.
90 */
91static const size_t nvcompBitcompRequiredCompressionAlignment = 8;
92
93/**
94 * @brief Get the minimum buffer alignment requirements for compression.
95 *
96 * @note Providing buffers with alignments above the minimum requirements
97 * (e.g., 16- or 32-byte alignment) may help improve performance.
98 *
99 * @param[in] compress_opts Compression options.
100 * @param[out] alignment_requirements The minimum buffer alignment requirements
101 * for compression.
102 *
103 * @return nvcompSuccess if successful, and an error code otherwise.
104 */
105NVCOMP_EXPORT
106nvcompStatus_t nvcompBatchedBitcompCompressGetRequiredAlignments(
107 nvcompBatchedBitcompCompressOpts_t compress_opts,
108 nvcompAlignmentRequirements_t* alignment_requirements);
109
110/**
111 * @brief Get the amount of temporary memory required on the GPU for compression
112 * asynchronously.
113 *
114 * @note This function does not interact with the device, its result can be used immediately.
115 *
116 * @param[in] num_chunks The number of chunks of memory in the batch.
117 * @param[in] max_uncompressed_chunk_bytes The maximum size of a chunk in the
118 * batch.
119 * @param[in] compress_opts Compression options.
120 * @param[out] temp_bytes The amount of GPU memory that will be temporarily
121 * required during compression. The value is returned on the host side.
122 * @param[in] max_total_uncompressed_bytes Upper bound on the total uncompressed
123 * size of all chunks
124 *
125 * @return nvcompSuccess if successful, and an error code otherwise.
126 */
127NVCOMP_EXPORT
128nvcompStatus_t nvcompBatchedBitcompCompressGetTempSizeAsync(
129 size_t num_chunks,
130 size_t max_uncompressed_chunk_bytes,
131 nvcompBatchedBitcompCompressOpts_t compress_opts,
132 size_t* temp_bytes,
133 size_t max_total_uncompressed_bytes);
134
135/**
136 * @brief Get the amount of temporary memory required on the GPU for compression.
137 * synchronously.
138 *
139 * @note This function may perform operations on the stream; if so, it will synchronize it internally.
140 * Therefore, it does not require additional synchronization after it returns,
141 * and the result can be used immediately.
142 *
143 * @param[in] device_uncompressed_chunk_ptrs Array with size \p num_chunks of pointers
144 * to the uncompressed data chunks. Both the pointers and the uncompressed data
145 * should reside in device-accessible memory.
146 * Each chunk must be aligned to the value in the `input` member of the
147 * \ref nvcompAlignmentRequirements_t object output by
148 * `nvcompBatchedBitcompCompressGetRequiredAlignments` when called with the same
149 * \p compress_opts.
150 * @param[in] device_uncompressed_chunk_bytes Array with size \p num_chunks of
151 * sizes of the uncompressed chunks in bytes.
152 * The sizes should reside in device-accessible memory.
153 * @param[in] num_chunks The number of chunks of memory in the batch.
154 * @param[in] max_uncompressed_chunk_bytes The maximum size of a chunk in the
155 * batch.
156 * @param[in] compress_opts Compression options.
157 * @param[out] temp_bytes The amount of GPU memory that will be temporarily
158 * required during compression. The value is returned on the host side.
159 * @param[in] max_total_uncompressed_bytes Upper bound on the total uncompressed
160 * size of all chunks
161 * @param[in] stream The CUDA stream to operate on.
162 *
163 * @return nvcompSuccess if successful, and an error code otherwise.
164 */
165NVCOMP_EXPORT
166nvcompStatus_t nvcompBatchedBitcompCompressGetTempSizeSync(
167 const void* const* const device_uncompressed_chunk_ptrs,
168 const size_t* const device_uncompressed_chunk_bytes,
169 size_t num_chunks,
170 size_t max_uncompressed_chunk_bytes,
171 nvcompBatchedBitcompCompressOpts_t compress_opts,
172 size_t* temp_bytes,
173 size_t max_total_uncompressed_bytes,
174 cudaStream_t stream);
175
176/**
177 * @brief Get the maximum size that a chunk of size at most max_uncompressed_chunk_bytes
178 * could compress to. That is, the minimum amount of output memory required to be given
179 * \ref nvcompBatchedBitcompCompressAsync for each chunk.
180 *
181 * @param[in] max_uncompressed_chunk_bytes The maximum size of a chunk before compression.
182 * @param[in] compress_opts Compression options.
183 * @param[out] max_compressed_chunk_bytes The maximum possible compressed size of the chunk.
184 *
185 * @return nvcompSuccess if successful, and an error code otherwise.
186 */
187NVCOMP_EXPORT
188nvcompStatus_t nvcompBatchedBitcompCompressGetMaxOutputChunkSize(
189 size_t max_uncompressed_chunk_bytes,
190 nvcompBatchedBitcompCompressOpts_t compress_opts,
191 size_t* max_compressed_chunk_bytes);
192
193/**
194 * @brief Perform batched asynchronous compression.
195 *
196 * @warning Violating any of the conditions listed in the parameter descriptions
197 * below may result in undefined behaviour.
198 *
199 * @note This function performs operations on the stream, and does not synchronize it,
200 * therefore, it requires synchronization or stream-ordered operations to use its results.
201 *
202 * @param[in] device_uncompressed_chunk_ptrs Array with size \p num_chunks of pointers
203 * to the uncompressed data chunks. Both the pointers and the uncompressed data
204 * should reside in device-accessible memory.
205 * Each chunk must be aligned to the value in the `input` member of the
206 * \ref nvcompAlignmentRequirements_t object output by
207 * `nvcompBatchedBitcompCompressGetRequiredAlignments` when called with the same
208 * \p compress_opts.
209 * @param[in] device_uncompressed_chunk_bytes Array with size \p num_chunks of
210 * sizes of the uncompressed chunks in bytes.
211 * The sizes should reside in device-accessible memory.
212 * Each chunk size must be a multiple of the size of the data type specified by
213 * compress_opts.data_type.
214 * @param[in] max_uncompressed_chunk_bytes The maximum size of a chunk in the
215 * batch. This parameter is currently unused.
216 * Set it to either the actual value or zero.
217 * @param[in] num_chunks Number of chunks of data to compress.
218 * @param[in] device_temp_ptr This argument is not used.
219 * @param[in] temp_bytes This argument is not used.
220 * @param[out] device_compressed_chunk_ptrs Array with size \p num_chunks of pointers
221 * to the output compressed buffers. Both the pointers and the compressed
222 * buffers should reside in device-accessible memory. Each compressed buffer
223 * should be preallocated with the size given by
224 * `nvcompBatchedBitcompCompressGetMaxOutputChunkSize`.
225 * Each compressed buffer must be aligned to the value in the `output` member of the
226 * \ref nvcompAlignmentRequirements_t object output by
227 * `nvcompBatchedBitcompCompressGetRequiredAlignments` when called with the same
228 * \p compress_opts.
229 * @param[out] device_compressed_chunk_bytes Array with size \p num_chunks,
230 * to be filled with the compressed sizes of each chunk.
231 * The buffer should be preallocated in device-accessible memory.
232 * @param[in] compress_opts Compression options. They must be valid.
233 * @param[out] device_statuses Array with size \p num_chunks of statuses in
234 * device-accessible memory. This argument needs to be preallocated. For each
235 * chunk, if the compression is successful, the status will be set to
236 * `nvcompSuccess`, and an error code otherwise.
237 * Can be NULL if desired, in which case error status is not reported.
238 * @param[in] stream The CUDA stream to operate on.
239 *
240 * @return nvcompSuccess if successfully launched, and an error code otherwise.
241 */
242NVCOMP_EXPORT
243nvcompStatus_t nvcompBatchedBitcompCompressAsync(
244 const void* const* device_uncompressed_chunk_ptrs,
245 const size_t* device_uncompressed_chunk_bytes,
246 size_t max_uncompressed_chunk_bytes, // not used
247 size_t num_chunks,
248 void* device_temp_ptr, // not used
249 size_t temp_bytes, // not used
250 void* const* device_compressed_chunk_ptrs,
251 size_t* device_compressed_chunk_bytes,
252 nvcompBatchedBitcompCompressOpts_t compress_opts,
253 nvcompStatus_t* device_statuses,
254 cudaStream_t stream);
255
256/**
257 * @brief The most restrictive of the minimum alignment requirements for void-type CUDA memory buffers
258 * used for input, output, or temporary memory, passed to decompression functions.
259 *
260 * @note In all cases, typed memory buffers must still be aligned to their type's size,
261 * e.g., 4 bytes for `int`.
262 */
263static const size_t nvcompBitcompRequiredDecompressionAlignment = 8;
264
265/**
266 * @brief Get the minimum buffer alignment requirements for decompression.
267 *
268 * @note Providing buffers with alignments above the minimum requirements
269 * (e.g., 16- or 32-byte alignment) may help improve performance.
270 *
271 * @param[in] decompress_opts Decompression options.
272 * @param[out] alignment_requirements The minimum buffer alignment requirements
273 * for decompression.
274 *
275 * @return nvcompSuccess if successful, and an error code otherwise.
276 */
277NVCOMP_EXPORT
278nvcompStatus_t nvcompBatchedBitcompDecompressGetRequiredAlignments(
279 nvcompBatchedBitcompDecompressOpts_t decompress_opts,
280 nvcompAlignmentRequirements_t* alignment_requirements);
281
282/**
283 * @brief Get the amount of temporary memory required on the GPU for decompression
284 * asynchronously.
285 *
286 * @note This function does not interact with the device, its result can be used immediately.
287 *
288 * @param[in] num_chunks Number of chunks of data to be decompressed.
289 * @param[in] max_uncompressed_chunk_bytes The size of the largest chunk in bytes
290 * when uncompressed.
291 * @param[in] decompress_opts Decompression options.
292 * @param[out] temp_bytes The amount of GPU memory that will be temporarily required
293 * during decompression. The value is returned on the host side.
294 * @param[in] max_total_uncompressed_bytes The total decompressed size of all the chunks.
295 *
296 * @return nvcompSuccess if successful, and an error code otherwise.
297 */
298NVCOMP_EXPORT
299nvcompStatus_t nvcompBatchedBitcompDecompressGetTempSizeAsync(
300 size_t num_chunks,
301 size_t max_uncompressed_chunk_bytes,
302 nvcompBatchedBitcompDecompressOpts_t decompress_opts,
303 size_t* temp_bytes,
304 size_t max_total_uncompressed_bytes);
305
306/**
307 * @brief Get the amount of temporary memory required on the GPU for decompression
308 * synchronously.
309 *
310 * @note This function may perform operations on the stream; if so, it will synchronize it internally.
311 * Therefore, it does not require additional synchronization after it returns,
312 * and the result can be used immediately.
313 *
314 * @param[in] device_compressed_chunk_ptrs Array with size \p num_chunks of pointers
315 * in device-accessible memory to device-accessible compressed buffers.
316 * Each chunk must be aligned to the value in the `input` member of the
317 * \ref nvcompAlignmentRequirements_t object output by
318 * `nvcompBatchedBitcompDecompressGetRequiredAlignments`.
319 * @param[in] device_compressed_chunk_bytes Array with size \p num_chunks of sizes of
320 * the compressed buffers in bytes. The sizes should reside in device-accessible memory.
321 * @param[in] num_chunks Number of chunks of data to be decompressed.
322 * @param[in] max_uncompressed_chunk_bytes The size of the largest chunk in bytes
323 * when uncompressed.
324 * @param[out] temp_bytes The amount of GPU memory that will be temporarily required
325 * during decompression. The value is returned on the host side.
326 * @param[in] max_total_uncompressed_bytes The total decompressed size of all the chunks.
327 * Unused in Bitcomp.
328 * @param[in] decompress_opts Decompression options.
329 * @param[out] device_statuses Array with size \p num_chunks of statuses in
330 * device-accessible memory. This argument needs to be preallocated. For each
331 * chunk, if the data can be parsed successfully, the status will be set to
332 * `nvcompSuccess`, and an error code otherwise.
333 * Can be NULL if desired, in which case error status is not reported.
334 * @param[in] stream The CUDA stream to operate on.
335 *
336 * @return nvcompSuccess if successful, and an error code otherwise.
337 */
338NVCOMP_EXPORT
339nvcompStatus_t nvcompBatchedBitcompDecompressGetTempSizeSync(
340 const void* const* const device_compressed_chunk_ptrs,
341 const size_t* const device_compressed_chunk_bytes,
342 size_t num_chunks,
343 size_t max_uncompressed_chunk_bytes,
344 size_t* temp_bytes,
345 size_t max_total_uncompressed_bytes,
346 nvcompBatchedBitcompDecompressOpts_t decompress_opts,
347 nvcompStatus_t* device_statuses,
348 cudaStream_t stream);
349
350/**
351 * @brief Asynchronously compute the number of bytes of uncompressed data for
352 * each compressed chunk.
353 *
354 * @warning Violating any of the conditions listed in the parameter descriptions
355 * below may result in undefined behaviour.
356 *
357 * @note This function performs operations on the stream, and does not synchronize it,
358 * therefore, it requires synchronization or stream-ordered operations to use its results.
359 *
360 * @param[in] device_compressed_chunk_ptrs Array with size \p num_chunks of
361 * pointers in device-accessible memory to compressed buffers.
362 * Each chunk must be aligned to the value in the `input` member of the
363 * \ref nvcompAlignmentRequirements_t object output by
364 * `nvcompBatchedBitcompDecompressGetRequiredAlignments`.
365 * @param[in] device_compressed_chunk_bytes This argument is not used.
366 * @param[out] device_uncompressed_chunk_bytes Array with size \p num_chunks
367 * to be filled with the sizes, in bytes, of each uncompressed data chunk.
368 * If there is an error when retrieving the size of a chunk, the
369 * uncompressed size of that chunk will be set to 0. This argument needs to
370 * be preallocated in device-accessible memory.
371 * @param[in] num_chunks Number of data chunks to compute sizes of.
372 * @param[in] stream The CUDA stream to operate on.
373 *
374 * @return nvcompSuccess if successful, and an error code otherwise.
375 */
376NVCOMP_EXPORT
377nvcompStatus_t nvcompBatchedBitcompGetDecompressSizeAsync(
378 const void* const* device_compressed_chunk_ptrs,
379 const size_t* device_compressed_chunk_bytes,
380 size_t* device_uncompressed_chunk_bytes,
381 size_t num_chunks,
382 cudaStream_t stream);
383
384/**
385 * @brief Perform batched asynchronous decompression.
386 *
387 * This function is used to decompress compressed buffers produced by
388 * \ref nvcompBatchedBitcompCompressAsync . It can also decompress buffers
389 * compressed with the native Bitcomp API.
390 *
391 * @warning Violating any of the conditions listed in the parameter descriptions
392 * below may result in undefined behaviour.
393 *
394 * @warning Providing a corrupt buffer for decompression will result in undefined
395 * behavior.
396 *
397 * @note The function is not completely asynchronous, as it needs to look
398 * at the compressed data in order to create the proper bitcomp handle.
399 * The stream is synchronized, the data is examined, then the asynchronous
400 * decompression is launched.
401 *
402 * @note An asynchronous, faster version of batched Bitcomp asynchronous decompression
403 * is available, and can be launched via the HLIF manager.
404 *
405 * @param[in] device_compressed_chunk_ptrs Array with size \p num_chunks of pointers
406 * in device-accessible memory to device-accessible compressed buffers.
407 * Each chunk must be aligned to the value in the `input` member of the
408 * \ref nvcompAlignmentRequirements_t object output by
409 * `nvcompBatchedBitcompDecompressGetRequiredAlignments`.
410 * @param[in] device_compressed_chunk_bytes This argument is not used.
411 * @param[in] device_uncompressed_buffer_bytes Array with size \p num_chunks of sizes,
412 * in bytes, of the output buffers to be filled with uncompressed data for each chunk.
413 * The sizes should reside in device-accessible memory. If a
414 * size is not large enough to hold all decompressed data, the decompressor
415 * will set the status in \p device_statuses corresponding to the
416 * overflow chunk to `nvcompErrorCannotDecompress`.
417 * @param[out] device_uncompressed_chunk_bytes Array with size \p num_chunks to
418 * be filled with the actual number of bytes decompressed for every chunk.
419 * This argument needs to be preallocated.
420 * @param[in] num_chunks Number of chunks of data to decompress.
421 * @param[in] device_temp_ptr Temporary scratch memory.
422 * @param[in] temp_bytes Size of temporary scratch memory.
423 * @param[out] device_uncompressed_chunk_ptrs Array with size \p num_chunks of
424 * pointers in device-accessible memory to decompressed data. Each uncompressed
425 * buffer needs to be preallocated in device-accessible memory, have the size
426 * specified by the corresponding entry in \p device_uncompressed_buffer_bytes,
427 * and be aligned to the value in the `output` member of the
428 * \ref nvcompAlignmentRequirements_t object output by
429 * `nvcompBatchedBitcompDecompressGetRequiredAlignments`.
430 * @param[in] decompress_opts Decompression options.
431 * @param[out] device_statuses Array with size \p num_chunks of statuses in
432 * device-accessible memory. This argument needs to be preallocated. For each
433 * chunk, if the decompression is successful, the status will be set to
434 * `nvcompSuccess`. Passing corrupt, invalid, or insufficient data leads to
435 * undefined behavior or out-of-bound errors. Error reporting cannot be guaranteed
436 * in this scenario as only a limited validation is performed to maintain performance.
437 * Can be NULL if desired, in which case error status is not reported.
438 * @param[in] stream The CUDA stream to operate on.
439 *
440 * @return nvcompSuccess if successfully launched, and an error code otherwise.
441 */
442NVCOMP_EXPORT
443nvcompStatus_t nvcompBatchedBitcompDecompressAsync(
444 const void* const* device_compressed_chunk_ptrs,
445 const size_t* device_compressed_chunk_bytes, // not used
446 const size_t* device_uncompressed_buffer_bytes,
447 size_t* device_uncompressed_chunk_bytes,
448 size_t num_chunks,
449 void* const device_temp_ptr,
450 size_t temp_bytes,
451 void* const* device_uncompressed_chunk_ptrs,
452 nvcompBatchedBitcompDecompressOpts_t decompress_opts,
453 nvcompStatus_t* device_statuses,
454 cudaStream_t stream);
455
456#ifdef __cplusplus
457}
458#endif
459
460#endif // NVCOMP_BITCOMP_H
461 