codekingpro/portable-devtools
114k
1/*
2 * SPDX-FileCopyrightText: Copyright (c) 2017-2025 NVIDIA CORPORATION & AFFILIATES.
3 * All rights reserved. SPDX-License-Identifier: LicenseRef-NvidiaProprietary
4 *
5 * NVIDIA CORPORATION, its affiliates and licensors retain all intellectual
6 * property and proprietary rights in and to this material, related
7 * documentation and any modifications thereto. Any use, reproduction,
8 * disclosure or distribution of this material and related documentation
9 * without an express license agreement from NVIDIA CORPORATION or
10 * its affiliates is strictly prohibited.
11*/
12
13#ifndef NVCOMP_GDEFLATE_H
14#define NVCOMP_GDEFLATE_H
15
16#include "nvcomp.h"
17
18#ifdef __cplusplus
19extern "C" {
20#endif
21
22/**
23 * @brief Gdeflate compression options for the low-level API
24 */
25typedef struct
26{
27 /**
28 * @brief Gdeflate algorithm options.
29 *
30 * - 0: highest-throughput, entropy-only compression (use for symmetric compression/decompression performance)
31 * - 1: high-throughput, low compression ratio (default)
32 * - 2: medium-througput, medium compression ratio, beat Zlib level 1 on the compression ratio
33 * - 3: placeholder for further compression level support, will fall into MEDIUM_COMPRESSION at this point
34 * - 4: lower-throughput, higher compression ratio, beat Zlib level 6 on the compression ratio
35 * - 5: lowest-throughput, highest compression ratio
36 */
37 int algorithm;
38 /**
39 * @brief These bytes are unused and must be zeroed. This ensures
40 * compatibility if additional fields are added in the future.
41 */
42 char reserved[60];
43} nvcompBatchedGdeflateCompressOpts_t;
44
45/**
46 * @brief Gdeflate decompression options for the low-level API
47 */
48typedef struct {
49 /**
50 * @brief Decompression backend to use.
51 */
52 nvcompDecompressBackend_t backend;
53 /**
54 * @brief These bytes are unused and must be zeroed. This ensures
55 * compatibility if additional fields are added in the future.
56 */
57 char reserved[60];
58} nvcompBatchedGdeflateDecompressOpts_t;
59
60/**
61 * @brief Default Gdeflate compression options
62 */
63static const nvcompBatchedGdeflateCompressOpts_t nvcompBatchedGdeflateCompressDefaultOpts = {1, {0}};
64
65/**
66 * @brief Default Gdeflate decompression options
67 */
68static const nvcompBatchedGdeflateDecompressOpts_t nvcompBatchedGdeflateDecompressDefaultOpts =
69 {NVCOMP_DECOMPRESS_BACKEND_DEFAULT, {0}};
70
71/**
72 * @brief The maximum supported uncompressed chunk size in bytes for the Gdeflate compressor.
73 *
74 * @note Although chunk sizes up to 2GB are theoretically possible, compression
75 * with large chunks may be very slow or use large amounts of temporary memory,
76 * so caution is advised when using chunk sizes above 64KB.
77 */
78static const size_t nvcompGdeflateCompressionMaxAllowedChunkSize = 1u << 31;
79
80/**
81 * @brief The maximum supported compressed and decompressed chunk size in bytes for the GDeflate decompressor.
82 * @note To maximize decompression performance, users are encouraged to compress in smaller chunks, for example 64KiB.
83 */
84static const size_t nvcompGdeflateDecompressionMaxAllowedChunkSize = (1ULL << 32) + 288;
85
86/**
87 * @brief The most restrictive of the minimum alignment requirements for void-type CUDA memory buffers
88 * used for input, output, or temporary memory, passed to compression functions.
89 *
90 * @note In all cases, typed memory buffers must still be aligned to their type's size,
91 * e.g., 4 bytes for `int`.
92 */
93static const size_t nvcompGdeflateRequiredCompressionAlignment = 8;
94
95/**
96 * @brief Get the minimum buffer alignment requirements for compression.
97 *
98 * @note Providing buffers with alignments above the minimum requirements
99 * (e.g., 16- or 32-byte alignment) may help improve performance.
100 *
101 * @param[in] compress_opts Compression options.
102 * @param[out] alignment_requirements The minimum buffer alignment requirements
103 * for compression.
104 *
105 * @return nvcompSuccess if successful, and an error code otherwise.
106 */
107NVCOMP_EXPORT
108nvcompStatus_t nvcompBatchedGdeflateCompressGetRequiredAlignments(
109 nvcompBatchedGdeflateCompressOpts_t compress_opts,
110 nvcompAlignmentRequirements_t* alignment_requirements);
111
112/**
113 * @brief Get the amount of temporary memory required on the GPU for compression
114 * asynchronously.
115 *
116 * @note This function does not interact with the device, its result can be used immediately.
117 *
118 * @note For best performance, a chunk size of 65536 bytes is recommended.
119 *
120 * @param[in] num_chunks The number of chunks of memory in the batch.
121 * @param[in] max_uncompressed_chunk_bytes The maximum size of a chunk in the
122 * batch.
123 * @param[in] compress_opts Compression options.
124 * @param[out] temp_bytes The amount of GPU memory that will be temporarily
125 * required during compression. The value is returned on the host side.
126 * @param[in] max_total_uncompressed_bytes Upper bound on the total uncompressed
127 * size of all chunks
128 *
129 * @return nvcompSuccess if successful, and an error code otherwise.
130 */
131NVCOMP_EXPORT
132nvcompStatus_t nvcompBatchedGdeflateCompressGetTempSizeAsync(
133 size_t num_chunks,
134 size_t max_uncompressed_chunk_bytes,
135 nvcompBatchedGdeflateCompressOpts_t compress_opts,
136 size_t* temp_bytes,
137 size_t max_total_uncompressed_bytes);
138
139/**
140 * @brief Get the amount of temporary memory required on the GPU for compression.
141 * synchronously.
142 *
143 * @note This function may perform operations on the stream; if so, it will synchronize it internally.
144 * Therefore, it does not require additional synchronization after it returns,
145 * and the result can be used immediately.
146 *
147 * @note For best performance, a chunk size of 65536 bytes is recommended.
148 *
149 * @param[in] device_uncompressed_chunk_ptrs Array with size \p num_chunks of pointers
150 * to the uncompressed data chunks. Both the pointers and the uncompressed data
151 * should reside in device-accessible memory.
152 * Each chunk must be aligned to the value in the `input` member of the
153 * \ref nvcompAlignmentRequirements_t object output by
154 * `nvcompBatchedGdeflateCompressGetRequiredAlignments` when called with the same
155 * \p compress_opts.
156 * @param[in] device_uncompressed_chunk_bytes Array with size \p num_chunks of
157 * sizes of the uncompressed chunks in bytes.
158 * The sizes should reside in device-accessible memory.
159 * @param[in] num_chunks The number of chunks of memory in the batch.
160 * @param[in] max_uncompressed_chunk_bytes The maximum size of a chunk in the
161 * batch.
162 * @param[in] compress_opts Compression options.
163 * @param[out] temp_bytes The amount of GPU memory that will be temporarily
164 * required during compression. The value is returned on the host side.
165 * @param[in] max_total_uncompressed_bytes Upper bound on the total uncompressed
166 * size of all chunks
167 * @param[in] stream The CUDA stream to operate on.
168 *
169 * @return nvcompSuccess if successful, and an error code otherwise.
170 */
171NVCOMP_EXPORT
172nvcompStatus_t nvcompBatchedGdeflateCompressGetTempSizeSync(
173 const void* const* const device_uncompressed_chunk_ptrs,
174 const size_t* const device_uncompressed_chunk_bytes,
175 size_t num_chunks,
176 size_t max_uncompressed_chunk_bytes,
177 nvcompBatchedGdeflateCompressOpts_t compress_opts,
178 size_t* temp_bytes,
179 size_t max_total_uncompressed_bytes,
180 cudaStream_t stream);
181
182/**
183 * @brief Get the maximum size that a chunk of size at most max_uncompressed_chunk_bytes
184 * could compress to. That is, the minimum amount of output memory required to be given
185 * \ref nvcompBatchedGdeflateCompressAsync for each chunk.
186 *
187 * @note For best performance, a chunk size of 65536 bytes is recommended.
188 *
189 * @param[in] max_uncompressed_chunk_bytes The maximum size of a chunk before compression.
190 * @param[in] compress_opts The GDeflate compression options to use.
191 * @param[out] max_compressed_chunk_bytes The maximum possible compressed size of the chunk.
192 *
193 * @return nvcompSuccess if successful, and an error code otherwise.
194 */
195NVCOMP_EXPORT
196nvcompStatus_t nvcompBatchedGdeflateCompressGetMaxOutputChunkSize(
197 size_t max_uncompressed_chunk_bytes,
198 nvcompBatchedGdeflateCompressOpts_t compress_opts,
199 size_t* max_compressed_chunk_bytes);
200
201/**
202 * @brief Perform batched asynchronous compression.
203 *
204 * @note For best performance, a chunk size of 65536 bytes is recommended.
205 *
206 * @warning Violating any of the conditions listed in the parameter descriptions
207 * below may result in undefined behaviour.
208 *
209 * @note This function performs operations on the stream, and does not synchronize it,
210 * therefore, it requires synchronization or stream-ordered operations to use its results.
211 *
212 * @param[in] device_uncompressed_chunk_ptrs Array with size \p num_chunks of pointers
213 * to the uncompressed data chunks. Both the pointers and the uncompressed data
214 * should reside in device-accessible memory.
215 * Each chunk must be aligned to the value in the `input` member of the
216 * \ref nvcompAlignmentRequirements_t object output by
217 * `nvcompBatchedGdeflateCompressGetRequiredAlignments` when called with the same
218 * \p compress_opts.
219 * @param[in] device_uncompressed_chunk_bytes Array with size \p num_chunks of
220 * sizes of the uncompressed chunks in bytes.
221 * The sizes should reside in device-accessible memory.
222 * Chunk sizes must not exceed 65536 bytes. For best performance, a chunk size
223 * of 65536 bytes is recommended.
224 * @param[in] max_uncompressed_chunk_bytes The size of the largest uncompressed chunk.
225 * @param[in] num_chunks Number of chunks of data to compress.
226 * @param[in] device_temp_ptr The temporary GPU workspace.
227 * Must be aligned to the value in the `temp` member of the
228 * \ref nvcompAlignmentRequirements_t object output by
229 * `nvcompBatchedGdeflateCompressGetRequiredAlignments` when called with the same
230 * \p compress_opts.
231 * @param[in] temp_bytes The size of the temporary GPU memory pointed to by
232 * `device_temp_ptr`.
233 * @param[out] device_compressed_chunk_ptrs Array with size \p num_chunks of pointers
234 * to the output compressed buffers. Both the pointers and the compressed
235 * buffers should reside in device-accessible memory. Each compressed buffer
236 * should be preallocated with the size given by
237 * `nvcompBatchedGdeflateCompressGetMaxOutputChunkSize`.
238 * Each compressed buffer must be aligned to the value in the `output` member of the
239 * \ref nvcompAlignmentRequirements_t object output by
240 * `nvcompBatchedGdeflateCompressGetRequiredAlignments` when called with the same
241 * \p compress_opts.
242 * @param[out] device_compressed_chunk_bytes Array with size \p num_chunks,
243 * to be filled with the compressed sizes of each chunk.
244 * The buffer should be preallocated in device-accessible memory.
245 * @param[in] compress_opts The GDeflate compression options to use.
246 * @param[out] device_statuses Array with size \p num_chunks of statuses in
247 * device-accessible memory. This argument needs to be preallocated. For each
248 * chunk, if the compression is successful, the status will be set to
249 * `nvcompSuccess`, and an error code otherwise.
250 * Can be NULL if desired, in which case error status is not reported.
251 * @param[in] stream The CUDA stream to operate on.
252 *
253 * @return nvcompSuccess if successfully launched, and an error code otherwise.
254 */
255NVCOMP_EXPORT
256nvcompStatus_t nvcompBatchedGdeflateCompressAsync(
257 const void* const* device_uncompressed_chunk_ptrs,
258 const size_t* device_uncompressed_chunk_bytes,
259 size_t max_uncompressed_chunk_bytes,
260 size_t num_chunks,
261 void* device_temp_ptr,
262 size_t temp_bytes,
263 void* const* device_compressed_chunk_ptrs,
264 size_t* device_compressed_chunk_bytes,
265 nvcompBatchedGdeflateCompressOpts_t compress_opts,
266 nvcompStatus_t* device_statuses,
267 cudaStream_t stream);
268
269/**
270 * @brief The most restrictive of the minimum alignment requirements for void-type CUDA memory buffers
271 * used for input, output, or temporary memory, passed to decompression functions.
272 *
273 * @note In all cases, typed memory buffers must still be aligned to their type's size,
274 * e.g., 4 bytes for `int`.
275 */
276static const size_t nvcompGdeflateRequiredDecompressionAlignment = 4;
277
278/**
279 * @brief Get the minimum buffer alignment requirements for decompression.
280 *
281 * @note Providing buffers with alignments above the minimum requirements
282 * (e.g., 16- or 32-byte alignment) may help improve performance.
283 *
284 * @param[in] decompress_opts Decompression options.
285 * @param[out] alignment_requirements The minimum buffer alignment requirements
286 * for decompression.
287 *
288 * @return nvcompSuccess if successful, and an error code otherwise.
289 */
290NVCOMP_EXPORT
291nvcompStatus_t nvcompBatchedGdeflateDecompressGetRequiredAlignments(
292 nvcompBatchedGdeflateDecompressOpts_t decompress_opts,
293 nvcompAlignmentRequirements_t* alignment_requirements);
294
295/**
296 * @brief Get the amount of temporary memory required on the GPU for decompression
297 * asynchronously.
298 *
299 * @note This function does not interact with the device, its result can be used immediately.
300 *
301 * @param[in] num_chunks Number of chunks of data to be decompressed.
302 * @param[in] max_uncompressed_chunk_bytes The size of the largest chunk in bytes
303 * when uncompressed.
304 * @param[in] decompress_opts Decompression options.
305 * @param[out] temp_bytes The amount of GPU memory that will be temporarily required
306 * during decompression. The value is returned on the host side.
307 * @param[in] max_total_uncompressed_bytes The total decompressed size of all the chunks.
308 *
309 * @return nvcompSuccess if successful, and an error code otherwise.
310 */
311NVCOMP_EXPORT
312nvcompStatus_t nvcompBatchedGdeflateDecompressGetTempSizeAsync(
313 size_t num_chunks,
314 size_t max_uncompressed_chunk_bytes,
315 nvcompBatchedGdeflateDecompressOpts_t decompress_opts,
316 size_t* temp_bytes,
317 size_t max_total_uncompressed_bytes);
318
319/**
320 * @brief Get the amount of temporary memory required on the GPU for decompression
321 * synchronously.
322 *
323 * @note This function may perform operations on the stream; if so, it will synchronize it internally.
324 * Therefore, it does not require additional synchronization after it returns,
325 * and the result can be used immediately.
326 *
327 * @param[in] device_compressed_chunk_ptrs Array with size \p num_chunks of pointers
328 * in device-accessible memory to device-accessible compressed buffers.
329 * Each chunk must be aligned to the value in the `input` member of the
330 * \ref nvcompAlignmentRequirements_t object output by
331 * `nvcompBatchedGdeflateDecompressGetRequiredAlignments`.
332 * @param[in] device_compressed_chunk_bytes Array with size \p num_chunks of sizes of
333 * the compressed buffers in bytes. The sizes should reside in device-accessible memory.
334 * @param[in] num_chunks Number of chunks of data to be decompressed.
335 * @param[in] max_uncompressed_chunk_bytes The size of the largest chunk in bytes
336 * when uncompressed.
337 * @param[out] temp_bytes The amount of GPU memory that will be temporarily required
338 * during decompression. The value is returned on the host side.
339 * @param[in] max_total_uncompressed_bytes The total decompressed size of all the chunks.
340 * @param[in] decompress_opts Decompression options.
341 * @param[out] device_statuses Array with size \p num_chunks of statuses in
342 * device-accessible memory. This argument needs to be preallocated. For each
343 * chunk, if the data can be parsed successfully, the status will be set to
344 * `nvcompSuccess`, and an error code otherwise.
345 * Can be NULL if desired, in which case error status is not reported.
346 * @param[in] stream The CUDA stream to operate on.
347 *
348 * @return nvcompSuccess if successful, and an error code otherwise.
349 */
350NVCOMP_EXPORT
351nvcompStatus_t nvcompBatchedGdeflateDecompressGetTempSizeSync(
352 const void* const* const device_compressed_chunk_ptrs,
353 const size_t* const device_compressed_chunk_bytes,
354 size_t num_chunks,
355 size_t max_uncompressed_chunk_bytes,
356 size_t* temp_bytes,
357 size_t max_total_uncompressed_bytes,
358 nvcompBatchedGdeflateDecompressOpts_t decompress_opts,
359 nvcompStatus_t* device_statuses,
360 cudaStream_t stream);
361
362/**
363 * @brief Asynchronously compute the number of bytes of uncompressed data for
364 * each compressed chunk.
365 *
366 * This is needed when we do not know the expected output size.
367 *
368 * @warning If the stream is corrupt, the calculated sizes will be invalid.
369 *
370 * @warning Violating any of the conditions listed in the parameter descriptions
371 * below may result in undefined behaviour.
372 *
373 * @note This function performs operations on the stream, and does not synchronize it,
374 * therefore, it requires synchronization or stream-ordered operations to use its results.
375 *
376 * @param[in] device_compressed_chunk_ptrs Array with size \p num_chunks of
377 * pointers in device-accessible memory to compressed buffers.
378 * Each chunk must be aligned to the value in the `input` member of the
379 * \ref nvcompAlignmentRequirements_t object output by
380 * `nvcompBatchedGdeflateDecompressGetRequiredAlignments`.
381 * @param[in] device_compressed_chunk_bytes Array with size \p num_chunks of sizes
382 * of the compressed buffers in bytes. The sizes should reside in device-accessible memory.
383 * @param[out] device_uncompressed_chunk_bytes Array with size \p num_chunks
384 * to be filled with the sizes, in bytes, of each uncompressed data chunk.
385 * @param[in] num_chunks Number of data chunks to compute sizes of.
386 * @param[in] stream The CUDA stream to operate on.
387 *
388 * @return nvcompSuccess if successful, and an error code otherwise.
389 */
390NVCOMP_EXPORT
391nvcompStatus_t nvcompBatchedGdeflateGetDecompressSizeAsync(
392 const void* const* device_compressed_chunk_ptrs,
393 const size_t* device_compressed_chunk_bytes,
394 size_t* device_uncompressed_chunk_bytes,
395 size_t num_chunks,
396 cudaStream_t stream);
397
398/**
399 * @brief Perform batched asynchronous decompression.
400 *
401 * @warning Violating any of the conditions listed in the parameter descriptions
402 * below may result in undefined behaviour.
403 *
404 * @warning In the case where a chunk of compressed data is not a valid GDeflate
405 * stream, the calculated sizes of the uncompressed chunk will be invalid and
406 * nvcompStatusCannotDecompress will be flagged for that chunk.
407 *
408 * @warning Providing a corrupt buffer for decompression will result in undefined
409 * behavior.
410 *
411 * @note This function performs operations on the stream, and does not synchronize it,
412 * therefore, it requires synchronization or stream-ordered operations to use its results.
413 *
414 * @param[in] device_compressed_chunk_ptrs Array with size \p num_chunks of pointers
415 * in device-accessible memory to device-accessible compressed buffers.
416 * Each chunk must be aligned to the value in the `input` member of the
417 * \ref nvcompAlignmentRequirements_t object output by
418 * `nvcompBatchedGdeflateDecompressGetRequiredAlignments`.
419 * @param[in] device_compressed_chunk_bytes Array with size \p num_chunks of sizes of
420 * the compressed buffers in bytes. The sizes should reside in device-accessible memory.
421 * @param[in] device_uncompressed_buffer_bytes Array with size \p num_chunks of sizes,
422 * in bytes, of the output buffers to be filled with uncompressed data for each chunk.
423 * The sizes should reside in device-accessible memory. If a
424 * size is not large enough to hold all decompressed data, the decompressor
425 * will set the status in \p device_statuses corresponding to the
426 * overflow chunk to `nvcompErrorCannotDecompress`.
427 * @param[out] device_uncompressed_chunk_bytes Array with size \p num_chunks to
428 * be filled with the actual number of bytes decompressed for every chunk.
429 * This argument needs to be preallocated, but can be NULL if desired,
430 * in which case the actual sizes are not reported.
431 * @param[in] num_chunks Number of chunks of data to decompress.
432 * @param[in] device_temp_ptr The temporary GPU space.
433 * Must be aligned to the value in the `temp` member of the
434 * \ref nvcompAlignmentRequirements_t object output by
435 * `nvcompBatchedGdeflateDecompressGetRequiredAlignments`.
436 * @param[in] temp_bytes The size of the temporary GPU space.
437 * @param[out] device_uncompressed_chunk_ptrs Array with size \p num_chunks of
438 * pointers in device-accessible memory to decompressed data. Each uncompressed
439 * buffer needs to be preallocated in device-accessible memory, have the size
440 * specified by the corresponding entry in \p device_uncompressed_buffer_bytes,
441 * and be aligned to the value in the `output` member of the
442 * \ref nvcompAlignmentRequirements_t object output by
443 * `nvcompBatchedGdeflateDecompressGetRequiredAlignments`.
444 * @param[in] decompress_opts Decompression options.
445 * @param[out] device_statuses Array with size \p num_chunks of statuses in
446 * device-accessible memory. This argument needs to be preallocated. For each
447 * chunk, if the decompression is successful, the status will be set to
448 * `nvcompSuccess`. Passing corrupt, invalid, or insufficient data leads to
449 * undefined behavior or out-of-bound errors. Error reporting cannot be guaranteed
450 * in this scenario as only a limited validation is performed to maintain performance.
451 * Can be NULL if desired, in which case error status is not reported.
452 * @param[in] stream The CUDA stream to operate on.
453 *
454 * @return nvcompSuccess if successfully launched, and an error code otherwise.
455 */
456NVCOMP_EXPORT
457nvcompStatus_t nvcompBatchedGdeflateDecompressAsync(
458 const void* const* device_compressed_chunk_ptrs,
459 const size_t* device_compressed_chunk_bytes,
460 const size_t* device_uncompressed_buffer_bytes,
461 size_t* device_uncompressed_chunk_bytes,
462 size_t num_chunks,
463 void* const device_temp_ptr,
464 size_t temp_bytes,
465 void* const* device_uncompressed_chunk_ptrs,
466 nvcompBatchedGdeflateDecompressOpts_t decompress_opts,
467 nvcompStatus_t* device_statuses,
468 cudaStream_t stream);
469
470#ifdef __cplusplus
471}
472#endif
473
474#endif // NVCOMP_GDEFLATE_H
475 