Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
gdeflate.h475 linesDownload Raw Back to nvcomp
1/*
2 * SPDX-FileCopyrightText: Copyright (c) 2017-2025 NVIDIA CORPORATION & AFFILIATES.
3 * All rights reserved. SPDX-License-Identifier: LicenseRef-NvidiaProprietary
4 *
5 * NVIDIA CORPORATION, its affiliates and licensors retain all intellectual
6 * property and proprietary rights in and to this material, related
7 * documentation and any modifications thereto. Any use, reproduction,
8 * disclosure or distribution of this material and related documentation
9 * without an express license agreement from NVIDIA CORPORATION or
10 * its affiliates is strictly prohibited.
11*/
12
13#ifndef NVCOMP_GDEFLATE_H
14#define NVCOMP_GDEFLATE_H
15
16#include "nvcomp.h"
17
18#ifdef __cplusplus
19extern "C" {
20#endif
21
22/**
23 * @brief Gdeflate compression options for the low-level API
24 */
25typedef struct
26{
27  /**
28   * @brief Gdeflate algorithm options.
29   *
30   * - 0: highest-throughput, entropy-only compression (use for symmetric compression/decompression performance)
31   * - 1: high-throughput, low compression ratio (default)
32   * - 2: medium-througput, medium compression ratio, beat Zlib level 1 on the compression ratio
33   * - 3: placeholder for further compression level support, will fall into MEDIUM_COMPRESSION at this point
34   * - 4: lower-throughput, higher compression ratio, beat Zlib level 6 on the compression ratio
35   * - 5: lowest-throughput, highest compression ratio
36   */
37  int algorithm;
38  /**
39   * @brief These bytes are unused and must be zeroed. This ensures
40   *        compatibility if additional fields are added in the future.
41   */
42  char reserved[60];
43} nvcompBatchedGdeflateCompressOpts_t;
44
45/**
46 * @brief Gdeflate decompression options for the low-level API
47 */
48typedef struct {
49   /**
50   * @brief Decompression backend to use.
51   */
52  nvcompDecompressBackend_t backend;
53  /**
54   * @brief These bytes are unused and must be zeroed. This ensures
55   *        compatibility if additional fields are added in the future.
56   */
57  char reserved[60];
58} nvcompBatchedGdeflateDecompressOpts_t;
59
60/**
61 * @brief Default Gdeflate compression options
62 */
63static const nvcompBatchedGdeflateCompressOpts_t nvcompBatchedGdeflateCompressDefaultOpts = {1, {0}};
64
65/**
66 * @brief Default Gdeflate decompression options
67 */
68static const nvcompBatchedGdeflateDecompressOpts_t nvcompBatchedGdeflateDecompressDefaultOpts =
69    {NVCOMP_DECOMPRESS_BACKEND_DEFAULT, {0}};
70
71/**
72 * @brief The maximum supported uncompressed chunk size in bytes for the Gdeflate compressor.
73 *
74 * @note Although chunk sizes up to 2GB are theoretically possible, compression
75 * with large chunks may be very slow or use large amounts of temporary memory,
76 * so caution is advised when using chunk sizes above 64KB.
77 */
78static const size_t nvcompGdeflateCompressionMaxAllowedChunkSize = 1u << 31;
79
80/**
81 * @brief The maximum supported compressed and decompressed chunk size in bytes for the GDeflate decompressor.
82 * @note To maximize decompression performance, users are encouraged to compress in smaller chunks, for example 64KiB.
83 */
84static const size_t nvcompGdeflateDecompressionMaxAllowedChunkSize = (1ULL << 32) + 288;
85
86/**
87 * @brief The most restrictive of the minimum alignment requirements for void-type CUDA memory buffers
88 * used for input, output, or temporary memory, passed to compression functions.
89 *
90 * @note In all cases, typed memory buffers must still be aligned to their type's size,
91 * e.g., 4 bytes for `int`.
92 */
93static const size_t nvcompGdeflateRequiredCompressionAlignment = 8;
94
95/**
96 * @brief Get the minimum buffer alignment requirements for compression.
97 *
98 * @note Providing buffers with alignments above the minimum requirements
99 * (e.g., 16- or 32-byte alignment) may help improve performance.
100 *
101 * @param[in] compress_opts Compression options.
102 * @param[out] alignment_requirements The minimum buffer alignment requirements
103 * for compression.
104 *
105 * @return nvcompSuccess if successful, and an error code otherwise.
106 */
107NVCOMP_EXPORT
108nvcompStatus_t nvcompBatchedGdeflateCompressGetRequiredAlignments(
109    nvcompBatchedGdeflateCompressOpts_t compress_opts,
110    nvcompAlignmentRequirements_t* alignment_requirements);
111
112/**
113 * @brief Get the amount of temporary memory required on the GPU for compression
114 * asynchronously.
115 *
116 * @note This function does not interact with the device, its result can be used immediately.
117 *
118 * @note For best performance, a chunk size of 65536 bytes is recommended.
119 *
120 * @param[in] num_chunks The number of chunks of memory in the batch.
121 * @param[in] max_uncompressed_chunk_bytes The maximum size of a chunk in the
122 * batch.
123 * @param[in] compress_opts Compression options.
124 * @param[out] temp_bytes The amount of GPU memory that will be temporarily
125 * required during compression. The value is returned on the host side.
126 * @param[in] max_total_uncompressed_bytes Upper bound on the total uncompressed
127 * size of all chunks
128 *
129 * @return nvcompSuccess if successful, and an error code otherwise.
130 */
131NVCOMP_EXPORT
132nvcompStatus_t nvcompBatchedGdeflateCompressGetTempSizeAsync(
133    size_t num_chunks,
134    size_t max_uncompressed_chunk_bytes,
135    nvcompBatchedGdeflateCompressOpts_t compress_opts,
136    size_t* temp_bytes,
137    size_t max_total_uncompressed_bytes);
138
139/**
140 * @brief Get the amount of temporary memory required on the GPU for compression.
141 * synchronously.
142 *
143 * @note This function may perform operations on the stream; if so, it will synchronize it internally.
144 * Therefore, it does not require additional synchronization after it returns,
145 * and the result can be used immediately.
146 *
147 * @note For best performance, a chunk size of 65536 bytes is recommended.
148 *
149 * @param[in] device_uncompressed_chunk_ptrs Array with size \p num_chunks of pointers
150 * to the uncompressed data chunks. Both the pointers and the uncompressed data
151 * should reside in device-accessible memory.
152 * Each chunk must be aligned to the value in the `input` member of the
153 * \ref nvcompAlignmentRequirements_t object output by
154 * `nvcompBatchedGdeflateCompressGetRequiredAlignments` when called with the same
155 * \p compress_opts.
156 * @param[in] device_uncompressed_chunk_bytes Array with size \p num_chunks of
157 * sizes of the uncompressed chunks in bytes.
158 * The sizes should reside in device-accessible memory.
159 * @param[in] num_chunks The number of chunks of memory in the batch.
160 * @param[in] max_uncompressed_chunk_bytes The maximum size of a chunk in the
161 * batch.
162 * @param[in] compress_opts Compression options.
163 * @param[out] temp_bytes The amount of GPU memory that will be temporarily
164 * required during compression. The value is returned on the host side.
165 * @param[in] max_total_uncompressed_bytes Upper bound on the total uncompressed
166 * size of all chunks
167 * @param[in] stream The CUDA stream to operate on.
168 *
169 * @return nvcompSuccess if successful, and an error code otherwise.
170 */
171NVCOMP_EXPORT
172nvcompStatus_t nvcompBatchedGdeflateCompressGetTempSizeSync(
173    const void* const* const device_uncompressed_chunk_ptrs,
174    const size_t* const device_uncompressed_chunk_bytes,
175    size_t num_chunks,
176    size_t max_uncompressed_chunk_bytes,
177    nvcompBatchedGdeflateCompressOpts_t compress_opts,
178    size_t* temp_bytes,
179    size_t max_total_uncompressed_bytes,
180    cudaStream_t stream);
181
182/**
183 * @brief Get the maximum size that a chunk of size at most max_uncompressed_chunk_bytes
184 * could compress to. That is, the minimum amount of output memory required to be given
185 * \ref nvcompBatchedGdeflateCompressAsync for each chunk.
186 *
187 * @note For best performance, a chunk size of 65536 bytes is recommended.
188 *
189 * @param[in] max_uncompressed_chunk_bytes The maximum size of a chunk before compression.
190 * @param[in] compress_opts The GDeflate compression options to use.
191 * @param[out] max_compressed_chunk_bytes The maximum possible compressed size of the chunk.
192 *
193 * @return nvcompSuccess if successful, and an error code otherwise.
194 */
195NVCOMP_EXPORT
196nvcompStatus_t nvcompBatchedGdeflateCompressGetMaxOutputChunkSize(
197    size_t max_uncompressed_chunk_bytes,
198    nvcompBatchedGdeflateCompressOpts_t compress_opts,
199    size_t* max_compressed_chunk_bytes);
200
201/**
202 * @brief Perform batched asynchronous compression.
203 *
204 * @note For best performance, a chunk size of 65536 bytes is recommended.
205 * 
206 * @warning Violating any of the conditions listed in the parameter descriptions
207 * below may result in undefined behaviour.
208 *
209 * @note This function performs operations on the stream, and does not synchronize it,
210 * therefore, it requires synchronization or stream-ordered operations to use its results.
211 *
212 * @param[in] device_uncompressed_chunk_ptrs Array with size \p num_chunks of pointers
213 * to the uncompressed data chunks. Both the pointers and the uncompressed data
214 * should reside in device-accessible memory.
215 * Each chunk must be aligned to the value in the `input` member of the
216 * \ref nvcompAlignmentRequirements_t object output by
217 * `nvcompBatchedGdeflateCompressGetRequiredAlignments` when called with the same
218 * \p compress_opts.
219 * @param[in] device_uncompressed_chunk_bytes Array with size \p num_chunks of
220 * sizes of the uncompressed chunks in bytes.
221 * The sizes should reside in device-accessible memory.
222 * Chunk sizes must not exceed 65536 bytes. For best performance, a chunk size
223 * of 65536 bytes is recommended.
224 * @param[in] max_uncompressed_chunk_bytes The size of the largest uncompressed chunk.
225 * @param[in] num_chunks Number of chunks of data to compress.
226 * @param[in] device_temp_ptr The temporary GPU workspace.
227 * Must be aligned to the value in the `temp` member of the
228 * \ref nvcompAlignmentRequirements_t object output by
229 * `nvcompBatchedGdeflateCompressGetRequiredAlignments` when called with the same
230 * \p compress_opts.
231 * @param[in] temp_bytes The size of the temporary GPU memory pointed to by
232 * `device_temp_ptr`.
233 * @param[out] device_compressed_chunk_ptrs Array with size \p num_chunks of pointers
234 * to the output compressed buffers. Both the pointers and the compressed
235 * buffers should reside in device-accessible memory. Each compressed buffer
236 * should be preallocated with the size given by
237 * `nvcompBatchedGdeflateCompressGetMaxOutputChunkSize`.
238 * Each compressed buffer must be aligned to the value in the `output` member of the
239 * \ref nvcompAlignmentRequirements_t object output by
240 * `nvcompBatchedGdeflateCompressGetRequiredAlignments` when called with the same
241 * \p compress_opts.
242 * @param[out] device_compressed_chunk_bytes Array with size \p num_chunks,
243 * to be filled with the compressed sizes of each chunk.
244 * The buffer should be preallocated in device-accessible memory.
245 * @param[in] compress_opts The GDeflate compression options to use.
246 * @param[out] device_statuses Array with size \p num_chunks of statuses in
247 * device-accessible memory. This argument needs to be preallocated. For each
248 * chunk, if the compression is successful, the status will be set to
249 * `nvcompSuccess`, and an error code otherwise.
250 * Can be NULL if desired, in which case error status is not reported.
251 * @param[in] stream The CUDA stream to operate on.
252 *
253 * @return nvcompSuccess if successfully launched, and an error code otherwise.
254 */
255NVCOMP_EXPORT
256nvcompStatus_t nvcompBatchedGdeflateCompressAsync(
257    const void* const* device_uncompressed_chunk_ptrs,
258    const size_t* device_uncompressed_chunk_bytes,
259    size_t max_uncompressed_chunk_bytes,
260    size_t num_chunks,
261    void* device_temp_ptr,
262    size_t temp_bytes,
263    void* const* device_compressed_chunk_ptrs,
264    size_t* device_compressed_chunk_bytes,
265    nvcompBatchedGdeflateCompressOpts_t compress_opts,
266    nvcompStatus_t* device_statuses,
267    cudaStream_t stream);
268
269/**
270 * @brief The most restrictive of the minimum alignment requirements for void-type CUDA memory buffers
271 * used for input, output, or temporary memory, passed to decompression functions.
272 *
273 * @note In all cases, typed memory buffers must still be aligned to their type's size,
274 * e.g., 4 bytes for `int`.
275 */
276static const size_t nvcompGdeflateRequiredDecompressionAlignment = 4;
277
278/**
279 * @brief Get the minimum buffer alignment requirements for decompression.
280 *
281 * @note Providing buffers with alignments above the minimum requirements
282 * (e.g., 16- or 32-byte alignment) may help improve performance.
283 *
284 * @param[in] decompress_opts Decompression options.
285 * @param[out] alignment_requirements The minimum buffer alignment requirements
286 * for decompression.
287 *
288 * @return nvcompSuccess if successful, and an error code otherwise.
289 */
290NVCOMP_EXPORT
291nvcompStatus_t nvcompBatchedGdeflateDecompressGetRequiredAlignments(
292    nvcompBatchedGdeflateDecompressOpts_t decompress_opts,
293    nvcompAlignmentRequirements_t* alignment_requirements);
294
295/**
296 * @brief Get the amount of temporary memory required on the GPU for decompression
297 * asynchronously.
298 *
299 * @note This function does not interact with the device, its result can be used immediately.
300 *
301 * @param[in] num_chunks Number of chunks of data to be decompressed.
302 * @param[in] max_uncompressed_chunk_bytes The size of the largest chunk in bytes
303 * when uncompressed.
304 * @param[in] decompress_opts Decompression options.
305 * @param[out] temp_bytes The amount of GPU memory that will be temporarily required
306 * during decompression. The value is returned on the host side.
307 * @param[in] max_total_uncompressed_bytes The total decompressed size of all the chunks.
308 *
309 * @return nvcompSuccess if successful, and an error code otherwise.
310 */
311NVCOMP_EXPORT
312nvcompStatus_t nvcompBatchedGdeflateDecompressGetTempSizeAsync(
313    size_t num_chunks,
314    size_t max_uncompressed_chunk_bytes,
315    nvcompBatchedGdeflateDecompressOpts_t decompress_opts,
316    size_t* temp_bytes,
317    size_t max_total_uncompressed_bytes);
318
319/**
320 * @brief Get the amount of temporary memory required on the GPU for decompression
321 * synchronously.
322 *
323 * @note This function may perform operations on the stream; if so, it will synchronize it internally.
324 * Therefore, it does not require additional synchronization after it returns,
325 * and the result can be used immediately.
326 *
327 * @param[in] device_compressed_chunk_ptrs Array with size \p num_chunks of pointers
328 * in device-accessible memory to device-accessible compressed buffers.
329 * Each chunk must be aligned to the value in the `input` member of the
330 * \ref nvcompAlignmentRequirements_t object output by
331 * `nvcompBatchedGdeflateDecompressGetRequiredAlignments`.
332 * @param[in] device_compressed_chunk_bytes Array with size \p num_chunks of sizes of
333 * the compressed buffers in bytes. The sizes should reside in device-accessible memory.
334 * @param[in] num_chunks Number of chunks of data to be decompressed.
335 * @param[in] max_uncompressed_chunk_bytes The size of the largest chunk in bytes
336 * when uncompressed.
337 * @param[out] temp_bytes The amount of GPU memory that will be temporarily required
338 * during decompression. The value is returned on the host side.
339 * @param[in] max_total_uncompressed_bytes  The total decompressed size of all the chunks.
340 * @param[in] decompress_opts Decompression options.
341 * @param[out] device_statuses Array with size \p num_chunks of statuses in
342 * device-accessible memory. This argument needs to be preallocated. For each
343 * chunk, if the data can be parsed successfully, the status will be set to
344 * `nvcompSuccess`, and an error code otherwise.
345 * Can be NULL if desired, in which case error status is not reported.
346 * @param[in] stream The CUDA stream to operate on.
347 *
348 * @return nvcompSuccess if successful, and an error code otherwise.
349 */
350NVCOMP_EXPORT
351nvcompStatus_t nvcompBatchedGdeflateDecompressGetTempSizeSync(
352    const void* const* const device_compressed_chunk_ptrs,
353    const size_t* const device_compressed_chunk_bytes,
354    size_t num_chunks,
355    size_t max_uncompressed_chunk_bytes,
356    size_t* temp_bytes,
357    size_t max_total_uncompressed_bytes,
358    nvcompBatchedGdeflateDecompressOpts_t decompress_opts,
359    nvcompStatus_t* device_statuses,
360    cudaStream_t stream);
361
362/**
363 * @brief Asynchronously compute the number of bytes of uncompressed data for
364 * each compressed chunk.
365 *
366 * This is needed when we do not know the expected output size.
367 *
368 * @warning If the stream is corrupt, the calculated sizes will be invalid.
369 *
370 * @warning Violating any of the conditions listed in the parameter descriptions
371 * below may result in undefined behaviour.
372 *
373 * @note This function performs operations on the stream, and does not synchronize it,
374 * therefore, it requires synchronization or stream-ordered operations to use its results.
375 *
376 * @param[in] device_compressed_chunk_ptrs Array with size \p num_chunks of
377 * pointers in device-accessible memory to compressed buffers.
378 * Each chunk must be aligned to the value in the `input` member of the
379 * \ref nvcompAlignmentRequirements_t object output by
380 * `nvcompBatchedGdeflateDecompressGetRequiredAlignments`.
381 * @param[in] device_compressed_chunk_bytes Array with size \p num_chunks of sizes
382 * of the compressed buffers in bytes. The sizes should reside in device-accessible memory.
383 * @param[out] device_uncompressed_chunk_bytes Array with size \p num_chunks
384 * to be filled with the sizes, in bytes, of each uncompressed data chunk.
385 * @param[in] num_chunks Number of data chunks to compute sizes of.
386 * @param[in] stream The CUDA stream to operate on.
387 *
388 * @return nvcompSuccess if successful, and an error code otherwise.
389 */
390NVCOMP_EXPORT
391nvcompStatus_t nvcompBatchedGdeflateGetDecompressSizeAsync(
392    const void* const* device_compressed_chunk_ptrs,
393    const size_t* device_compressed_chunk_bytes,
394    size_t* device_uncompressed_chunk_bytes,
395    size_t num_chunks,
396    cudaStream_t stream);
397
398/**
399 * @brief Perform batched asynchronous decompression.
400 *
401 * @warning Violating any of the conditions listed in the parameter descriptions
402 * below may result in undefined behaviour.
403 *
404 * @warning In the case where a chunk of compressed data is not a valid GDeflate
405 * stream, the calculated sizes of the uncompressed chunk will be invalid and
406 * nvcompStatusCannotDecompress will be flagged for that chunk.
407 *
408 * @warning Providing a corrupt buffer for decompression will result in undefined
409 * behavior.
410 *
411 * @note This function performs operations on the stream, and does not synchronize it,
412 * therefore, it requires synchronization or stream-ordered operations to use its results.
413 *
414 * @param[in] device_compressed_chunk_ptrs Array with size \p num_chunks of pointers
415 * in device-accessible memory to device-accessible compressed buffers.
416 * Each chunk must be aligned to the value in the `input` member of the
417 * \ref nvcompAlignmentRequirements_t object output by
418 * `nvcompBatchedGdeflateDecompressGetRequiredAlignments`.
419 * @param[in] device_compressed_chunk_bytes Array with size \p num_chunks of sizes of
420 * the compressed buffers in bytes. The sizes should reside in device-accessible memory.
421 * @param[in] device_uncompressed_buffer_bytes Array with size \p num_chunks of sizes,
422 * in bytes, of the output buffers to be filled with uncompressed data for each chunk.
423 * The sizes should reside in device-accessible memory. If a
424 * size is not large enough to hold all decompressed data, the decompressor
425 * will set the status in \p device_statuses corresponding to the
426 * overflow chunk to `nvcompErrorCannotDecompress`.
427 * @param[out] device_uncompressed_chunk_bytes Array with size \p num_chunks to
428 * be filled with the actual number of bytes decompressed for every chunk.
429 * This argument needs to be preallocated, but can be NULL if desired,
430 * in which case the actual sizes are not reported.
431 * @param[in] num_chunks Number of chunks of data to decompress.
432 * @param[in] device_temp_ptr The temporary GPU space.
433 * Must be aligned to the value in the `temp` member of the
434 * \ref nvcompAlignmentRequirements_t object output by
435 * `nvcompBatchedGdeflateDecompressGetRequiredAlignments`.
436 * @param[in] temp_bytes The size of the temporary GPU space.
437 * @param[out] device_uncompressed_chunk_ptrs Array with size \p num_chunks of
438 * pointers in device-accessible memory to decompressed data. Each uncompressed
439 * buffer needs to be preallocated in device-accessible memory, have the size
440 * specified by the corresponding entry in \p device_uncompressed_buffer_bytes,
441 * and be aligned to the value in the `output` member of the
442 * \ref nvcompAlignmentRequirements_t object output by
443 * `nvcompBatchedGdeflateDecompressGetRequiredAlignments`.
444 * @param[in] decompress_opts Decompression options.
445 * @param[out] device_statuses Array with size \p num_chunks of statuses in
446 * device-accessible memory. This argument needs to be preallocated. For each
447 * chunk, if the decompression is successful, the status will be set to
448 * `nvcompSuccess`. Passing corrupt, invalid, or insufficient data leads to
449 * undefined behavior or out-of-bound errors. Error reporting cannot be guaranteed
450 * in this scenario as only a limited validation is performed to maintain performance.
451 * Can be NULL if desired, in which case error status is not reported.
452 * @param[in] stream The CUDA stream to operate on.
453 *
454 * @return nvcompSuccess if successfully launched, and an error code otherwise.
455 */
456NVCOMP_EXPORT
457nvcompStatus_t nvcompBatchedGdeflateDecompressAsync(
458    const void* const* device_compressed_chunk_ptrs,
459    const size_t* device_compressed_chunk_bytes,
460    const size_t* device_uncompressed_buffer_bytes,
461    size_t* device_uncompressed_chunk_bytes,
462    size_t num_chunks,
463    void* const device_temp_ptr,
464    size_t temp_bytes,
465    void* const* device_uncompressed_chunk_ptrs,
466    nvcompBatchedGdeflateDecompressOpts_t decompress_opts,
467    nvcompStatus_t* device_statuses,
468    cudaStream_t stream);
469
470#ifdef __cplusplus
471}
472#endif
473
474#endif // NVCOMP_GDEFLATE_H
475 
codekingpro/portable-devtools · Team Ai