Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
zstd.h457 linesDownload Raw Back to nvcomp
1/*
2 * SPDX-FileCopyrightText: Copyright (c) 2022-2025 NVIDIA CORPORATION & AFFILIATES.
3 * All rights reserved. SPDX-License-Identifier: LicenseRef-NvidiaProprietary
4 *
5 * NVIDIA CORPORATION, its affiliates and licensors retain all intellectual
6 * property and proprietary rights in and to this material, related
7 * documentation and any modifications thereto. Any use, reproduction,
8 * disclosure or distribution of this material and related documentation
9 * without an express license agreement from NVIDIA CORPORATION or
10 * its affiliates is strictly prohibited.
11*/
12
13#ifndef NVCOMP_ZSTD_H
14#define NVCOMP_ZSTD_H
15
16#include "nvcomp.h"
17
18#ifdef __cplusplus
19extern "C" {
20#endif
21
22/**
23 * @brief Zstd compression options for the low-level API
24 */
25typedef struct
26{
27  /**
28   * @brief These bytes are unused and must be zeroed. This ensures
29   *        compatibility if additional fields are added in the future.
30   */
31  char reserved[64];
32} nvcompBatchedZstdCompressOpts_t;
33
34/**
35 * @brief Zstd decompression options for the low-level API
36 */
37typedef struct {
38  /**
39   * @brief Decompression backend to use.
40   */
41  nvcompDecompressBackend_t backend;
42  /**
43   * @brief These bytes are unused and must be zeroed. This ensures
44   *        compatibility if additional fields are added in the future.
45   */
46  char reserved[60];
47} nvcompBatchedZstdDecompressOpts_t;
48
49/**
50 * @brief Default Zstd compression options
51 */
52static const nvcompBatchedZstdCompressOpts_t nvcompBatchedZstdCompressDefaultOpts = {{0}};
53
54/**
55 * @brief Default Zstd decompression options
56 */
57static const nvcompBatchedZstdDecompressOpts_t nvcompBatchedZstdDecompressDefaultOpts =
58    {NVCOMP_DECOMPRESS_BACKEND_DEFAULT, {0}};
59
60/**
61 * @brief The maximum supported uncompressed chunk size in bytes for the Zstd compressor.
62 */
63static const size_t nvcompZstdCompressionMaxAllowedChunkSize = (1UL << 31) - 1;
64
65/**
66 * @brief The maximum supported compressed and decompressed chunk size in bytes for the Zstd decompressor.
67 * @note To maximize decompression performance, users are encouraged to compress in smaller chunks, for example 64KiB.
68 */
69static const size_t nvcompZstdDecompressionMaxAllowedChunkSize = (1ull << 31) - 1;
70
71/**
72 * @brief The most restrictive of the minimum alignment requirements for void-type CUDA memory buffers
73 * used for input, output, or temporary memory, passed to compression functions.
74 *
75 * @note In all cases, typed memory buffers must still be aligned to their type's size,
76 * e.g., 4 bytes for `int`.
77 */
78static const size_t nvcompZstdRequiredCompressionAlignment = 4;
79
80/**
81 * @brief Get the minimum buffer alignment requirements for compression.
82 *
83 * @note Providing buffers with alignments above the minimum requirements
84 * (e.g., 16- or 32-byte alignment) may help improve performance.
85 *
86 * @param[in] compress_opts Compression options.
87 * @param[out] alignment_requirements The minimum buffer alignment requirements
88 * for compression.
89 *
90 * @return nvcompSuccess if successful, and an error code otherwise.
91 */
92NVCOMP_EXPORT
93nvcompStatus_t nvcompBatchedZstdCompressGetRequiredAlignments(
94    nvcompBatchedZstdCompressOpts_t compress_opts,
95    nvcompAlignmentRequirements_t* alignment_requirements);
96
97/**
98 * @brief Get the amount of temporary memory required on the GPU for compression
99 * asynchronously.
100 *
101 * @note For best performance, a chunk size of 65536 bytes is recommended.
102 * 
103 * @note This function must be called using the same CUDA device as the 
104 * subsequent compression operations, since temporary storage requirements 
105 * are architecture-specific.
106 *
107 * @note This function does not interact asynchronously with the device,
108 * its result can be used immediately.
109 * 
110 * @param[in] num_chunks The number of chunks of memory in the batch.
111 * @param[in] max_uncompressed_chunk_bytes The maximum size of a chunk in the
112 * batch.
113 * @param[in] compress_opts Compression options.
114 * @param[out] temp_bytes The amount of GPU memory that will be temporarily
115 * required during compression. The value is returned on the host side.
116 * @param[in] max_total_uncompressed_bytes Upper bound on the total uncompressed
117 * size of all chunks
118 *
119 * @return nvcompSuccess if successful, and an error code otherwise.
120 */
121NVCOMP_EXPORT
122nvcompStatus_t nvcompBatchedZstdCompressGetTempSizeAsync(
123    size_t num_chunks,
124    size_t max_uncompressed_chunk_bytes,
125    nvcompBatchedZstdCompressOpts_t compress_opts,
126    size_t* temp_bytes,
127    size_t max_total_uncompressed_bytes);
128
129/**
130 * @brief Get the amount of temporary memory required on the GPU for compression.
131 * synchronously.
132 *
133 * @note This function may perform operations on the stream; if so, it will synchronize it internally.
134 * Therefore, it does not require additional synchronization after it returns,
135 * and the result can be used immediately.
136 *
137 * @note For best performance, a chunk size of 65536 bytes is recommended.
138 *
139 * @param[in] device_uncompressed_chunk_ptrs Array with size \p num_chunks of pointers
140 * to the uncompressed data chunks. Both the pointers and the uncompressed data
141 * should reside in device-accessible memory.
142 * Each chunk must be aligned to the value in the `input` member of the
143 * \ref nvcompAlignmentRequirements_t object output by
144 * `nvcompBatchedZstdCompressGetRequiredAlignments` when called with the same
145 * \p compress_opts.
146 * @param[in] device_uncompressed_chunk_bytes Array with size \p num_chunks of
147 * sizes of the uncompressed chunks in bytes.
148 * The sizes should reside in device-accessible memory.
149 * @param[in] num_chunks The number of chunks of memory in the batch.
150 * @param[in] max_uncompressed_chunk_bytes The maximum size of a chunk in the
151 * batch.
152 * @param[in] compress_opts Compression options.
153 * @param[out] temp_bytes The amount of GPU memory that will be temporarily
154 * required during compression. The value is returned on the host side.
155 * @param[in] max_total_uncompressed_bytes Upper bound on the total uncompressed
156 * size of all chunks
157 * @param[in] stream The CUDA stream to operate on.
158 *
159 * @return nvcompSuccess if successful, and an error code otherwise.
160 */
161NVCOMP_EXPORT
162nvcompStatus_t nvcompBatchedZstdCompressGetTempSizeSync(
163    const void* const* const device_uncompressed_chunk_ptrs,
164    const size_t* const device_uncompressed_chunk_bytes,
165    size_t num_chunks,
166    size_t max_uncompressed_chunk_bytes,
167    nvcompBatchedZstdCompressOpts_t compress_opts,
168    size_t* temp_bytes,
169    size_t max_total_uncompressed_bytes,
170    cudaStream_t stream);
171
172/**
173 * @brief Get the maximum size that a chunk of size at most max_uncompressed_chunk_bytes
174 * could compress to. That is, the minimum amount of output memory required to be given
175 * \ref nvcompBatchedZstdCompressAsync for each chunk.
176 *
177 * @note For best performance, a chunk size of 65536 bytes is recommended.
178 *
179 * @param[in] max_uncompressed_chunk_bytes The maximum size of a chunk before compression.
180 * @param[in] compress_opts The Zstd compression options to use. Currently empty.
181 * @param[out] max_compressed_chunk_bytes The maximum possible compressed size of the chunk.
182 *
183 * @return nvcompSuccess if successful, and an error code otherwise.
184 */
185NVCOMP_EXPORT
186nvcompStatus_t nvcompBatchedZstdCompressGetMaxOutputChunkSize(
187    size_t max_uncompressed_chunk_bytes,
188    nvcompBatchedZstdCompressOpts_t compress_opts,
189    size_t* max_compressed_chunk_bytes);
190
191/**
192 * @brief Perform batched asynchronous compression.
193 *
194 * @note For best performance, a chunk size of 65536 bytes is recommended.
195 * 
196 * @warning Violating any of the conditions listed in the parameter descriptions
197 * below may result in undefined behaviour.
198 *
199 * @note This function performs operations on the stream, and does not synchronize it,
200 * therefore, it requires synchronization or stream-ordered operations to use its results.
201 *
202 * @param[in] device_uncompressed_chunk_ptrs Array with size \p num_chunks of pointers
203 * to the uncompressed data chunks. Both the pointers and the uncompressed data
204 * should reside in device-accessible memory.
205 * Each chunk must be aligned to the value in the `input` member of the
206 * \ref nvcompAlignmentRequirements_t object output by
207 * `nvcompBatchedZstdCompressGetRequiredAlignments` when called with the same
208 * \p compress_opts.
209 * @param[in] device_uncompressed_chunk_bytes Array with size \p num_chunks of
210 * sizes of the uncompressed chunks in bytes.
211 * The sizes should reside in device-accessible memory.
212 * Chunk sizes must not exceed 16 MB. For best performance, a chunk size of
213 * 64 KB is recommended.
214 * @param[in] max_uncompressed_chunk_bytes The size of the largest uncompressed chunk.
215 * @param[in] num_chunks Number of chunks of data to compress.
216 * @param[in] device_temp_ptr The temporary GPU workspace, could be NULL in case
217 * temporary memory is not needed.
218 * Must be aligned to the value in the `temp` member of the
219 * \ref nvcompAlignmentRequirements_t object output by
220 * `nvcompBatchedZstdCompressGetRequiredAlignments` when called with the same
221 * \p compress_opts.
222 * @param[in] temp_bytes The size of the temporary GPU memory pointed to by
223 * `device_temp_ptr`.
224 * @param[out] device_compressed_chunk_ptrs Array with size \p num_chunks of pointers
225 * to the output compressed buffers. Both the pointers and the compressed
226 * buffers should reside in device-accessible memory. Each compressed buffer
227 * should be preallocated with the size given by
228 * `nvcompBatchedZstdCompressGetMaxOutputChunkSize`.
229 * Each compressed buffer must be aligned to the value in the `output` member of the
230 * \ref nvcompAlignmentRequirements_t object output by
231 * `nvcompBatchedZstdCompressGetRequiredAlignments` when called with the same
232 * \p compress_opts.
233 * @param[out] device_compressed_chunk_bytes Array with size \p num_chunks,
234 * to be filled with the compressed sizes of each chunk.
235 * The buffer should be preallocated in device-accessible memory.
236 * @param[in] compress_opts The Zstd compression options to use. Currently empty.
237 * @param[out] device_statuses Array with size \p num_chunks of statuses in
238 * device-accessible memory. This argument needs to be preallocated. For each
239 * chunk, if the decompression is successful, the status will be set to
240 * `nvcompSuccess`, and an error code otherwise.
241 * Can be NULL if desired, in which case error status is not reported.
242 * @param[in] stream The CUDA stream to operate on.
243 *
244 * @return nvcompSuccess if successfully launched, and an error code otherwise.
245 */
246NVCOMP_EXPORT
247nvcompStatus_t nvcompBatchedZstdCompressAsync(
248    const void* const* device_uncompressed_chunk_ptrs,
249    const size_t* device_uncompressed_chunk_bytes,
250    size_t max_uncompressed_chunk_bytes,
251    size_t num_chunks,
252    void* device_temp_ptr,
253    size_t temp_bytes,
254    void* const* device_compressed_chunk_ptrs,
255    size_t* device_compressed_chunk_bytes,
256    nvcompBatchedZstdCompressOpts_t compress_opts,
257    nvcompStatus_t* device_statuses,
258    cudaStream_t stream);
259
260/**
261 * @brief The most restrictive of the minimum alignment requirements for void-type CUDA memory buffers
262 * used for input, output, or temporary memory, passed to decompression functions.
263 *
264 * @note In all cases, typed memory buffers must still be aligned to their type's size,
265 * e.g., 4 bytes for `int`.
266 */
267static const size_t nvcompZstdRequiredDecompressionAlignment = 8;
268
269/**
270 * @brief Get the minimum buffer alignment requirements for decompression.
271 *
272 * @note Providing buffers with alignments above the minimum requirements
273 * (e.g., 16- or 32-byte alignment) may help improve performance.
274 *
275 * @param[in] decompress_opts Decompression options.
276 * @param[out] alignment_requirements The minimum buffer alignment requirements
277 * for decompression.
278 *
279 * @return nvcompSuccess if successful, and an error code otherwise.
280 */
281NVCOMP_EXPORT
282nvcompStatus_t nvcompBatchedZstdDecompressGetRequiredAlignments(
283    nvcompBatchedZstdDecompressOpts_t decompress_opts,
284    nvcompAlignmentRequirements_t* alignment_requirements);
285
286/**
287 * @brief Get the amount of temporary memory required on the GPU for decompression
288 * asynchronously.
289 *
290 * @note This function does not interact with the device, its result can be used immediately.
291 *
292 * @param[in] num_chunks Number of chunks of data to be decompressed.
293 * @param[in] max_uncompressed_chunk_bytes The size of the largest chunk in bytes
294 * when uncompressed.
295 * @param[in] decompress_opts Decompression options.
296 * @param[out] temp_bytes The amount of GPU memory that will be temporarily required
297 * during decompression. The value is returned on the host side.
298 * @param[in] max_total_uncompressed_bytes The total decompressed size of all the chunks.
299 *
300 * @return nvcompSuccess if successful, and an error code otherwise.
301 */
302NVCOMP_EXPORT
303nvcompStatus_t nvcompBatchedZstdDecompressGetTempSizeAsync(
304    size_t num_chunks,
305    size_t max_uncompressed_chunk_bytes,
306    nvcompBatchedZstdDecompressOpts_t decompress_opts,
307    size_t* temp_bytes,
308    size_t max_total_uncompressed_bytes);
309
310/**
311 * @brief Get the amount of temporary memory required on the GPU for decompression
312 * synchronously.
313 *
314 * @note This function may perform operations on the stream; if so, it will synchronize it internally.
315 * Therefore, it does not require additional synchronization after it returns,
316 * and the result can be used immediately.
317 *
318 * @param[in] device_compressed_chunk_ptrs Array with size \p num_chunks of pointers
319 * in device-accessible memory to device-accessible compressed buffers.
320 * Each chunk must be aligned to the value in the `input` member of the
321 * \ref nvcompAlignmentRequirements_t object output by
322 * `nvcompBatchedZstdDecompressGetRequiredAlignments`.
323 * @param[in] device_compressed_chunk_bytes Array with size \p num_chunks of sizes of
324 * the compressed buffers in bytes. The sizes should reside in device-accessible memory.
325 * @param[in] num_chunks Number of chunks of data to be decompressed.
326 * @param[in] max_uncompressed_chunk_bytes The size of the largest chunk in bytes
327 * when uncompressed.
328 * @param[out] temp_bytes The amount of GPU memory that will be temporarily required
329 * during decompression. The value is returned on the host side.
330 * @param[in] max_total_uncompressed_bytes  The total decompressed size of all the chunks.
331 * @param[in] decompress_opts Decompression options.
332 * @param[out] device_statuses Array with size \p num_chunks of statuses in
333 * device-accessible memory. This argument needs to be preallocated. For each
334 * chunk, if the data can be parsed successfully, the status will be set to
335 * `nvcompSuccess`, and an error code otherwise.
336 * Can be NULL if desired, in which case error status is not reported.
337 * @param[in] stream The CUDA stream to operate on.
338 *
339 * @return nvcompSuccess if successful, and an error code otherwise.
340 */
341NVCOMP_EXPORT
342nvcompStatus_t nvcompBatchedZstdDecompressGetTempSizeSync(
343    const void* const* const device_compressed_chunk_ptrs,
344    const size_t* const device_compressed_chunk_bytes,
345    size_t num_chunks,
346    size_t max_uncompressed_chunk_bytes,
347    size_t* temp_bytes,
348    size_t max_total_uncompressed_bytes,
349    nvcompBatchedZstdDecompressOpts_t decompress_opts,
350    nvcompStatus_t* device_statuses,
351    cudaStream_t stream);
352
353/**
354 * @brief Asynchronously compute the number of bytes of uncompressed data for
355 * each compressed chunk.
356 *
357 * @warning Violating any of the conditions listed in the parameter descriptions
358 * below may result in undefined behaviour.
359 *
360 * @note This function performs operations on the stream, and does not synchronize it,
361 * therefore, it requires synchronization or stream-ordered operations to use its results.
362 *
363 * @param[in] device_compressed_chunk_ptrs Array with size \p num_chunks of
364 * pointers in device-accessible memory to compressed buffers.
365 * Each chunk must be aligned to the value in the `input` member of the
366 * \ref nvcompAlignmentRequirements_t object output by
367 * `nvcompBatchedZstdDecompressGetRequiredAlignments`.
368 * @param[in] device_compressed_chunk_bytes Array with size \p num_chunks of sizes
369 * of the compressed buffers in bytes. The sizes should reside in device-accessible memory.
370 * @param[out] device_uncompressed_chunk_bytes Array with size \p num_chunks
371 * to be filled with the sizes, in bytes, of each uncompressed data chunk.
372 * This argument needs to be preallocated in device-accessible memory.
373 * @param[in] num_chunks Number of data chunks to compute sizes of.
374 * @param[in] stream The CUDA stream to operate on.
375 *
376 * @return nvcompSuccess if successful, and an error code otherwise.
377 */
378NVCOMP_EXPORT
379nvcompStatus_t nvcompBatchedZstdGetDecompressSizeAsync(
380    const void* const* device_compressed_chunk_ptrs,
381    const size_t* device_compressed_chunk_bytes,
382    size_t* device_uncompressed_chunk_bytes,
383    size_t num_chunks,
384    cudaStream_t stream);
385
386/**
387 * @brief Perform batched asynchronous decompression.
388 *
389 * @warning Violating any of the conditions listed in the parameter descriptions
390 * below may result in undefined behaviour.
391 *
392 * @warning Providing a corrupt buffer for decompression will result in undefined
393 * behavior.
394 *
395 * @note This function performs operations on the stream, and does not synchronize it,
396 * therefore, it requires synchronization or stream-ordered operations to use its results.
397 *
398 * @param[in] device_compressed_chunk_ptrs Array with size \p num_chunks of pointers
399 * in device-accessible memory to device-accessible compressed buffers.
400 * Each chunk must be aligned to the value in the `input` member of the
401 * \ref nvcompAlignmentRequirements_t object output by
402 * `nvcompBatchedZstdDecompressGetRequiredAlignments`.
403 * @param[in] device_compressed_chunk_bytes Array with size \p num_chunks of sizes of
404 * the compressed buffers in bytes. The sizes should reside in device-accessible memory.
405 * @param[in] device_uncompressed_buffer_bytes Array with size \p num_chunks of sizes,
406 * in bytes, of the output buffers to be filled with uncompressed data for each chunk.
407 * The sizes should reside in device-accessible memory. If a
408 * size is not large enough to hold all decompressed data, the decompressor
409 * will set the status in \p device_statuses corresponding to the
410 * overflow chunk to `nvcompErrorCannotDecompress`.
411 * @param[out] device_uncompressed_chunk_bytes Array with size \p num_chunks to
412 * be filled with the actual number of bytes decompressed for every chunk.
413 * @param[in] num_chunks Number of chunks of data to decompress.
414 * @param[in] device_temp_ptr The temporary GPU space, could be NULL in case temporary space is not needed.
415  * Must be aligned to the value in the `temp` member of the
416 * \ref nvcompAlignmentRequirements_t object output by
417 * `nvcompBatchedZstdDecompressGetRequiredAlignments`.
418 * @param[in] temp_bytes The size of the temporary GPU space.
419 * @param[out] device_uncompressed_chunk_ptrs Array with size \p num_chunks of
420 * pointers in device-accessible memory to decompressed data. Each uncompressed
421 * buffer needs to be preallocated in device-accessible memory, have the size
422 * specified by the corresponding entry in \p device_uncompressed_buffer_bytes,
423 * and be aligned to the value in the `output` member of the
424 * \ref nvcompAlignmentRequirements_t object output by
425 * `nvcompBatchedZstdDecompressGetRequiredAlignments`.
426 * @param[in] decompress_opts Decompression options.
427 * @param[out] device_statuses Array with size \p num_chunks of statuses in
428 * device-accessible memory. This argument needs to be preallocated. For each
429 * chunk, if the decompression is successful, the status will be set to
430 * `nvcompSuccess`. Passing corrupt, invalid, or insufficient data leads to
431 * undefined behavior or out-of-bound errors. Error reporting cannot be guaranteed
432 * in this scenario as only a limited validation is performed to maintain performance.
433 * Can be NULL if desired, in which case error status is not reported.
434 * @param[in] stream The CUDA stream to operate on.
435 *
436 * @return nvcompSuccess if successfully launched, and an error code otherwise.
437 */
438NVCOMP_EXPORT
439nvcompStatus_t nvcompBatchedZstdDecompressAsync(
440    const void* const* device_compressed_chunk_ptrs,
441    const size_t* device_compressed_chunk_bytes,
442    const size_t* device_uncompressed_buffer_bytes,
443    size_t* device_uncompressed_chunk_bytes,
444    size_t num_chunks,
445    void* const device_temp_ptr,
446    size_t temp_bytes,
447    void* const* device_uncompressed_chunk_ptrs,
448    nvcompBatchedZstdDecompressOpts_t decompress_opts,
449    nvcompStatus_t* device_statuses,
450    cudaStream_t stream);
451
452#ifdef __cplusplus
453}
454#endif
455
456#endif // NVCOMP_ZSTD_H
457 
codekingpro/portable-devtools · Team Ai