Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
snappy.h455 linesDownload Raw Back to nvcomp
1/*
2 * SPDX-FileCopyrightText: Copyright (c) 2017-2025 NVIDIA CORPORATION & AFFILIATES.
3 * All rights reserved. SPDX-License-Identifier: LicenseRef-NvidiaProprietary
4 *
5 * NVIDIA CORPORATION, its affiliates and licensors retain all intellectual
6 * property and proprietary rights in and to this material, related
7 * documentation and any modifications thereto. Any use, reproduction,
8 * disclosure or distribution of this material and related documentation
9 * without an express license agreement from NVIDIA CORPORATION or
10 * its affiliates is strictly prohibited.
11*/
12
13#ifndef NVCOMP_SNAPPY_H
14#define NVCOMP_SNAPPY_H
15
16#include "nvcomp.h"
17
18#ifdef __cplusplus
19extern "C" {
20#endif
21
22/**
23 * @brief Snappy compression options for the low-level API
24 */
25typedef struct
26{
27  /**
28   * @brief These bytes are unused and must be zeroed. This ensures
29   *        compatibility if additional fields are added in the future.
30   */
31  char reserved[64];
32} nvcompBatchedSnappyCompressOpts_t;
33
34/**
35 * @brief Snappy decompression options for the low-level API
36 */
37typedef struct {
38  /**
39   * @brief Decompression backend to use.
40   */
41  nvcompDecompressBackend_t backend;
42  /**
43   * @brief Whether to sort chunks before hardware decompression for better load balancing.
44   *        Only used when the backend is the hardware decompression engine.
45   */
46  int sort_before_hw_decompress;
47  /**
48   * @brief These bytes are unused and must be zeroed. This ensures
49   *        compatibility if additional fields are added in the future.
50   */
51  char reserved[56];
52} nvcompBatchedSnappyDecompressOpts_t;
53
54/**
55 * @brief Default Snappy compression options
56 */
57static const nvcompBatchedSnappyCompressOpts_t nvcompBatchedSnappyCompressDefaultOpts = {{0}};
58
59/**
60 * @brief Default Snappy decompression options
61 */
62static const nvcompBatchedSnappyDecompressOpts_t nvcompBatchedSnappyDecompressDefaultOpts =
63    {NVCOMP_DECOMPRESS_BACKEND_DEFAULT, 0 /* sort_before_hw_decompress */, {0}};
64
65/**
66 * @brief The maximum supported uncompressed chunk size in bytes for the Snappy compressor.
67 */
68static const size_t nvcompSnappyCompressionMaxAllowedChunkSize = 1 << 24;
69
70/**
71 * @brief The maximum supported compressed and decompressed chunk size in bytes for the Snappy decompressor.
72 * @note To maximize decompression performance, users are encouraged to compress in smaller chunks, for example 64KiB.
73 * @note The hardware Decompression Engine (DE) may have different maximum chunk size limits.
74 * This can be queried through `cuDeviceGetAttribute` with `CU_DEVICE_ATTRIBUTE_MEM_DECOMPRESS_MAXIMUM_LENGTH`.
75 */
76static const size_t nvcompSnappyDecompressionMaxAllowedChunkSize = (1ull<<31)-1;
77
78/**
79 * @brief The most restrictive of the minimum alignment requirements for void-type CUDA memory buffers
80 * used for input, output, or temporary memory, passed to compression functions.
81 *
82 * @note In all cases, typed memory buffers must still be aligned to their type's size,
83 * e.g., 4 bytes for `int`.
84 */
85static const size_t nvcompSnappyRequiredCompressionAlignment = 1;
86
87/**
88 * @brief Get the minimum buffer alignment requirements for compression.
89 *
90 * @note Providing buffers with alignments above the minimum requirements
91 * (e.g., 16- or 32-byte alignment) may help improve performance.
92 *
93 * @param[in] compress_opts Compression options.
94 * @param[out] alignment_requirements The minimum buffer alignment requirements
95 * for compression.
96 *
97 * @return nvcompSuccess if successful, and an error code otherwise.
98 */
99NVCOMP_EXPORT
100nvcompStatus_t nvcompBatchedSnappyCompressGetRequiredAlignments(
101    nvcompBatchedSnappyCompressOpts_t compress_opts,
102    nvcompAlignmentRequirements_t* alignment_requirements);
103
104/**
105 * @brief Get the amount of temporary memory required on the GPU for compression
106 * asynchronously.
107 *
108 * @note This function does not interact with the device, its result can be used immediately.
109 *
110 * @param[in] num_chunks The number of chunks of memory in the batch.
111 * @param[in] max_uncompressed_chunk_bytes The maximum size of a chunk in the
112 * batch.
113 * @param[in] compress_opts Compression options.
114 * @param[out] temp_bytes The amount of GPU memory that will be temporarily
115 * required during compression. The value is returned on the host side.
116 * @param[in] max_total_uncompressed_bytes Upper bound on the total uncompressed
117 * size of all chunks
118 *
119 * @return nvcompSuccess if successful, and an error code otherwise.
120 */
121NVCOMP_EXPORT
122nvcompStatus_t nvcompBatchedSnappyCompressGetTempSizeAsync(
123    size_t num_chunks,
124    size_t max_uncompressed_chunk_bytes,
125    nvcompBatchedSnappyCompressOpts_t compress_opts,
126    size_t* temp_bytes,
127    size_t max_total_uncompressed_bytes);
128
129/**
130 * @brief Get the amount of temporary memory required on the GPU for compression.
131 * synchronously.
132 *
133 * @note This function may perform operations on the stream; if so, it will synchronize it internally.
134 * Therefore, it does not require additional synchronization after it returns,
135 * and the result can be used immediately.
136 *
137 * @param[in] device_uncompressed_chunk_ptrs Array with size \p num_chunks of pointers
138 * to the uncompressed data chunks. Both the pointers and the uncompressed data
139 * should reside in device-accessible memory.
140 * Each chunk must be aligned to the value in the `input` member of the
141 * \ref nvcompAlignmentRequirements_t object output by
142 * `nvcompBatchedSnappyCompressGetRequiredAlignments` when called with the same
143 * \p compress_opts.
144 * @param[in] device_uncompressed_chunk_bytes Array with size \p num_chunks of
145 * sizes of the uncompressed chunks in bytes.
146 * The sizes should reside in device-accessible memory.
147 * @param[in] num_chunks The number of chunks of memory in the batch.
148 * @param[in] max_uncompressed_chunk_bytes The maximum size of a chunk in the
149 * batch.
150 * @param[in] compress_opts Compression options.
151 * @param[out] temp_bytes The amount of GPU memory that will be temporarily
152 * required during compression. The value is returned on the host side.
153 * @param[in] max_total_uncompressed_bytes Upper bound on the total uncompressed
154 * size of all chunks
155 * @param[in] stream The CUDA stream to operate on.
156 *
157 * @return nvcompSuccess if successful, and an error code otherwise.
158 */
159NVCOMP_EXPORT
160nvcompStatus_t nvcompBatchedSnappyCompressGetTempSizeSync(
161    const void* const* const device_uncompressed_chunk_ptrs,
162    const size_t* const device_uncompressed_chunk_bytes,
163    size_t num_chunks,
164    size_t max_uncompressed_chunk_bytes,
165    nvcompBatchedSnappyCompressOpts_t compress_opts,
166    size_t* temp_bytes,
167    size_t max_total_uncompressed_bytes,
168    cudaStream_t stream);
169
170/**
171 * @brief Get the maximum size that a chunk of size at most max_uncompressed_chunk_bytes
172 * could compress to. That is, the minimum amount of output memory required to be given
173 * \ref nvcompBatchedSnappyCompressAsync for each chunk.
174 *
175 * @param[in] max_uncompressed_chunk_bytes The maximum size of a chunk before compression.
176 * @param[in] compress_opts Snappy compression options.
177 * @param[out] max_compressed_chunk_bytes The maximum possible compressed size of the chunk.
178 *
179 * @return nvcompSuccess if successful, and an error code otherwise.
180 */
181NVCOMP_EXPORT
182nvcompStatus_t nvcompBatchedSnappyCompressGetMaxOutputChunkSize(
183    size_t max_uncompressed_chunk_bytes,
184    nvcompBatchedSnappyCompressOpts_t compress_opts,
185    size_t* max_compressed_chunk_bytes);
186
187/**
188 * @brief Perform batched asynchronous compression.
189 *
190 * @warning Violating any of the conditions listed in the parameter descriptions
191 * below may result in undefined behaviour.
192 *
193 * @note This function performs operations on the stream, and does not synchronize it,
194 * therefore, it requires synchronization or stream-ordered operations to use its results.
195 *
196 * @param[in] device_uncompressed_chunk_ptrs Array with size \p num_chunks of pointers
197 * to the uncompressed data chunks. Both the pointers and the uncompressed data
198 * should reside in device-accessible memory.
199 * Each chunk must be aligned to the value in the `input` member of the
200 * \ref nvcompAlignmentRequirements_t object output by
201 * `nvcompBatchedSnappyCompressGetRequiredAlignments` when called with the same
202 * \p compress_opts.
203 * @param[in] device_uncompressed_chunk_bytes Array with size \p num_chunks of
204 * sizes of the uncompressed chunks in bytes.
205 * The sizes should reside in device-accessible memory.
206 * @param[in] max_uncompressed_chunk_bytes The size of the largest uncompressed chunk.
207 * This parameter is currently unused. Set it to either the actual value
208 * or zero.
209 * @param[in] num_chunks Number of chunks of data to compress.
210 * @param[in] device_temp_ptr The temporary GPU workspace, could be NULL in case
211 * temporary memory is not needed.
212 * Must be aligned to the value in the `temp` member of the
213 * \ref nvcompAlignmentRequirements_t object output by
214 * `nvcompBatchedSnappyCompressGetRequiredAlignments` when called with the same
215 * \p compress_opts.
216 * @param[in] temp_bytes The size of the temporary GPU memory pointed to by
217 * `device_temp_ptr`.
218 * @param[out] device_compressed_chunk_ptrs Array with size \p num_chunks of pointers
219 * to the output compressed buffers. Both the pointers and the compressed
220 * buffers should reside in device-accessible memory. Each compressed buffer
221 * should be preallocated with the size given by
222 * `nvcompBatchedSnappyCompressGetMaxOutputChunkSize`.
223 * Each compressed buffer must be aligned to the value in the `output` member of the
224 * \ref nvcompAlignmentRequirements_t object output by
225 * `nvcompBatchedSnappyCompressGetRequiredAlignments` when called with the same
226 * \p compress_opts.
227 * @param[out] device_compressed_chunk_bytes Array with size \p num_chunks,
228 * to be filled with the compressed sizes of each chunk.
229 * The buffer should be preallocated in device-accessible memory.
230 * @param[in] compress_opts Snappy compression options.
231 * @param[out] device_statuses Array with size \p num_chunks of statuses in
232 * device-accessible memory. This argument needs to be preallocated. For each
233 * chunk, if the compression is successful, the status will be set to
234 * `nvcompSuccess`, and an error code otherwise.
235 * Can be NULL if desired, in which case error status is not reported.
236 * @param[in] stream The CUDA stream to operate on.
237 *
238 * @return nvcompSuccess if successfully launched, and an error code otherwise.
239 */
240NVCOMP_EXPORT
241nvcompStatus_t nvcompBatchedSnappyCompressAsync(
242    const void* const* device_uncompressed_chunk_ptrs,
243    const size_t* device_uncompressed_chunk_bytes,
244    size_t max_uncompressed_chunk_bytes,
245    size_t num_chunks,
246    void* device_temp_ptr,
247    size_t temp_bytes,
248    void* const* device_compressed_chunk_ptrs,
249    size_t* device_compressed_chunk_bytes,
250    nvcompBatchedSnappyCompressOpts_t compress_opts,
251    nvcompStatus_t* device_statuses,
252    cudaStream_t stream);
253
254/**
255 * @brief The most restrictive of the minimum alignment requirements for void-type CUDA memory buffers
256 * used for input, output, or temporary memory, passed to decompression functions.
257 *
258 * @note In all cases, typed memory buffers must still be aligned to their type's size,
259 * e.g., 4 bytes for `int`.
260 */
261static const size_t nvcompSnappyRequiredDecompressionAlignment = 1;
262
263/**
264 * @brief Get the minimum buffer alignment requirements for decompression.
265 *
266 * @note Providing buffers with alignments above the minimum requirements
267 * (e.g., 16- or 32-byte alignment) may help improve performance.
268 *
269 * @param[in] decompress_opts Decompression options.
270 * @param[out] alignment_requirements The minimum buffer alignment requirements
271 * for decompression.
272 *
273 * @return nvcompSuccess if successful, and an error code otherwise.
274 */
275NVCOMP_EXPORT
276nvcompStatus_t nvcompBatchedSnappyDecompressGetRequiredAlignments(
277    nvcompBatchedSnappyDecompressOpts_t decompress_opts,
278    nvcompAlignmentRequirements_t* alignment_requirements);
279
280/**
281 * @brief Get the amount of temporary memory required on the GPU for decompression
282 * asynchronously.
283 *
284 * @note This function does not interact with the device, its result can be used immediately.
285 *
286 * @param[in] num_chunks Number of chunks of data to be decompressed.
287 * @param[in] max_uncompressed_chunk_bytes The size of the largest chunk in bytes
288 * when uncompressed.
289 * @param[in] decompress_opts Decompression options.
290 * @param[out] temp_bytes The amount of GPU memory that will be temporarily required
291 * during decompression. The value is returned on the host side.
292 * @param[in] max_total_uncompressed_bytes The total decompressed size of all the chunks.
293 *
294 * @return nvcompSuccess if successful, and an error code otherwise.
295 */
296NVCOMP_EXPORT
297nvcompStatus_t nvcompBatchedSnappyDecompressGetTempSizeAsync(
298    size_t num_chunks,
299    size_t max_uncompressed_chunk_bytes,
300    nvcompBatchedSnappyDecompressOpts_t decompress_opts,
301    size_t* temp_bytes,
302    size_t max_total_uncompressed_bytes);
303
304/**
305 * @brief Get the amount of temporary memory required on the GPU for decompression
306 * synchronously.
307 *
308 * @note This function may perform operations on the stream; if so, it will synchronize it internally.
309 * Therefore, it does not require additional synchronization after it returns,
310 * and the result can be used immediately.
311 *
312 * @param[in] device_compressed_chunk_ptrs Array with size \p num_chunks of pointers
313 * in device-accessible memory to device-accessible compressed buffers.
314 * Each chunk must be aligned to the value in the `input` member of the
315 * \ref nvcompAlignmentRequirements_t object output by
316 * `nvcompBatchedSnappyDecompressGetRequiredAlignments`.
317 * @param[in] device_compressed_chunk_bytes Array with size \p num_chunks of sizes of
318 * the compressed buffers in bytes. The sizes should reside in device-accessible memory.
319 * @param[in] num_chunks Number of chunks of data to be decompressed.
320 * @param[in] max_uncompressed_chunk_bytes The size of the largest chunk in bytes
321 * when uncompressed.
322 * @param[out] temp_bytes The amount of GPU memory that will be temporarily required
323 * during decompression. The value is returned on the host side.
324 * @param[in] max_total_uncompressed_bytes  The total decompressed size of all the chunks.
325 * @param[in] decompress_opts Decompression options.
326 * @param[out] device_statuses Array with size \p num_chunks of statuses in
327 * device-accessible memory. This argument needs to be preallocated. For each
328 * chunk, if the data can be parsed successfully, the status will be set to
329 * `nvcompSuccess`, and an error code otherwise.
330 * Can be NULL if desired, in which case error status is not reported.
331 * @param[in] stream The CUDA stream to operate on.
332 *
333 * @return nvcompSuccess if successful, and an error code otherwise.
334 */
335NVCOMP_EXPORT
336nvcompStatus_t nvcompBatchedSnappyDecompressGetTempSizeSync(
337    const void* const* const device_compressed_chunk_ptrs,
338    const size_t* const device_compressed_chunk_bytes,
339    size_t num_chunks,
340    size_t max_uncompressed_chunk_bytes,
341    size_t* temp_bytes,
342    size_t max_total_uncompressed_bytes,
343    nvcompBatchedSnappyDecompressOpts_t decompress_opts,
344    nvcompStatus_t* device_statuses,
345    cudaStream_t stream);
346
347/**
348 * @brief Asynchronously compute the number of bytes of uncompressed data for
349 * each compressed chunk.
350 *
351 * @warning Violating any of the conditions listed in the parameter descriptions
352 * below may result in undefined behaviour.
353 *
354 * @note This function performs operations on the stream, and does not synchronize it,
355 * therefore, it requires synchronization or stream-ordered operations to use its results.
356 *
357 * @param[in] device_compressed_chunk_ptrs Array with size \p num_chunks of
358 * pointers in device-accessible memory to compressed buffers.
359 * Each chunk must be aligned to the value in the `input` member of the
360 * \ref nvcompAlignmentRequirements_t object output by
361 * `nvcompBatchedSnappyDecompressGetRequiredAlignments`.
362 * @param[in] device_compressed_chunk_bytes Array with size \p num_chunks of sizes
363 * of the compressed buffers in bytes. The sizes should reside in device-accessible memory.
364 * @param[out] device_uncompressed_chunk_bytes Array with size \p num_chunks
365 * to be filled with the sizes, in bytes, of each uncompressed data chunk.
366 * This argument needs to be preallocated in device-accessible memory.
367 * @param[in] num_chunks Number of data chunks to compute sizes of.
368 * @param[in] stream The CUDA stream to operate on.
369 *
370 * @return nvcompSuccess if successful, and an error code otherwise.
371 */
372NVCOMP_EXPORT
373nvcompStatus_t nvcompBatchedSnappyGetDecompressSizeAsync(
374    const void* const* device_compressed_chunk_ptrs,
375    const size_t* device_compressed_chunk_bytes,
376    size_t* device_uncompressed_chunk_bytes,
377    size_t num_chunks,
378    cudaStream_t stream);
379
380/**
381 * @brief Perform batched asynchronous decompression.
382 *
383 * @warning Violating any of the conditions listed in the parameter descriptions
384 * below may result in undefined behaviour.
385 * 
386 * @warning Providing a corrupt buffer for decompression will result in undefined 
387 * behavior irrespective of the decompression backend used.
388 *
389 * @note This function performs operations on the stream, and does not synchronize it,
390 * therefore, it requires synchronization or stream-ordered operations to use its results.
391 *
392 * @param[in] device_compressed_chunk_ptrs Array with size \p num_chunks of pointers
393 * in device-accessible memory to device-accessible compressed buffers.
394 * Each chunk must be aligned to the value in the `input` member of the
395 * \ref nvcompAlignmentRequirements_t object output by
396 * `nvcompBatchedSnappyDecompressGetRequiredAlignments`.
397 * @param[in] device_compressed_chunk_bytes Array with size \p num_chunks of sizes of
398 * the compressed buffers in bytes. The sizes should reside in device-accessible memory.
399 * @param[in] device_uncompressed_buffer_bytes Array with size \p num_chunks of sizes,
400 * in bytes, of the output buffers to be filled with uncompressed data for each chunk.
401 * The sizes should reside in device-accessible memory. If a
402 * size is not large enough to hold all decompressed data, the decompressor
403 * will set the status in \p device_statuses corresponding to the
404 * overflow chunk to `nvcompErrorCannotDecompress`.
405 * @param[out] device_uncompressed_chunk_bytes Array with size \p num_chunks to
406 * be filled with the actual number of bytes decompressed for every chunk.
407 * This argument needs to be preallocated.
408 * When `NVCOMP_DECOMPRESS_BACKEND_HARDWARE` is specified in \p decompress_opts.backend,
409 * this parameter is required. For `NVCOMP_DECOMPRESS_BACKEND_CUDA`, it is optional
410 * and may be set to NULL if reporting the actual sizes is not necessary.
411 * @param[in] num_chunks Number of chunks of data to decompress.
412 * @param[in] device_temp_ptr The temporary GPU space, could be NULL in case temporary space is not needed.
413 * Must be aligned to the value in the `temp` member of the
414 * \ref nvcompAlignmentRequirements_t object output by
415 * `nvcompBatchedSnappyDecompressGetRequiredAlignments`.
416 * @param[in] temp_bytes The size of the temporary GPU space.
417 * @param[out] device_uncompressed_chunk_ptrs Array with size \p num_chunks of
418 * pointers in device-accessible memory to decompressed data. Each uncompressed
419 * buffer needs to be preallocated in device-accessible memory, have the size
420 * specified by the corresponding entry in \p device_uncompressed_buffer_bytes,
421 * and be aligned to the value in the `output` member of the
422 * \ref nvcompAlignmentRequirements_t object output by
423 * `nvcompBatchedSnappyDecompressGetRequiredAlignments`.
424 * @param[in] decompress_opts Decompression options.
425 * @param[out] device_statuses Array with size \p num_chunks of statuses in
426 * device-accessible memory. This argument needs to be preallocated. For each
427 * chunk, if the decompression is successful, the status will be set to
428 * `nvcompSuccess`. Passing corrupt, invalid, or insufficient data leads to
429 * undefined behavior or out-of-bound errors. Error reporting cannot be guaranteed
430 * in this scenario as only a limited validation is performed to maintain performance.
431 * Can be NULL if desired, in which case error status is not reported.
432 * @param[in] stream The CUDA stream to operate on.
433 *
434 * @return nvcompSuccess if successfully launched, and an error code otherwise.
435 */
436NVCOMP_EXPORT
437nvcompStatus_t nvcompBatchedSnappyDecompressAsync(
438    const void* const* device_compressed_chunk_ptrs,
439    const size_t* device_compressed_chunk_bytes,
440    const size_t* device_uncompressed_buffer_bytes,
441    size_t* device_uncompressed_chunk_bytes,
442    size_t num_chunks,
443    void* const device_temp_ptr,
444    size_t temp_bytes,
445    void* const* device_uncompressed_chunk_ptrs,
446    nvcompBatchedSnappyDecompressOpts_t decompress_opts,
447    nvcompStatus_t* device_statuses,
448    cudaStream_t stream);
449
450#ifdef __cplusplus
451}
452#endif
453
454#endif // NVCOMP_SNAPPY_H
455 
codekingpro/portable-devtools · Team Ai