|
FZGPUModules 2.0
GPU-accelerated modular compression pipelines
|
Enumerations | |
| enum class | StageType : uint16_t { } |
| Stage type identifiers written into the FZM header. More... | |
| enum class | DataType : uint8_t { } |
| Element data type identifiers used in buffer and stage descriptors. More... | |
| enum class | LogLevel : int { TRACE = 0 , DEBUG = 1 , INFO = 2 , WARN = 3 , SILENT = 255 } |
| enum class | MemoryStrategy { MINIMAL , PREALLOCATE } |
| enum class | HuffmanEncodeMode { Coarse , Fine } |
| enum class | HuffmanBookSource { PerBlock , Fixed , Adaptive } |
| enum class | HuffmanBookModel { Gaussian , Laplace , GeneralizedNormal , Uniform } |
| enum class | ErrorBoundMode : uint8_t { ABS = 0 , REL = 1 , NOA = 2 , PREL = 3 } |
| enum class | FusionMode : uint8_t { NEGABINARY = 0 , ZIGZAG = 1 } |
| enum class | ADMDtype : uint8_t |
Functions | |
| const char * | getBackendErrorString (error_t err) |
| size_t | getPoolUsedMemCurrent (mempool_t pool) |
| size_t | getPoolUsedMemHigh (mempool_t pool) |
| constexpr uint8_t | fzmVersionMajor (uint16_t v) |
| constexpr uint8_t | fzmVersionMinor (uint16_t v) |
| size_t | getDataTypeSize (DataType type) |
| std::string | dataTypeToString (DataType type) |
| std::string | stageTypeToString (StageType type) |
| std::vector< std::string > | registeredStageTypes () |
Every stage type string loadConfig() accepts, in registry order. | |
| std::vector< StageFingerprintInfo > | stageFingerprints () |
| Per-stage source fingerprints for THIS build. | |
| template<typename T > | |
| ReconstructionStats | calculateStatistics (const T *d_original, const T *d_decompressed, size_t n) |
| Stage * | createStage (StageType type, const uint8_t *config, size_t config_size) |
| template<typename T > | |
| constexpr size_t | rleValuesOffset () |
| template<typename T > | |
| constexpr size_t | rleChunkedValuesOffset (size_t num_chunks) |
| template<typename T > | |
| void | launchAdaptiveLorenzoForward (const T *d_input, T *d_residuals, uint8_t *d_modes, T *d_means, size_t n, uint32_t tile_size, bool enable_order2, bool enable_centering, fz::stream_t stream) |
| Forward: select the best variant per tile and emit its residuals. | |
| template<typename T > | |
| void | launchAdaptiveLorenzoInverse (const T *d_residuals, const uint8_t *d_modes, const T *d_means, T *d_output, size_t n, uint32_t tile_size, fz::stream_t stream) |
| Inverse: replay each tile's recorded variant. | |
| ErrorBoundMode | resolveApproxRelMode (ErrorBoundMode mode, const char *stage_name) |
| template<typename TInput , typename TCode > | |
| void | launchLorenzoKernel (const TInput *d_input, size_t n, TInput ebx2_r, TCode quant_radius, TCode *d_codes, TInput *d_outlier_errors, uint32_t *d_outlier_indices, uint32_t *d_outlier_count, size_t max_outliers, int grid_size, bool zigzag_codes, fz::stream_t stream, TInput *d_means=nullptr) |
| template<typename TInput , typename TCode > | |
| void | launchLorenzoInverseKernel (const TCode *d_codes, const TInput *d_outlier_errors, const uint32_t *d_outlier_indices, uint32_t outlier_n, size_t n, TInput ebx2, TCode quant_radius, TInput *d_output, bool zigzag_codes, fz::stream_t stream, MemoryPool *pool, const TInput *d_means=nullptr) |
| template<typename TInput , typename TCode > | |
| void | launchLorenzoKernel2D (const TInput *d_input, size_t nx, size_t ny, TInput ebx2_r, TCode quant_radius, TCode *d_codes, TInput *d_outlier_errors, uint32_t *d_outlier_indices, uint32_t *d_outlier_count, size_t max_outliers, bool zigzag_codes, fz::stream_t stream) |
2-D forward Lorenzo kernel launcher. nx is the fast (x) dimension. | |
| template<typename TInput , typename TCode > | |
| void | launchLorenzoInverseKernel2D (const TCode *d_codes, const TInput *d_outlier_errors, const uint32_t *d_outlier_indices, uint32_t outlier_n, size_t nx, size_t ny, TInput ebx2, TCode quant_radius, TInput *d_output, bool zigzag_codes, fz::stream_t stream, MemoryPool *pool) |
| 2-D inverse Lorenzo kernel launcher. | |
| template<typename TInput , typename TCode > | |
| void | launchLorenzoKernel3D (const TInput *d_input, size_t nx, size_t ny, size_t nz, TInput ebx2_r, TCode quant_radius, TCode *d_codes, TInput *d_outlier_errors, uint32_t *d_outlier_indices, uint32_t *d_outlier_count, size_t max_outliers, bool zigzag_codes, fz::stream_t stream) |
| 3-D forward Lorenzo kernel launcher. | |
| template<typename TInput , typename TCode > | |
| void | launchLorenzoInverseKernel3D (const TCode *d_codes, const TInput *d_outlier_errors, const uint32_t *d_outlier_indices, uint32_t outlier_n, size_t nx, size_t ny, size_t nz, TInput ebx2, TCode quant_radius, TInput *d_output, bool zigzag_codes, fz::stream_t stream, MemoryPool *pool) |
| 3-D inverse Lorenzo kernel launcher. | |
| template<typename T > | |
| void | launchLorenzoDeltaCentered1D (const T *d_input, T *d_output, T *d_means, size_t n, fz::stream_t stream, unsigned block_threads) |
| template<typename T > | |
| void | launchLorenzoSegmentedScan (const T *d_input, const T *d_means, T *d_output, size_t n, fz::stream_t stream, unsigned block_threads, int passes) |
| template<typename T > | |
| void | launchLorenzo2Delta1D (const T *d_input, T *d_output, T *d_means, size_t n, fz::stream_t stream, unsigned block_threads) |
Variables | |
| constexpr bool | kBackendSupportsGraphCapture = true |
| constexpr uint32_t | FZM_MAGIC = 0x464D5A32 |
| constexpr uint8_t | FZM_VERSION_MAJOR = 3 |
| constexpr size_t | FZM_LEGACY_HEADER_CORE_SIZE = 72 |
| constexpr uint16_t | FZM_FLAG_HAS_DATA_CHECKSUM = 0x0001u |
| data_checksum field is valid | |
| constexpr uint16_t | FZM_FLAG_HAS_HEADER_CHECKSUM = 0x0002u |
| header_checksum field is valid | |
| constexpr size_t | FZM_MAX_BUFFERS = 32 |
| Maximum pipeline output buffers per file. | |
| constexpr size_t | FZM_MAX_NAME_LEN = 64 |
| Maximum output port name length (bytes, null-terminated) | |
| constexpr size_t | FZM_STAGE_CONFIG_SIZE = 128 |
| Per-stage serialized config slot (bytes) | |
| constexpr size_t | FZM_MAX_SOURCES = 4 |
| Maximum source stages per pipeline. | |
Copyright (c) Meta Platforms, Inc. and affiliates.
This source code is licensed under the MIT license found in the LICENSE file in the root directory of this source tree.
Adapted for FZGPUModules: namespace renamed from multibyte_ans to fz::ans.
Copyright (c) Meta Platforms, Inc. and affiliates.
This source code is licensed under the MIT license found in the LICENSE file in the root directory of this source tree.
Adapted for FZGPUModules: namespace renamed from multibyte_ans to fz::ans; stripped ansDecode() host function (called directly from ans_stage.cu instead).
Copyright (c) Meta Platforms, Inc. and affiliates.
This source code is licensed under the MIT license found in the LICENSE file in the root directory of this source tree.
Adapted for FZGPUModules: namespace renamed from multibyte_ans to fz::ans; stripped ansEncode() host function (called directly from ans_stage.cu instead).
Copyright (c) Meta Platforms, Inc. and affiliates.
This source code is licensed under the MIT license found in the LICENSE file in the root directory of this source tree.
Adapted for FZGPUModules: namespace renamed from multibyte_ans to fz::ans; stripped histogramSingle, histogramBatch, ansHistogramBatch (replaced by fz::module::GPU_histogram_generic in ans_stage.cu).
|
strong |
Stage type identifiers written into the FZM header.
Each concrete Stage subclass reports one of these values via getStageTypeId(). Used by StageFactory::createStage() to reconstruct the pipeline during decompression.
|
strong |
Element data type identifiers used in buffer and stage descriptors.
Returned by Stage::getOutputDataType() and Stage::getInputDataType(). UNKNOWN is returned by byte-transparent stages (Bitshuffle, RZE) to opt out of Pipeline::finalize() type-compatibility checking.
| Enumerator | |
|---|---|
| UNKNOWN | Byte-transparent stages: skip type checking at finalize() |
|
strong |
| Enumerator | |
|---|---|
| TRACE | Per-stage execute(), per-chunk details — very verbose. |
| DEBUG | Pipeline construction, buffer allocation, data stats. |
| INFO | High-level milestones: finalize, compress, decompress. |
| WARN | Unexpected but recoverable: outlier overflow, fallbacks. |
| SILENT | Compile-time sentinel — do not pass to log() |
|
strong |
|
strong |
Selects the PHF encode algorithm used by HuffmanStage on the forward path.
| Enumerator | |
|---|---|
| Coarse | Multi-kernel coarse path; CPU prefix-sum sync in phase 3 (default). |
| Fine | EXPERIMENTAL. ReVISIT-lite single kernel, no mid-encode CPU sync, but requires all codes ≤ 8 bits and so does not engage on realistic data; falls back to Coarse. See HuffmanStage::setEncodeMode(). |
|
strong |
Selects where the Huffman codebook comes from on the forward path.
PerBlock is the classic path: histogram the input, copy the frequencies to the host, build a canonical tree, copy the codebook back. Fixed builds one book up front and reuses it for every call, which removes the histogram kernel, the frequency D2H, the host stream sync, and the host tree build from every compress.
Prior work: CEAZ (Xiong et al., ICS'22) generates canonical codewords offline from representative scientific data; Shah et al., Lightweight Huffman Coding for Efficient GPU Compression (ICS'23) precomputes a dictionary of codebooks fitted to cuSZ's quantization-code distribution and selects one at runtime. The win grows as the per-call payload shrinks, so Fixed matters most for small-chunk workloads.
|
strong |
|
strong |
Interpretation of the user-specified error bound.
|x_orig - x_recon| <= eb (default).|error| / |x_orig| <= eb for every element. Requires per-element log-space quantization and is therefore implemented only by QuantizerStage. Predictor-fused stages (LorenzoQuantStage, GInterpStage) cannot honour it — they accept it as a deprecated alias for PREL and emit a warning.abs_eb = eb × max(|data|), then treated as ABS. This is the cheap global approximation of REL used by predictor-fused stages. It bounds |error| / max(|x|), not |error| / |x|: any element with |x| < max(|data|) sees a proportionally looser effective relative error, and elements near zero are effectively unbounded in relative terms. Named PREL (not REL) precisely so that this is impossible to use by accident.abs_eb = eb × (max(data) - min(data)). Equivalent to what most other compressors call "relative". Differs from PREL only in the scan statistic (range vs. max magnitude); for data that straddles zero the two are within 2×. | Enumerator | |
|---|---|
| ABS | Absolute error bound. |
| REL | Exact per-element point-wise relative bound (QuantizerStage only). |
| NOA | Value-range relative bound (norm-of-absolute). |
| PREL | Pseudo-relative: |
|
strong |
Fusion applied at the final write of the forward difference kernel (and undone as the first step of the inverse kernel) when TOut is the unsigned counterpart of a signed T. Both transforms are O(1), bitwise-only, and size-preserving (see modules/transforms/negabinary/negabinary.h and modules/transforms/zigzag/zigzag.h); neither dominates universally — negabinary tends to produce denser zero runs at high bit-planes for smooth, symmetric-around-zero residuals, but zigzag/TCMS can win on other residual distributions. This mirrors the LC framework's decision to keep DIFFNB and DIFFMS as two separate searchable components rather than picking one.
| Enumerator | |
|---|---|
| NEGABINARY | LC's DIFFNB — Negabinary<T>::encode/decode. |
| ZIGZAG | LC's DIFFMS — Zigzag<T>::encode/decode (sign-magnitude/TCMS). |
|
strong |
Input element type for ADMStage.
|
inline |
Returns a human-readable description of a backend error code.
|
inline |
Current live bytes in pool (0 if pool is null).
|
inline |
Peak live bytes in pool since the attribute was last reset (0 if pool is null).
|
constexpr |
Extract major version from a raw on-disk version field. Pre-split files stored small integers (e.g. 3); values ≤ 0xFF are treated as (major=value, minor=0).
|
constexpr |
Extract minor version from a raw on-disk version field (see fzmVersionMajor).
|
inline |
Returns the size in bytes of the given DataType. Throws for DataType::UNKNOWN.
|
inline |
Returns a human-readable string for the given DataType (e.g. "float32").
|
inline |
Returns a human-readable string for the given StageType (e.g. "LorenzoQuant").
| std::vector< std::string > fz::registeredStageTypes | ( | ) |
Every stage type string loadConfig() accepts, in registry order.
Reads the one stage registry that also drives TOML load and save dispatch, so it is correct by construction: adding a stage per the procedure in config.cpp updates this automatically, and no second list can drift out of sync.
Exposed because consumers need the inventory, not just whatever happened to execute — a downstream benchmark harness invalidating cached results per stage has to know a stage exists even when no current pipeline uses it.
| std::vector< StageFingerprintInfo > fz::stageFingerprints | ( | ) |
Per-stage source fingerprints for THIS build.
Each fingerprint is a sha256 (truncated to 16 hex chars) over the stage's own sources plus the transitive closure of its repo-local #includes, generated at build time by scripts/gen_stage_fingerprints.py.
The transitive part is what makes it useful: stages share infrastructure and include each other, so hashing only a stage's own directory would miss a change to the memory pool or to a transform it inlines. A change to a shared header moves every dependent stage's fingerprint; a change to one kernel moves exactly one.
Intended use is cache invalidation: a consumer that recorded these alongside a result can re-run only the entries whose stages have since changed, instead of re-running everything or trusting a stale number. Compare fingerprints for equality only — they carry no ordering.
Deliberately conservative: comment and formatting edits move the fingerprint too, because proving an edit is semantically inert is not something a hash can do, and a needless re-run is much cheaper than a wrong cached result.
| ReconstructionStats fz::calculateStatistics | ( | const T * | d_original, |
| const T * | d_decompressed, | ||
| size_t | n | ||
| ) |
Compute reconstruction statistics between two device arrays.
| d_original | Device pointer to original data. |
| d_decompressed | Device pointer to reconstructed data. |
| n | Number of elements. |
Reconstruct a Stage from a serialized FZM header. Used by the decompressor to rebuild the inverse pipeline from the file.
| type | Stage type read from FZMStageInfo. |
| config | Serialized config bytes. |
| config_size | Number of valid bytes in config. |
|
constexpr |
Byte offset of the values section within the packed RLE wire format, rounded up to alignof(T). The 4-byte num_runs header alone only guarantees 4-byte alignment; for 8-byte T (int64_t/uint64_t) the values section must start on an 8-byte boundary or the reinterpret_cast<T*> reads/writes in rle_pack_kernel/execute() fault with an unaligned 64-bit load (found via the RLE_8 word-size round-trip test).
|
constexpr |
Byte offset of the values section within the chunked wire format, given the chunk count. The offset table is num_chunks + 1 uint32_t entries following the num_chunks header word; the values section is then rounded up to alignof(T) for the same reason as rleValuesOffset<T>().
|
inline |
Resolve an error-bound mode for a stage that has no exact point-wise REL path.
LorenzoQuantStage and GInterpStage quantize prediction residuals against one global tolerance, so a per-element relative bound cannot be threaded through them. Historically both accepted REL and silently applied the eb × max(|data|) approximation; that mode is now spelled PREL. REL is still accepted here as a deprecated alias so existing configs keep running, but it warns — if you need the real guarantee, use QuantizerStage.
| mode | Requested mode. |
| stage_name | Stage name, for the warning message. |
PREL when mode == REL, otherwise mode unchanged. | void fz::launchLorenzoKernel | ( | const TInput * | d_input, |
| size_t | n, | ||
| TInput | ebx2_r, | ||
| TCode | quant_radius, | ||
| TCode * | d_codes, | ||
| TInput * | d_outlier_errors, | ||
| uint32_t * | d_outlier_indices, | ||
| uint32_t * | d_outlier_count, | ||
| size_t | max_outliers, | ||
| int | grid_size, | ||
| bool | zigzag_codes, | ||
| fz::stream_t | stream, | ||
| TInput * | d_means = nullptr |
||
| ) |
| d_means | Per-tile means output (one per 1024-element tile), or nullptr to disable adaptive centering. |
| void fz::launchLorenzoInverseKernel | ( | const TCode * | d_codes, |
| const TInput * | d_outlier_errors, | ||
| const uint32_t * | d_outlier_indices, | ||
| uint32_t | outlier_n, | ||
| size_t | n, | ||
| TInput | ebx2, | ||
| TCode | quant_radius, | ||
| TInput * | d_output, | ||
| bool | zigzag_codes, | ||
| fz::stream_t | stream, | ||
| MemoryPool * | pool, | ||
| const TInput * | d_means = nullptr |
||
| ) |
| d_means | Per-tile means from the forward pass, or nullptr if centering is off. |
| void fz::launchLorenzoDeltaCentered1D | ( | const T * | d_input, |
| T * | d_output, | ||
| T * | d_means, | ||
| size_t | n, | ||
| fz::stream_t | stream, | ||
| unsigned | block_threads | ||
| ) |
Block-mode forward with per-block mean centering. Writes one mean per block to d_means (ceil(n / block_threads) elements) and centers only the first residual of each block.
| void fz::launchLorenzoSegmentedScan | ( | const T * | d_input, |
| const T * | d_means, | ||
| T * | d_output, | ||
| size_t | n, | ||
| fz::stream_t | stream, | ||
| unsigned | block_threads, | ||
| int | passes | ||
| ) |
Unified block-mode inverse: passes segmented prefix sums (1 = LZ1, 2 = LZ2) followed by a uniform + mu when d_means is non-null. One CTA per reset segment with several elements per thread, so the CTA width no longer tracks the segment length.
| void fz::launchLorenzo2Delta1D | ( | const T * | d_input, |
| T * | d_output, | ||
| T * | d_means, | ||
| size_t | n, | ||
| fz::stream_t | stream, | ||
| unsigned | block_threads | ||
| ) |
Block-mode second-order (LZ2) forward. d_means may be nullptr (no centering); when non-null it also writes one mean per block.
|
inlineconstexpr |
True for backends with a mature CUDA-Graph-equivalent capture API (CUDA, HIP).
|
constexpr |
FZM magic number ("FZM2" in little-endian).
|
constexpr |
Version encoding: high byte = major, low byte = minor.
Major mismatch → throw. Minor mismatch → warn and continue. Pre-split files stored a bare integer (e.g. 3); those are treated as major = value, minor = 0, so FZM_VERSION = 0x0300 is backward-compatible.
v3.0 → v3.1: FZMHeaderCore grew from 72 to 80 bytes; added flags, data_checksum, and header_checksum fields.
|
constexpr |
FZMHeaderCore size for v3.0 files (before checksums). Used by readHeader() to avoid overrunning the stage array.