FZGPUModules 2.0
GPU-accelerated modular compression pipelines
Loading...
Searching...
No Matches
fz Namespace Reference

Classes

struct  AdaptiveBitpackConfig
 
class  AdaptiveBitpackStage
 
struct  AdaptiveLorenzoConfig
 Serialized config stored in FZMBufferEntry.stage_config. More...
 
class  AdaptiveLorenzoStage
 
struct  AllocationInfo
 
class  BitpackStage
 
class  BitplaneRZEStage
 
class  BitshuffleStage
 
struct  BufferInfo
 
class  CLOGStage
 
class  CompressionDAG
 
struct  DAGNode
 
class  DifferenceStage
 
struct  FZMBufferEntry
 Per-buffer metadata record written into the FZM header (256 bytes). More...
 
struct  FZMHeaderCore
 Fixed-size FZM file header core (80 bytes). More...
 
struct  FZMStageInfo
 Per-stage metadata record written into the FZM header (256 bytes). More...
 
struct  GInterpConfig
 
class  GInterpStage
 
class  GPULZStage
 
class  HCLOGStage
 
struct  HuffmanBookSpec
 
class  HuffmanStage
 
struct  LevelTimingResult
 
class  Logger
 
struct  LogTransformConfig
 
class  LogTransformStage
 
struct  LorenzoConfig
 
struct  LorenzoQuantConfig
 
class  LorenzoQuantStage
 
class  LorenzoStage
 
class  MemoryPool
 
struct  MemoryPoolConfig
 
class  MergeStage
 
struct  Negabinary
 
class  NegabinaryStage
 
struct  PersistentAllocInfo
 
class  Pipeline
 
struct  PipelinePerfResult
 
struct  QuantizerConfig
 
class  QuantizerStage
 
class  RAREStage
 
class  RAZEStage
 
struct  ReconstructionStats
 
class  RLEStage
 
class  RREStage
 
class  RZEStage
 
class  Stage
 
struct  StageFingerprintInfo
 A stage's name paired with a hash of the source that implements it. More...
 
struct  StageTimingResult
 
struct  TiledLorenzoConfig
 
class  TiledLorenzoStage
 
class  TUPLStage
 
struct  Zigzag
 
class  ZigzagStage
 

Enumerations

enum class  StageType : uint16_t {
}
 Stage type identifiers written into the FZM header. More...
 
enum class  DataType : uint8_t { }
 Element data type identifiers used in buffer and stage descriptors. More...
 
enum class  LogLevel : int {
  TRACE = 0 , DEBUG = 1 , INFO = 2 , WARN = 3 ,
  SILENT = 255
}
 
enum class  MemoryStrategy { MINIMAL , PREALLOCATE }
 
enum class  HuffmanEncodeMode { Coarse , Fine }
 
enum class  HuffmanBookSource { PerBlock , Fixed , Adaptive }
 
enum class  HuffmanBookModel { Gaussian , Laplace , GeneralizedNormal , Uniform }
 
enum class  ErrorBoundMode : uint8_t { ABS = 0 , REL = 1 , NOA = 2 , PREL = 3 }
 
enum class  FusionMode : uint8_t { NEGABINARY = 0 , ZIGZAG = 1 }
 
enum class  ADMDtype : uint8_t
 

Functions

const char * getBackendErrorString (error_t err)
 
size_t getPoolUsedMemCurrent (mempool_t pool)
 
size_t getPoolUsedMemHigh (mempool_t pool)
 
constexpr uint8_t fzmVersionMajor (uint16_t v)
 
constexpr uint8_t fzmVersionMinor (uint16_t v)
 
size_t getDataTypeSize (DataType type)
 
std::string dataTypeToString (DataType type)
 
std::string stageTypeToString (StageType type)
 
std::vector< std::string > registeredStageTypes ()
 Every stage type string loadConfig() accepts, in registry order.
 
std::vector< StageFingerprintInfostageFingerprints ()
 Per-stage source fingerprints for THIS build.
 
template<typename T >
ReconstructionStats calculateStatistics (const T *d_original, const T *d_decompressed, size_t n)
 
StagecreateStage (StageType type, const uint8_t *config, size_t config_size)
 
template<typename T >
constexpr size_t rleValuesOffset ()
 
template<typename T >
constexpr size_t rleChunkedValuesOffset (size_t num_chunks)
 
template<typename T >
void launchAdaptiveLorenzoForward (const T *d_input, T *d_residuals, uint8_t *d_modes, T *d_means, size_t n, uint32_t tile_size, bool enable_order2, bool enable_centering, fz::stream_t stream)
 Forward: select the best variant per tile and emit its residuals.
 
template<typename T >
void launchAdaptiveLorenzoInverse (const T *d_residuals, const uint8_t *d_modes, const T *d_means, T *d_output, size_t n, uint32_t tile_size, fz::stream_t stream)
 Inverse: replay each tile's recorded variant.
 
ErrorBoundMode resolveApproxRelMode (ErrorBoundMode mode, const char *stage_name)
 
template<typename TInput , typename TCode >
void launchLorenzoKernel (const TInput *d_input, size_t n, TInput ebx2_r, TCode quant_radius, TCode *d_codes, TInput *d_outlier_errors, uint32_t *d_outlier_indices, uint32_t *d_outlier_count, size_t max_outliers, int grid_size, bool zigzag_codes, fz::stream_t stream, TInput *d_means=nullptr)
 
template<typename TInput , typename TCode >
void launchLorenzoInverseKernel (const TCode *d_codes, const TInput *d_outlier_errors, const uint32_t *d_outlier_indices, uint32_t outlier_n, size_t n, TInput ebx2, TCode quant_radius, TInput *d_output, bool zigzag_codes, fz::stream_t stream, MemoryPool *pool, const TInput *d_means=nullptr)
 
template<typename TInput , typename TCode >
void launchLorenzoKernel2D (const TInput *d_input, size_t nx, size_t ny, TInput ebx2_r, TCode quant_radius, TCode *d_codes, TInput *d_outlier_errors, uint32_t *d_outlier_indices, uint32_t *d_outlier_count, size_t max_outliers, bool zigzag_codes, fz::stream_t stream)
 2-D forward Lorenzo kernel launcher. nx is the fast (x) dimension.
 
template<typename TInput , typename TCode >
void launchLorenzoInverseKernel2D (const TCode *d_codes, const TInput *d_outlier_errors, const uint32_t *d_outlier_indices, uint32_t outlier_n, size_t nx, size_t ny, TInput ebx2, TCode quant_radius, TInput *d_output, bool zigzag_codes, fz::stream_t stream, MemoryPool *pool)
 2-D inverse Lorenzo kernel launcher.
 
template<typename TInput , typename TCode >
void launchLorenzoKernel3D (const TInput *d_input, size_t nx, size_t ny, size_t nz, TInput ebx2_r, TCode quant_radius, TCode *d_codes, TInput *d_outlier_errors, uint32_t *d_outlier_indices, uint32_t *d_outlier_count, size_t max_outliers, bool zigzag_codes, fz::stream_t stream)
 3-D forward Lorenzo kernel launcher.
 
template<typename TInput , typename TCode >
void launchLorenzoInverseKernel3D (const TCode *d_codes, const TInput *d_outlier_errors, const uint32_t *d_outlier_indices, uint32_t outlier_n, size_t nx, size_t ny, size_t nz, TInput ebx2, TCode quant_radius, TInput *d_output, bool zigzag_codes, fz::stream_t stream, MemoryPool *pool)
 3-D inverse Lorenzo kernel launcher.
 
template<typename T >
void launchLorenzoDeltaCentered1D (const T *d_input, T *d_output, T *d_means, size_t n, fz::stream_t stream, unsigned block_threads)
 
template<typename T >
void launchLorenzoSegmentedScan (const T *d_input, const T *d_means, T *d_output, size_t n, fz::stream_t stream, unsigned block_threads, int passes)
 
template<typename T >
void launchLorenzo2Delta1D (const T *d_input, T *d_output, T *d_means, size_t n, fz::stream_t stream, unsigned block_threads)
 

Variables

constexpr bool kBackendSupportsGraphCapture = true
 
constexpr uint32_t FZM_MAGIC = 0x464D5A32
 
constexpr uint8_t FZM_VERSION_MAJOR = 3
 
constexpr size_t FZM_LEGACY_HEADER_CORE_SIZE = 72
 
constexpr uint16_t FZM_FLAG_HAS_DATA_CHECKSUM = 0x0001u
 data_checksum field is valid
 
constexpr uint16_t FZM_FLAG_HAS_HEADER_CHECKSUM = 0x0002u
 header_checksum field is valid
 
constexpr size_t FZM_MAX_BUFFERS = 32
 Maximum pipeline output buffers per file.
 
constexpr size_t FZM_MAX_NAME_LEN = 64
 Maximum output port name length (bytes, null-terminated)
 
constexpr size_t FZM_STAGE_CONFIG_SIZE = 128
 Per-stage serialized config slot (bytes)
 
constexpr size_t FZM_MAX_SOURCES = 4
 Maximum source stages per pipeline.
 

Detailed Description

Copyright (c) Meta Platforms, Inc. and affiliates.

This source code is licensed under the MIT license found in the LICENSE file in the root directory of this source tree.

Adapted for FZGPUModules: namespace renamed from multibyte_ans to fz::ans.

Copyright (c) Meta Platforms, Inc. and affiliates.

This source code is licensed under the MIT license found in the LICENSE file in the root directory of this source tree.

Adapted for FZGPUModules: namespace renamed from multibyte_ans to fz::ans; stripped ansDecode() host function (called directly from ans_stage.cu instead).

Copyright (c) Meta Platforms, Inc. and affiliates.

This source code is licensed under the MIT license found in the LICENSE file in the root directory of this source tree.

Adapted for FZGPUModules: namespace renamed from multibyte_ans to fz::ans; stripped ansEncode() host function (called directly from ans_stage.cu instead).

Copyright (c) Meta Platforms, Inc. and affiliates.

This source code is licensed under the MIT license found in the LICENSE file in the root directory of this source tree.

Adapted for FZGPUModules: namespace renamed from multibyte_ans to fz::ans; stripped histogramSingle, histogramBatch, ansHistogramBatch (replaced by fz::module::GPU_histogram_generic in ans_stage.cu).

Enumeration Type Documentation

◆ StageType

enum class fz::StageType : uint16_t
strong

Stage type identifiers written into the FZM header.

Each concrete Stage subclass reports one of these values via getStageTypeId(). Used by StageFactory::createStage() to reconstruct the pipeline during decompression.

Enumerator
ANS 

rANS entropy coder (GPU, via dietGPU)

ADM 

Adaptive Data Mapping transform (MANS)

G_INTERP 

Spline interpolation predictor + quantizer (cuSZ-Hi G-Interp)

BITPLANE_RZE 

Fused bitplane transpose + zero-group RZE (FZ-GPU lossless encoder)

ADAPTIVE_BITPACK 

Per-block adaptive fixed-rate bit-plane coder (cuSZp plain mode)

TILED_LORENZO 

Dimension-aware (tiled separable) Lorenzo predictor (cuSZp3 delta)

RRE 

Repetition-Reduction Encoding (LC framework lossless component)

RARE 

Repetition-Adaptive Reduction Encoding (LC framework, auto-k generalization of RRE)

RAZE 

Zero-Adaptive Reduction Encoding (LC framework, auto-k generalization of RZE)

CLOG 

Compressed-Logarithm adaptive bit-width coding (LC framework lossless component)

HCLOG 

Compressed-Logarithm coding with per-subchunk TCMS fallback (LC framework lossless component)

TUPL 

Tuple deinterleave (AoS -> SoA) transpose (LC framework lossless component)

GPULZ 

TODO: describe this stage.

LOG_TRANSFORM 

Log-space transform for point-wise relative bounds (Liang et al., CLUSTER'18)

ADAPTIVE_LORENZO 

Per-tile adaptive multi-order Lorenzo + centering (FSZ prediction stage)

◆ DataType

enum class fz::DataType : uint8_t
strong

Element data type identifiers used in buffer and stage descriptors.

Returned by Stage::getOutputDataType() and Stage::getInputDataType(). UNKNOWN is returned by byte-transparent stages (Bitshuffle, RZE) to opt out of Pipeline::finalize() type-compatibility checking.

Enumerator
UNKNOWN 

Byte-transparent stages: skip type checking at finalize()

◆ LogLevel

enum class fz::LogLevel : int
strong
Enumerator
TRACE 

Per-stage execute(), per-chunk details — very verbose.

DEBUG 

Pipeline construction, buffer allocation, data stats.

INFO 

High-level milestones: finalize, compress, decompress.

WARN 

Unexpected but recoverable: outlier overflow, fallbacks.

SILENT 

Compile-time sentinel — do not pass to log()

◆ MemoryStrategy

enum class fz::MemoryStrategy
strong

Memory allocation strategy for pipeline execution.

Enumerator
MINIMAL 

Allocate on-demand, free at last consumer. Lowest peak memory.

PREALLOCATE 

Allocate everything upfront at finalize(). Required for graph mode.

◆ HuffmanEncodeMode

enum class fz::HuffmanEncodeMode
strong

Selects the PHF encode algorithm used by HuffmanStage on the forward path.

Enumerator
Coarse 

Multi-kernel coarse path; CPU prefix-sum sync in phase 3 (default).

Fine 

EXPERIMENTAL. ReVISIT-lite single kernel, no mid-encode CPU sync, but requires all codes ≤ 8 bits and so does not engage on realistic data; falls back to Coarse. See HuffmanStage::setEncodeMode().

◆ HuffmanBookSource

enum class fz::HuffmanBookSource
strong

Selects where the Huffman codebook comes from on the forward path.

PerBlock is the classic path: histogram the input, copy the frequencies to the host, build a canonical tree, copy the codebook back. Fixed builds one book up front and reuses it for every call, which removes the histogram kernel, the frequency D2H, the host stream sync, and the host tree build from every compress.

Prior work: CEAZ (Xiong et al., ICS'22) generates canonical codewords offline from representative scientific data; Shah et al., Lightweight Huffman Coding for Efficient GPU Compression (ICS'23) precomputes a dictionary of codebooks fitted to cuSZ's quantization-code distribution and selects one at runtime. The win grows as the per-call payload shrinks, so Fixed matters most for small-chunk workloads.

Enumerator
PerBlock 

Histogram + build a fresh codebook on every forward call (default).

Fixed 

Build one codebook up front and reuse it for every forward call.

Adaptive 

Histogram the first call only, then reuse that codebook forever.

◆ HuffmanBookModel

enum class fz::HuffmanBookModel
strong

Analytic symbol distribution used to synthesize a fixed codebook.

Enumerator
Gaussian 

exp(-((i-center)/scale)^2 / 2)

Laplace 

exp(-|i-center|/scale)

GeneralizedNormal 

exp(-(|i-center|/scale)^shape)

Uniform 

flat; every symbol equally likely

◆ ErrorBoundMode

enum class fz::ErrorBoundMode : uint8_t
strong

Interpretation of the user-specified error bound.

  • ABS|x_orig - x_recon| <= eb (default).
  • RELguaranteed point-wise relative (PFPL): |error| / |x_orig| <= eb for every element. Requires per-element log-space quantization and is therefore implemented only by QuantizerStage. Predictor-fused stages (LorenzoQuantStage, GInterpStage) cannot honour it — they accept it as a deprecated alias for PREL and emit a warning.
  • PRELpseudo-relative: abs_eb = eb × max(|data|), then treated as ABS. This is the cheap global approximation of REL used by predictor-fused stages. It bounds |error| / max(|x|), not |error| / |x|: any element with |x| < max(|data|) sees a proportionally looser effective relative error, and elements near zero are effectively unbounded in relative terms. Named PREL (not REL) precisely so that this is impossible to use by accident.
  • NOA — norm-of-absolute / value-range relative (PFPL): abs_eb = eb × (max(data) - min(data)). Equivalent to what most other compressors call "relative". Differs from PREL only in the scan statistic (range vs. max magnitude); for data that straddles zero the two are within 2×.
Enumerator
ABS 

Absolute error bound.

REL 

Exact per-element point-wise relative bound (QuantizerStage only).

NOA 

Value-range relative bound (norm-of-absolute).

PREL 

Pseudo-relative: eb × max(|data|), applied as a single ABS bound.

◆ FusionMode

enum class fz::FusionMode : uint8_t
strong

Fusion applied at the final write of the forward difference kernel (and undone as the first step of the inverse kernel) when TOut is the unsigned counterpart of a signed T. Both transforms are O(1), bitwise-only, and size-preserving (see modules/transforms/negabinary/negabinary.h and modules/transforms/zigzag/zigzag.h); neither dominates universally — negabinary tends to produce denser zero runs at high bit-planes for smooth, symmetric-around-zero residuals, but zigzag/TCMS can win on other residual distributions. This mirrors the LC framework's decision to keep DIFFNB and DIFFMS as two separate searchable components rather than picking one.

Enumerator
NEGABINARY 

LC's DIFFNB — Negabinary<T>::encode/decode.

ZIGZAG 

LC's DIFFMS — Zigzag<T>::encode/decode (sign-magnitude/TCMS).

◆ ADMDtype

enum class fz::ADMDtype : uint8_t
strong

Input element type for ADMStage.

Function Documentation

◆ getBackendErrorString()

const char * fz::getBackendErrorString ( error_t  err)
inline

Returns a human-readable description of a backend error code.

◆ getPoolUsedMemCurrent()

size_t fz::getPoolUsedMemCurrent ( mempool_t  pool)
inline

Current live bytes in pool (0 if pool is null).

◆ getPoolUsedMemHigh()

size_t fz::getPoolUsedMemHigh ( mempool_t  pool)
inline

Peak live bytes in pool since the attribute was last reset (0 if pool is null).

◆ fzmVersionMajor()

constexpr uint8_t fz::fzmVersionMajor ( uint16_t  v)
constexpr

Extract major version from a raw on-disk version field. Pre-split files stored small integers (e.g. 3); values ≤ 0xFF are treated as (major=value, minor=0).

◆ fzmVersionMinor()

constexpr uint8_t fz::fzmVersionMinor ( uint16_t  v)
constexpr

Extract minor version from a raw on-disk version field (see fzmVersionMajor).

◆ getDataTypeSize()

size_t fz::getDataTypeSize ( DataType  type)
inline

Returns the size in bytes of the given DataType. Throws for DataType::UNKNOWN.

◆ dataTypeToString()

std::string fz::dataTypeToString ( DataType  type)
inline

Returns a human-readable string for the given DataType (e.g. "float32").

◆ stageTypeToString()

std::string fz::stageTypeToString ( StageType  type)
inline

Returns a human-readable string for the given StageType (e.g. "LorenzoQuant").

◆ registeredStageTypes()

std::vector< std::string > fz::registeredStageTypes ( )

Every stage type string loadConfig() accepts, in registry order.

Reads the one stage registry that also drives TOML load and save dispatch, so it is correct by construction: adding a stage per the procedure in config.cpp updates this automatically, and no second list can drift out of sync.

Exposed because consumers need the inventory, not just whatever happened to execute — a downstream benchmark harness invalidating cached results per stage has to know a stage exists even when no current pipeline uses it.

Returns
Stage type names, e.g. {"Lorenzo", "LorenzoQuant", "Quantizer", ...}.

◆ stageFingerprints()

std::vector< StageFingerprintInfo > fz::stageFingerprints ( )

Per-stage source fingerprints for THIS build.

Each fingerprint is a sha256 (truncated to 16 hex chars) over the stage's own sources plus the transitive closure of its repo-local #includes, generated at build time by scripts/gen_stage_fingerprints.py.

The transitive part is what makes it useful: stages share infrastructure and include each other, so hashing only a stage's own directory would miss a change to the memory pool or to a transform it inlines. A change to a shared header moves every dependent stage's fingerprint; a change to one kernel moves exactly one.

Intended use is cache invalidation: a consumer that recorded these alongside a result can re-run only the entries whose stages have since changed, instead of re-running everything or trusting a stale number. Compare fingerprints for equality only — they carry no ordering.

Deliberately conservative: comment and formatting edits move the fingerprint too, because proving an edit is semantically inert is not something a hash can do, and a needless re-run is much cheaper than a wrong cached result.

Returns
One entry per registered stage, in registry order.

◆ calculateStatistics()

template<typename T >
ReconstructionStats fz::calculateStatistics ( const T *  d_original,
const T *  d_decompressed,
size_t  n 
)

Compute reconstruction statistics between two device arrays.

Parameters
d_originalDevice pointer to original data.
d_decompressedDevice pointer to reconstructed data.
nNumber of elements.

◆ createStage()

Stage * fz::createStage ( StageType  type,
const uint8_t *  config,
size_t  config_size 
)
inline

Reconstruct a Stage from a serialized FZM header. Used by the decompressor to rebuild the inverse pipeline from the file.

Parameters
typeStage type read from FZMStageInfo.
configSerialized config bytes.
config_sizeNumber of valid bytes in config.
Returns
Heap-allocated Stage; caller takes ownership.

◆ rleValuesOffset()

template<typename T >
constexpr size_t fz::rleValuesOffset ( )
constexpr

Byte offset of the values section within the packed RLE wire format, rounded up to alignof(T). The 4-byte num_runs header alone only guarantees 4-byte alignment; for 8-byte T (int64_t/uint64_t) the values section must start on an 8-byte boundary or the reinterpret_cast<T*> reads/writes in rle_pack_kernel/execute() fault with an unaligned 64-bit load (found via the RLE_8 word-size round-trip test).

◆ rleChunkedValuesOffset()

template<typename T >
constexpr size_t fz::rleChunkedValuesOffset ( size_t  num_chunks)
constexpr

Byte offset of the values section within the chunked wire format, given the chunk count. The offset table is num_chunks + 1 uint32_t entries following the num_chunks header word; the values section is then rounded up to alignof(T) for the same reason as rleValuesOffset<T>().

◆ resolveApproxRelMode()

ErrorBoundMode fz::resolveApproxRelMode ( ErrorBoundMode  mode,
const char *  stage_name 
)
inline

Resolve an error-bound mode for a stage that has no exact point-wise REL path.

LorenzoQuantStage and GInterpStage quantize prediction residuals against one global tolerance, so a per-element relative bound cannot be threaded through them. Historically both accepted REL and silently applied the eb × max(|data|) approximation; that mode is now spelled PREL. REL is still accepted here as a deprecated alias so existing configs keep running, but it warns — if you need the real guarantee, use QuantizerStage.

Parameters
modeRequested mode.
stage_nameStage name, for the warning message.
Returns
PREL when mode == REL, otherwise mode unchanged.

◆ launchLorenzoKernel()

template<typename TInput , typename TCode >
void fz::launchLorenzoKernel ( const TInput *  d_input,
size_t  n,
TInput  ebx2_r,
TCode  quant_radius,
TCode *  d_codes,
TInput *  d_outlier_errors,
uint32_t *  d_outlier_indices,
uint32_t *  d_outlier_count,
size_t  max_outliers,
int  grid_size,
bool  zigzag_codes,
fz::stream_t  stream,
TInput *  d_means = nullptr 
)
Parameters
d_meansPer-tile means output (one per 1024-element tile), or nullptr to disable adaptive centering.

◆ launchLorenzoInverseKernel()

template<typename TInput , typename TCode >
void fz::launchLorenzoInverseKernel ( const TCode *  d_codes,
const TInput *  d_outlier_errors,
const uint32_t *  d_outlier_indices,
uint32_t  outlier_n,
size_t  n,
TInput  ebx2,
TCode  quant_radius,
TInput *  d_output,
bool  zigzag_codes,
fz::stream_t  stream,
MemoryPool pool,
const TInput *  d_means = nullptr 
)
Parameters
d_meansPer-tile means from the forward pass, or nullptr if centering is off.

◆ launchLorenzoDeltaCentered1D()

template<typename T >
void fz::launchLorenzoDeltaCentered1D ( const T *  d_input,
T *  d_output,
T *  d_means,
size_t  n,
fz::stream_t  stream,
unsigned  block_threads 
)

Block-mode forward with per-block mean centering. Writes one mean per block to d_means (ceil(n / block_threads) elements) and centers only the first residual of each block.

◆ launchLorenzoSegmentedScan()

template<typename T >
void fz::launchLorenzoSegmentedScan ( const T *  d_input,
const T *  d_means,
T *  d_output,
size_t  n,
fz::stream_t  stream,
unsigned  block_threads,
int  passes 
)

Unified block-mode inverse: passes segmented prefix sums (1 = LZ1, 2 = LZ2) followed by a uniform + mu when d_means is non-null. One CTA per reset segment with several elements per thread, so the CTA width no longer tracks the segment length.

◆ launchLorenzo2Delta1D()

template<typename T >
void fz::launchLorenzo2Delta1D ( const T *  d_input,
T *  d_output,
T *  d_means,
size_t  n,
fz::stream_t  stream,
unsigned  block_threads 
)

Block-mode second-order (LZ2) forward. d_means may be nullptr (no centering); when non-null it also writes one mean per block.

Variable Documentation

◆ kBackendSupportsGraphCapture

constexpr bool fz::kBackendSupportsGraphCapture = true
inlineconstexpr

True for backends with a mature CUDA-Graph-equivalent capture API (CUDA, HIP).

◆ FZM_MAGIC

constexpr uint32_t fz::FZM_MAGIC = 0x464D5A32
constexpr

FZM magic number ("FZM2" in little-endian).

◆ FZM_VERSION_MAJOR

constexpr uint8_t fz::FZM_VERSION_MAJOR = 3
constexpr

Version encoding: high byte = major, low byte = minor.

Major mismatch → throw. Minor mismatch → warn and continue. Pre-split files stored a bare integer (e.g. 3); those are treated as major = value, minor = 0, so FZM_VERSION = 0x0300 is backward-compatible.

v3.0 → v3.1: FZMHeaderCore grew from 72 to 80 bytes; added flags, data_checksum, and header_checksum fields.

◆ FZM_LEGACY_HEADER_CORE_SIZE

constexpr size_t fz::FZM_LEGACY_HEADER_CORE_SIZE = 72
constexpr

FZMHeaderCore size for v3.0 files (before checksums). Used by readHeader() to avoid overrunning the stage array.