structured_zstd/encoding/mod.rs
1//! Zstandard encoder — frame compression, streaming, dictionary support.
2//!
3//! Four entry points cover the common use cases:
4//!
5//! * [`compress`] — one-shot helper that builds a self-contained
6//! Zstandard frame from a `Read` source to a `Write` sink. The
7//! input is consumed incrementally from `Read`, so input buffering
8//! stays bounded; however, the compressed output is buffered in
9//! memory until the frame is complete so the Frame Content Size
10//! field can be filled in the header — peak memory is
11//! `O(compressed_size)` (worst-case `O(input_size)` for
12//! incompressible payloads, plus a small frame overhead). The
13//! savings vs [`compress_to_vec`] come from not materialising the
14//! input alongside the output.
15//! * [`compress_to_vec`] — same one-shot path as [`compress`] but
16//! the input is eagerly drained into an internal `Vec` first
17//! (`read_to_end`) so the encoder can be handed a `&[u8]` and a
18//! precise source-size hint. Peak memory is therefore ≈
19//! `input_size + output_size`; prefer [`compress`] or
20//! [`StreamingEncoder`] when the input is large or unbounded.
21//! * [`StreamingEncoder`] — implements [`crate::io::Write`], which
22//! re-exports [`std::io::Write`] under the `std` feature and falls
23//! back to a `no_std`-friendly trait otherwise. Accepts bytes
24//! incrementally and flushes compressed output as blocks fill.
25//! Requires `set_pledged_content_size` before the first write if
26//! the Frame Content Size field is to be populated.
27//! * [`FrameCompressor`] — lower-level builder that owns the matcher and
28//! the per-frame configuration; the streaming and one-shot helpers are
29//! thin wrappers over it. Reach for it when you need to swap in a custom
30//! [`Matcher`] implementation or share the matcher across frames.
31//!
32//! Compression intensity is selected via [`CompressionLevel`], which
33//! provides both named presets (`Fastest`, `Default`, `Better`, `Best`) and
34//! numeric levels (`from_level(n)`) that mirror C zstd's level numbering
35//! (negative for ultra-fast, `0` = default, `1..=22` for the standard
36//! range).
37//!
38//! All produced frames are valid RFC 8878 Zstandard streams and decode
39//! through both this crate's [`crate::decoding`] module and upstream C zstd.
40//!
41//! For memory budgeting, [`estimated_compression_workspace_bytes`] reports
42//! the approximate steady-state heap footprint of a one-shot compression at
43//! a given level (window + match-finder tables + block staging).
44
45pub(crate) mod block_header;
46pub(crate) mod blocks;
47pub(crate) mod cparams;
48pub(crate) mod dict_attach;
49pub(crate) mod fastpath;
50pub(crate) mod frame_header;
51pub(crate) mod incompressible;
52pub(crate) mod match_generator;
53pub(crate) mod util;
54
55// `#111` encoder architecture rewrite. `cost_model`, `opt`,
56// `strategy`, `dfast`, `row`, and `simple` host the relocated
57// cost-model types, the optimal-parser plain-data types, the
58// const-generic [`strategy::Strategy`] trait + per-level [`strategy::
59// StrategyTag`] dispatcher, and the Dfast / Row / Simple matchers
60// respectively. `match_table::helpers` hosts the shared match-finder
61// primitives. The rewrite plan is tracked in
62// <https://github.com/structured-world/structured-zstd/issues/111>;
63// per-phase boundaries are `perf/post-pr-110-baseline` (start),
64// `perf/post-pr-121-baseline` (post-Phase-2).
65pub(crate) mod bt;
66pub(crate) mod cost_model;
67pub(crate) mod dfast;
68pub(crate) mod hc;
69pub(crate) mod lazy_parse;
70// LDM hashes each `min_match_length` window with XXH64 (upstream zstd
71// `zstd_ldm.c:315`), so the `ldm` feature implies `hash` and the
72// `twox-hash` dependency it pulls in. `BtMatcher::ldm_producer` and the
73// `cfg(feature = "ldm")` blocks inside `BtMatcher::prepare_ldm_candidates` /
74// `BtMatcher::reset` carry the same gate; the call site in
75// `hc::optimal::HcMatchGenerator::start_matching_optimal` invokes
76// `prepare_ldm_candidates` unconditionally because the gating is internal to
77// the method body (without the feature it shrinks to an
78// `ldm_sequences.clear()` stub).
79#[cfg(feature = "ldm")]
80pub(crate) mod ldm;
81pub(crate) mod match_table;
82pub(crate) mod opt;
83pub(crate) mod row;
84pub(crate) mod simple;
85pub(crate) mod strategy;
86
87pub(crate) mod frame_compressor;
88#[cfg(feature = "lsm")]
89pub mod frame_emit_info;
90mod levels;
91pub(crate) mod parameters;
92#[cfg(feature = "bench-internals")]
93pub mod sequence_capture;
94mod streaming_encoder;
95pub use frame_compressor::{EncoderDictionary, FrameCompressor};
96#[cfg(feature = "lsm")]
97pub use frame_emit_info::{BlockType, FrameBlock, FrameEmitInfo};
98pub use levels::config::{
99 estimated_bt_strategy_extra_bytes, estimated_compression_workspace_bytes,
100 estimated_compression_workspace_bytes_for_run,
101 estimated_compression_workspace_bytes_for_source,
102};
103pub use match_generator::MatchGeneratorDriver;
104pub use parameters::{
105 Bounds, CParameter, CompressionParameters, CompressionParametersBuilder, ParameterError,
106 Strategy,
107};
108pub use streaming_encoder::StreamingEncoder;
109
110use crate::io::{Read, Write};
111use alloc::vec::Vec;
112
113/// Convenience function to compress some source into a target without reusing any resources of the compressor
114/// ```rust
115/// use structured_zstd::encoding::{compress, CompressionLevel};
116/// let data: &[u8] = &[0,0,0,0,0,0,0,0,0,0,0,0];
117/// let mut target = Vec::new();
118/// compress(data, &mut target, CompressionLevel::Fastest);
119/// ```
120pub fn compress<R: Read, W: Write>(source: R, target: W, level: CompressionLevel) {
121 let mut frame_enc = FrameCompressor::new(level);
122 frame_enc.set_source(source);
123 frame_enc.set_drain(target);
124 frame_enc.compress();
125}
126
127/// Convenience function to compress some source into a Vec without reusing any resources of the compressor.
128///
129/// This helper eagerly buffers the full input (`Read`) before compression so it
130/// can provide a source-size hint to the one-shot encoder path. Peak memory can
131/// therefore be roughly `input_size + output_size`. For very large payloads or
132/// tighter memory budgets, prefer streaming APIs such as [`StreamingEncoder`].
133///
134/// **This is NOT a streaming API.** The source is fully buffered
135/// into a `Vec<u8>` before any compression work begins, so peak input
136/// memory is bounded by `source.len()` (not "constant regardless of
137/// payload size" as a stream-shaped encoder would offer). If the
138/// source is large enough that holding it in memory is not acceptable,
139/// use [`StreamingEncoder`] which consumes chunks incrementally
140/// without the up-front Vec build.
141///
142/// This helper drives `read_to_end` to materialize the full source
143/// into a `Vec<u8>` before forwarding the slice to
144/// [`compress_slice_to_vec`]. For a `Read` whose size is unknown ahead
145/// of time, `read_to_end` grows that input `Vec` via power-of-two
146/// doubling: peak input allocation can be up to 2× the final source
147/// length transiently. The live working set on this entry point is
148/// roughly `input.capacity()` plus the block-accumulation buffer and
149/// per-block scratch carried by [`compress_slice_to_vec`], plus the
150/// exactly-sized output `Vec`. [`StreamingEncoder`] avoids the input
151/// materialization step entirely and is the right entry point when
152/// the source is large or unbounded.
153///
154/// ```rust
155/// use structured_zstd::encoding::{compress_to_vec, CompressionLevel};
156/// let data: &[u8] = &[0,0,0,0,0,0,0,0,0,0,0,0];
157/// let compressed = compress_to_vec(data, CompressionLevel::Fastest);
158/// ```
159pub fn compress_to_vec<R: Read>(source: R, level: CompressionLevel) -> Vec<u8> {
160 let mut source = source;
161 let mut input = Vec::new();
162 source.read_to_end(&mut input).unwrap();
163 compress_slice_to_vec(input.as_slice(), level)
164}
165
166/// Compress a contiguous byte slice into a fresh `Vec<u8>` without the
167/// input-buffering step that [`compress_to_vec`] performs to adapt a
168/// `Read` source.
169///
170/// One-shot wrapper over
171/// [`FrameCompressor::compress_independent_frame`]: the input is read by
172/// reference (the eligible Fast path scans it in place, no per-block
173/// history copy), and the returned `Vec` is allocated exactly once at the
174/// final frame size after compression. Peak transient memory is the
175/// block-accumulation buffer (grown via amortized doubling, ≈ 2× current
176/// compressed size at the last realloc) plus the exactly-sized output. The
177/// worst-case compressed-size bound is never pinned upfront, so a highly
178/// compressible 100 MiB input does not charge ~100 MiB of worst-case
179/// expansion against peak.
180///
181/// To compress many slices, construct one [`FrameCompressor`] and call
182/// [`compress_independent_frame_into`](FrameCompressor::compress_independent_frame_into)
183/// in a loop instead, which reuses the matcher tables, scratch, and output
184/// buffer across frames (this function allocates and primes from scratch
185/// each call).
186///
187/// # Panics
188///
189/// Panics on encoder error (matches the failure surface of
190/// [`compress_to_vec`], which this function backs). Out-of-memory during
191/// the output / per-block scratch allocations is handled by the global
192/// allocator's abort policy. The slice/Vec entry points mirror the upstream zstd
193/// `ZSTD_compress` shape (no error return on the bulk path).
194///
195/// ```rust
196/// use structured_zstd::encoding::{compress_slice_to_vec, CompressionLevel};
197/// let data: &[u8] = &[0,0,0,0,0,0,0,0,0,0,0,0];
198/// let compressed = compress_slice_to_vec(data, CompressionLevel::Fastest);
199/// ```
200pub fn compress_slice_to_vec(source: &[u8], level: CompressionLevel) -> Vec<u8> {
201 // Bare `FrameCompressor` resolves all three type params to their
202 // defaults (`&'static [u8]` reader, `Vec<u8>` drain, MatchGeneratorDriver);
203 // neither the reader nor the drain is used by the in-place
204 // `compress_independent_frame` path.
205 let mut enc: FrameCompressor = FrameCompressor::new(level);
206 enc.compress_independent_frame(source)
207}
208
209/// Worst-case compressed-frame size for an input of `src_size` bytes.
210///
211/// A destination buffer of this size is always large enough to hold the
212/// output of [`compress_slice_to_vec`] (or any single-frame compression) for
213/// an input of `src_size` bytes, so a caller sizing a fixed buffer once (the
214/// shape the C `ZSTD_compress` entry point needs) never has to grow it.
215///
216/// Mirrors the upstream `ZSTD_COMPRESSBOUND` formula exactly:
217/// `src_size + (src_size >> 8) + margin`, where `margin` is
218/// `(128 KiB - src_size) >> 11` for inputs below 128 KiB and `0` otherwise.
219/// The margin guarantees `bound(a) + bound(b) <= bound(a + b)` for blocks of
220/// at least 128 KiB, which keeps multi-frame concatenation sizing sound.
221///
222/// Saturates at [`usize::MAX`] if the formula would overflow on a
223/// pathologically large `src_size` — no allocation that large can exist, so
224/// the saturated value is the correct "cannot fit" sentinel rather than a
225/// masked wrap.
226///
227/// ```rust
228/// use structured_zstd::encoding::{compress_bound, compress_slice_to_vec, CompressionLevel};
229/// let data = [7u8; 4096];
230/// assert!(compress_slice_to_vec(&data, CompressionLevel::Default).len() <= compress_bound(data.len()));
231/// ```
232pub const fn compress_bound(src_size: usize) -> usize {
233 const LOWER: usize = 128 * 1024;
234 let margin = if src_size < LOWER {
235 (LOWER - src_size) >> 11
236 } else {
237 0
238 };
239 // Saturating is the correct UPPER-BOUND semantic here, not a masked bug:
240 // this is a public API over an arbitrary `usize`, and the largest meaningful
241 // bound is `usize::MAX`. A real slice is at most `isize::MAX` bytes, so the
242 // `* 1.004 + margin` cannot overflow for genuine inputs; the saturation only
243 // caps a pathological caller-supplied size at the representable ceiling.
244 src_size
245 .saturating_add(src_size >> 8)
246 .saturating_add(margin)
247}
248
249/// Compress a byte slice into a fresh `Vec<u8>` using fine-grained
250/// [`CompressionParameters`] (#27) instead of a bare
251/// [`CompressionLevel`].
252///
253/// One-shot wrapper over [`FrameCompressor::set_parameters`] +
254/// [`FrameCompressor::compress_independent_frame`]. The produced frame is
255/// a valid RFC 8878 stream regardless of the knobs chosen.
256///
257/// ```rust
258/// use structured_zstd::encoding::{
259/// compress_with_parameters, CompressionLevel, CompressionParameters, Strategy,
260/// };
261/// let data: &[u8] = b"the quick brown fox jumps over the lazy dog";
262/// let params = CompressionParameters::builder(CompressionLevel::Level(5))
263/// .strategy(Strategy::Greedy)
264/// .build()
265/// .unwrap();
266/// let compressed = compress_with_parameters(data, ¶ms);
267/// assert!(!compressed.is_empty());
268/// ```
269pub fn compress_with_parameters(source: &[u8], params: &CompressionParameters) -> Vec<u8> {
270 let mut enc: FrameCompressor = FrameCompressor::new(params.level());
271 enc.set_parameters(params);
272 enc.compress_independent_frame(source)
273}
274
275/// The compression mode used impacts the speed of compression,
276/// and resulting compression ratios. Faster compression will result
277/// in worse compression ratios, and vice versa.
278#[derive(Copy, Clone, Debug, PartialEq, Eq)]
279pub enum CompressionLevel {
280 /// This level does not compress the data at all, and simply wraps
281 /// it in a Zstandard frame.
282 Uncompressed,
283 /// This level is roughly equivalent to Zstd compression level 1
284 Fastest,
285 /// This level uses the crate's dedicated `dfast`-style matcher to
286 /// target a better speed/ratio tradeoff than [`CompressionLevel::Fastest`].
287 ///
288 /// It represents this crate's "default" compression setting and may
289 /// evolve in future versions as the implementation moves closer to
290 /// reference zstd level 3 behavior.
291 Default,
292 /// This level is roughly equivalent to Zstd level 7.
293 ///
294 /// Uses the hash-chain matcher with a lazy2 matching strategy: the encoder
295 /// evaluates up to two positions ahead before committing to a match,
296 /// trading speed for a better compression ratio than [`CompressionLevel::Default`].
297 Better,
298 /// This level is equivalent to Zstd level 13.
299 ///
300 /// Uses the lazy2 parse over the binary-tree match finder (`btlazy2`),
301 /// the first level of the deep band that strictly dominates every level
302 /// below it on ratio; compared to [`CompressionLevel::Better`] it
303 /// trades speed for the best ratio of the named presets.
304 Best,
305 /// Numeric compression level.
306 ///
307 /// Levels 1–22 correspond to the C zstd level numbering. Higher values
308 /// produce smaller output at the cost of more CPU time. Negative values
309 /// select ultra-fast modes that trade ratio for speed. Level 0 is
310 /// treated as [`DEFAULT_LEVEL`](Self::DEFAULT_LEVEL), matching C zstd
311 /// semantics.
312 ///
313 /// Named variants map to specific numeric levels:
314 /// [`Fastest`](Self::Fastest) = 1, [`Default`](Self::Default) = 3,
315 /// [`Better`](Self::Better) = 7, [`Best`](Self::Best) = 13.
316 /// [`Best`](Self::Best) remains the highest-ratio named preset, but
317 /// [`Level`](Self::Level) values above 13 can target stronger (slower)
318 /// tuning than the named hierarchy.
319 ///
320 /// Levels above 13 use progressively larger windows and deeper search.
321 /// Levels 16–17 use a `btopt`-style price parser, 18 uses `btultra`,
322 /// and 19–22 use a `btultra2`-style two-pass selection profile.
323 ///
324 /// Semver note: this variant was added after the initial enum shape and
325 /// is a breaking API change for downstream crates that exhaustively
326 /// `match` on [`CompressionLevel`] without a wildcard arm.
327 Level(i32),
328}
329
330impl CompressionLevel {
331 /// The minimum supported numeric compression level (ultra-fast mode).
332 pub const MIN_LEVEL: i32 = -131072;
333 /// The maximum supported numeric compression level.
334 pub const MAX_LEVEL: i32 = 22;
335 /// The default numeric compression level (equivalent to [`Default`](Self::Default)).
336 pub const DEFAULT_LEVEL: i32 = 3;
337
338 /// Create a compression level from a numeric value.
339 ///
340 /// Returns named variants for canonical levels (`0`/`3`, `1`, `7`, `13`)
341 /// and [`Level`](Self::Level) for all other values.
342 ///
343 /// With the default matcher backend (`MatchGeneratorDriver`), values
344 /// outside [`MIN_LEVEL`](Self::MIN_LEVEL)..=[`MAX_LEVEL`](Self::MAX_LEVEL)
345 /// are silently clamped during built-in level parameter resolution.
346 pub const fn from_level(level: i32) -> Self {
347 match level {
348 0 | Self::DEFAULT_LEVEL => Self::Default,
349 1 => Self::Fastest,
350 7 => Self::Better,
351 13 => Self::Best,
352 _ => Self::Level(level),
353 }
354 }
355}
356
357/// The sizes of a dictionary handed to [`Matcher::set_dictionary_size_hint`].
358///
359/// Upstream picks the CDict's cParams tier from the size of the serialized
360/// dictionary buffer (`ZSTD_createCDict(dictBuffer, dictSize, level)`, header
361/// and entropy tables included), while the dictionary tables and the attach
362/// cutoffs are sized from the content that is actually indexed.
363///
364/// # Examples
365/// ```
366/// use structured_zstd::encoding::DictionarySizes;
367/// let sizes = DictionarySizes::raw_content(4096);
368/// assert_eq!(sizes.serialized, sizes.content);
369/// ```
370#[derive(Clone, Copy, Debug, PartialEq, Eq)]
371pub struct DictionarySizes {
372 /// Bytes of dictionary content the matcher indexes.
373 pub content: usize,
374 /// Bytes of the serialized dictionary (the CDict cParams tier key); equal
375 /// to `content` for a raw-content dictionary.
376 pub serialized: usize,
377}
378
379impl DictionarySizes {
380 /// Sizes of a raw-content dictionary: nothing but the content is
381 /// serialized.
382 pub const fn raw_content(len: usize) -> Self {
383 Self {
384 content: len,
385 serialized: len,
386 }
387 }
388}
389
390/// Trait used by the encoder that users can use to extend the matching facilities with their own algorithm
391/// making their own tradeoffs between runtime, memory usage and compression ratio
392///
393/// This trait operates on buffers that represent the chunks of data the matching algorithm wants to work on.
394/// Each one of these buffers is referred to as a *space*. One or more of these buffers represent the window
395/// the decoder will need to decode the data again.
396///
397/// This library asks the Matcher for a new buffer using `get_next_space` to allow reusing of allocated buffers when they are no longer part of the
398/// window of data that is being used for matching.
399///
400/// The library fills the buffer with data that is to be compressed and commits them back to the matcher using `commit_space`.
401///
402/// Then it will either call `start_matching` or, if the space is deemed not worth compressing, `skip_matching` is called.
403///
404/// This is repeated until no more data is left to be compressed.
405pub trait Matcher {
406 /// Get a space where we can put data to be matched on. Will be encoded as one block. The maximum allowed size is 128 kB.
407 fn get_next_space(&mut self) -> alloc::vec::Vec<u8>;
408 /// Get a reference to the last committed space
409 fn get_last_space(&mut self) -> &[u8];
410 /// Commit a space to the matcher so it can be matched against
411 fn commit_space(&mut self, space: alloc::vec::Vec<u8>);
412 /// Read the next block straight into the matcher's own history buffer,
413 /// skipping the scratch buffer that [`commit_space`](Self::commit_space)
414 /// otherwise has to copy in.
415 ///
416 /// `fill` is handed the history buffer with room reserved for `capacity`
417 /// more bytes and returns `(appended, eof)`. The bytes are readable through
418 /// [`uncommitted_input`](Self::uncommitted_input) but are NOT yet part of
419 /// the match window: the caller chooses the block boundary (the pre-split
420 /// pass needs the bytes to decide) and then calls
421 /// [`commit_filled`](Self::commit_filled). Whatever is left over stays in
422 /// the buffer and becomes the head of the next block, so a carried split
423 /// remainder costs no copy.
424 ///
425 /// Returns `None` if this matcher has no in-place ingest, which is the
426 /// default: the caller then keeps the staged-copy path.
427 fn fill_in_place(
428 &mut self,
429 _capacity: usize,
430 _fill: &mut dyn FnMut(&mut alloc::vec::Vec<u8>) -> (usize, bool),
431 ) -> Option<(usize, bool)> {
432 None
433 }
434 /// Bytes ingested by [`fill_in_place`](Self::fill_in_place) that no block
435 /// has claimed yet. Empty unless that hook is implemented.
436 fn uncommitted_input(&self) -> &[u8] {
437 &[]
438 }
439 /// Claim `len` bytes from the head of
440 /// [`uncommitted_input`](Self::uncommitted_input) as the next block.
441 fn commit_filled(&mut self, _len: usize) {}
442 /// Size the ingest buffer for a frame of `bytes` up front, so filling it
443 /// block by block doesn't walk a doubling chain of reallocations. Clamped
444 /// internally to the buffer's eviction ceiling, so an over-long or absent
445 /// hint can never reserve more than a bounded window. No-op unless
446 /// [`fill_in_place`](Self::fill_in_place) is implemented.
447 fn reserve_for_frame(&mut self, _bytes: usize) {}
448 /// Just process the data in the last committed space for future matching.
449 fn skip_matching(&mut self);
450 /// Hint-aware skip path used internally to thread a precomputed block
451 /// incompressibility verdict to matcher backends.
452 ///
453 /// Default implementation preserves backwards compatibility for external
454 /// custom matchers by delegating to [`skip_matching`](Self::skip_matching).
455 fn skip_matching_with_hint(&mut self, _incompressible_hint: Option<bool>) {
456 self.skip_matching();
457 }
458 /// Process the data in the last committed space for future matching AND generate matches for the data
459 fn start_matching(&mut self, handle_sequence: impl for<'a> FnMut(Sequence<'a>));
460 /// Reset this matcher so it can be used for the next new frame
461 fn reset(&mut self, level: CompressionLevel);
462 /// Provide a hint about the total uncompressed size for the next frame.
463 ///
464 /// Implementations may use this to select smaller hash tables and windows
465 /// for small inputs, matching the C zstd source-size-class behavior.
466 /// Called before [`reset`](Self::reset) when the caller knows the input
467 /// size (e.g. from pledged content size or file metadata).
468 ///
469 /// The default implementation is a no-op for custom matchers and
470 /// test stubs. The built-in runtime matcher (`MatchGeneratorDriver`)
471 /// overrides this hook and applies the hint during level resolution.
472 fn set_source_size_hint(&mut self, _size: u64) {}
473 /// Hint the sizes of the dictionary that will be primed into the next
474 /// frame. The built-in runtime matcher resolves the frame's cParams from
475 /// the dictionary's CDict tier (upstream `ZSTD_createCDict`, keyed by the
476 /// serialized size) and sizes its dictionary tables from the content.
477 /// Default no-op for custom matchers and test stubs; consumed at the next
478 /// [`reset`](Self::reset).
479 fn set_dictionary_size_hint(&mut self, _sizes: DictionarySizes) {}
480 /// Drop any per-frame fine-grained parameter overrides installed via
481 /// the public parameter API, reverting to plain level-based geometry
482 /// at the next [`reset`](Self::reset). Called by
483 /// [`FrameCompressor::set_compression_level`](crate::encoding::FrameCompressor::set_compression_level)
484 /// so switching back to a bare level after a customized frame does not
485 /// keep the old overrides sticky. Default no-op for custom matchers.
486 fn clear_param_overrides(&mut self) {}
487 /// Prime matcher state with dictionary history before compressing the next frame.
488 /// Default implementation is a no-op for custom matchers that do not support this.
489 fn prime_with_dictionary(&mut self, _dict_content: &[u8], _offset_hist: [u32; 3]) {}
490 /// Whether the most recent [`reset`](Self::reset) re-borrowed a resident
491 /// attach-mode dictionary (kept the dict bytes + cached index in place).
492 /// When `true` the caller MUST skip [`Self::prime_with_dictionary`] and only
493 /// reapply the offset history via [`Self::reapply_resident_dictionary`].
494 fn dictionary_is_resident(&self) -> bool {
495 false
496 }
497 /// Reapply the dictionary's offset history to a re-borrowed frame — the cheap
498 /// tail of priming, without the dict commit / re-index. Default no-op.
499 fn reapply_resident_dictionary(&mut self, _offset_hist: [u32; 3]) {}
500 /// CDict-equivalent fast path for repeated frames sharing one dictionary.
501 /// Restore the matcher state captured by [`Self::capture_primed_dictionary`]
502 /// at the SAME level (a table copy) instead of re-running
503 /// [`Self::prime_with_dictionary`] (which re-hashes every dictionary
504 /// position). Returns `true` when a matching snapshot was restored;
505 /// `false` (the default) means the caller must prime then capture.
506 fn restore_primed_dictionary(&mut self, _level: CompressionLevel) -> bool {
507 false
508 }
509 /// Snapshot the post-prime matcher state for the given level so later
510 /// frames can [`Self::restore_primed_dictionary`] it. Default no-op.
511 fn capture_primed_dictionary(&mut self, _level: CompressionLevel) {}
512 /// Drop any captured prime snapshot (dictionary or level changed).
513 /// Default no-op.
514 fn invalidate_primed_dictionary(&mut self) {}
515 /// Seed matcher cost model with dictionary entropy tables before the next frame.
516 /// Default implementation is a no-op for custom matchers.
517 fn seed_dictionary_entropy(
518 &mut self,
519 _huff: Option<&crate::huff0::huff0_encoder::HuffmanTable>,
520 _ll: Option<&crate::fse::fse_encoder::FSETable>,
521 _ml: Option<&crate::fse::fse_encoder::FSETable>,
522 _of: Option<&crate::fse::fse_encoder::FSETable>,
523 ) {
524 }
525 /// Returns whether this matcher can consume dictionary priming state and produce
526 /// dictionary-dependent sequences. Defaults to `false` for custom matchers.
527 fn supports_dictionary_priming(&self) -> bool {
528 false
529 }
530 /// Whether a sample of `block` hashes to a match in an attached dictionary.
531 /// The raw-fast-path uses this to avoid skipping the scan on a block that
532 /// looks incompressible but compresses against the dictionary (an external
533 /// match the block's own content cannot reveal). Defaults to `false` for
534 /// custom matchers (and the no-dict case), leaving the content-only verdict.
535 fn block_samples_match_dict(&self, _block: &[u8]) -> bool {
536 false
537 }
538 /// Heap bytes this matcher's allocations hold (tables, history, scratch),
539 /// excluding the inline struct itself. Lets a context report its true
540 /// footprint via `ZSTD_sizeof_CCtx`. Defaults to `0` for custom matchers.
541 fn heap_size(&self) -> usize {
542 0
543 }
544 /// The size of the window the decoder will need to execute all sequences produced by this matcher.
545 ///
546 /// Must return a positive (non-zero) value; returning 0 causes
547 /// [`StreamingEncoder`] to reject the first write with an invalid-input error
548 /// (`InvalidInput` with `std`, `Other` with `no_std`).
549 ///
550 /// Must remain stable for the lifetime of a frame.
551 /// It may change only after `reset()` is called for the next frame
552 /// (for example because the compression level changed).
553 fn window_size(&self) -> u64;
554}
555
556#[derive(PartialEq, Eq, Debug)]
557/// Sequences that a [`Matcher`] can produce
558pub enum Sequence<'data> {
559 /// Is encoded as a sequence for the decoder sequence execution.
560 ///
561 /// First the literals will be copied to the decoded data,
562 /// then `match_len` bytes are copied from `offset` bytes back in the decoded data
563 Triple {
564 literals: &'data [u8],
565 offset: usize,
566 match_len: usize,
567 },
568 /// This is returned as the last sequence in a block
569 ///
570 /// These literals will just be copied at the end of the sequence execution by the decoder
571 Literals { literals: &'data [u8] },
572}
573
574#[cfg(test)]
575mod compress_bound_tests;