Skip to main content

ferrox_core/
kv_signature.rs

1//! Compatibility marking for stored KV blocks: whether a block found
2//! in a cache may be *used*, as opposed to merely found.
3//!
4//! A [`BlockHash`](crate::kv_block::BlockHash) says which tokens a block
5//! covers. It says nothing about the shape of the tensors stored under
6//! it -- how many layers, how many KV heads, what head dimension, what
7//! dtype, how many token positions are really there. A cache that
8//! survives a restart will eventually be read by a process configured
9//! differently from the one that wrote it, and reading a block whose
10//! layout does not match is not a miss: it is silently wrong attention
11//! state, which produces confident wrong tokens.
12//!
13//! The one rule this module exists to enforce:
14//!
15//! > **A signature is derived from the block's own payload, never from
16//! > the manager's expectation, and an unmarked block is rejected
17//! > rather than trusted.**
18//!
19//! So [`CacheSignature::from_payload`] measures the tensors; it takes no
20//! "expected" shape to fill gaps with. [`UnverifiedBlock::verify`] then
21//! makes three separate checks, in order:
22//!
23//! 1. the block carries a signature at all ([`SignatureError::Unmarked`]
24//!    otherwise -- absence is never treated as agreement);
25//! 2. the recorded signature matches what the payload actually is
26//!    ([`SignatureError::PayloadMismatch`]) -- a stamp may not vouch for
27//!    a width the payload does not have;
28//! 3. only then, that this shape is the shape the reader wants
29//!    ([`SignatureError::Incompatible`]).
30//!
31//! Step 2 is the one that is easy to skip and expensive to skip: without
32//! it, "the stamp says 32 heads" and "there are 32 heads" are different
33//! claims that a cache would be treating as one.
34//!
35//! # The block layout is part of the signature
36//!
37//! A signature also carries the [`BlockLayout`] the block was cut
38//! under: its block size, and the sliding window that block size had to
39//! divide (see [`kv_swa`](crate::kv_swa) for why it must). Both are
40//! here rather than checked once at startup because a *durable* cache
41//! outlives the configuration that filled it. A block written by a
42//! build running gpt-oss at window 128 must not be handed to a build
43//! running it at window 256, and the failure mode if it were is not a
44//! crash -- it is an attention mask silently wider or narrower than the
45//! model's.
46//!
47//! `block_size` is payload-checkable and is checked: a stored block is
48//! exactly one whole block (`kv_block::chain` refuses to name a partial
49//! tail), so a stamp claiming `block_size = 64` over a 48-token payload
50//! is refused the same way a stamp claiming 32 heads over 16 is. The
51//! window is not provable from tensors -- like `model` -- so it is
52//! carried and compared against the reader's expectation.
53
54use crate::cache::KvCache;
55use crate::kv_swa::{BlockLayout, BlockLayoutError};
56
57/// Layout version of a stored block payload. A reader accepts only the
58/// versions in [`READABLE_FORMAT_VERSIONS`]; anything else is rejected
59/// rather than guessed at.
60pub const BLOCK_FORMAT_VERSION: u32 = 2;
61
62/// Versions this build can read. Kept explicit (rather than `<=
63/// BLOCK_FORMAT_VERSION`) so dropping support for an old layout is a
64/// deliberate edit and not an accident of arithmetic.
65///
66/// Version 1 is deliberately **not** readable. Its header had no block
67/// size and no sliding window, so a v1 block cannot say what layout it
68/// was cut under -- and this module's rule is that absence is never
69/// agreement. Reading one would mean assuming it happened to be
70/// aligned, which is the exact assumption `kv_swa` exists to refuse.
71pub const READABLE_FORMAT_VERSIONS: &[u32] = &[2];
72
73/// Element type of the stored K/V tensors.
74///
75/// Only `F32` exists today, because [`KvCache`] stores `Vec<f32>`. The
76/// enum is here so a future f16/quantized KV tier changes the signature
77/// -- and therefore invalidates blocks written by an f32 build -- rather
78/// than reinterpreting their bytes.
79#[derive(Clone, Copy, Debug, PartialEq, Eq)]
80pub enum KvDtype {
81    F32,
82}
83
84impl KvDtype {
85    pub fn as_str(self) -> &'static str {
86        match self {
87            KvDtype::F32 => "f32",
88        }
89    }
90}
91
92/// What a block's payload *is*: the shape any reader must match to use
93/// it. Construct it from a payload with [`Self::from_payload`], or as a
94/// reader's requirement with [`Self::expected`].
95#[derive(Clone, Debug, PartialEq, Eq)]
96pub struct CacheSignature {
97    pub format_version: u32,
98    /// Identifies the weights the KV state was computed under. The one
99    /// field no payload can prove about itself -- which is exactly why
100    /// it is checked against the reader's expectation in step 3.
101    pub model: String,
102    pub n_layers: usize,
103    pub n_kv_heads: usize,
104    pub head_dim: usize,
105    pub dtype: KvDtype,
106    /// Token positions actually stored, per layer. Equal to
107    /// `layout.block_size()` for any block this module will stamp or
108    /// verify.
109    pub tokens: usize,
110    /// How the sequence was cut into blocks, and the sliding window
111    /// that cut had to line up with. See the module note.
112    pub layout: BlockLayout,
113}
114
115impl CacheSignature {
116    /// Derives a signature by *measuring* `layers`. There is
117    /// deliberately no parameter to fill a gap from: every shape field
118    /// except `model` and the sliding window comes from the tensors
119    /// themselves, and even the declared `layout`'s block size is
120    /// checked against the depth the tensors actually have.
121    ///
122    /// Fails if the payload cannot describe itself coherently: no
123    /// layers at all, layers that disagree with each other, a layer
124    /// whose buffers do not match its own declared shape, or a token
125    /// depth that is not the block size the layout claims.
126    pub fn from_payload(
127        model: &str,
128        layout: BlockLayout,
129        layers: &[KvCache],
130    ) -> Result<Self, SignatureError> {
131        let first = layers.first().ok_or(SignatureError::EmptyPayload)?;
132        let n_kv_heads = first.n_kv_heads;
133        let head_dim = first.head_dim;
134        if n_kv_heads == 0 || head_dim == 0 {
135            return Err(SignatureError::DegenerateLayer {
136                layer: 0,
137                n_kv_heads,
138                head_dim,
139            });
140        }
141        let per_token = n_kv_heads * head_dim;
142        let tokens = measure_layer(0, first, per_token)?;
143
144        for (index, layer) in layers.iter().enumerate().skip(1) {
145            if layer.n_kv_heads != n_kv_heads || layer.head_dim != head_dim {
146                return Err(SignatureError::RaggedPayload {
147                    layer: index,
148                    field: "layer shape",
149                    expected: format!("{n_kv_heads}x{head_dim}"),
150                    found: format!("{}x{}", layer.n_kv_heads, layer.head_dim),
151                });
152            }
153            let layer_tokens = measure_layer(index, layer, per_token)?;
154            if layer_tokens != tokens {
155                return Err(SignatureError::RaggedPayload {
156                    layer: index,
157                    field: "token count",
158                    expected: tokens.to_string(),
159                    found: layer_tokens.to_string(),
160                });
161            }
162        }
163
164        // The block size is a claim like any other, and this one the
165        // payload can settle: a stored block is one whole block.
166        if tokens != layout.block_size() {
167            return Err(SignatureError::BlockSizeMismatch {
168                block_size: layout.block_size(),
169                tokens,
170            });
171        }
172
173        Ok(CacheSignature {
174            format_version: BLOCK_FORMAT_VERSION,
175            model: model.to_string(),
176            n_layers: layers.len(),
177            n_kv_heads,
178            head_dim,
179            dtype: KvDtype::F32,
180            tokens,
181            layout,
182        })
183    }
184
185    /// A reader's requirement: the shape this process would compute
186    /// itself. Never stamped onto a block -- only compared against one.
187    pub fn expected(
188        model: &str,
189        layout: BlockLayout,
190        n_layers: usize,
191        n_kv_heads: usize,
192        head_dim: usize,
193        tokens: usize,
194    ) -> Self {
195        CacheSignature {
196            format_version: BLOCK_FORMAT_VERSION,
197            model: model.to_string(),
198            n_layers,
199            n_kv_heads,
200            head_dim,
201            dtype: KvDtype::F32,
202            tokens,
203            layout,
204        }
205    }
206
207    /// Field-by-field comparison, naming the first field that differs
208    /// so an operator learns *what* changed rather than "cache miss".
209    fn compare(
210        &self,
211        other: &CacheSignature,
212        mismatch: fn(&'static str, String, String) -> SignatureError,
213    ) -> Result<(), SignatureError> {
214        if self.format_version != other.format_version {
215            return Err(mismatch(
216                "format_version",
217                self.format_version.to_string(),
218                other.format_version.to_string(),
219            ));
220        }
221        if self.model != other.model {
222            return Err(mismatch("model", self.model.clone(), other.model.clone()));
223        }
224        if self.n_layers != other.n_layers {
225            return Err(mismatch(
226                "n_layers",
227                self.n_layers.to_string(),
228                other.n_layers.to_string(),
229            ));
230        }
231        if self.n_kv_heads != other.n_kv_heads {
232            return Err(mismatch(
233                "n_kv_heads",
234                self.n_kv_heads.to_string(),
235                other.n_kv_heads.to_string(),
236            ));
237        }
238        if self.head_dim != other.head_dim {
239            return Err(mismatch(
240                "head_dim",
241                self.head_dim.to_string(),
242                other.head_dim.to_string(),
243            ));
244        }
245        if self.dtype != other.dtype {
246            return Err(mismatch(
247                "dtype",
248                self.dtype.as_str().to_string(),
249                other.dtype.as_str().to_string(),
250            ));
251        }
252        if self.tokens != other.tokens {
253            return Err(mismatch(
254                "tokens",
255                self.tokens.to_string(),
256                other.tokens.to_string(),
257            ));
258        }
259        if self.layout.block_size() != other.layout.block_size() {
260            return Err(mismatch(
261                "block_size",
262                self.layout.block_size().to_string(),
263                other.layout.block_size().to_string(),
264            ));
265        }
266        if self.layout.sliding_window() != other.layout.sliding_window() {
267            return Err(mismatch(
268                "sliding_window",
269                describe_window(self.layout.sliding_window()),
270                describe_window(other.layout.sliding_window()),
271            ));
272        }
273        Ok(())
274    }
275}
276
277/// Renders a window for an error message. `None` is spelled out rather
278/// than printed as an empty string: "this build uses no sliding window"
279/// and "this build did not say" must not look the same in a log.
280fn describe_window(window: Option<usize>) -> String {
281    match window {
282        Some(w) => w.to_string(),
283        None => "none (full causal)".to_string(),
284    }
285}
286
287/// Measures one layer, rejecting a layer whose buffers disagree with
288/// its own declared `seq_len` -- `seq_len` is a claim, `k.len()` is the
289/// evidence.
290fn measure_layer(index: usize, layer: &KvCache, per_token: usize) -> Result<usize, SignatureError> {
291    if !layer.k.len().is_multiple_of(per_token) {
292        return Err(SignatureError::RaggedPayload {
293            layer: index,
294            field: "k length",
295            expected: format!("a multiple of {per_token}"),
296            found: layer.k.len().to_string(),
297        });
298    }
299    if layer.v.len() != layer.k.len() {
300        return Err(SignatureError::RaggedPayload {
301            layer: index,
302            field: "v length",
303            expected: layer.k.len().to_string(),
304            found: layer.v.len().to_string(),
305        });
306    }
307    let tokens = layer.k.len() / per_token;
308    // Positions against rows. They are equal for anything this codec
309    // can currently produce, and a payload where they disagree is
310    // ragged rather than windowed. When a store learns to evict (#61)
311    // a restored windowed layer will legitimately carry more positions
312    // than rows, and THIS CHECK IS WHERE THAT HAS TO BE TAUGHT: the
313    // codec will need to serialize the position count separately
314    // instead of deriving it from the byte length.
315    if layer.positions() != tokens {
316        return Err(SignatureError::RaggedPayload {
317            layer: index,
318            field: "positions",
319            expected: tokens.to_string(),
320            found: layer.positions().to_string(),
321        });
322    }
323    Ok(tokens)
324}
325
326/// Why a stored block was refused. Every variant is a refusal to guess.
327#[derive(Clone, Debug, PartialEq, Eq)]
328pub enum SignatureError {
329    /// The block carries no signature. Not treated as "probably fine":
330    /// an unmarked block was written by something whose layout is
331    /// unknown, which is precisely the case that must not be trusted.
332    Unmarked,
333    /// A block with no layers describes nothing and can vouch for
334    /// nothing.
335    EmptyPayload,
336    /// A layer with no heads or zero head dimension.
337    DegenerateLayer {
338        layer: usize,
339        n_kv_heads: usize,
340        head_dim: usize,
341    },
342    /// The payload does not agree with itself: layers of different
343    /// shapes or lengths, or a layer whose buffers contradict its own
344    /// `seq_len`.
345    RaggedPayload {
346        layer: usize,
347        field: &'static str,
348        expected: String,
349        found: String,
350    },
351    /// The recorded signature claims something the payload is not. The
352    /// stamp is wrong (or was written by a build with a different
353    /// layout); the payload is the truth.
354    PayloadMismatch {
355        field: &'static str,
356        recorded: String,
357        actual: String,
358    },
359    /// The payload is coherent and honestly stamped, but it is not what
360    /// this reader needs -- a different model, a config change, a
361    /// different block size.
362    Incompatible {
363        field: &'static str,
364        expected: String,
365        found: String,
366    },
367    /// The stamp claims a block size the payload's token depth is not.
368    /// A stored block is exactly one whole block, so these are the same
369    /// number or the stamp is wrong.
370    BlockSizeMismatch { block_size: usize, tokens: usize },
371    /// The block layout itself is not usable -- most importantly, a
372    /// block size that does not divide the sliding window. See
373    /// [`kv_swa`](crate::kv_swa).
374    BadLayout(BlockLayoutError),
375    /// Written by a build whose payload layout this one cannot read.
376    UnsupportedFormat {
377        found: u32,
378        readable: &'static [u32],
379    },
380}
381
382impl From<BlockLayoutError> for SignatureError {
383    fn from(err: BlockLayoutError) -> Self {
384        SignatureError::BadLayout(err)
385    }
386}
387
388impl std::fmt::Display for SignatureError {
389    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
390        match self {
391            SignatureError::Unmarked => write!(
392                f,
393                "KV block carries no cache signature; refusing to trust an unmarked block"
394            ),
395            SignatureError::EmptyPayload => {
396                write!(f, "KV block has no layers; nothing to verify")
397            }
398            SignatureError::DegenerateLayer {
399                layer,
400                n_kv_heads,
401                head_dim,
402            } => write!(
403                f,
404                "KV block layer {layer} is degenerate: {n_kv_heads} kv heads x {head_dim} head dim"
405            ),
406            SignatureError::RaggedPayload {
407                layer,
408                field,
409                expected,
410                found,
411            } => write!(
412                f,
413                "KV block payload is inconsistent at layer {layer}: {field} is {found}, expected {expected}"
414            ),
415            SignatureError::PayloadMismatch {
416                field,
417                recorded,
418                actual,
419            } => write!(
420                f,
421                "KV block signature vouches for {field}={recorded} but its payload has {field}={actual}"
422            ),
423            SignatureError::Incompatible {
424                field,
425                expected,
426                found,
427            } => write!(
428                f,
429                "KV block is incompatible: {field} is {found}, this server needs {expected}"
430            ),
431            SignatureError::BlockSizeMismatch { block_size, tokens } => write!(
432                f,
433                "KV block signature declares a block size of {block_size} but its payload holds \
434                 {tokens} token positions; a stored block is exactly one whole block"
435            ),
436            SignatureError::BadLayout(err) => write!(f, "KV block layout is unusable: {err}"),
437            SignatureError::UnsupportedFormat { found, readable } => write!(
438                f,
439                "KV block format version {found} is not readable by this build (readable: {readable:?})"
440            ),
441        }
442    }
443}
444
445impl std::error::Error for SignatureError {}
446
447/// A block whose signature has been verified against its own payload
448/// and against the reader's expectation. Only way to get one is
449/// [`KvBlock::stamp`] (writing) or [`UnverifiedBlock::verify`]
450/// (reading), so holding one is itself the proof.
451pub struct KvBlock {
452    signature: CacheSignature,
453    layers: Vec<KvCache>,
454}
455
456/// Summarizes rather than dumping tensors: a block's `Debug` is for a
457/// log line or a failing assertion, and printing every f32 in a KV
458/// block helps nobody.
459impl std::fmt::Debug for KvBlock {
460    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
461        f.debug_struct("KvBlock")
462            .field("signature", &self.signature)
463            .field("layers", &self.layers.len())
464            .finish()
465    }
466}
467
468impl std::fmt::Debug for UnverifiedBlock {
469    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
470        f.debug_struct("UnverifiedBlock")
471            .field("signature", &self.signature)
472            .field("layers", &self.layers.len())
473            .finish()
474    }
475}
476
477impl KvBlock {
478    /// Stamps a block from its own payload. The caller supplies the
479    /// model identity and the layout the block was cut under; every
480    /// shape field is measured, and the layout's block size is checked
481    /// against the depth measured.
482    pub fn stamp(
483        model: &str,
484        layout: BlockLayout,
485        layers: Vec<KvCache>,
486    ) -> Result<Self, SignatureError> {
487        let signature = CacheSignature::from_payload(model, layout, &layers)?;
488        Ok(KvBlock { signature, layers })
489    }
490
491    /// The layout this block was cut under.
492    pub fn layout(&self) -> BlockLayout {
493        self.signature.layout
494    }
495
496    pub fn signature(&self) -> &CacheSignature {
497        &self.signature
498    }
499
500    pub fn tokens(&self) -> usize {
501        self.signature.tokens
502    }
503
504    pub fn layers(&self) -> &[KvCache] {
505        &self.layers
506    }
507
508    pub fn into_layers(self) -> Vec<KvCache> {
509        self.layers
510    }
511}
512
513/// A block as it comes back from an untrusted source: a disk tier, a
514/// shared cache, or a build that is not this one. `signature` is an
515/// `Option` on purpose -- "no signature recorded" is a state a real
516/// stored block can be in, and it must be representable so it can be
517/// refused.
518pub struct UnverifiedBlock {
519    pub signature: Option<CacheSignature>,
520    pub layers: Vec<KvCache>,
521}
522
523impl UnverifiedBlock {
524    pub fn new(signature: Option<CacheSignature>, layers: Vec<KvCache>) -> Self {
525        UnverifiedBlock { signature, layers }
526    }
527
528    /// Verifies in the order the module doc describes: marked, honest,
529    /// then compatible. `expected` is used only in the last step -- it
530    /// never contributes a field to the signature being checked.
531    pub fn verify(self, expected: &CacheSignature) -> Result<KvBlock, SignatureError> {
532        let recorded = self.signature.ok_or(SignatureError::Unmarked)?;
533        if !READABLE_FORMAT_VERSIONS.contains(&recorded.format_version) {
534            return Err(SignatureError::UnsupportedFormat {
535                found: recorded.format_version,
536                readable: READABLE_FORMAT_VERSIONS,
537            });
538        }
539        // Measured from the payload. `recorded.model` is carried over
540        // because no payload can prove which weights produced it; the
541        // model check happens against `expected` below, where a wrong
542        // model is caught as an incompatibility.
543        // `recorded.layout` is carried over for the same reason
544        // `recorded.model` is: the reader's expectation must not get a
545        // vote before the payload has been checked against the stamp.
546        // Carrying it is not trusting it -- `from_payload` rejects a
547        // block size the token depth contradicts, and the window is
548        // settled against `expected` below.
549        let actual = CacheSignature::from_payload(&recorded.model, recorded.layout, &self.layers)?;
550        recorded.compare(&actual, |field, recorded, actual| {
551            SignatureError::PayloadMismatch {
552                field,
553                recorded,
554                actual,
555            }
556        })?;
557        expected.compare(&actual, |field, expected, found| {
558            SignatureError::Incompatible {
559                field,
560                expected,
561                found,
562            }
563        })?;
564        Ok(KvBlock {
565            signature: actual,
566            layers: self.layers,
567        })
568    }
569}
570
571#[cfg(test)]
572mod tests {
573    use super::*;
574
575    fn layer(n_kv_heads: usize, head_dim: usize, tokens: usize) -> KvCache {
576        let mut cache = KvCache::new(n_kv_heads, head_dim);
577        let step = vec![0.5f32; n_kv_heads * head_dim];
578        for _ in 0..tokens {
579            cache.push(&step, &step).expect("unpooled push cannot fail");
580        }
581        cache
582    }
583
584    fn payload(n_layers: usize, n_kv_heads: usize, head_dim: usize, tokens: usize) -> Vec<KvCache> {
585        (0..n_layers)
586            .map(|_| layer(n_kv_heads, head_dim, tokens))
587            .collect()
588    }
589
590    /// A full-causal layout whose block size is the payload depth --
591    /// what every test that is not about SWA wants.
592    fn flat(block_size: usize) -> BlockLayout {
593        BlockLayout::full_attention(block_size).expect("positive block size")
594    }
595
596    /// The core rule, in its positive form: every shape field comes
597    /// from the tensors. `stamp` is given a model name and nothing else.
598    #[test]
599    fn signature_is_measured_from_the_payload() {
600        let block = KvBlock::stamp("model-a", flat(4), payload(3, 2, 8, 4)).expect("stamp");
601        let sig = block.signature();
602        assert_eq!(sig.n_layers, 3);
603        assert_eq!(sig.n_kv_heads, 2);
604        assert_eq!(sig.head_dim, 8);
605        assert_eq!(sig.tokens, 4);
606        assert_eq!(sig.dtype, KvDtype::F32);
607        assert_eq!(sig.format_version, BLOCK_FORMAT_VERSION);
608        assert_eq!(block.tokens(), 4);
609        assert_eq!(block.layers().len(), 3);
610    }
611
612    #[test]
613    fn a_stamped_block_round_trips_through_verification() {
614        let layers = payload(3, 2, 8, 4);
615        let signature =
616            CacheSignature::from_payload("model-a", flat(4), &layers).expect("signature");
617        let expected = CacheSignature::expected("model-a", flat(4), 3, 2, 8, 4);
618        let block = UnverifiedBlock::new(Some(signature), layers)
619            .verify(&expected)
620            .expect("a block that is what it says it is must verify");
621        assert_eq!(block.layers().len(), 3);
622        assert_eq!(block.into_layers().len(), 3);
623    }
624
625    /// Absence of a signature is not agreement. This is the difference
626    /// between a persistent cache and silent corruption after a config
627    /// change: an unmarked block came from something whose layout is
628    /// unknown by definition.
629    #[test]
630    fn an_unmarked_block_is_rejected_not_trusted() {
631        let expected = CacheSignature::expected("model-a", flat(4), 3, 2, 8, 4);
632        let err = UnverifiedBlock::new(None, payload(3, 2, 8, 4))
633            .verify(&expected)
634            .expect_err("an unmarked block must be refused");
635        assert_eq!(err, SignatureError::Unmarked);
636    }
637
638    /// The rule the plan states as "a signature must never vouch for a
639    /// width the payload does not have". Here the stamp is exactly what
640    /// the reader expects -- and the payload is not. A cache that
641    /// trusted the stamp (or, equivalently, stamped from the manager's
642    /// expectation) would hand back tensors of the wrong width and
643    /// produce confident wrong tokens.
644    #[test]
645    fn a_signature_that_overstates_its_payload_is_rejected() {
646        let expected = CacheSignature::expected("model-a", flat(4), 3, 2, 16, 4);
647        let mut lying = expected.clone();
648        assert_eq!(lying.head_dim, 16);
649        let err = UnverifiedBlock::new(Some(lying.clone()), payload(3, 2, 8, 4))
650            .verify(&expected)
651            .expect_err("stamp claims head_dim 16 over an 8-wide payload");
652        assert_eq!(
653            err,
654            SignatureError::PayloadMismatch {
655                field: "head_dim",
656                recorded: "16".into(),
657                actual: "8".into(),
658            }
659        );
660
661        // Same shape, overstated depth: 8 token positions claimed over
662        // a 4-position payload.
663        lying.head_dim = 8;
664        lying.tokens = 8;
665        let expected = CacheSignature::expected("model-a", flat(8), 3, 2, 8, 8);
666        let err = UnverifiedBlock::new(Some(lying.clone()), payload(3, 2, 8, 4))
667            .verify(&expected)
668            .expect_err("stamp claims 8 tokens over a 4-token payload");
669        assert_eq!(
670            err,
671            SignatureError::PayloadMismatch {
672                field: "tokens",
673                recorded: "8".into(),
674                actual: "4".into(),
675            }
676        );
677
678        // And overstated layer count.
679        lying.tokens = 4;
680        lying.n_layers = 4;
681        let expected = CacheSignature::expected("model-a", flat(4), 4, 2, 8, 4);
682        let err = UnverifiedBlock::new(Some(lying), payload(3, 2, 8, 4))
683            .verify(&expected)
684            .expect_err("stamp claims 4 layers over a 3-layer payload");
685        assert_eq!(
686            err,
687            SignatureError::PayloadMismatch {
688                field: "n_layers",
689                recorded: "4".into(),
690                actual: "3".into(),
691            }
692        );
693    }
694
695    /// A payload that lies to itself: `seq_len` says 4, the buffers
696    /// hold 3. `seq_len` is a claim; `k.len()` is the evidence.
697    #[test]
698    fn a_layer_whose_seq_len_contradicts_its_buffers_is_rejected() {
699        let mut layers = payload(2, 2, 8, 4);
700        // The contradiction is the thing under test.
701        layers[1].force_positions_for_test(7);
702        let err = CacheSignature::from_payload("model-a", flat(4), &layers)
703            .expect_err("seq_len must be verified, not believed");
704        assert_eq!(
705            err,
706            SignatureError::RaggedPayload {
707                layer: 1,
708                field: "positions",
709                expected: "4".into(),
710                found: "7".into(),
711            }
712        );
713    }
714
715    #[test]
716    fn a_ragged_payload_is_rejected() {
717        let mut layers = payload(3, 2, 8, 4);
718        layers[2] = layer(2, 4, 4);
719        let err = CacheSignature::from_payload("model-a", flat(4), &layers)
720            .expect_err("shape disagreement");
721        assert!(matches!(
722            err,
723            SignatureError::RaggedPayload {
724                layer: 2,
725                field: "layer shape",
726                ..
727            }
728        ));
729
730        let mut layers = payload(3, 2, 8, 4);
731        layers[1] = layer(2, 8, 3);
732        let err = CacheSignature::from_payload("model-a", flat(4), &layers)
733            .expect_err("depth disagreement");
734        assert!(matches!(
735            err,
736            SignatureError::RaggedPayload {
737                layer: 1,
738                field: "token count",
739                ..
740            }
741        ));
742
743        let mut layers = payload(2, 2, 8, 4);
744        layers[0].v.truncate(8);
745        let err = CacheSignature::from_payload("model-a", flat(4), &layers)
746            .expect_err("k/v disagreement");
747        assert!(matches!(
748            err,
749            SignatureError::RaggedPayload {
750                layer: 0,
751                field: "v length",
752                ..
753            }
754        ));
755    }
756
757    #[test]
758    fn an_empty_payload_is_rejected() {
759        assert_eq!(
760            CacheSignature::from_payload("model-a", flat(4), &[])
761                .expect_err("nothing to vouch for"),
762            SignatureError::EmptyPayload
763        );
764    }
765
766    /// An honest block of the wrong shape is a *miss*, reported as an
767    /// incompatibility naming the field that changed -- not a
768    /// corruption, and not a silent fallback.
769    #[test]
770    fn an_honest_block_from_a_different_config_is_incompatible() {
771        let layers = payload(3, 2, 8, 4);
772        let signature =
773            CacheSignature::from_payload("model-a", flat(4), &layers).expect("signature");
774        let err = UnverifiedBlock::new(Some(signature.clone()), layers)
775            .verify(&CacheSignature::expected("model-b", flat(4), 3, 2, 8, 4))
776            .expect_err("a different model must not share KV state");
777        assert_eq!(
778            err,
779            SignatureError::Incompatible {
780                field: "model",
781                expected: "model-b".into(),
782                found: "model-a".into(),
783            }
784        );
785
786        let layers = payload(3, 2, 8, 4);
787        let err = UnverifiedBlock::new(Some(signature), layers)
788            .verify(&CacheSignature::expected("model-a", flat(4), 3, 4, 8, 4))
789            .expect_err("a different KV head count must not be reused");
790        assert_eq!(
791            err,
792            SignatureError::Incompatible {
793                field: "n_kv_heads",
794                expected: "4".into(),
795                found: "2".into(),
796            }
797        );
798    }
799
800    #[test]
801    fn an_unreadable_format_version_is_rejected() {
802        let layers = payload(2, 2, 8, 4);
803        let mut signature =
804            CacheSignature::from_payload("model-a", flat(4), &layers).expect("signature");
805        signature.format_version = 99;
806        let err = UnverifiedBlock::new(Some(signature), layers)
807            .verify(&CacheSignature::expected("model-a", flat(4), 2, 2, 8, 4))
808            .expect_err("an unknown layout must not be guessed at");
809        assert_eq!(
810            err,
811            SignatureError::UnsupportedFormat {
812                found: 99,
813                readable: READABLE_FORMAT_VERSIONS,
814            }
815        );
816    }
817
818    /// The `kv-swa-block-alignment` invariant at the signature layer.
819    ///
820    /// Two builds of the same model, same tensors, same block size --
821    /// one configured with a 128-token sliding window and one with 256.
822    /// The payload cannot tell them apart, which is precisely why the
823    /// window is stamped: without this check the second build reads the
824    /// first build's blocks back and runs a mask the model never had.
825    #[test]
826    fn a_block_written_under_a_different_window_is_refused_not_reused() {
827        let layout_128 = BlockLayout::new(4, Some(128)).expect("4 divides 128");
828        let layout_256 = BlockLayout::new(4, Some(256)).expect("4 divides 256");
829        let layers = payload(3, 2, 8, 4);
830        let signature =
831            CacheSignature::from_payload("model-a", layout_128, &layers).expect("signature");
832
833        let err = UnverifiedBlock::new(Some(signature.clone()), layers)
834            .verify(&CacheSignature::expected("model-a", layout_256, 3, 2, 8, 4))
835            .expect_err("a window change must invalidate the block, not be ignored");
836        assert_eq!(
837            err,
838            SignatureError::Incompatible {
839                field: "sliding_window",
840                expected: "256".into(),
841                found: "128".into(),
842            }
843        );
844
845        // And the same block under the same window still verifies --
846        // the check must invalidate on change, not on principle.
847        let layers = payload(3, 2, 8, 4);
848        UnverifiedBlock::new(Some(signature), layers)
849            .verify(&CacheSignature::expected("model-a", layout_128, 3, 2, 8, 4))
850            .expect("unchanged config must still hit");
851    }
852
853    /// Turning SWA off (or on) is a config change of exactly the same
854    /// kind, and `None` must not read as "matches anything".
855    #[test]
856    fn a_full_causal_reader_will_not_take_a_sliding_window_block() {
857        let sliding = BlockLayout::new(4, Some(128)).expect("aligned");
858        let layers = payload(2, 2, 8, 4);
859        let signature =
860            CacheSignature::from_payload("model-a", sliding, &layers).expect("signature");
861        let err = UnverifiedBlock::new(Some(signature), layers)
862            .verify(&CacheSignature::expected("model-a", flat(4), 2, 2, 8, 4))
863            .expect_err("no window and a 128 window are different configurations");
864        assert_eq!(
865            err,
866            SignatureError::Incompatible {
867                field: "sliding_window",
868                expected: "none (full causal)".into(),
869                found: "128".into(),
870            }
871        );
872    }
873
874    /// A block cut at a different block size cannot be spliced into a
875    /// sequence cut at this one: the hash chain would not line up and
876    /// the eviction unit would not either.
877    #[test]
878    fn a_block_cut_at_a_different_block_size_is_incompatible() {
879        let layers = payload(2, 2, 8, 4);
880        let signature =
881            CacheSignature::from_payload("model-a", flat(4), &layers).expect("signature");
882        let err = UnverifiedBlock::new(Some(signature), layers)
883            .verify(&CacheSignature::expected("model-a", flat(2), 2, 2, 8, 4))
884            .expect_err("a 4-token block is not a 2-token block");
885        assert_eq!(
886            err,
887            SignatureError::Incompatible {
888                field: "block_size",
889                expected: "2".into(),
890                found: "4".into(),
891            }
892        );
893    }
894
895    /// `block_size` is the one new field the payload can settle, so it
896    /// is settled: a stamp claiming 8-token blocks over a 4-token
897    /// payload is a lying stamp, exactly like an overstated head_dim.
898    #[test]
899    fn a_stamp_may_not_claim_a_block_size_the_payload_lacks() {
900        let err = KvBlock::stamp("model-a", flat(8), payload(2, 2, 8, 4))
901            .expect_err("8-token blocks over a 4-token payload");
902        assert_eq!(
903            err,
904            SignatureError::BlockSizeMismatch {
905                block_size: 8,
906                tokens: 4,
907            }
908        );
909
910        // Same on the read path, where the stamp comes from a file
911        // rather than from this process.
912        let honest =
913            CacheSignature::from_payload("model-a", flat(4), &payload(2, 2, 8, 4)).expect("sig");
914        let mut lying = honest.clone();
915        lying.layout = flat(8);
916        lying.tokens = 8;
917        let err = UnverifiedBlock::new(Some(lying), payload(2, 2, 8, 4))
918            .verify(&CacheSignature::expected("model-a", flat(8), 2, 2, 8, 8))
919            .expect_err("the payload settles the block size, not the stamp");
920        assert_eq!(
921            err,
922            SignatureError::BlockSizeMismatch {
923                block_size: 8,
924                tokens: 4,
925            }
926        );
927    }
928
929    /// v1 blocks recorded no window at all. Reading one would mean
930    /// assuming it was aligned -- the assumption this whole item
931    /// exists to refuse -- so the readable-set drops it.
932    #[test]
933    fn blocks_from_the_pre_layout_format_are_not_readable() {
934        assert!(!READABLE_FORMAT_VERSIONS.contains(&1));
935        let layers = payload(2, 2, 8, 4);
936        let mut signature =
937            CacheSignature::from_payload("model-a", flat(4), &layers).expect("signature");
938        signature.format_version = 1;
939        let err = UnverifiedBlock::new(Some(signature), layers)
940            .verify(&CacheSignature::expected("model-a", flat(4), 2, 2, 8, 4))
941            .expect_err("a v1 block cannot say what layout it was cut under");
942        assert_eq!(
943            err,
944            SignatureError::UnsupportedFormat {
945                found: 1,
946                readable: READABLE_FORMAT_VERSIONS,
947            }
948        );
949    }
950
951    #[test]
952    fn errors_name_the_field_that_changed() {
953        let text = SignatureError::Incompatible {
954            field: "head_dim",
955            expected: "128".into(),
956            found: "64".into(),
957        }
958        .to_string();
959        assert!(text.contains("head_dim"), "{text}");
960        assert!(text.contains("64"), "{text}");
961        assert!(text.contains("128"), "{text}");
962        assert!(SignatureError::Unmarked.to_string().contains("unmarked"));
963    }
964}