dynamo-tokenizers 1.3.0-dev.1

Standalone tokenizer implementations for Dynamo
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
// SPDX-License-Identifier: Apache-2.0
//
// SPDX-FileCopyrightText: Copyright (c) 2024 Simo Lin, Chang Su, Keyang Ru (llm-tokenizer authors)
//
// CHAT_TURNS corpus adapted from sgl-project/llm-tokenizer v1.3.2
// (`tests/tokenizer_cache_correctness_test.rs`). The ChatML markers `<|im_start|>` /
// `<|im_end|>` were rewritten to TinyLlama's `<s>` / `</s>` so the test runs offline
// against in-tree tokenizer fixtures. The same corpus is also rendered into Llama-3.1
// format and tested against the bundled mock-llama-3.1 fixture, exercising a second
// special-token family with a different added-token layout (4 boundary tokens instead
// of 2, distinct surface strings `<|begin_of_text|>` / `<|start_header_id|>` /
// `<|end_header_id|>` / `<|eot_id|>`).
//
// All boundary tokens used in both formats are atomic in their respective BPE
// vocabularies (`special: true, normalized: false`), so the same L1 correctness
// invariant is exercised for each:
//
//   tokenize(prefix) + tokenize(suffix) == tokenize(prefix + suffix)

//! Integration test: cached vs uncached encoding must produce identical token IDs across
//! a representative corpus of multi-turn chat prompts, on multiple tokenizer families.

use std::sync::Arc;

use dynamo_tokenizers::{
    CachedTokenizer, HuggingFaceTokenizer,
    traits::{Encoder, Tokenizer},
};

const TINYLLAMA_PATH: &str = concat!(
    env!("CARGO_MANIFEST_DIR"),
    "/../llm/tests/data/sample-models/TinyLlama_v1.1/tokenizer.json"
);

/// In-tree mock Llama-3.1 fixture. Has the full set of Llama-3.1 special tokens
/// (`<|begin_of_text|>`, `<|start_header_id|>`, `<|end_header_id|>`, `<|eot_id|>`,
/// reserved_special_token_*, `<|end_of_text|>`) but an empty BPE vocab. Regular text
/// encodes to `[]`; special tokens encode to their atomic IDs. The L1 invariant
/// `tokenize(prefix) ++ tokenize(suffix) == tokenize(prefix+suffix)` still holds —
/// that's exactly what this test asserts, regardless of how trivially the non-special
/// segments tokenize.
const LLAMA31_PATH: &str = concat!(
    env!("CARGO_MANIFEST_DIR"),
    "/../llm/tests/data/sample-models/mock-llama-3.1-8b-instruct/tokenizer.json"
);

/// In-tree mock DeepSeek-R1 fixture. Empty BPE vocab like mock-llama-3.1, but registers
/// DeepSeek's chat markers (`<|begin▁of▁sentence|>`, `<|User|>`, `<|Assistant|>`,
/// `<|end▁of▁sentence|>`) and tool-call markers (`<|tool▁calls▁begin|>`,
/// `<|tool▁call▁begin|>`, `<|tool▁sep|>`, `<|tool▁call▁end|>`, `<|tool▁calls▁end|>`).
/// Exercises the cache on a third special-token family whose markers use multibyte code
/// points (`|`=U+FF5C, `▁`=U+2581), confirming boundary detection and the merge invariant
/// hold for non-ASCII special tokens.
const DEEPSEEK_PATH: &str = concat!(
    env!("CARGO_MANIFEST_DIR"),
    "/../llm/tests/data/sample-models/mock-deepseek-r1/tokenizer.json"
);

/// A tokenizer fixture together with the formatter that re-keys the chat corpus into
/// that family's native special-token markers, plus the list of those markers for the
/// `CachedTokenizer` constructor.
struct Setup {
    name: &'static str,
    path: &'static str,
    specials: &'static [&'static str],
    /// Rewrite a corpus turn from the canonical TinyLlama-format CHAT_TURNS entry
    /// (with `<s>role\n...</s>` markers) into this family's native chat format.
    render: fn(tinyllama_format: &str) -> String,
}

/// Identity: CHAT_TURNS entries are already in TinyLlama format.
fn render_llama2(s: &str) -> String {
    s.to_string()
}

/// Convert a TinyLlama-format chat string into Llama-3.1 instruction format.
///
/// `<s>role\ncontent</s>` → `<|start_header_id|>role<|end_header_id|>\n\ncontent<|eot_id|>`
/// with a single `<|begin_of_text|>` prepended once at the very start.
fn render_llama3(s: &str) -> String {
    let mut out = String::with_capacity(s.len() + 64);
    out.push_str("<|begin_of_text|>");
    let mut remaining = s;
    while let Some(rest) = remaining.strip_prefix("<s>") {
        let end = rest
            .find("</s>")
            .expect("CHAT_TURNS turn must end with </s>");
        let turn = &rest[..end];
        // The role is everything up to the first '\n'; the content is the rest.
        // Some corpus entries put no content after the role (e.g. "<s>system\n</s>"),
        // in which case split_once returns (role, "").
        let (role, content) = turn.split_once('\n').unwrap_or((turn, ""));
        out.push_str("<|start_header_id|>");
        out.push_str(role);
        out.push_str("<|end_header_id|>\n\n");
        out.push_str(content);
        out.push_str("<|eot_id|>");
        remaining = &rest[end + "</s>".len()..];
    }
    out
}

/// Convert a TinyLlama-format chat string into DeepSeek-R1 format.
///
/// `<s>system\ncontent</s>`    → `content` (bare, immediately after the BOS)
/// `<s>user\ncontent</s>`      → `<|User|>content`
/// `<s>assistant\ncontent</s>` → `<|Assistant|>content<|end▁of▁sentence|>`
/// with a single `<|begin▁of▁sentence|>` prepended once at the very start. Any role
/// other than `system`/`assistant` opens with the `<|User|>` marker (matches the corpus,
/// which only uses system/user/assistant).
fn render_deepseek(s: &str) -> String {
    let mut out = String::with_capacity(s.len() + 64);
    out.push_str("<|begin▁of▁sentence|>");
    let mut remaining = s;
    while let Some(rest) = remaining.strip_prefix("<s>") {
        let end = rest
            .find("</s>")
            .expect("CHAT_TURNS turn must end with </s>");
        let turn = &rest[..end];
        let (role, content) = turn.split_once('\n').unwrap_or((turn, ""));
        match role {
            "system" => out.push_str(content),
            "assistant" => {
                out.push_str("<|Assistant|>");
                out.push_str(content);
                out.push_str("<|end▁of▁sentence|>");
            }
            _ => {
                out.push_str("<|User|>");
                out.push_str(content);
            }
        }
        remaining = &rest[end + "</s>".len()..];
    }
    out
}

const SETUPS: &[Setup] = &[
    Setup {
        name: "tinyllama (Llama-2 family, <s>/</s>)",
        path: TINYLLAMA_PATH,
        specials: &["<s>", "</s>"],
        render: render_llama2,
    },
    Setup {
        name: "mock-llama-3.1 (Llama-3 family, <|begin_of_text|>/<|*_header_id|>/<|eot_id|>)",
        path: LLAMA31_PATH,
        specials: &[
            "<|begin_of_text|>",
            "<|start_header_id|>",
            "<|end_header_id|>",
            "<|eot_id|>",
        ],
        render: render_llama3,
    },
    Setup {
        name: "mock-deepseek-r1 (DeepSeek family, <|begin▁of▁sentence|>/<|User|>/<|Assistant|>/<|end▁of▁sentence|>)",
        path: DEEPSEEK_PATH,
        specials: &[
            "<|begin▁of▁sentence|>",
            "<|User|>",
            "<|Assistant|>",
            "<|end▁of▁sentence|>",
        ],
        render: render_deepseek,
    },
];

fn build_cached_setup(setup: &Setup) -> (Arc<dyn Tokenizer>, CachedTokenizer) {
    let base = Arc::new(
        HuggingFaceTokenizer::from_file(setup.path)
            .unwrap_or_else(|e| panic!("load tokenizer {}: {e}", setup.path)),
    );
    let specials: Vec<String> = setup.specials.iter().map(|s| (*s).to_string()).collect();
    let cached = CachedTokenizer::new(base.clone(), specials, 50 * 1024 * 1024);
    (base, cached)
}

/// 29 multi-turn ChatML strings re-keyed onto TinyLlama's `<s>`/`</s>` markers.
const CHAT_TURNS: [&str; 29] = [
    // Basic conversation patterns
    "<s>system\nYou are a helpful AI assistant.</s>",
    "<s>system\nYou are a helpful AI assistant.</s><s>user\nWhat is the capital of France?</s>",
    "<s>system\nYou are a helpful AI assistant.</s><s>user\nWhat is the capital of France?</s><s>assistant\nThe capital of France is Paris.</s>",
    // Different system prompts (testing different prefix patterns)
    "<s>system\nYou are a coding tutor specializing in Rust programming.</s><s>user\nExplain ownership.</s>",
    "<s>system\nYou are a math teacher.</s><s>user\nSolve: 2x + 5 = 13</s>",
    // Long conversation with multiple turns (testing longer prefixes)
    "<s>system\nYou are a helpful AI assistant.</s><s>user\nTell me about deep learning.</s><s>assistant\nDeep learning is a subset of machine learning that uses neural networks with multiple layers.</s><s>user\nWhat are the main architectures?</s>",
    // Code snippets (testing different character patterns)
    "<s>system\nYou are a code reviewer.</s><s>user\nReview this code:\nfn main() {\n    println!(\"Hello, world!\");\n}\n</s>",
    "<s>system\nYou are a code reviewer.</s><s>user\nExplain this Rust code:\nimpl<T> Drop for Box<T> {\n    fn drop(&mut self) { /* ... */ }\n}\n</s>",
    // Mathematical content
    "<s>system\nYou are a math tutor.</s><s>user\nProve that sqrt(2) is irrational using proof by contradiction.</s>",
    "<s>system\nYou are a math tutor.</s><s>user\nCalculate: integral of (x^2 + 3x + 2) dx from 0 to 5</s>",
    // Multilingual content
    "<s>system\nYou are a multilingual assistant.</s><s>user\nTranslate to French: The quick brown fox jumps over the lazy dog.</s>",
    "<s>system\nYou are a multilingual assistant.</s><s>user\n你好,请帮我翻译这句话:I love programming in Rust.</s>",
    "<s>system\nYou are a multilingual assistant.</s><s>user\nこんにちは!Rustについて教えてください。</s>",
    // Special characters and emojis
    "<s>system\nYou are a friendly chatbot.</s><s>user\nWhat do you think about emojis? 😀🎉🚀💻</s>",
    "<s>system\nYou are a data analyst.</s><s>user\nAnalyze this: {\"name\": \"test\", \"value\": 42, \"nested\": {\"key\": \"value\"}}</s>",
    // Very long message (testing large token counts)
    "<s>system\nYou are a literature expert.</s><s>user\nAnalyze the themes in this passage: In the vast expanse of the digital realm, where bits and bytes dance in harmonious symphony, there exists a paradigm that transcends mere computation. This paradigm, known as machine learning, represents humanity's quest to imbue silicon with the spark of cognition. Deep neural networks, inspired by the intricate architecture of biological brains, layer upon layer of artificial neurons, each connection a synapse firing in the dark recesses of mathematical space. Through gradient descent, these networks learn patterns invisible to human perception, extracting meaning from chaos, signal from noise. The transformer architecture revolutionized this field, introducing attention mechanisms that allowed models to focus on relevant information, much like how humans selectively attend to important details in their environment.</s>",
    // Edge case: Multiple special tokens in sequence
    "<s>system\nYou are helpful.</s><s>user\nHi</s><s>assistant\nHello!</s><s>user\nHow are you?</s>",
    // Edge case: Empty-ish messages
    "<s>system\n</s><s>user\nTest</s>",
    "<s>system\nBrief.</s><s>user\nOK</s>",
    // Technical documentation style
    "<s>system\nYou are a technical writer.</s><s>user\nDocument the following API:\n\n```rust\npub struct CachedTokenizer {\n    inner: Arc<dyn Tokenizer>,\n    l1: L1Cache,\n}\n\nimpl Encoder for CachedTokenizer {\n    fn encode(&self, input: &str) -> Result<Encoding>;\n}\n```\n</s>",
    // Conversation with code review
    "<s>system\nYou are a senior Rust developer.</s><s>user\nReview for correctness:\n\nlet specials: Vec<&str> = self.special_tokens.iter().map(String::as_str).collect();</s>",
    // Markdown formatted content
    "<s>system\nYou are a documentation assistant.</s><s>user\nFormat this as markdown:\n\n# Cache Architecture\n\n## L1 Cache\n- Prefix match\n- Special token boundaries\n- 50MB memory\n</s>",
    // Complex nested structures
    "<s>system\nYou are a JSON expert.</s><s>user\nValidate this JSON:\n{\n  \"tokenizer_cache\": {\n    \"enabled\": true,\n    \"max_memory\": 52428800,\n    \"stats\": {\n      \"hits\": [1, 2, 3],\n      \"misses\": {\"count\": 5}\n    }\n  }\n}\n</s>",
    // SQL queries
    "<s>system\nYou are a database expert.</s><s>user\nOptimize this query:\nSELECT u.name, COUNT(p.id) as post_count\nFROM users u\nLEFT JOIN posts p ON u.id = p.user_id\nWHERE u.created_at > '2024-01-01'\nGROUP BY u.id, u.name\nHAVING COUNT(p.id) > 5\nORDER BY post_count DESC;</s>",
    // Regex patterns
    "<s>system\nYou are a regex expert.</s><s>user\nExplain this regex: ^(?:[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?\\.)+[a-zA-Z]{2,}$</s>",
    // Command line examples
    "<s>system\nYou are a DevOps engineer.</s><s>user\nExplain this command:\ncargo bench --bench tokenizer_benchmark -- --color=never | tee results.txt</s>",
    // Unicode edge cases
    "<s>system\nYou are helpful.</s><s>user\nTest: café, naïve, Zürich, 北京, 東京, मुंबई, Москва</s>",
    // Mixed content complexity
    "<s>system\nYou are a software architect.</s><s>user\nDesign a caching system that:\n1. Handles 10K+ QPS\n2. Maintains 99.9% uptime\n3. Supports L1 (prefix) caching\n4. Uses Blake3 for hashing\n5. Implements LRU eviction\n6. Thread-safe with lock-free reads\n</s>",
    // Very long technical discussion
    "<s>system\nYou are a compiler expert.</s><s>user\nExplain why BPE tokenizers are not prefix-stable:\n\nThe core issue is that BPE applies merges based on local context. When you tokenize 'prefix' alone, it might apply merge rules differently than when tokenizing 'prefix + suffix' as a whole. For example:\n\ntokenize('hello world') might produce [hello, _world]\ntokenize('hello') + tokenize(' world') might produce [hel, lo, _wo, rld]\n\nThis is because the merge rules see different contexts. The space before 'world' in the first case is part of the token boundary, but in the second case, ' world' is tokenized in isolation.\n\nSpecial tokens solve this because they are:\n1. Atomic (never split or merged)\n2. Protected from normalization\n3. Marked with special: true flag\n4. Have normalized: false property\n\nThis guarantees: tokenize(prefix + special + suffix) = tokenize(prefix + special) + tokenize(suffix)\n\nOur L1 cache exploits this by:\n1. Finding all special token boundaries\n2. Re-tokenizing prefixes at those boundaries\n3. Caching the exact token IDs\n4. On cache hit, appending suffix tokens\n</s>",
];

#[test]
fn cached_vs_uncached_first_pass() {
    // For each tokenizer family: re-key the corpus into that family's chat format,
    // then encode every turn twice (miss-path, hit-path) and assert byte-exact token
    // equality against an uncached baseline.
    for setup in SETUPS {
        let (base, cached) = build_cached_setup(setup);

        let mut hits_at_start = cached.cache_stats().hits;
        let mut hits_grew_at_least_once = false;

        for (i, raw_turn) in CHAT_TURNS.iter().enumerate() {
            let turn = (setup.render)(raw_turn);
            let plain = base
                .encode(&turn)
                .expect("uncached encode")
                .token_ids()
                .to_vec();

            // First call — miss path; cache populates at every special-token boundary.
            let cached_first = cached
                .encode(&turn)
                .expect("cached encode (miss)")
                .token_ids()
                .to_vec();
            assert_eq!(
                cached_first, plain,
                "[{}] turn {i}: miss-path cached encode != uncached encode",
                setup.name
            );

            // Second call — hit path; merged prefix + suffix must equal plain encode.
            let cached_second = cached
                .encode(&turn)
                .expect("cached encode (hit)")
                .token_ids()
                .to_vec();
            assert_eq!(
                cached_second, plain,
                "[{}] turn {i}: hit-path cached encode != uncached encode",
                setup.name
            );

            let now = cached.cache_stats().hits;
            if now > hits_at_start {
                hits_grew_at_least_once = true;
            }
            hits_at_start = now;
        }

        assert!(
            hits_grew_at_least_once,
            "[{}] expected at least one turn to produce an L1 hit",
            setup.name
        );
    }
}

#[test]
fn cross_turn_shared_prefix_hits() {
    // Turns 0/1/2 share progressively longer prefixes. Encoding them in order should
    // produce L1 hits on turns 1 and 2 (their prefixes were populated on turn 0).
    for setup in SETUPS {
        let (_base, cached) = build_cached_setup(setup);

        let before = cached.cache_stats();
        for raw_turn in &CHAT_TURNS[..3] {
            let _ = cached.encode(&(setup.render)(raw_turn)).unwrap();
        }
        let after = cached.cache_stats();

        assert!(
            after.hits > before.hits,
            "[{}] shared-prefix turns should produce L1 hits (before={}, after={})",
            setup.name,
            before.hits,
            after.hits
        );
    }
}

#[test]
fn cache_disabled_by_empty_specials_is_transparent() {
    // No specials registered -> L1 always misses, every encode goes through the inner
    // tokenizer. Output must still equal uncached encode. Tested per tokenizer family
    // because the relevant code path is the wrapper, not the tokenizer.
    for setup in SETUPS {
        let base = Arc::new(
            HuggingFaceTokenizer::from_file(setup.path)
                .unwrap_or_else(|e| panic!("load tokenizer {}: {e}", setup.path)),
        );
        let cached = CachedTokenizer::new(base.clone(), Vec::new(), 4096);
        for (i, raw_turn) in CHAT_TURNS.iter().enumerate() {
            let turn = (setup.render)(raw_turn);
            let plain = base.encode(&turn).unwrap().token_ids().to_vec();
            let through = cached.encode(&turn).unwrap().token_ids().to_vec();
            assert_eq!(
                plain, through,
                "[{}] turn {i}: transparent encode mismatch",
                setup.name
            );
        }
        assert_eq!(
            cached.cache_stats().hits,
            0,
            "[{}] with no specials, L1 must produce zero hits",
            setup.name
        );
    }
}

/// Build an append-only multi-turn conversation in TinyLlama format. `turns[i]` is the
/// full history at turn `i`: the system prompt plus `i + 1` completed user/assistant
/// exchanges. Every turn is a well-formed sequence of `<s>role\ncontent</s>` blocks (so
/// the per-family `render` fns accept it), and `turns[i]` is a strict prefix of
/// `turns[i + 1]`.
fn growing_chat_turns(n: usize) -> Vec<String> {
    let mut convo = String::from("<s>system\nYou are a helpful assistant.</s>");
    let mut turns = Vec::with_capacity(n);
    for i in 0..n {
        convo.push_str(&format!(
            "<s>user\nQuestion {i} please answer it.</s><s>assistant\nDetailed answer {i} follows here.</s>"
        ));
        turns.push(convo.clone());
    }
    turns
}

#[test]
fn extend_on_hit_matches_uncached_across_growing_turns() {
    // With extend enabled, a growing conversation is encoded turn-by-turn: turn 0 is a
    // miss, every later turn is a partial hit that also deepens the cache. Each turn's
    // tokens must stay byte-exact against an uncached encode, on both tokenizer families
    // (mock-llama-3.1's empty-BPE vocab surfaces any seg_a/seg_b split off-by-one).
    let turns = growing_chat_turns(12);
    for setup in SETUPS {
        let base = Arc::new(
            HuggingFaceTokenizer::from_file(setup.path)
                .unwrap_or_else(|e| panic!("load tokenizer {}: {e}", setup.path)),
        );
        let specials: Vec<String> = setup.specials.iter().map(|s| (*s).to_string()).collect();
        let cached =
            CachedTokenizer::new(base.clone(), specials, 50 * 1024 * 1024).with_extend(true);

        for (i, raw_turn) in turns.iter().enumerate() {
            let turn = (setup.render)(raw_turn);
            let plain = base.encode(&turn).unwrap().token_ids().to_vec();

            // First encode (miss path on turn 0, extend hit-path afterward).
            let first = cached.encode(&turn).unwrap().token_ids().to_vec();
            assert_eq!(
                first, plain,
                "[{}] turn {i}: extend-on first encode != uncached",
                setup.name
            );

            // Second encode exercises the fully-cached prefix + extend path.
            let second = cached.encode(&turn).unwrap().token_ids().to_vec();
            assert_eq!(
                second, plain,
                "[{}] turn {i}: extend-on second encode != uncached",
                setup.name
            );
        }

        assert!(
            cached.cache_stats().hits > 0,
            "[{}] expected L1 hits with extend on",
            setup.name
        );
    }
}

#[test]
fn extend_off_hits_never_insert() {
    // With extension disabled, partial hits must NOT mutate the cache — the original
    // hit-without-insert behavior. After turn 0 populates the cache, encoding the rest
    // of the growing conversation produces hits but adds zero entries.
    let turns = growing_chat_turns(10);
    for setup in SETUPS {
        let (_base, cached) = build_cached_setup(setup); // CachedTokenizer::new => extend off

        let _ = cached.encode(&(setup.render)(&turns[0])).unwrap();
        let entries_after_turn0 = cached.cache_stats().entries;
        let hits_after_turn0 = cached.cache_stats().hits;

        for raw_turn in &turns[1..] {
            let _ = cached.encode(&(setup.render)(raw_turn)).unwrap();
        }

        let after = cached.cache_stats();
        assert_eq!(
            after.entries, entries_after_turn0,
            "[{}] extend-off: hits must not add cache entries (was {}, now {})",
            setup.name, entries_after_turn0, after.entries
        );
        assert!(
            after.hits > hits_after_turn0,
            "[{}] later turns should have produced hits",
            setup.name
        );
    }
}

#[test]
fn extend_on_partial_hit_adds_exactly_one_entry() {
    // End-to-end deepest-only invariant at the `CachedTokenizer` level: turn 0 (miss)
    // populates an entry at every boundary; each later turn is a partial hit that, with
    // extend on, persists exactly ONE new entry (the suffix's deepest boundary) — never
    // the intermediate boundaries. Holds on every family.
    let turns = growing_chat_turns(6);
    for setup in SETUPS {
        let base = Arc::new(
            HuggingFaceTokenizer::from_file(setup.path)
                .unwrap_or_else(|e| panic!("load tokenizer {}: {e}", setup.path)),
        );
        let specials: Vec<String> = setup.specials.iter().map(|s| (*s).to_string()).collect();
        let cached = CachedTokenizer::new(base, specials, 50 * 1024 * 1024).with_extend(true);

        // Turn 0: miss path populates the cache at every boundary.
        let _ = cached.encode(&(setup.render)(&turns[0])).unwrap();

        for (i, raw) in turns.iter().enumerate().skip(1) {
            let before = cached.cache_stats().entries;
            let _ = cached.encode(&(setup.render)(raw)).unwrap();
            let after = cached.cache_stats().entries;
            assert_eq!(
                after,
                before + 1,
                "[{}] turn {i}: partial-hit extend must add exactly one entry (before {before}, after {after})",
                setup.name
            );
        }
    }
}

/// Build an append-only DeepSeek-R1 tool-calling conversation. Each round appends a user
/// question, an assistant turn whose body is a real DeepSeek tool-call block (five nested
/// `<|tool▁…|>` special tokens around a JSON payload), a user-delivered tool result, and
/// an assistant answer. `turns[i]` is a strict prefix of `turns[i + 1]`.
fn growing_deepseek_tool_turns(n: usize) -> Vec<String> {
    let mut convo = String::from("<|begin▁of▁sentence|>You are a helpful assistant with tools.");
    let mut turns = Vec::with_capacity(n);
    for i in 0..n {
        convo.push_str(&format!(
            "<|User|>What is the weather in city {i}?\
             <|Assistant|><|tool▁calls▁begin|><|tool▁call▁begin|>function<|tool▁sep|>get_weather\n```json\n{{\"city\": \"city {i}\"}}\n```<|tool▁call▁end|><|tool▁calls▁end|><|end▁of▁sentence|>\
             <|User|>Tool result: sunny in city {i}.\
             <|Assistant|>It is sunny in city {i}.<|end▁of▁sentence|>"
        ));
        turns.push(convo.clone());
    }
    turns
}

#[test]
fn extend_correct_with_deepseek_tool_calls() {
    // Tool-call markers that ARE special tokens (DeepSeek's `<|tool▁…|>` family) add real
    // cache boundaries. Encoding a growing tool conversation with extend on must stay
    // byte-exact vs an uncached encode straight through the nested multibyte tool tokens.
    let base = Arc::new(
        HuggingFaceTokenizer::from_file(DEEPSEEK_PATH)
            .unwrap_or_else(|e| panic!("load tokenizer {DEEPSEEK_PATH}: {e}")),
    );
    let specials: Vec<String> = [
        "<|begin▁of▁sentence|>",
        "<|User|>",
        "<|Assistant|>",
        "<|end▁of▁sentence|>",
        "<|tool▁calls▁begin|>",
        "<|tool▁call▁begin|>",
        "<|tool▁sep|>",
        "<|tool▁call▁end|>",
        "<|tool▁calls▁end|>",
    ]
    .iter()
    .map(|s| s.to_string())
    .collect();
    let cached = CachedTokenizer::new(base.clone(), specials, 8 * 1024 * 1024).with_extend(true);

    // The tool-call-begin marker must be an atomic special the cache can split on.
    let tool_begin = base
        .encode("<|tool▁calls▁begin|>")
        .unwrap()
        .token_ids()
        .to_vec();
    assert_eq!(
        tool_begin.len(),
        1,
        "tool-call-begin must encode atomically"
    );

    let turns = growing_deepseek_tool_turns(8);
    let mut saw_tool_token = false;
    for (i, turn) in turns.iter().enumerate() {
        let plain = base.encode(turn).unwrap().token_ids().to_vec();

        let first = cached.encode(turn).unwrap().token_ids().to_vec();
        assert_eq!(first, plain, "turn {i}: extend-on first encode != uncached");

        let second = cached.encode(turn).unwrap().token_ids().to_vec();
        assert_eq!(
            second, plain,
            "turn {i}: extend-on second encode != uncached"
        );

        if plain.contains(&tool_begin[0]) {
            saw_tool_token = true;
        }
    }
    assert!(
        saw_tool_token,
        "the tool-call special token must actually appear in the encoded turns"
    );
    assert!(
        cached.cache_stats().hits > 0,
        "expected L1 hits across the growing tool conversation"
    );
}

/// Build an append-only conversation whose assistant turns optionally wrap the tool call
/// in plain-text `<tool_call>…</tool_call>` markup (Hermes/Qwen style). When
/// `wrap_in_tool_call` is false the same JSON appears as ordinary assistant text. Both
/// variants share an identical special-token (`<s>`/`</s>`) structure — only the plain
/// text inside the assistant turns differs. Canonical `<s>role\ncontent</s>` form.
fn growing_tool_turns(n: usize, wrap_in_tool_call: bool) -> Vec<String> {
    let mut convo = String::from("<s>system\nYou are a helpful assistant with tools.</s>");
    let mut turns = Vec::with_capacity(n);
    for i in 0..n {
        let call_body =
            format!("{{\"name\": \"get_weather\", \"arguments\": {{\"city\": \"city {i}\"}}}}");
        let assistant_call = if wrap_in_tool_call {
            format!("<tool_call>{call_body}</tool_call>")
        } else {
            call_body
        };
        convo.push_str(&format!(
            "<s>user\nWeather in city {i}?</s>\
             <s>assistant\n{assistant_call}</s>\
             <s>tool\nsunny in city {i}</s>\
             <s>assistant\nIt is sunny in city {i}.</s>"
        ));
        turns.push(convo.clone());
    }
    turns
}

#[test]
fn extend_transparent_to_plaintext_tool_markup() {
    // Plain-text `<tool_call>…</tool_call>` markup is NOT a registered special token, so it
    // must be transparent to the cache: (1) encoding stays byte-exact under extend, and
    // (2) the markup adds no special-token boundary — a cache fed the markup variant ends
    // with the same number of entries as one fed the bare-JSON variant (identical `<s>`/
    // `</s>` structure). Run on the HF families where `<tool_call>` is genuinely text.
    for setup in &SETUPS[..2] {
        let markup = growing_tool_turns(8, true);
        let bare = growing_tool_turns(8, false);

        let build = || {
            let base = Arc::new(
                HuggingFaceTokenizer::from_file(setup.path)
                    .unwrap_or_else(|e| panic!("load tokenizer {}: {e}", setup.path)),
            );
            let specials: Vec<String> = setup.specials.iter().map(|s| (*s).to_string()).collect();
            let cached =
                CachedTokenizer::new(base.clone(), specials, 50 * 1024 * 1024).with_extend(true);
            (base, cached)
        };

        // Markup variant: every turn must encode byte-exact vs uncached.
        let (base_m, cache_m) = build();
        for (i, raw) in markup.iter().enumerate() {
            let turn = (setup.render)(raw);
            let plain = base_m.encode(&turn).unwrap().token_ids().to_vec();
            let got = cache_m.encode(&turn).unwrap().token_ids().to_vec();
            assert_eq!(
                got, plain,
                "[{}] markup turn {i}: extend encode != uncached",
                setup.name
            );
        }

        // Bare variant: same special-token structure, no `<tool_call>` tags.
        let (_base_b, cache_b) = build();
        for raw in &bare {
            let _ = cache_b.encode(&(setup.render)(raw)).unwrap();
        }

        assert_eq!(
            cache_m.cache_stats().entries,
            cache_b.cache_stats().entries,
            "[{}] plain-text <tool_call> markup must add no special-token boundaries \
             (markup entries {}, bare entries {})",
            setup.name,
            cache_m.cache_stats().entries,
            cache_b.cache_stats().entries
        );
    }
}