Skip to main content

vtcode_config/core/
prompt_cache.rs

1use crate::constants::prompt_cache;
2use crate::env_helpers::default_true;
3use serde::{Deserialize, Serialize};
4use std::path::{Path, PathBuf};
5use vtcode_commons::VtCodePaths;
6
7/// Global prompt caching configuration loaded from vtcode.toml
8#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
9#[derive(Debug, Clone, Deserialize, Serialize)]
10pub struct PromptCachingConfig {
11    /// Enable prompt caching features globally
12    #[serde(default = "default_enabled")]
13    pub enabled: bool,
14
15    /// Base directory for local prompt cache storage (supports `~` expansion)
16    #[serde(default = "default_cache_dir")]
17    pub cache_dir: String,
18
19    /// Maximum number of cached prompt entries to retain on disk
20    #[serde(default = "default_max_entries")]
21    pub max_entries: usize,
22
23    /// Maximum age (in days) before cached entries are purged
24    #[serde(default = "default_max_age_days")]
25    pub max_age_days: u64,
26
27    /// Automatically evict stale entries on startup/shutdown
28    #[serde(default = "default_auto_cleanup")]
29    pub enable_auto_cleanup: bool,
30
31    /// Minimum quality score required before persisting an entry
32    #[serde(default = "default_min_quality_threshold")]
33    pub min_quality_threshold: f64,
34
35    /// Enable prompt-shaping optimizations that improve provider-side cache locality.
36    /// When enabled, VT Code keeps volatile runtime context at the end of prompt text.
37    #[serde(default = "default_cache_friendly_prompt_shaping")]
38    pub cache_friendly_prompt_shaping: bool,
39
40    /// Keep the wire tool catalog identical across planning and execution
41    /// turns (union of both modes' tools). Planning toggles then do not
42    /// rewrite the tool prefix. The fail-closed execution gate still blocks
43    /// mutations during planning.
44    #[serde(default = "default_true")]
45    pub stable_tool_catalog_across_modes: bool,
46
47    /// Warn before a request when the pause since the previous request likely
48    /// exceeded the provider prompt cache lifetime (advisory only).
49    #[serde(default = "default_true")]
50    pub gap_warning_enabled: bool,
51
52    /// Override for the cache-gap warning threshold in seconds. When unset,
53    /// a provider-specific default is used (Anthropic 300s, OpenAI 600s).
54    #[serde(default)]
55    pub gap_warning_threshold_secs: Option<u64>,
56
57    /// Provider specific overrides
58    #[serde(default)]
59    pub providers: ProviderPromptCachingConfig,
60}
61
62impl Default for PromptCachingConfig {
63    fn default() -> Self {
64        let cache_dir = default_cache_dir();
65        Self {
66            enabled: default_enabled() && !cache_dir.is_empty(),
67            cache_dir,
68            max_entries: default_max_entries(),
69            max_age_days: default_max_age_days(),
70            enable_auto_cleanup: default_auto_cleanup(),
71            min_quality_threshold: default_min_quality_threshold(),
72            cache_friendly_prompt_shaping: default_cache_friendly_prompt_shaping(),
73            stable_tool_catalog_across_modes: true,
74            gap_warning_enabled: default_true(),
75            gap_warning_threshold_secs: None,
76            providers: ProviderPromptCachingConfig::default(),
77        }
78    }
79}
80
81impl PromptCachingConfig {
82    /// Resolve the configured cache directory to an absolute path
83    ///
84    /// - `~` is expanded to the user's home directory when available
85    /// - Relative paths are resolved against the provided workspace root when supplied
86    /// - Falls back to the configured string when neither applies
87    pub fn resolve_cache_dir(&self, workspace_root: Option<&Path>) -> PathBuf {
88        resolve_path(&self.cache_dir, workspace_root)
89    }
90
91    /// Returns true when prompt caching is active for the given provider runtime name.
92    pub fn is_provider_enabled(&self, provider_name: &str) -> bool {
93        if !self.enabled {
94            return false;
95        }
96
97        match provider_name.to_ascii_lowercase().as_str() {
98            "openai" | "merge-gateway" => self.providers.openai.enabled,
99            "anthropic" | "minimax" => self.providers.anthropic.enabled,
100            "gemini" => {
101                self.providers.gemini.enabled && !matches!(self.providers.gemini.mode, GeminiPromptCacheMode::Off)
102            }
103            "openrouter" => self.providers.openrouter.enabled,
104            "moonshot" => self.providers.moonshot.enabled,
105            "deepseek" => self.providers.deepseek.enabled,
106            "zai" => self.providers.zai.enabled,
107            _ => false,
108        }
109    }
110
111    /// Resolve the cache-gap warning threshold in seconds for a provider.
112    ///
113    /// Returns `None` when the warning should not fire: gap warnings disabled,
114    /// caching disabled for the provider, or OpenAI configured with 24h
115    /// extended retention (where a short pause cannot expire the cache).
116    pub fn gap_threshold_secs(&self, provider_name: &str) -> Option<u64> {
117        if !self.gap_warning_enabled || !self.is_provider_enabled(provider_name) {
118            return None;
119        }
120
121        let provider = provider_name.to_ascii_lowercase();
122        if provider == "openai"
123            && matches!(self.providers.openai.prompt_cache_retention, Some(PromptCacheRetention::H24))
124        {
125            return None;
126        }
127
128        if let Some(threshold) = self.gap_warning_threshold_secs {
129            return Some(threshold);
130        }
131
132        Some(match provider.as_str() {
133            "anthropic" | "minimax" => prompt_cache::ANTHROPIC_CACHE_GAP_WARNING_SECONDS,
134            "openai" => prompt_cache::OPENAI_CACHE_GAP_WARNING_SECONDS,
135            _ => prompt_cache::DEFAULT_CACHE_GAP_WARNING_SECONDS,
136        })
137    }
138}
139
140/// Per-provider configuration overrides
141#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
142#[derive(Debug, Clone, Deserialize, Serialize, Default)]
143pub struct ProviderPromptCachingConfig {
144    #[serde(default = "OpenAIPromptCacheSettings::default")]
145    pub openai: OpenAIPromptCacheSettings,
146
147    #[serde(default = "AnthropicPromptCacheSettings::default")]
148    pub anthropic: AnthropicPromptCacheSettings,
149
150    #[serde(default = "GeminiPromptCacheSettings::default")]
151    pub gemini: GeminiPromptCacheSettings,
152
153    #[serde(default = "OpenRouterPromptCacheSettings::default")]
154    pub openrouter: OpenRouterPromptCacheSettings,
155
156    #[serde(default = "MoonshotPromptCacheSettings::default")]
157    moonshot: MoonshotPromptCacheSettings,
158
159    #[serde(default = "DeepSeekPromptCacheSettings::default")]
160    pub deepseek: DeepSeekPromptCacheSettings,
161
162    #[serde(default = "ZaiPromptCacheSettings::default")]
163    zai: ZaiPromptCacheSettings,
164}
165
166/// OpenAI prompt cache retention policy.
167/// Controls the model-side prompt caching window.
168#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
169#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, Default)]
170pub enum PromptCacheRetention {
171    /// In-memory caching (short-lived, free)
172    #[serde(rename = "in_memory")]
173    InMemory,
174    /// 24-hour persistent caching
175    #[default]
176    #[serde(rename = "24h")]
177    H24,
178    /// Forward-compatible catch-all for unknown retention values.
179    ///
180    /// The explicit `rename` keeps the serialized form (`"unknown"`) aligned
181    /// with [`PromptCacheRetention::as_str`]; bare `#[serde(other)]` would
182    /// serialize the variant name as `"Unknown"`.
183    #[serde(other, rename = "unknown")]
184    Unknown,
185}
186
187impl PromptCacheRetention {
188    /// Returns the string representation for the OpenAI API wire format.
189    pub fn as_str(&self) -> &str {
190        match self {
191            Self::InMemory => "in_memory",
192            Self::H24 => "24h",
193            Self::Unknown => "unknown",
194        }
195    }
196}
197
198impl std::fmt::Display for PromptCacheRetention {
199    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
200        f.write_str(self.as_str())
201    }
202}
203
204/// OpenAI prompt caching controls (automatic with metrics)
205#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
206#[derive(Debug, Clone, Deserialize, Serialize)]
207pub struct OpenAIPromptCacheSettings {
208    #[serde(default = "default_true")]
209    pub enabled: bool,
210
211    #[serde(default = "default_openai_min_prefix_tokens")]
212    min_prefix_tokens: u32,
213
214    #[serde(default = "default_openai_idle_expiration")]
215    idle_expiration_seconds: u64,
216
217    #[serde(default = "default_true")]
218    pub surface_metrics: bool,
219
220    /// Strategy for generating OpenAI `prompt_cache_key`.
221    /// Session-scoped keys derive one stable key per VT Code conversation.
222    #[serde(default = "default_openai_prompt_cache_key_mode")]
223    pub prompt_cache_key_mode: OpenAIPromptCacheKeyMode,
224
225    /// Optional prompt cache retention policy to pass directly into OpenAI Responses API.
226    /// Supported values are "in_memory" and "24h". If set, VT Code will include
227    /// `prompt_cache_retention` in the request body to extend the model-side prompt
228    /// caching window or request the explicit in-memory policy.
229    #[serde(default)]
230    pub prompt_cache_retention: Option<PromptCacheRetention>,
231}
232
233impl Default for OpenAIPromptCacheSettings {
234    fn default() -> Self {
235        Self {
236            enabled: default_true(),
237            min_prefix_tokens: default_openai_min_prefix_tokens(),
238            idle_expiration_seconds: default_openai_idle_expiration(),
239            surface_metrics: default_true(),
240            prompt_cache_key_mode: default_openai_prompt_cache_key_mode(),
241            prompt_cache_retention: None,
242        }
243    }
244}
245
246impl OpenAIPromptCacheSettings {
247    /// Validate OpenAI provider prompt cache settings.
248    /// With the typed enum, invalid values are caught at deserialization time.
249    fn validate(&self) -> anyhow::Result<()> {
250        if let Some(PromptCacheRetention::Unknown) = self.prompt_cache_retention {
251            anyhow::bail!("prompt_cache_retention must be one of: in_memory, 24h");
252        }
253        Ok(())
254    }
255}
256
257/// Build a stable OpenAI `prompt_cache_key` for requests that should share
258/// provider-side cache routing.
259#[must_use]
260pub fn build_openai_prompt_cache_key(
261    prompt_cache_enabled: bool,
262    prompt_cache_key_mode: &OpenAIPromptCacheKeyMode,
263    lineage_id: Option<&str>,
264) -> Option<String> {
265    if !prompt_cache_enabled {
266        return None;
267    }
268
269    let lineage_id = lineage_id.map(str::trim).filter(|value| !value.is_empty());
270    match prompt_cache_key_mode {
271        OpenAIPromptCacheKeyMode::Session => lineage_id.map(|lineage_id| format!("vtcode:openai:{lineage_id}")),
272        OpenAIPromptCacheKeyMode::Off => None,
273    }
274}
275
276/// Rewrite an OpenAI-style cache key for routes that share the wire shape but
277/// need distinct routing stickiness. Keeps the stable session identifier while
278/// namespacing gateway/session-affinity traffic apart from native OpenAI.
279#[must_use]
280pub fn map_prompt_cache_key_for_provider(provider_name: &str, key: String) -> String {
281    let provider = provider_name.trim().to_ascii_lowercase();
282    match provider.as_str() {
283        "merge-gateway" => key.replacen("vtcode:openai:", "vtcode:merge:", 1),
284        "openrouter" => key.replacen("vtcode:openai:", "vtcode:openrouter:", 1),
285        "xai" => key.replacen("vtcode:openai:", "vtcode:xai:", 1),
286        _ => key,
287    }
288}
289
290/// Providers that send lineage-stable session/cache identity on the wire.
291#[must_use]
292pub fn session_affinity_provider(provider_name: &str) -> bool {
293    matches!(
294        provider_name.trim().to_ascii_lowercase().as_str(),
295        "openai" | "merge-gateway" | "openrouter" | "xai"
296    )
297}
298
299/// Whether `LLMRequest.prompt_cache_key` should carry session lineage for this provider.
300///
301/// OpenAI/Merge keep the OpenAI prompt-cache enablement gate.
302/// OpenRouter/xAI need lineage for documented session affinity even when the
303/// OpenAI-specific cache block is off (global prompt-cache master switch still applies).
304#[must_use]
305pub fn session_affinity_key_enabled(
306    provider_name: &str,
307    global_prompt_cache_enabled: bool,
308    openai_prompt_cache_enabled: bool,
309) -> bool {
310    if !global_prompt_cache_enabled {
311        return false;
312    }
313    match provider_name.trim().to_ascii_lowercase().as_str() {
314        "openai" | "merge-gateway" => openai_prompt_cache_enabled,
315        "openrouter" | "xai" => true,
316        _ => false,
317    }
318}
319
320/// Build the namespaced session-affinity / prompt-cache key for a request.
321///
322/// OpenAI/Merge honor [`OpenAIPromptCacheKeyMode`]. OpenRouter/xAI always use a
323/// stable session namespace when the affinity gate is on: sticky routing is
324/// independent of the OpenAI prompt-cache key mode.
325#[must_use]
326pub fn build_session_affinity_prompt_cache_key(
327    provider_name: &str,
328    session_affinity_enabled: bool,
329    prompt_cache_key_mode: &OpenAIPromptCacheKeyMode,
330    lineage_id: Option<&str>,
331) -> Option<String> {
332    if !session_affinity_enabled {
333        return None;
334    }
335    let provider = provider_name.trim().to_ascii_lowercase();
336    let lineage = lineage_id.map(str::trim).filter(|value| !value.is_empty())?;
337    match provider.as_str() {
338        "openai" | "merge-gateway" => match prompt_cache_key_mode {
339            OpenAIPromptCacheKeyMode::Session => {
340                Some(map_prompt_cache_key_for_provider(&provider, format!("vtcode:openai:{lineage}")))
341            }
342            OpenAIPromptCacheKeyMode::Off => None,
343        },
344        "openrouter" | "xai" => Some(format!("vtcode:{provider}:{lineage}")),
345        _ => None,
346    }
347}
348
349/// OpenAI prompt cache key derivation mode.
350#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
351#[derive(Debug, Clone, Deserialize, Serialize, PartialEq, Eq, Default)]
352#[serde(rename_all = "snake_case")]
353pub enum OpenAIPromptCacheKeyMode {
354    /// Do not send `prompt_cache_key` in OpenAI requests.
355    Off,
356    /// Send one stable `prompt_cache_key` per VT Code session.
357    #[default]
358    Session,
359}
360
361/// Anthropic Claude cache control settings
362#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
363#[derive(Debug, Clone, Deserialize, Serialize)]
364pub struct AnthropicPromptCacheSettings {
365    #[serde(default = "default_true")]
366    pub enabled: bool,
367
368    /// Default TTL in seconds for the first cache breakpoint (tools/system).
369    /// Anthropic only supports "5m" (300s) or "1h" (3600s) TTL formats.
370    /// Set to >= 3600 for 1-hour cache on tools and system prompts.
371    /// Default: 3600 (1 hour) - recommended for stable tool definitions
372    #[serde(default = "default_anthropic_tools_ttl")]
373    pub tools_ttl_seconds: u64,
374
375    /// TTL for subsequent cache breakpoints (messages).
376    /// Set to >= 3600 for 1-hour cache on messages.
377    /// Default: 300 (5 minutes) - recommended for frequently changing messages
378    #[serde(default = "default_anthropic_messages_ttl")]
379    pub messages_ttl_seconds: u64,
380
381    /// Maximum number of cache breakpoints to use (max 4 per Anthropic spec).
382    /// Default: 4
383    #[serde(default = "default_anthropic_max_breakpoints")]
384    pub max_breakpoints: u8,
385
386    /// Apply cache control to system prompts by default
387    #[serde(default = "default_true")]
388    pub cache_system_messages: bool,
389
390    /// Apply cache control to user messages exceeding threshold
391    #[serde(default = "default_true")]
392    pub cache_user_messages: bool,
393
394    /// Apply cache control to tool definitions by default
395    /// Default: true (tools are typically stable and benefit from longer caching)
396    #[serde(default = "default_true")]
397    pub cache_tool_definitions: bool,
398
399    /// Minimum message length (in characters) before applying cache control
400    /// to avoid caching very short messages that don't benefit from caching.
401    /// Default: 256 characters (~64 tokens)
402    #[serde(default = "default_min_message_length")]
403    pub min_message_length_for_cache: usize,
404
405    /// Extended TTL for Anthropic prompt caching (in seconds)
406    /// Set to >= 3600 for 1-hour cache on messages
407    #[serde(default = "default_anthropic_extended_ttl")]
408    pub extended_ttl_seconds: Option<u64>,
409
410    /// Prefer the 1h extended TTL for tools/system/messages even when the
411    /// per-breakpoint TTLs are 5m. Use when sessions idle past the 5m cache
412    /// window. Opt-in; 1h writes cost 2x base input.
413    #[serde(default)]
414    pub prefer_extended_ttl: bool,
415}
416
417impl Default for AnthropicPromptCacheSettings {
418    fn default() -> Self {
419        Self {
420            enabled: default_true(),
421            tools_ttl_seconds: default_anthropic_tools_ttl(),
422            messages_ttl_seconds: default_anthropic_messages_ttl(),
423            max_breakpoints: default_anthropic_max_breakpoints(),
424            cache_system_messages: default_true(),
425            cache_user_messages: default_true(),
426            cache_tool_definitions: default_true(),
427            min_message_length_for_cache: default_min_message_length(),
428            extended_ttl_seconds: default_anthropic_extended_ttl(),
429            prefer_extended_ttl: false,
430        }
431    }
432}
433
434/// Gemini API caching preferences
435#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
436#[derive(Debug, Clone, Deserialize, Serialize)]
437pub struct GeminiPromptCacheSettings {
438    #[serde(default = "default_true")]
439    pub enabled: bool,
440
441    #[serde(default = "default_gemini_mode")]
442    pub mode: GeminiPromptCacheMode,
443
444    #[serde(default = "default_gemini_min_prefix_tokens")]
445    min_prefix_tokens: u32,
446
447    /// TTL for explicit caches (ignored in implicit mode)
448    #[serde(default = "default_gemini_explicit_ttl")]
449    pub explicit_ttl_seconds: Option<u64>,
450}
451
452impl Default for GeminiPromptCacheSettings {
453    fn default() -> Self {
454        Self {
455            enabled: default_true(),
456            mode: GeminiPromptCacheMode::default(),
457            min_prefix_tokens: default_gemini_min_prefix_tokens(),
458            explicit_ttl_seconds: default_gemini_explicit_ttl(),
459        }
460    }
461}
462
463/// Gemini prompt caching mode selection
464#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
465#[derive(Debug, Clone, Deserialize, Serialize, PartialEq, Eq)]
466#[serde(rename_all = "snake_case")]
467#[derive(Default)]
468pub enum GeminiPromptCacheMode {
469    #[default]
470    Implicit,
471    Explicit,
472    Off,
473}
474
475/// OpenRouter passthrough caching controls
476#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
477#[derive(Debug, Clone, Deserialize, Serialize)]
478pub struct OpenRouterPromptCacheSettings {
479    #[serde(default = "default_true")]
480    pub enabled: bool,
481
482    /// Propagate provider cache instructions automatically
483    #[serde(default = "default_true")]
484    propagate_provider_capabilities: bool,
485
486    /// Surface cache savings reported by OpenRouter
487    #[serde(default = "default_true")]
488    pub report_savings: bool,
489}
490
491impl Default for OpenRouterPromptCacheSettings {
492    fn default() -> Self {
493        Self {
494            enabled: default_true(),
495            propagate_provider_capabilities: default_true(),
496            report_savings: default_true(),
497        }
498    }
499}
500
501/// Moonshot prompt caching configuration (leverages server-side reuse)
502#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
503#[derive(Debug, Clone, Deserialize, Serialize)]
504pub struct MoonshotPromptCacheSettings {
505    #[serde(default = "default_moonshot_enabled")]
506    enabled: bool,
507}
508
509impl Default for MoonshotPromptCacheSettings {
510    fn default() -> Self {
511        Self { enabled: default_moonshot_enabled() }
512    }
513}
514
515/// DeepSeek prompt caching configuration (automatic KV cache reuse)
516#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
517#[derive(Debug, Clone, Deserialize, Serialize)]
518pub struct DeepSeekPromptCacheSettings {
519    #[serde(default = "default_true")]
520    pub enabled: bool,
521
522    /// Emit cache hit/miss metrics from responses when available
523    #[serde(default = "default_true")]
524    pub surface_metrics: bool,
525}
526
527impl Default for DeepSeekPromptCacheSettings {
528    fn default() -> Self {
529        Self {
530            enabled: default_true(),
531            surface_metrics: default_true(),
532        }
533    }
534}
535
536/// Z.AI prompt caching configuration (disabled until platform exposes metrics)
537#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
538#[derive(Debug, Clone, Deserialize, Serialize)]
539pub struct ZaiPromptCacheSettings {
540    #[serde(default = "default_zai_enabled")]
541    enabled: bool,
542}
543
544impl Default for ZaiPromptCacheSettings {
545    fn default() -> Self {
546        Self { enabled: default_zai_enabled() }
547    }
548}
549
550fn default_enabled() -> bool {
551    prompt_cache::DEFAULT_ENABLED
552}
553
554fn default_cache_dir() -> String {
555    VtCodePaths::resolve()
556        .map(|paths| paths.cache_dir().join("prompts").display().to_string())
557        .unwrap_or_default()
558}
559
560fn default_max_entries() -> usize {
561    prompt_cache::DEFAULT_MAX_ENTRIES
562}
563
564fn default_max_age_days() -> u64 {
565    prompt_cache::DEFAULT_MAX_AGE_DAYS
566}
567
568fn default_auto_cleanup() -> bool {
569    prompt_cache::DEFAULT_AUTO_CLEANUP
570}
571
572fn default_min_quality_threshold() -> f64 {
573    prompt_cache::DEFAULT_MIN_QUALITY_THRESHOLD
574}
575
576fn default_cache_friendly_prompt_shaping() -> bool {
577    prompt_cache::DEFAULT_CACHE_FRIENDLY_PROMPT_SHAPING
578}
579
580fn default_openai_min_prefix_tokens() -> u32 {
581    prompt_cache::OPENAI_MIN_PREFIX_TOKENS
582}
583
584fn default_openai_idle_expiration() -> u64 {
585    prompt_cache::OPENAI_IDLE_EXPIRATION_SECONDS
586}
587
588fn default_openai_prompt_cache_key_mode() -> OpenAIPromptCacheKeyMode {
589    OpenAIPromptCacheKeyMode::Session
590}
591
592#[allow(dead_code, reason = "Intentional compatibility, platform, or test-only suppression.")]
593fn default_anthropic_extended_ttl() -> Option<u64> {
594    Some(prompt_cache::ANTHROPIC_EXTENDED_TTL_SECONDS)
595}
596
597fn default_anthropic_tools_ttl() -> u64 {
598    prompt_cache::ANTHROPIC_TOOLS_TTL_SECONDS
599}
600
601fn default_anthropic_messages_ttl() -> u64 {
602    prompt_cache::ANTHROPIC_MESSAGES_TTL_SECONDS
603}
604
605fn default_anthropic_max_breakpoints() -> u8 {
606    prompt_cache::ANTHROPIC_MAX_BREAKPOINTS
607}
608
609#[allow(dead_code, reason = "Intentional compatibility, platform, or test-only suppression.")]
610fn default_min_message_length() -> usize {
611    prompt_cache::ANTHROPIC_MIN_MESSAGE_LENGTH_FOR_CACHE
612}
613
614fn default_gemini_min_prefix_tokens() -> u32 {
615    prompt_cache::GEMINI_MIN_PREFIX_TOKENS
616}
617
618fn default_gemini_explicit_ttl() -> Option<u64> {
619    Some(prompt_cache::GEMINI_EXPLICIT_DEFAULT_TTL_SECONDS)
620}
621
622fn default_gemini_mode() -> GeminiPromptCacheMode {
623    GeminiPromptCacheMode::Implicit
624}
625
626fn default_zai_enabled() -> bool {
627    prompt_cache::ZAI_CACHE_ENABLED
628}
629
630fn default_moonshot_enabled() -> bool {
631    prompt_cache::MOONSHOT_CACHE_ENABLED
632}
633
634fn resolve_path(input: &str, workspace_root: Option<&Path>) -> PathBuf {
635    let trimmed = input.trim();
636    if trimmed.is_empty() {
637        return resolve_default_cache_dir();
638    }
639
640    if let Some(stripped) = trimmed.strip_prefix("~/").or_else(|| trimmed.strip_prefix("~\\")) {
641        if let Some(home) = dirs::home_dir() {
642            return home.join(stripped);
643        }
644        return PathBuf::from(stripped);
645    }
646
647    let candidate = Path::new(trimmed);
648    if candidate.is_absolute() {
649        return candidate.to_path_buf();
650    }
651
652    if let Some(root) = workspace_root {
653        return root.join(candidate);
654    }
655
656    candidate.to_path_buf()
657}
658
659fn resolve_default_cache_dir() -> PathBuf {
660    VtCodePaths::resolve()
661        .map(|paths| paths.cache_dir().join("prompts"))
662        .unwrap_or_default()
663}
664
665impl PromptCachingConfig {
666    /// Validate prompt cache config and provider overrides
667    pub(crate) fn validate(&self) -> anyhow::Result<()> {
668        // Validate OpenAI provider settings
669        self.providers.openai.validate()?;
670        Ok(())
671    }
672}
673
674#[cfg(test)]
675mod tests {
676    use super::*;
677    use assert_fs::TempDir;
678    use std::fs;
679
680    /// Walk up from `CARGO_MANIFEST_DIR` to the directory whose `Cargo.toml`
681    /// contains `[workspace]`. Hardcoding `parent()` breaks when crates are
682    /// nested (e.g. `crates/codegen/<crate>` — three levels deep, not one).
683    fn find_workspace_root() -> PathBuf {
684        let manifest = PathBuf::from(env!("CARGO_MANIFEST_DIR"));
685        let mut current = manifest.as_path();
686        while let Some(parent) = current.parent() {
687            let cargo_toml = parent.join("Cargo.toml");
688            if cargo_toml.is_file() && fs::read_to_string(&cargo_toml).is_ok_and(|c| c.contains("[workspace]")) {
689                return parent.to_path_buf();
690            }
691            current = parent;
692        }
693        panic!("could not find workspace root from CARGO_MANIFEST_DIR");
694    }
695
696    #[test]
697    fn prompt_caching_defaults_align_with_constants() {
698        let cfg = PromptCachingConfig::default();
699        assert!(cfg.enabled);
700        assert_eq!(cfg.max_entries, prompt_cache::DEFAULT_MAX_ENTRIES);
701        assert_eq!(cfg.max_age_days, prompt_cache::DEFAULT_MAX_AGE_DAYS);
702        assert!((cfg.min_quality_threshold - prompt_cache::DEFAULT_MIN_QUALITY_THRESHOLD).abs() < f64::EPSILON);
703        assert_eq!(cfg.cache_friendly_prompt_shaping, prompt_cache::DEFAULT_CACHE_FRIENDLY_PROMPT_SHAPING);
704        assert!(cfg.providers.openai.enabled);
705        assert_eq!(cfg.providers.openai.min_prefix_tokens, prompt_cache::OPENAI_MIN_PREFIX_TOKENS);
706        assert_eq!(cfg.providers.openai.prompt_cache_key_mode, OpenAIPromptCacheKeyMode::Session);
707        assert_eq!(cfg.providers.anthropic.extended_ttl_seconds, Some(prompt_cache::ANTHROPIC_EXTENDED_TTL_SECONDS));
708        assert_eq!(cfg.providers.gemini.mode, GeminiPromptCacheMode::Implicit);
709        assert!(cfg.providers.moonshot.enabled);
710        assert_eq!(cfg.providers.openai.prompt_cache_retention, None);
711    }
712
713    #[test]
714    fn resolve_cache_dir_expands_home() {
715        let cfg = PromptCachingConfig {
716            cache_dir: "~/.custom/cache".to_string(),
717            ..PromptCachingConfig::default()
718        };
719        let resolved = cfg.resolve_cache_dir(None);
720        if let Some(home) = dirs::home_dir() {
721            assert!(resolved.starts_with(home));
722        } else {
723            assert_eq!(resolved, PathBuf::from(".custom/cache"));
724        }
725    }
726
727    #[test]
728    fn resolve_cache_dir_uses_workspace_when_relative() {
729        let temp = TempDir::new().unwrap();
730        let workspace = temp.path();
731        let cfg = PromptCachingConfig {
732            cache_dir: "relative/cache".to_string(),
733            ..PromptCachingConfig::default()
734        };
735        let resolved = cfg.resolve_cache_dir(Some(workspace));
736        assert_eq!(resolved, workspace.join("relative/cache"));
737    }
738
739    #[test]
740    fn prompt_cache_retention_deserializes_valid_values() {
741        let cfg: OpenAIPromptCacheSettings = toml::from_str(
742            r#"
743            prompt_cache_retention = "in_memory"
744            "#,
745        )
746        .unwrap();
747        assert_eq!(cfg.prompt_cache_retention, Some(PromptCacheRetention::InMemory));
748
749        let cfg2: OpenAIPromptCacheSettings = toml::from_str(
750            r#"
751            prompt_cache_retention = "24h"
752            "#,
753        )
754        .unwrap();
755        assert_eq!(cfg2.prompt_cache_retention, Some(PromptCacheRetention::H24));
756    }
757
758    #[test]
759    fn prompt_cache_retention_unknown_values_deserialize_as_unknown() {
760        let cfg: OpenAIPromptCacheSettings = toml::from_str(
761            r#"
762            prompt_cache_retention = "5m"
763            "#,
764        )
765        .unwrap();
766        assert_eq!(cfg.prompt_cache_retention, Some(PromptCacheRetention::Unknown));
767    }
768
769    /// `PromptCacheRetention` exposes the same value through `as_str()`/
770    /// `Display` and the serde wire form. Both the config table and the OpenAI
771    /// API wire format read this enum, so a one-sided change (e.g. a bare
772    /// `#[serde(other)]` that serializes `Unknown` as `"Unknown"`) would desync
773    /// them. Pin every variant.
774    #[test]
775    fn prompt_cache_retention_as_str_matches_serde() {
776        for retention in [
777            PromptCacheRetention::InMemory,
778            PromptCacheRetention::H24,
779            PromptCacheRetention::Unknown,
780        ] {
781            assert_eq!(
782                serde_json::to_value(retention).unwrap(),
783                serde_json::json!(retention.as_str()),
784                "serde wire form drifted from as_str() for {retention:?}"
785            );
786        }
787    }
788
789    #[test]
790    fn validate_prompt_cache_rejects_unknown_retention() {
791        let mut cfg = PromptCachingConfig::default();
792        cfg.providers.openai.prompt_cache_retention = Some(PromptCacheRetention::Unknown);
793        assert!(cfg.validate().is_err());
794    }
795
796    #[test]
797    fn prompt_cache_key_mode_parses_from_toml() {
798        let parsed: PromptCachingConfig = toml::from_str(
799            r#"
800[providers.openai]
801prompt_cache_key_mode = "off"
802"#,
803        )
804        .expect("prompt cache config should parse");
805
806        assert_eq!(parsed.providers.openai.prompt_cache_key_mode, OpenAIPromptCacheKeyMode::Off);
807    }
808
809    #[test]
810    fn build_openai_prompt_cache_key_uses_trimmed_lineage_id() {
811        let key = build_openai_prompt_cache_key(true, &OpenAIPromptCacheKeyMode::Session, Some(" lineage-abc "));
812
813        assert_eq!(key.as_deref(), Some("vtcode:openai:lineage-abc"));
814    }
815
816    #[test]
817    fn map_prompt_cache_key_namespaces_merge_gateway() {
818        assert_eq!(
819            map_prompt_cache_key_for_provider("merge-gateway", "vtcode:openai:abc".to_string()),
820            "vtcode:merge:abc"
821        );
822        assert_eq!(map_prompt_cache_key_for_provider("openai", "vtcode:openai:abc".to_string()), "vtcode:openai:abc");
823        assert_eq!(
824            map_prompt_cache_key_for_provider("openrouter", "vtcode:openai:abc".to_string()),
825            "vtcode:openrouter:abc"
826        );
827        assert_eq!(map_prompt_cache_key_for_provider("xai", "vtcode:openai:abc".to_string()), "vtcode:xai:abc");
828    }
829
830    #[test]
831    fn session_affinity_key_gate_covers_openrouter_and_xai() {
832        assert!(session_affinity_key_enabled("openrouter", true, false));
833        assert!(session_affinity_key_enabled("xai", true, false));
834        assert!(session_affinity_key_enabled("openai", true, true));
835        assert!(!session_affinity_key_enabled("openai", true, false));
836        assert!(!session_affinity_key_enabled("openrouter", false, true));
837        assert!(!session_affinity_key_enabled("anthropic", true, true));
838        assert!(session_affinity_provider("openrouter"));
839        assert!(session_affinity_provider("xai"));
840        assert!(!session_affinity_provider("anthropic"));
841    }
842
843    #[test]
844    fn build_session_affinity_prompt_cache_key_namespaces_providers() {
845        let mode = OpenAIPromptCacheKeyMode::Session;
846        assert_eq!(
847            build_session_affinity_prompt_cache_key("openai", true, &mode, Some("lineage-1")).as_deref(),
848            Some("vtcode:openai:lineage-1")
849        );
850        assert_eq!(
851            build_session_affinity_prompt_cache_key("merge-gateway", true, &mode, Some("lineage-1")).as_deref(),
852            Some("vtcode:merge:lineage-1")
853        );
854        assert_eq!(
855            build_session_affinity_prompt_cache_key("openrouter", true, &mode, Some("lineage-1")).as_deref(),
856            Some("vtcode:openrouter:lineage-1")
857        );
858        assert_eq!(
859            build_session_affinity_prompt_cache_key("xai", true, &mode, Some("lineage-1")).as_deref(),
860            Some("vtcode:xai:lineage-1")
861        );
862        // OpenAI/Merge honor Off mode; OpenRouter/xAI ignore it for sticky routing.
863        assert_eq!(
864            build_session_affinity_prompt_cache_key("openai", true, &OpenAIPromptCacheKeyMode::Off, Some("lineage-1")),
865            None
866        );
867        assert_eq!(
868            build_session_affinity_prompt_cache_key(
869                "openrouter",
870                true,
871                &OpenAIPromptCacheKeyMode::Off,
872                Some("lineage-1")
873            )
874            .as_deref(),
875            Some("vtcode:openrouter:lineage-1")
876        );
877        // Disabled gate or blank lineage yields None.
878        assert_eq!(build_session_affinity_prompt_cache_key("openrouter", false, &mode, Some("lineage-1")), None);
879        assert_eq!(build_session_affinity_prompt_cache_key("openrouter", true, &mode, Some("  ")), None);
880        assert_eq!(build_session_affinity_prompt_cache_key("anthropic", true, &mode, Some("lineage-1")), None);
881    }
882
883    #[test]
884    fn build_openai_prompt_cache_key_honors_disabled_or_off_mode() {
885        assert_eq!(build_openai_prompt_cache_key(false, &OpenAIPromptCacheKeyMode::Session, Some("id")), None);
886        assert_eq!(build_openai_prompt_cache_key(true, &OpenAIPromptCacheKeyMode::Off, Some("id")), None);
887        assert_eq!(build_openai_prompt_cache_key(true, &OpenAIPromptCacheKeyMode::Session, Some("  ")), None);
888    }
889
890    #[test]
891    fn provider_enablement_respects_global_and_provider_flags() {
892        let mut cfg = PromptCachingConfig { enabled: true, ..PromptCachingConfig::default() };
893        cfg.providers.openai.enabled = true;
894        assert!(cfg.is_provider_enabled("openai"));
895        assert!(cfg.is_provider_enabled("merge-gateway"));
896
897        cfg.enabled = false;
898        assert!(!cfg.is_provider_enabled("openai"));
899        assert!(!cfg.is_provider_enabled("merge-gateway"));
900    }
901
902    #[test]
903    fn provider_enablement_handles_aliases_and_modes() {
904        let mut cfg = PromptCachingConfig { enabled: true, ..PromptCachingConfig::default() };
905
906        cfg.providers.anthropic.enabled = true;
907        assert!(cfg.is_provider_enabled("minimax"));
908
909        cfg.providers.gemini.enabled = true;
910        cfg.providers.gemini.mode = GeminiPromptCacheMode::Off;
911        assert!(!cfg.is_provider_enabled("gemini"));
912    }
913
914    #[test]
915    fn gap_threshold_uses_provider_defaults() {
916        let cfg = PromptCachingConfig::default();
917        assert_eq!(cfg.gap_threshold_secs("anthropic"), Some(prompt_cache::ANTHROPIC_CACHE_GAP_WARNING_SECONDS));
918        assert_eq!(cfg.gap_threshold_secs("minimax"), Some(prompt_cache::ANTHROPIC_CACHE_GAP_WARNING_SECONDS));
919        assert_eq!(cfg.gap_threshold_secs("openai"), Some(prompt_cache::OPENAI_CACHE_GAP_WARNING_SECONDS));
920        assert_eq!(cfg.gap_threshold_secs("gemini"), Some(prompt_cache::DEFAULT_CACHE_GAP_WARNING_SECONDS));
921    }
922
923    #[test]
924    fn gap_threshold_respects_disable_switches() {
925        let cfg = PromptCachingConfig { gap_warning_enabled: false, ..Default::default() };
926        assert_eq!(cfg.gap_threshold_secs("anthropic"), None);
927
928        let mut cfg = PromptCachingConfig::default();
929        cfg.providers.anthropic.enabled = false;
930        assert_eq!(cfg.gap_threshold_secs("anthropic"), None);
931        assert_eq!(cfg.gap_threshold_secs("minimax"), None);
932
933        let cfg = PromptCachingConfig { enabled: false, ..Default::default() };
934        assert_eq!(cfg.gap_threshold_secs("openai"), None);
935    }
936
937    #[test]
938    fn gap_threshold_skips_openai_with_extended_retention() {
939        let mut cfg = PromptCachingConfig::default();
940        cfg.providers.openai.prompt_cache_retention = Some(PromptCacheRetention::H24);
941        assert_eq!(cfg.gap_threshold_secs("openai"), None);
942
943        cfg.providers.openai.prompt_cache_retention = Some(PromptCacheRetention::InMemory);
944        assert_eq!(cfg.gap_threshold_secs("openai"), Some(prompt_cache::OPENAI_CACHE_GAP_WARNING_SECONDS));
945    }
946
947    #[test]
948    fn gap_threshold_honors_explicit_override() {
949        let cfg = PromptCachingConfig {
950            gap_warning_threshold_secs: Some(42),
951            ..Default::default()
952        };
953        assert_eq!(cfg.gap_threshold_secs("anthropic"), Some(42));
954        assert_eq!(cfg.gap_threshold_secs("openai"), Some(42));
955    }
956
957    #[test]
958    fn gap_warning_fields_parse_from_toml() {
959        let parsed: PromptCachingConfig = toml::from_str(
960            r#"
961gap_warning_enabled = false
962gap_warning_threshold_secs = 120
963"#,
964        )
965        .expect("prompt cache config should parse");
966
967        assert!(!parsed.gap_warning_enabled);
968        assert_eq!(parsed.gap_warning_threshold_secs, Some(120));
969
970        let defaults: PromptCachingConfig = toml::from_str("").expect("empty config");
971        assert!(defaults.gap_warning_enabled);
972        assert_eq!(defaults.gap_warning_threshold_secs, None);
973    }
974
975    #[test]
976    fn bundled_config_templates_match_prompt_cache_defaults() {
977        let manifest_dir = PathBuf::from(env!("CARGO_MANIFEST_DIR"));
978        let bootstrap_template =
979            fs::read_to_string(manifest_dir.join("data/default_config.toml")).expect("bootstrap config template");
980        let config: crate::VTCodeConfig = toml::from_str(&bootstrap_template).expect("parse bootstrap template");
981        assert!(config.prompt_cache.enabled);
982        assert!(config.prompt_cache.cache_friendly_prompt_shaping);
983        assert!(bootstrap_template.contains("# prompt_cache_retention = \"24h\""));
984
985        let workspace_root = find_workspace_root();
986        let example_config =
987            fs::read_to_string(workspace_root.join("vtcode.toml.example")).expect("vtcode.toml.example");
988        if example_config.contains("[prompt_cache]") {
989            assert!(example_config.contains("enabled = true"));
990            assert!(example_config.contains("cache_friendly_prompt_shaping = true"));
991            assert!(example_config.contains("# prompt_cache_retention = \"24h\""));
992        }
993
994        let prompt_cache_guide = fs::read_to_string(workspace_root.join("docs/tools/PROMPT_CACHING_GUIDE.md"))
995            .expect("prompt caching guide");
996        // The guide is markdownlint-wrapped, so sentences may be reflowed across
997        // lines; match against whitespace-normalized text instead of raw lines.
998        let normalized_guide = vtcode_commons::formatting::collapse_whitespace(&prompt_cache_guide);
999        assert!(
1000            normalized_guide
1001                .contains("VT Code enables `prompt_cache.cache_friendly_prompt_shaping = true` by default.")
1002        );
1003        assert!(
1004            normalized_guide
1005                .contains("Default: `None` (opt-in) - VT Code does not set prompt_cache_retention by default;")
1006        );
1007
1008        let field_reference = fs::read_to_string(workspace_root.join("docs/config/CONFIG_FIELD_REFERENCE.md"))
1009            .expect("config field reference");
1010        assert!(field_reference.contains("`prompt_cache.cache_friendly_prompt_shaping`"));
1011        assert!(field_reference.contains("`prompt_cache.providers.openai.prompt_cache_retention`"));
1012    }
1013
1014    #[test]
1015    fn bundled_example_config_parses_and_validates() {
1016        // Guards the tracked template against stale keys and schema drift:
1017        // the shipped example must load as VTCodeConfig and pass validation.
1018        let workspace_root = find_workspace_root();
1019        let example = fs::read_to_string(workspace_root.join("vtcode.toml.example")).expect("vtcode.toml.example");
1020        let config: crate::VTCodeConfig = toml::from_str(&example).expect("example config should parse");
1021        config.validate().expect("example config should validate");
1022    }
1023}