Skip to main content

vtcode_config/core/
prompt_cache.rs

1use crate::constants::prompt_cache;
2use crate::env_helpers::default_true;
3use serde::{Deserialize, Serialize};
4use std::path::{Path, PathBuf};
5use vtcode_commons::VtCodePaths;
6
7/// Global prompt caching configuration loaded from vtcode.toml
8#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
9#[derive(Debug, Clone, Deserialize, Serialize)]
10pub struct PromptCachingConfig {
11    /// Enable prompt caching features globally
12    #[serde(default = "default_enabled")]
13    pub enabled: bool,
14
15    /// Base directory for local prompt cache storage (supports `~` expansion)
16    #[serde(default = "default_cache_dir")]
17    pub cache_dir: String,
18
19    /// Maximum number of cached prompt entries to retain on disk
20    #[serde(default = "default_max_entries")]
21    pub max_entries: usize,
22
23    /// Maximum age (in days) before cached entries are purged
24    #[serde(default = "default_max_age_days")]
25    pub max_age_days: u64,
26
27    /// Automatically evict stale entries on startup/shutdown
28    #[serde(default = "default_auto_cleanup")]
29    pub enable_auto_cleanup: bool,
30
31    /// Minimum quality score required before persisting an entry
32    #[serde(default = "default_min_quality_threshold")]
33    pub min_quality_threshold: f64,
34
35    /// Enable prompt-shaping optimizations that improve provider-side cache locality.
36    /// When enabled, VT Code keeps volatile runtime context at the end of prompt text.
37    #[serde(default = "default_cache_friendly_prompt_shaping")]
38    pub cache_friendly_prompt_shaping: bool,
39
40    /// Keep the wire tool catalog identical across planning and execution
41    /// turns (union of both modes' tools). Planning toggles then do not
42    /// rewrite the tool prefix. The fail-closed execution gate still blocks
43    /// mutations during planning.
44    #[serde(default = "default_true")]
45    pub stable_tool_catalog_across_modes: bool,
46
47    /// Warn before a request when the pause since the previous request likely
48    /// exceeded the provider prompt cache lifetime (advisory only).
49    #[serde(default = "default_true")]
50    pub gap_warning_enabled: bool,
51
52    /// Override for the cache-gap warning threshold in seconds. When unset,
53    /// a provider-specific default is used (Anthropic 300s, OpenAI 600s).
54    #[serde(default)]
55    pub gap_warning_threshold_secs: Option<u64>,
56
57    /// Provider specific overrides
58    #[serde(default)]
59    pub providers: ProviderPromptCachingConfig,
60}
61
62impl Default for PromptCachingConfig {
63    fn default() -> Self {
64        let cache_dir = default_cache_dir();
65        Self {
66            enabled: default_enabled() && !cache_dir.is_empty(),
67            cache_dir,
68            max_entries: default_max_entries(),
69            max_age_days: default_max_age_days(),
70            enable_auto_cleanup: default_auto_cleanup(),
71            min_quality_threshold: default_min_quality_threshold(),
72            cache_friendly_prompt_shaping: default_cache_friendly_prompt_shaping(),
73            stable_tool_catalog_across_modes: true,
74            gap_warning_enabled: default_true(),
75            gap_warning_threshold_secs: None,
76            providers: ProviderPromptCachingConfig::default(),
77        }
78    }
79}
80
81impl PromptCachingConfig {
82    /// Resolve the configured cache directory to an absolute path
83    ///
84    /// - `~` is expanded to the user's home directory when available
85    /// - Relative paths are resolved against the provided workspace root when supplied
86    /// - Falls back to the configured string when neither applies
87    pub fn resolve_cache_dir(&self, workspace_root: Option<&Path>) -> PathBuf {
88        resolve_path(&self.cache_dir, workspace_root)
89    }
90
91    /// Returns true when prompt caching is active for the given provider runtime name.
92    pub fn is_provider_enabled(&self, provider_name: &str) -> bool {
93        if !self.enabled {
94            return false;
95        }
96
97        match provider_name.to_ascii_lowercase().as_str() {
98            "openai" | "merge-gateway" => self.providers.openai.enabled,
99            "anthropic" | "minimax" => self.providers.anthropic.enabled,
100            "gemini" => {
101                self.providers.gemini.enabled && !matches!(self.providers.gemini.mode, GeminiPromptCacheMode::Off)
102            }
103            "openrouter" => self.providers.openrouter.enabled,
104            "moonshot" => self.providers.moonshot.enabled,
105            "deepseek" => self.providers.deepseek.enabled,
106            "zai" => self.providers.zai.enabled,
107            _ => false,
108        }
109    }
110
111    /// Resolve the cache-gap warning threshold in seconds for a provider.
112    ///
113    /// Returns `None` when the warning should not fire: gap warnings disabled,
114    /// caching disabled for the provider, or OpenAI configured with 24h
115    /// extended retention (where a short pause cannot expire the cache).
116    pub fn gap_threshold_secs(&self, provider_name: &str) -> Option<u64> {
117        if !self.gap_warning_enabled || !self.is_provider_enabled(provider_name) {
118            return None;
119        }
120
121        let provider = provider_name.to_ascii_lowercase();
122        if provider == "openai"
123            && matches!(self.providers.openai.prompt_cache_retention, Some(PromptCacheRetention::H24))
124        {
125            return None;
126        }
127
128        if let Some(threshold) = self.gap_warning_threshold_secs {
129            return Some(threshold);
130        }
131
132        Some(match provider.as_str() {
133            "anthropic" | "minimax" => prompt_cache::ANTHROPIC_CACHE_GAP_WARNING_SECONDS,
134            "openai" => prompt_cache::OPENAI_CACHE_GAP_WARNING_SECONDS,
135            _ => prompt_cache::DEFAULT_CACHE_GAP_WARNING_SECONDS,
136        })
137    }
138}
139
140/// Per-provider configuration overrides
141#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
142#[derive(Debug, Clone, Deserialize, Serialize, Default)]
143pub struct ProviderPromptCachingConfig {
144    #[serde(default = "OpenAIPromptCacheSettings::default")]
145    pub openai: OpenAIPromptCacheSettings,
146
147    #[serde(default = "AnthropicPromptCacheSettings::default")]
148    pub anthropic: AnthropicPromptCacheSettings,
149
150    #[serde(default = "GeminiPromptCacheSettings::default")]
151    pub gemini: GeminiPromptCacheSettings,
152
153    #[serde(default = "OpenRouterPromptCacheSettings::default")]
154    pub openrouter: OpenRouterPromptCacheSettings,
155
156    #[serde(default = "MoonshotPromptCacheSettings::default")]
157    moonshot: MoonshotPromptCacheSettings,
158
159    #[serde(default = "DeepSeekPromptCacheSettings::default")]
160    pub deepseek: DeepSeekPromptCacheSettings,
161
162    #[serde(default = "ZaiPromptCacheSettings::default")]
163    zai: ZaiPromptCacheSettings,
164}
165
166/// OpenAI prompt cache retention policy.
167/// Controls the model-side prompt caching window.
168#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
169#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, Default)]
170pub enum PromptCacheRetention {
171    /// In-memory caching (short-lived, free)
172    #[serde(rename = "in_memory")]
173    InMemory,
174    /// 24-hour persistent caching
175    #[default]
176    #[serde(rename = "24h")]
177    H24,
178    /// Forward-compatible catch-all for unknown retention values
179    #[serde(other)]
180    Unknown,
181}
182
183impl PromptCacheRetention {
184    /// Returns the string representation for the OpenAI API wire format.
185    pub fn as_str(&self) -> &str {
186        match self {
187            Self::InMemory => "in_memory",
188            Self::H24 => "24h",
189            Self::Unknown => "unknown",
190        }
191    }
192}
193
194impl std::fmt::Display for PromptCacheRetention {
195    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
196        f.write_str(self.as_str())
197    }
198}
199
200/// OpenAI prompt caching controls (automatic with metrics)
201#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
202#[derive(Debug, Clone, Deserialize, Serialize)]
203pub struct OpenAIPromptCacheSettings {
204    #[serde(default = "default_true")]
205    pub enabled: bool,
206
207    #[serde(default = "default_openai_min_prefix_tokens")]
208    min_prefix_tokens: u32,
209
210    #[serde(default = "default_openai_idle_expiration")]
211    idle_expiration_seconds: u64,
212
213    #[serde(default = "default_true")]
214    pub surface_metrics: bool,
215
216    /// Strategy for generating OpenAI `prompt_cache_key`.
217    /// Session-scoped keys derive one stable key per VT Code conversation.
218    #[serde(default = "default_openai_prompt_cache_key_mode")]
219    pub prompt_cache_key_mode: OpenAIPromptCacheKeyMode,
220
221    /// Optional prompt cache retention policy to pass directly into OpenAI Responses API.
222    /// Supported values are "in_memory" and "24h". If set, VT Code will include
223    /// `prompt_cache_retention` in the request body to extend the model-side prompt
224    /// caching window or request the explicit in-memory policy.
225    #[serde(default)]
226    pub prompt_cache_retention: Option<PromptCacheRetention>,
227}
228
229impl Default for OpenAIPromptCacheSettings {
230    fn default() -> Self {
231        Self {
232            enabled: default_true(),
233            min_prefix_tokens: default_openai_min_prefix_tokens(),
234            idle_expiration_seconds: default_openai_idle_expiration(),
235            surface_metrics: default_true(),
236            prompt_cache_key_mode: default_openai_prompt_cache_key_mode(),
237            prompt_cache_retention: None,
238        }
239    }
240}
241
242impl OpenAIPromptCacheSettings {
243    /// Validate OpenAI provider prompt cache settings.
244    /// With the typed enum, invalid values are caught at deserialization time.
245    fn validate(&self) -> anyhow::Result<()> {
246        if let Some(PromptCacheRetention::Unknown) = self.prompt_cache_retention {
247            anyhow::bail!("prompt_cache_retention must be one of: in_memory, 24h");
248        }
249        Ok(())
250    }
251}
252
253/// Build a stable OpenAI `prompt_cache_key` for requests that should share
254/// provider-side cache routing.
255#[must_use]
256pub fn build_openai_prompt_cache_key(
257    prompt_cache_enabled: bool,
258    prompt_cache_key_mode: &OpenAIPromptCacheKeyMode,
259    lineage_id: Option<&str>,
260) -> Option<String> {
261    if !prompt_cache_enabled {
262        return None;
263    }
264
265    let lineage_id = lineage_id.map(str::trim).filter(|value| !value.is_empty());
266    match prompt_cache_key_mode {
267        OpenAIPromptCacheKeyMode::Session => lineage_id.map(|lineage_id| format!("vtcode:openai:{lineage_id}")),
268        OpenAIPromptCacheKeyMode::Off => None,
269    }
270}
271
272/// Rewrite an OpenAI-style cache key for routes that share the wire shape but
273/// need distinct routing stickiness. Keeps the stable session identifier while
274/// namespacing gateway/session-affinity traffic apart from native OpenAI.
275#[must_use]
276pub fn map_prompt_cache_key_for_provider(provider_name: &str, key: String) -> String {
277    let provider = provider_name.trim().to_ascii_lowercase();
278    match provider.as_str() {
279        "merge-gateway" => key.replacen("vtcode:openai:", "vtcode:merge:", 1),
280        "openrouter" => key.replacen("vtcode:openai:", "vtcode:openrouter:", 1),
281        "xai" => key.replacen("vtcode:openai:", "vtcode:xai:", 1),
282        _ => key,
283    }
284}
285
286/// Providers that send lineage-stable session/cache identity on the wire.
287#[must_use]
288pub fn session_affinity_provider(provider_name: &str) -> bool {
289    matches!(
290        provider_name.trim().to_ascii_lowercase().as_str(),
291        "openai" | "merge-gateway" | "openrouter" | "xai"
292    )
293}
294
295/// Whether `LLMRequest.prompt_cache_key` should carry session lineage for this provider.
296///
297/// OpenAI/Merge keep the OpenAI prompt-cache enablement gate.
298/// OpenRouter/xAI need lineage for documented session affinity even when the
299/// OpenAI-specific cache block is off (global prompt-cache master switch still applies).
300#[must_use]
301pub fn session_affinity_key_enabled(
302    provider_name: &str,
303    global_prompt_cache_enabled: bool,
304    openai_prompt_cache_enabled: bool,
305) -> bool {
306    if !global_prompt_cache_enabled {
307        return false;
308    }
309    match provider_name.trim().to_ascii_lowercase().as_str() {
310        "openai" | "merge-gateway" => openai_prompt_cache_enabled,
311        "openrouter" | "xai" => true,
312        _ => false,
313    }
314}
315
316/// Build the namespaced session-affinity / prompt-cache key for a request.
317///
318/// OpenAI/Merge honor [`OpenAIPromptCacheKeyMode`]. OpenRouter/xAI always use a
319/// stable session namespace when the affinity gate is on: sticky routing is
320/// independent of the OpenAI prompt-cache key mode.
321#[must_use]
322pub fn build_session_affinity_prompt_cache_key(
323    provider_name: &str,
324    session_affinity_enabled: bool,
325    prompt_cache_key_mode: &OpenAIPromptCacheKeyMode,
326    lineage_id: Option<&str>,
327) -> Option<String> {
328    if !session_affinity_enabled {
329        return None;
330    }
331    let provider = provider_name.trim().to_ascii_lowercase();
332    let lineage = lineage_id.map(str::trim).filter(|value| !value.is_empty())?;
333    match provider.as_str() {
334        "openai" | "merge-gateway" => match prompt_cache_key_mode {
335            OpenAIPromptCacheKeyMode::Session => {
336                Some(map_prompt_cache_key_for_provider(&provider, format!("vtcode:openai:{lineage}")))
337            }
338            OpenAIPromptCacheKeyMode::Off => None,
339        },
340        "openrouter" | "xai" => Some(format!("vtcode:{provider}:{lineage}")),
341        _ => None,
342    }
343}
344
345/// OpenAI prompt cache key derivation mode.
346#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
347#[derive(Debug, Clone, Deserialize, Serialize, PartialEq, Eq, Default)]
348#[serde(rename_all = "snake_case")]
349pub enum OpenAIPromptCacheKeyMode {
350    /// Do not send `prompt_cache_key` in OpenAI requests.
351    Off,
352    /// Send one stable `prompt_cache_key` per VT Code session.
353    #[default]
354    Session,
355}
356
357/// Anthropic Claude cache control settings
358#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
359#[derive(Debug, Clone, Deserialize, Serialize)]
360pub struct AnthropicPromptCacheSettings {
361    #[serde(default = "default_true")]
362    pub enabled: bool,
363
364    /// Default TTL in seconds for the first cache breakpoint (tools/system).
365    /// Anthropic only supports "5m" (300s) or "1h" (3600s) TTL formats.
366    /// Set to >= 3600 for 1-hour cache on tools and system prompts.
367    /// Default: 3600 (1 hour) - recommended for stable tool definitions
368    #[serde(default = "default_anthropic_tools_ttl")]
369    pub tools_ttl_seconds: u64,
370
371    /// TTL for subsequent cache breakpoints (messages).
372    /// Set to >= 3600 for 1-hour cache on messages.
373    /// Default: 300 (5 minutes) - recommended for frequently changing messages
374    #[serde(default = "default_anthropic_messages_ttl")]
375    pub messages_ttl_seconds: u64,
376
377    /// Maximum number of cache breakpoints to use (max 4 per Anthropic spec).
378    /// Default: 4
379    #[serde(default = "default_anthropic_max_breakpoints")]
380    pub max_breakpoints: u8,
381
382    /// Apply cache control to system prompts by default
383    #[serde(default = "default_true")]
384    pub cache_system_messages: bool,
385
386    /// Apply cache control to user messages exceeding threshold
387    #[serde(default = "default_true")]
388    pub cache_user_messages: bool,
389
390    /// Apply cache control to tool definitions by default
391    /// Default: true (tools are typically stable and benefit from longer caching)
392    #[serde(default = "default_true")]
393    pub cache_tool_definitions: bool,
394
395    /// Minimum message length (in characters) before applying cache control
396    /// to avoid caching very short messages that don't benefit from caching.
397    /// Default: 256 characters (~64 tokens)
398    #[serde(default = "default_min_message_length")]
399    pub min_message_length_for_cache: usize,
400
401    /// Extended TTL for Anthropic prompt caching (in seconds)
402    /// Set to >= 3600 for 1-hour cache on messages
403    #[serde(default = "default_anthropic_extended_ttl")]
404    pub extended_ttl_seconds: Option<u64>,
405
406    /// Prefer the 1h extended TTL for tools/system/messages even when the
407    /// per-breakpoint TTLs are 5m. Use when sessions idle past the 5m cache
408    /// window. Opt-in; 1h writes cost 2x base input.
409    #[serde(default)]
410    pub prefer_extended_ttl: bool,
411}
412
413impl Default for AnthropicPromptCacheSettings {
414    fn default() -> Self {
415        Self {
416            enabled: default_true(),
417            tools_ttl_seconds: default_anthropic_tools_ttl(),
418            messages_ttl_seconds: default_anthropic_messages_ttl(),
419            max_breakpoints: default_anthropic_max_breakpoints(),
420            cache_system_messages: default_true(),
421            cache_user_messages: default_true(),
422            cache_tool_definitions: default_true(),
423            min_message_length_for_cache: default_min_message_length(),
424            extended_ttl_seconds: default_anthropic_extended_ttl(),
425            prefer_extended_ttl: false,
426        }
427    }
428}
429
430/// Gemini API caching preferences
431#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
432#[derive(Debug, Clone, Deserialize, Serialize)]
433pub struct GeminiPromptCacheSettings {
434    #[serde(default = "default_true")]
435    pub enabled: bool,
436
437    #[serde(default = "default_gemini_mode")]
438    pub mode: GeminiPromptCacheMode,
439
440    #[serde(default = "default_gemini_min_prefix_tokens")]
441    min_prefix_tokens: u32,
442
443    /// TTL for explicit caches (ignored in implicit mode)
444    #[serde(default = "default_gemini_explicit_ttl")]
445    pub explicit_ttl_seconds: Option<u64>,
446}
447
448impl Default for GeminiPromptCacheSettings {
449    fn default() -> Self {
450        Self {
451            enabled: default_true(),
452            mode: GeminiPromptCacheMode::default(),
453            min_prefix_tokens: default_gemini_min_prefix_tokens(),
454            explicit_ttl_seconds: default_gemini_explicit_ttl(),
455        }
456    }
457}
458
459/// Gemini prompt caching mode selection
460#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
461#[derive(Debug, Clone, Deserialize, Serialize, PartialEq, Eq)]
462#[serde(rename_all = "snake_case")]
463#[derive(Default)]
464pub enum GeminiPromptCacheMode {
465    #[default]
466    Implicit,
467    Explicit,
468    Off,
469}
470
471/// OpenRouter passthrough caching controls
472#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
473#[derive(Debug, Clone, Deserialize, Serialize)]
474pub struct OpenRouterPromptCacheSettings {
475    #[serde(default = "default_true")]
476    pub enabled: bool,
477
478    /// Propagate provider cache instructions automatically
479    #[serde(default = "default_true")]
480    propagate_provider_capabilities: bool,
481
482    /// Surface cache savings reported by OpenRouter
483    #[serde(default = "default_true")]
484    pub report_savings: bool,
485}
486
487impl Default for OpenRouterPromptCacheSettings {
488    fn default() -> Self {
489        Self {
490            enabled: default_true(),
491            propagate_provider_capabilities: default_true(),
492            report_savings: default_true(),
493        }
494    }
495}
496
497/// Moonshot prompt caching configuration (leverages server-side reuse)
498#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
499#[derive(Debug, Clone, Deserialize, Serialize)]
500pub struct MoonshotPromptCacheSettings {
501    #[serde(default = "default_moonshot_enabled")]
502    enabled: bool,
503}
504
505impl Default for MoonshotPromptCacheSettings {
506    fn default() -> Self {
507        Self { enabled: default_moonshot_enabled() }
508    }
509}
510
511/// DeepSeek prompt caching configuration (automatic KV cache reuse)
512#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
513#[derive(Debug, Clone, Deserialize, Serialize)]
514pub struct DeepSeekPromptCacheSettings {
515    #[serde(default = "default_true")]
516    pub enabled: bool,
517
518    /// Emit cache hit/miss metrics from responses when available
519    #[serde(default = "default_true")]
520    pub surface_metrics: bool,
521}
522
523impl Default for DeepSeekPromptCacheSettings {
524    fn default() -> Self {
525        Self {
526            enabled: default_true(),
527            surface_metrics: default_true(),
528        }
529    }
530}
531
532/// Z.AI prompt caching configuration (disabled until platform exposes metrics)
533#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
534#[derive(Debug, Clone, Deserialize, Serialize)]
535pub struct ZaiPromptCacheSettings {
536    #[serde(default = "default_zai_enabled")]
537    enabled: bool,
538}
539
540impl Default for ZaiPromptCacheSettings {
541    fn default() -> Self {
542        Self { enabled: default_zai_enabled() }
543    }
544}
545
546fn default_enabled() -> bool {
547    prompt_cache::DEFAULT_ENABLED
548}
549
550fn default_cache_dir() -> String {
551    VtCodePaths::resolve()
552        .map(|paths| paths.cache_dir().join("prompts").display().to_string())
553        .unwrap_or_default()
554}
555
556fn default_max_entries() -> usize {
557    prompt_cache::DEFAULT_MAX_ENTRIES
558}
559
560fn default_max_age_days() -> u64 {
561    prompt_cache::DEFAULT_MAX_AGE_DAYS
562}
563
564fn default_auto_cleanup() -> bool {
565    prompt_cache::DEFAULT_AUTO_CLEANUP
566}
567
568fn default_min_quality_threshold() -> f64 {
569    prompt_cache::DEFAULT_MIN_QUALITY_THRESHOLD
570}
571
572fn default_cache_friendly_prompt_shaping() -> bool {
573    prompt_cache::DEFAULT_CACHE_FRIENDLY_PROMPT_SHAPING
574}
575
576fn default_openai_min_prefix_tokens() -> u32 {
577    prompt_cache::OPENAI_MIN_PREFIX_TOKENS
578}
579
580fn default_openai_idle_expiration() -> u64 {
581    prompt_cache::OPENAI_IDLE_EXPIRATION_SECONDS
582}
583
584fn default_openai_prompt_cache_key_mode() -> OpenAIPromptCacheKeyMode {
585    OpenAIPromptCacheKeyMode::Session
586}
587
588#[allow(dead_code, reason = "Intentional compatibility, platform, or test-only suppression.")]
589fn default_anthropic_extended_ttl() -> Option<u64> {
590    Some(prompt_cache::ANTHROPIC_EXTENDED_TTL_SECONDS)
591}
592
593fn default_anthropic_tools_ttl() -> u64 {
594    prompt_cache::ANTHROPIC_TOOLS_TTL_SECONDS
595}
596
597fn default_anthropic_messages_ttl() -> u64 {
598    prompt_cache::ANTHROPIC_MESSAGES_TTL_SECONDS
599}
600
601fn default_anthropic_max_breakpoints() -> u8 {
602    prompt_cache::ANTHROPIC_MAX_BREAKPOINTS
603}
604
605#[allow(dead_code, reason = "Intentional compatibility, platform, or test-only suppression.")]
606fn default_min_message_length() -> usize {
607    prompt_cache::ANTHROPIC_MIN_MESSAGE_LENGTH_FOR_CACHE
608}
609
610fn default_gemini_min_prefix_tokens() -> u32 {
611    prompt_cache::GEMINI_MIN_PREFIX_TOKENS
612}
613
614fn default_gemini_explicit_ttl() -> Option<u64> {
615    Some(prompt_cache::GEMINI_EXPLICIT_DEFAULT_TTL_SECONDS)
616}
617
618fn default_gemini_mode() -> GeminiPromptCacheMode {
619    GeminiPromptCacheMode::Implicit
620}
621
622fn default_zai_enabled() -> bool {
623    prompt_cache::ZAI_CACHE_ENABLED
624}
625
626fn default_moonshot_enabled() -> bool {
627    prompt_cache::MOONSHOT_CACHE_ENABLED
628}
629
630fn resolve_path(input: &str, workspace_root: Option<&Path>) -> PathBuf {
631    let trimmed = input.trim();
632    if trimmed.is_empty() {
633        return resolve_default_cache_dir();
634    }
635
636    if let Some(stripped) = trimmed.strip_prefix("~/").or_else(|| trimmed.strip_prefix("~\\")) {
637        if let Some(home) = dirs::home_dir() {
638            return home.join(stripped);
639        }
640        return PathBuf::from(stripped);
641    }
642
643    let candidate = Path::new(trimmed);
644    if candidate.is_absolute() {
645        return candidate.to_path_buf();
646    }
647
648    if let Some(root) = workspace_root {
649        return root.join(candidate);
650    }
651
652    candidate.to_path_buf()
653}
654
655fn resolve_default_cache_dir() -> PathBuf {
656    VtCodePaths::resolve()
657        .map(|paths| paths.cache_dir().join("prompts"))
658        .unwrap_or_default()
659}
660
661impl PromptCachingConfig {
662    /// Validate prompt cache config and provider overrides
663    pub(crate) fn validate(&self) -> anyhow::Result<()> {
664        // Validate OpenAI provider settings
665        self.providers.openai.validate()?;
666        Ok(())
667    }
668}
669
670#[cfg(test)]
671mod tests {
672    use super::*;
673    use assert_fs::TempDir;
674    use std::fs;
675
676    /// Walk up from `CARGO_MANIFEST_DIR` to the directory whose `Cargo.toml`
677    /// contains `[workspace]`. Hardcoding `parent()` breaks when crates are
678    /// nested (e.g. `crates/codegen/<crate>` — three levels deep, not one).
679    fn find_workspace_root() -> PathBuf {
680        let manifest = PathBuf::from(env!("CARGO_MANIFEST_DIR"));
681        let mut current = manifest.as_path();
682        while let Some(parent) = current.parent() {
683            let cargo_toml = parent.join("Cargo.toml");
684            if cargo_toml.is_file() && fs::read_to_string(&cargo_toml).is_ok_and(|c| c.contains("[workspace]")) {
685                return parent.to_path_buf();
686            }
687            current = parent;
688        }
689        panic!("could not find workspace root from CARGO_MANIFEST_DIR");
690    }
691
692    #[test]
693    fn prompt_caching_defaults_align_with_constants() {
694        let cfg = PromptCachingConfig::default();
695        assert!(cfg.enabled);
696        assert_eq!(cfg.max_entries, prompt_cache::DEFAULT_MAX_ENTRIES);
697        assert_eq!(cfg.max_age_days, prompt_cache::DEFAULT_MAX_AGE_DAYS);
698        assert!((cfg.min_quality_threshold - prompt_cache::DEFAULT_MIN_QUALITY_THRESHOLD).abs() < f64::EPSILON);
699        assert_eq!(cfg.cache_friendly_prompt_shaping, prompt_cache::DEFAULT_CACHE_FRIENDLY_PROMPT_SHAPING);
700        assert!(cfg.providers.openai.enabled);
701        assert_eq!(cfg.providers.openai.min_prefix_tokens, prompt_cache::OPENAI_MIN_PREFIX_TOKENS);
702        assert_eq!(cfg.providers.openai.prompt_cache_key_mode, OpenAIPromptCacheKeyMode::Session);
703        assert_eq!(cfg.providers.anthropic.extended_ttl_seconds, Some(prompt_cache::ANTHROPIC_EXTENDED_TTL_SECONDS));
704        assert_eq!(cfg.providers.gemini.mode, GeminiPromptCacheMode::Implicit);
705        assert!(cfg.providers.moonshot.enabled);
706        assert_eq!(cfg.providers.openai.prompt_cache_retention, None);
707    }
708
709    #[test]
710    fn resolve_cache_dir_expands_home() {
711        let cfg = PromptCachingConfig {
712            cache_dir: "~/.custom/cache".to_string(),
713            ..PromptCachingConfig::default()
714        };
715        let resolved = cfg.resolve_cache_dir(None);
716        if let Some(home) = dirs::home_dir() {
717            assert!(resolved.starts_with(home));
718        } else {
719            assert_eq!(resolved, PathBuf::from(".custom/cache"));
720        }
721    }
722
723    #[test]
724    fn resolve_cache_dir_uses_workspace_when_relative() {
725        let temp = TempDir::new().unwrap();
726        let workspace = temp.path();
727        let cfg = PromptCachingConfig {
728            cache_dir: "relative/cache".to_string(),
729            ..PromptCachingConfig::default()
730        };
731        let resolved = cfg.resolve_cache_dir(Some(workspace));
732        assert_eq!(resolved, workspace.join("relative/cache"));
733    }
734
735    #[test]
736    fn prompt_cache_retention_deserializes_valid_values() {
737        let cfg: OpenAIPromptCacheSettings = toml::from_str(
738            r#"
739            prompt_cache_retention = "in_memory"
740            "#,
741        )
742        .unwrap();
743        assert_eq!(cfg.prompt_cache_retention, Some(PromptCacheRetention::InMemory));
744
745        let cfg2: OpenAIPromptCacheSettings = toml::from_str(
746            r#"
747            prompt_cache_retention = "24h"
748            "#,
749        )
750        .unwrap();
751        assert_eq!(cfg2.prompt_cache_retention, Some(PromptCacheRetention::H24));
752    }
753
754    #[test]
755    fn prompt_cache_retention_unknown_values_deserialize_as_unknown() {
756        let cfg: OpenAIPromptCacheSettings = toml::from_str(
757            r#"
758            prompt_cache_retention = "5m"
759            "#,
760        )
761        .unwrap();
762        assert_eq!(cfg.prompt_cache_retention, Some(PromptCacheRetention::Unknown));
763    }
764
765    #[test]
766    fn validate_prompt_cache_rejects_unknown_retention() {
767        let mut cfg = PromptCachingConfig::default();
768        cfg.providers.openai.prompt_cache_retention = Some(PromptCacheRetention::Unknown);
769        assert!(cfg.validate().is_err());
770    }
771
772    #[test]
773    fn prompt_cache_key_mode_parses_from_toml() {
774        let parsed: PromptCachingConfig = toml::from_str(
775            r#"
776[providers.openai]
777prompt_cache_key_mode = "off"
778"#,
779        )
780        .expect("prompt cache config should parse");
781
782        assert_eq!(parsed.providers.openai.prompt_cache_key_mode, OpenAIPromptCacheKeyMode::Off);
783    }
784
785    #[test]
786    fn build_openai_prompt_cache_key_uses_trimmed_lineage_id() {
787        let key = build_openai_prompt_cache_key(true, &OpenAIPromptCacheKeyMode::Session, Some(" lineage-abc "));
788
789        assert_eq!(key.as_deref(), Some("vtcode:openai:lineage-abc"));
790    }
791
792    #[test]
793    fn map_prompt_cache_key_namespaces_merge_gateway() {
794        assert_eq!(
795            map_prompt_cache_key_for_provider("merge-gateway", "vtcode:openai:abc".to_string()),
796            "vtcode:merge:abc"
797        );
798        assert_eq!(map_prompt_cache_key_for_provider("openai", "vtcode:openai:abc".to_string()), "vtcode:openai:abc");
799        assert_eq!(
800            map_prompt_cache_key_for_provider("openrouter", "vtcode:openai:abc".to_string()),
801            "vtcode:openrouter:abc"
802        );
803        assert_eq!(map_prompt_cache_key_for_provider("xai", "vtcode:openai:abc".to_string()), "vtcode:xai:abc");
804    }
805
806    #[test]
807    fn session_affinity_key_gate_covers_openrouter_and_xai() {
808        assert!(session_affinity_key_enabled("openrouter", true, false));
809        assert!(session_affinity_key_enabled("xai", true, false));
810        assert!(session_affinity_key_enabled("openai", true, true));
811        assert!(!session_affinity_key_enabled("openai", true, false));
812        assert!(!session_affinity_key_enabled("openrouter", false, true));
813        assert!(!session_affinity_key_enabled("anthropic", true, true));
814        assert!(session_affinity_provider("openrouter"));
815        assert!(session_affinity_provider("xai"));
816        assert!(!session_affinity_provider("anthropic"));
817    }
818
819    #[test]
820    fn build_session_affinity_prompt_cache_key_namespaces_providers() {
821        let mode = OpenAIPromptCacheKeyMode::Session;
822        assert_eq!(
823            build_session_affinity_prompt_cache_key("openai", true, &mode, Some("lineage-1")).as_deref(),
824            Some("vtcode:openai:lineage-1")
825        );
826        assert_eq!(
827            build_session_affinity_prompt_cache_key("merge-gateway", true, &mode, Some("lineage-1")).as_deref(),
828            Some("vtcode:merge:lineage-1")
829        );
830        assert_eq!(
831            build_session_affinity_prompt_cache_key("openrouter", true, &mode, Some("lineage-1")).as_deref(),
832            Some("vtcode:openrouter:lineage-1")
833        );
834        assert_eq!(
835            build_session_affinity_prompt_cache_key("xai", true, &mode, Some("lineage-1")).as_deref(),
836            Some("vtcode:xai:lineage-1")
837        );
838        // OpenAI/Merge honor Off mode; OpenRouter/xAI ignore it for sticky routing.
839        assert_eq!(
840            build_session_affinity_prompt_cache_key("openai", true, &OpenAIPromptCacheKeyMode::Off, Some("lineage-1")),
841            None
842        );
843        assert_eq!(
844            build_session_affinity_prompt_cache_key(
845                "openrouter",
846                true,
847                &OpenAIPromptCacheKeyMode::Off,
848                Some("lineage-1")
849            )
850            .as_deref(),
851            Some("vtcode:openrouter:lineage-1")
852        );
853        // Disabled gate or blank lineage yields None.
854        assert_eq!(build_session_affinity_prompt_cache_key("openrouter", false, &mode, Some("lineage-1")), None);
855        assert_eq!(build_session_affinity_prompt_cache_key("openrouter", true, &mode, Some("  ")), None);
856        assert_eq!(build_session_affinity_prompt_cache_key("anthropic", true, &mode, Some("lineage-1")), None);
857    }
858
859    #[test]
860    fn build_openai_prompt_cache_key_honors_disabled_or_off_mode() {
861        assert_eq!(build_openai_prompt_cache_key(false, &OpenAIPromptCacheKeyMode::Session, Some("id")), None);
862        assert_eq!(build_openai_prompt_cache_key(true, &OpenAIPromptCacheKeyMode::Off, Some("id")), None);
863        assert_eq!(build_openai_prompt_cache_key(true, &OpenAIPromptCacheKeyMode::Session, Some("  ")), None);
864    }
865
866    #[test]
867    fn provider_enablement_respects_global_and_provider_flags() {
868        let mut cfg = PromptCachingConfig { enabled: true, ..PromptCachingConfig::default() };
869        cfg.providers.openai.enabled = true;
870        assert!(cfg.is_provider_enabled("openai"));
871        assert!(cfg.is_provider_enabled("merge-gateway"));
872
873        cfg.enabled = false;
874        assert!(!cfg.is_provider_enabled("openai"));
875        assert!(!cfg.is_provider_enabled("merge-gateway"));
876    }
877
878    #[test]
879    fn provider_enablement_handles_aliases_and_modes() {
880        let mut cfg = PromptCachingConfig { enabled: true, ..PromptCachingConfig::default() };
881
882        cfg.providers.anthropic.enabled = true;
883        assert!(cfg.is_provider_enabled("minimax"));
884
885        cfg.providers.gemini.enabled = true;
886        cfg.providers.gemini.mode = GeminiPromptCacheMode::Off;
887        assert!(!cfg.is_provider_enabled("gemini"));
888    }
889
890    #[test]
891    fn gap_threshold_uses_provider_defaults() {
892        let cfg = PromptCachingConfig::default();
893        assert_eq!(cfg.gap_threshold_secs("anthropic"), Some(prompt_cache::ANTHROPIC_CACHE_GAP_WARNING_SECONDS));
894        assert_eq!(cfg.gap_threshold_secs("minimax"), Some(prompt_cache::ANTHROPIC_CACHE_GAP_WARNING_SECONDS));
895        assert_eq!(cfg.gap_threshold_secs("openai"), Some(prompt_cache::OPENAI_CACHE_GAP_WARNING_SECONDS));
896        assert_eq!(cfg.gap_threshold_secs("gemini"), Some(prompt_cache::DEFAULT_CACHE_GAP_WARNING_SECONDS));
897    }
898
899    #[test]
900    fn gap_threshold_respects_disable_switches() {
901        let cfg = PromptCachingConfig { gap_warning_enabled: false, ..Default::default() };
902        assert_eq!(cfg.gap_threshold_secs("anthropic"), None);
903
904        let mut cfg = PromptCachingConfig::default();
905        cfg.providers.anthropic.enabled = false;
906        assert_eq!(cfg.gap_threshold_secs("anthropic"), None);
907        assert_eq!(cfg.gap_threshold_secs("minimax"), None);
908
909        let cfg = PromptCachingConfig { enabled: false, ..Default::default() };
910        assert_eq!(cfg.gap_threshold_secs("openai"), None);
911    }
912
913    #[test]
914    fn gap_threshold_skips_openai_with_extended_retention() {
915        let mut cfg = PromptCachingConfig::default();
916        cfg.providers.openai.prompt_cache_retention = Some(PromptCacheRetention::H24);
917        assert_eq!(cfg.gap_threshold_secs("openai"), None);
918
919        cfg.providers.openai.prompt_cache_retention = Some(PromptCacheRetention::InMemory);
920        assert_eq!(cfg.gap_threshold_secs("openai"), Some(prompt_cache::OPENAI_CACHE_GAP_WARNING_SECONDS));
921    }
922
923    #[test]
924    fn gap_threshold_honors_explicit_override() {
925        let cfg = PromptCachingConfig {
926            gap_warning_threshold_secs: Some(42),
927            ..Default::default()
928        };
929        assert_eq!(cfg.gap_threshold_secs("anthropic"), Some(42));
930        assert_eq!(cfg.gap_threshold_secs("openai"), Some(42));
931    }
932
933    #[test]
934    fn gap_warning_fields_parse_from_toml() {
935        let parsed: PromptCachingConfig = toml::from_str(
936            r#"
937gap_warning_enabled = false
938gap_warning_threshold_secs = 120
939"#,
940        )
941        .expect("prompt cache config should parse");
942
943        assert!(!parsed.gap_warning_enabled);
944        assert_eq!(parsed.gap_warning_threshold_secs, Some(120));
945
946        let defaults: PromptCachingConfig = toml::from_str("").expect("empty config");
947        assert!(defaults.gap_warning_enabled);
948        assert_eq!(defaults.gap_warning_threshold_secs, None);
949    }
950
951    #[test]
952    fn bundled_config_templates_match_prompt_cache_defaults() {
953        let manifest_dir = PathBuf::from(env!("CARGO_MANIFEST_DIR"));
954        let bootstrap_template =
955            fs::read_to_string(manifest_dir.join("data/default_config.toml")).expect("bootstrap config template");
956        let config: crate::VTCodeConfig = toml::from_str(&bootstrap_template).expect("parse bootstrap template");
957        assert!(config.prompt_cache.enabled);
958        assert!(config.prompt_cache.cache_friendly_prompt_shaping);
959        assert!(bootstrap_template.contains("# prompt_cache_retention = \"24h\""));
960
961        let workspace_root = find_workspace_root();
962        let example_config =
963            fs::read_to_string(workspace_root.join("vtcode.toml.example")).expect("vtcode.toml.example");
964        if example_config.contains("[prompt_cache]") {
965            assert!(example_config.contains("enabled = true"));
966            assert!(example_config.contains("cache_friendly_prompt_shaping = true"));
967            assert!(example_config.contains("# prompt_cache_retention = \"24h\""));
968        }
969
970        let prompt_cache_guide = fs::read_to_string(workspace_root.join("docs/tools/PROMPT_CACHING_GUIDE.md"))
971            .expect("prompt caching guide");
972        // The guide is markdownlint-wrapped, so sentences may be reflowed across
973        // lines; match against whitespace-normalized text instead of raw lines.
974        let normalized_guide = vtcode_commons::formatting::collapse_whitespace(&prompt_cache_guide);
975        assert!(
976            normalized_guide
977                .contains("VT Code enables `prompt_cache.cache_friendly_prompt_shaping = true` by default.")
978        );
979        assert!(
980            normalized_guide
981                .contains("Default: `None` (opt-in) - VT Code does not set prompt_cache_retention by default;")
982        );
983
984        let field_reference = fs::read_to_string(workspace_root.join("docs/config/CONFIG_FIELD_REFERENCE.md"))
985            .expect("config field reference");
986        assert!(field_reference.contains("`prompt_cache.cache_friendly_prompt_shaping`"));
987        assert!(field_reference.contains("`prompt_cache.providers.openai.prompt_cache_retention`"));
988    }
989
990    #[test]
991    fn bundled_example_config_parses_and_validates() {
992        // Guards the tracked template against stale keys and schema drift:
993        // the shipped example must load as VTCodeConfig and pass validation.
994        let workspace_root = find_workspace_root();
995        let example = fs::read_to_string(workspace_root.join("vtcode.toml.example")).expect("vtcode.toml.example");
996        let config: crate::VTCodeConfig = toml::from_str(&example).expect("example config should parse");
997        config.validate().expect("example config should validate");
998    }
999}