Skip to main content

navi_core/
model.rs

1use crate::tool::{ToolDefinition, ToolInvocation};
2use anyhow::Result;
3use async_trait::async_trait;
4use futures_util::StreamExt;
5use futures_util::stream::BoxStream;
6use serde::{Deserialize, Serialize};
7
8/// A single part of a multimodal message content.
9///
10/// Models like GPT-4o, Claude, and Gemini accept messages with mixed
11/// text and attachment parts. When a [`ModelMessage`] contains non-empty
12/// [`ModelMessage::content_parts`], providers serialize each part
13/// according to their native wire format instead of using the plain
14/// [`ModelMessage::content`] string.
15#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
16#[serde(tag = "type", rename_all = "snake_case")]
17pub enum ContentPart {
18    /// A plain text content block.
19    Text {
20        /// The text content.
21        text: String,
22    },
23    /// An inline image (base64-encoded).
24    Image {
25        /// MIME type of the image (e.g. `"image/png"`, `"image/jpeg"`).
26        media_type: String,
27        /// Base64-encoded image data (no data-URL prefix, raw base64 only).
28        data: String,
29    },
30    /// An inline audio attachment (base64-encoded).
31    Audio {
32        /// MIME type of the audio (e.g. `"audio/mpeg"`, `"audio/wav"`).
33        media_type: String,
34        /// Base64-encoded audio data (no data-URL prefix, raw base64 only).
35        data: String,
36        /// Optional filename or user-facing label.
37        #[serde(default, skip_serializing_if = "Option::is_none")]
38        name: Option<String>,
39    },
40    /// An inline video attachment (base64-encoded).
41    Video {
42        /// MIME type of the video (e.g. `"video/mp4"`).
43        media_type: String,
44        /// Base64-encoded video data (no data-URL prefix, raw base64 only).
45        data: String,
46        /// Optional filename or user-facing label.
47        #[serde(default, skip_serializing_if = "Option::is_none")]
48        name: Option<String>,
49    },
50    /// An inline document attachment (base64-encoded).
51    Document {
52        /// MIME type of the document (e.g. `"application/pdf"`, `"text/plain"`).
53        media_type: String,
54        /// Base64-encoded document data (no data-URL prefix, raw base64 only).
55        data: String,
56        /// Optional filename or user-facing label.
57        #[serde(default, skip_serializing_if = "Option::is_none")]
58        name: Option<String>,
59    },
60}
61
62impl ContentPart {
63    /// Returns `true` if this is a text part.
64    pub fn is_text(&self) -> bool {
65        matches!(self, Self::Text { .. })
66    }
67
68    /// Returns `true` if this is an image part.
69    pub fn is_image(&self) -> bool {
70        matches!(self, Self::Image { .. })
71    }
72
73    /// Returns `true` if this is an audio part.
74    pub fn is_audio(&self) -> bool {
75        matches!(self, Self::Audio { .. })
76    }
77
78    /// Returns `true` if this is a video part.
79    pub fn is_video(&self) -> bool {
80        matches!(self, Self::Video { .. })
81    }
82
83    /// Returns `true` if this is a document part.
84    pub fn is_document(&self) -> bool {
85        matches!(self, Self::Document { .. })
86    }
87
88    /// Returns the attachment kind, if this part is an attachment.
89    pub fn attachment_kind(&self) -> Option<AttachmentKind> {
90        match self {
91            Self::Text { .. } => None,
92            Self::Image { .. } => Some(AttachmentKind::Image),
93            Self::Audio { .. } => Some(AttachmentKind::Audio),
94            Self::Video { .. } => Some(AttachmentKind::Video),
95            Self::Document { .. } => Some(AttachmentKind::Document),
96        }
97    }
98
99    /// Returns the MIME type for attachment parts.
100    pub fn media_type(&self) -> Option<&str> {
101        match self {
102            Self::Text { .. } => None,
103            Self::Image { media_type, .. }
104            | Self::Audio { media_type, .. }
105            | Self::Video { media_type, .. }
106            | Self::Document { media_type, .. } => Some(media_type),
107        }
108    }
109
110    /// Returns base64 data for attachment parts.
111    pub fn data(&self) -> Option<&str> {
112        match self {
113            Self::Text { .. } => None,
114            Self::Image { data, .. }
115            | Self::Audio { data, .. }
116            | Self::Video { data, .. }
117            | Self::Document { data, .. } => Some(data),
118        }
119    }
120
121    /// Optional filename/label for audio, video, or document attachments.
122    pub fn name(&self) -> Option<&str> {
123        match self {
124            Self::Audio { name, .. } | Self::Video { name, .. } | Self::Document { name, .. } => {
125                name.as_deref()
126            }
127            Self::Text { .. } | Self::Image { .. } => None,
128        }
129    }
130
131    /// Extracts the text content if this is a text part.
132    pub fn as_text(&self) -> Option<&str> {
133        match self {
134            Self::Text { text } => Some(text),
135            _ => None,
136        }
137    }
138
139    /// Returns all text content from a slice of parts, concatenated.
140    pub fn text_from_parts(parts: &[ContentPart]) -> String {
141        parts
142            .iter()
143            .filter_map(|p| p.as_text())
144            .collect::<Vec<_>>()
145            .join("")
146    }
147}
148
149/// Attachment modalities NAVI can route to specialized models.
150#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
151#[serde(rename_all = "snake_case")]
152pub enum AttachmentKind {
153    Image,
154    Audio,
155    Video,
156    Document,
157}
158
159impl AttachmentKind {
160    pub fn as_str(self) -> &'static str {
161        match self {
162            Self::Image => "image",
163            Self::Audio => "audio",
164            Self::Video => "video",
165            Self::Document => "document",
166        }
167    }
168}
169
170/// Trait for model provider backends that can stream and complete requests.
171///
172/// Implementors handle the wire protocol for a specific API (OpenAI, Anthropic,
173/// Gemini, etc.) while the engine works with the generic [`ModelRequest`] and
174/// [`ModelStreamEvent`] types.
175#[async_trait]
176pub trait ModelProvider: Send + Sync {
177    /// Starts a streaming request and returns a stream of [`ModelStreamEvent`].
178    fn stream(&self, request: ModelRequest) -> ModelStream;
179
180    /// Completes a request by consuming the full stream and returning the
181    /// accumulated text response. Default implementation calls [`Self::stream`].
182    async fn complete(&self, request: ModelRequest) -> Result<ModelResponse> {
183        let mut stream = self.stream(request);
184        let mut text = String::new();
185
186        while let Some(event) = stream.next().await {
187            match event? {
188                ModelStreamEvent::TextDelta { text: delta } => text.push_str(&delta),
189                ModelStreamEvent::Done => break,
190                ModelStreamEvent::Status { .. }
191                | ModelStreamEvent::Usage { .. }
192                | ModelStreamEvent::ThinkingDelta { .. }
193                | ModelStreamEvent::ToolCall(_) => {}
194            }
195        }
196
197        Ok(ModelResponse { text })
198    }
199
200    /// Lists available model identifiers from this provider.
201    ///
202    /// Returns an error if the provider does not support model listing.
203    async fn list_models(&self) -> Result<Vec<String>> {
204        anyhow::bail!("listing models is not supported by this provider")
205    }
206}
207
208/// A boxed async stream of [`ModelStreamEvent`] results from a provider.
209pub type ModelStream = BoxStream<'static, Result<ModelStreamEvent>>;
210
211/// A request to a model provider containing the conversation, model name,
212/// thinking configuration, and available tool definitions.
213#[derive(Debug, Clone, Serialize, Deserialize)]
214pub struct ModelRequest {
215    /// The model identifier to use (e.g. `"gpt-5.5"`, `"claude-sonnet-4-20250514"`).
216    pub model: String,
217    /// Stable base instructions sent in the provider's `instructions` field
218    /// (Responses API) or as the first system message (Chat Completions,
219    /// Anthropic, Gemini). Kept separate from [`Self::messages`] so that
220    /// dynamic context blocks (developer messages) don't invalidate the
221    /// provider's prompt cache for this prefix.
222    #[serde(default, skip_serializing_if = "Option::is_none")]
223    pub instructions: Option<String>,
224    /// The conversation messages to send to the model.
225    pub messages: Vec<ModelMessage>,
226    /// The thinking/reasoning effort level to request.
227    pub thinking: ThinkingConfig,
228    /// Tool definitions the model may invoke.
229    #[serde(default)]
230    pub tools: Vec<ToolDefinition>,
231    /// Stable session id for provider-side prompt-cache affinity.
232    ///
233    /// Providers such as Charm Hyper use this to set `x-session-id` and
234    /// `x-session-affinity` so consecutive turns of the same agent session
235    /// hit the same KV-cache shard. Not serialized to disk/transcripts.
236    #[serde(default, skip_serializing, skip_deserializing)]
237    pub session_id: Option<String>,
238}
239
240/// A single message in a model conversation.
241///
242/// Messages carry role, content, and optional tool-related metadata so the
243/// same type can represent system prompts, user input, assistant responses,
244/// tool calls, and tool results.
245#[derive(Debug, Clone, Serialize, Deserialize)]
246pub struct ModelMessage {
247    /// The conversational role of this message.
248    pub role: ModelRole,
249    /// The text content of the message.
250    pub content: String,
251    /// Multimodal content parts (text + images).
252    ///
253    /// When non-empty, providers use these parts instead of the plain
254    /// [`content`](Self::content) field to build the wire-format message.
255    /// This allows attaching images alongside text in user messages.
256    #[serde(default, skip_serializing_if = "Vec::is_empty")]
257    pub content_parts: Vec<ContentPart>,
258    /// For tool-result messages, the id of the tool call being answered.
259    #[serde(default)]
260    pub tool_call_id: Option<String>,
261    /// For tool-result messages, the name of the tool that produced this result.
262    #[serde(default)]
263    pub tool_name: Option<String>,
264    /// For assistant messages, the tool invocations requested by the model.
265    #[serde(default)]
266    pub tool_calls: Vec<ToolInvocation>,
267    /// Creation timestamp in milliseconds since Unix epoch (not serialized).
268    #[serde(default, skip_serializing, skip_deserializing)]
269    pub created_at: Option<u64>,
270    /// Optional thinking/reasoning content from the model.
271    #[serde(default, skip_serializing_if = "Option::is_none")]
272    pub thinking_content: Option<String>,
273}
274
275/// The conversational role of a [`ModelMessage`].
276#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
277#[serde(rename_all = "lowercase")]
278pub enum ModelRole {
279    /// System-level instructions (system prompt).
280    System,
281    /// Developer-level instructions (context blocks injected separately from
282    /// the base system prompt for provider cache efficiency).
283    Developer,
284    /// End-user input.
285    User,
286    /// Model-generated response.
287    Assistant,
288    /// A tool result returned to the model.
289    Tool,
290}
291
292/// A completed model response containing the full text output.
293#[derive(Debug, Clone, Serialize, Deserialize)]
294pub struct ModelResponse {
295    /// The full text content of the model's response.
296    pub text: String,
297}
298
299/// A single event from a model provider's streaming response.
300#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
301pub enum ModelStreamEvent {
302    /// An incremental text delta from the assistant.
303    TextDelta {
304        /// The incremental text content.
305        text: String,
306    },
307    /// An incremental thinking/reasoning delta.
308    ThinkingDelta {
309        /// The incremental thinking content.
310        text: String,
311    },
312    /// A status message from the provider (e.g. "processing", "queued").
313    Status {
314        /// Human-readable status label.
315        label: String,
316    },
317    /// Token usage information reported by the provider.
318    Usage {
319        /// Number of input/prompt tokens consumed, if reported.
320        input_tokens: Option<u64>,
321        /// Number of output/completion tokens produced, if reported.
322        output_tokens: Option<u64>,
323        /// Number of tokens written to the prompt cache (Anthropic).
324        cache_creation_tokens: Option<u64>,
325        /// Number of tokens read from the prompt cache (Anthropic).
326        cache_read_tokens: Option<u64>,
327    },
328    /// The model requested a tool invocation.
329    ToolCall(ToolInvocation),
330    /// The stream has ended.
331    Done,
332}
333
334impl ModelMessage {
335    /// Creates a system-role message.
336    pub fn system(content: impl Into<String>) -> Self {
337        Self::new(ModelRole::System, content)
338    }
339
340    /// Creates a developer-role message (context block injected separately
341    /// from the base system prompt for provider cache efficiency).
342    pub fn developer(content: impl Into<String>) -> Self {
343        Self::new(ModelRole::Developer, content)
344    }
345
346    /// Creates a user-role message.
347    pub fn user(content: impl Into<String>) -> Self {
348        Self::new(ModelRole::User, content)
349    }
350
351    /// Creates a user-role message with text and optional image attachments.
352    pub fn user_multimodal(content: impl Into<String>, parts: Vec<ContentPart>) -> Self {
353        Self {
354            role: ModelRole::User,
355            content: content.into(),
356            content_parts: parts,
357            tool_call_id: None,
358            tool_name: None,
359            tool_calls: Vec::new(),
360            created_at: Some(current_unix_millis()),
361            thinking_content: None,
362        }
363    }
364
365    /// Creates an assistant-role message without thinking content.
366    pub fn assistant(content: impl Into<String>) -> Self {
367        Self {
368            thinking_content: None,
369            ..Self::new(ModelRole::Assistant, content)
370        }
371    }
372
373    /// Creates an assistant-role message with optional thinking content.
374    pub fn assistant_with_thinking(content: impl Into<String>, thinking: Option<String>) -> Self {
375        Self {
376            thinking_content: thinking,
377            ..Self::new(ModelRole::Assistant, content)
378        }
379    }
380
381    /// Creates a tool-result message responding to a specific tool call.
382    pub fn tool_result(
383        tool_call_id: impl Into<String>,
384        tool_name: impl Into<String>,
385        content: impl Into<String>,
386    ) -> Self {
387        Self::tool_result_with_parts(tool_call_id, tool_name, content, Vec::new())
388    }
389
390    /// Creates a tool-result message with optional multimodal content parts
391    /// (e.g. images from `view_image` for vision-capable models).
392    pub fn tool_result_with_parts(
393        tool_call_id: impl Into<String>,
394        tool_name: impl Into<String>,
395        content: impl Into<String>,
396        content_parts: Vec<ContentPart>,
397    ) -> Self {
398        Self {
399            role: ModelRole::Tool,
400            content: content.into(),
401            content_parts,
402            tool_call_id: Some(tool_call_id.into()),
403            tool_name: Some(tool_name.into()),
404            tool_calls: Vec::new(),
405            created_at: Some(current_unix_millis()),
406            thinking_content: None,
407        }
408    }
409
410    /// Creates an assistant message that requests a single tool invocation.
411    pub fn assistant_tool_call(invocation: ToolInvocation) -> Self {
412        Self::assistant_tool_call_with_context(invocation, String::new(), None)
413    }
414
415    /// Creates an assistant message that requests a tool invocation with
416    /// accompanying text content and optional thinking.
417    pub fn assistant_tool_call_with_context(
418        invocation: ToolInvocation,
419        content: impl Into<String>,
420        thinking: Option<String>,
421    ) -> Self {
422        Self::assistant_tool_calls_with_context(vec![invocation], content, thinking)
423    }
424
425    pub fn assistant_tool_calls_with_context(
426        invocations: Vec<ToolInvocation>,
427        content: impl Into<String>,
428        thinking: Option<String>,
429    ) -> Self {
430        Self {
431            role: ModelRole::Assistant,
432            content: content.into(),
433            content_parts: Vec::new(),
434            tool_call_id: None,
435            tool_name: None,
436            tool_calls: invocations,
437            created_at: Some(current_unix_millis()),
438            thinking_content: thinking,
439        }
440    }
441
442    fn new(role: ModelRole, content: impl Into<String>) -> Self {
443        Self {
444            role,
445            content: content.into(),
446            content_parts: Vec::new(),
447            tool_call_id: None,
448            tool_name: None,
449            tool_calls: Vec::new(),
450            created_at: Some(current_unix_millis()),
451            thinking_content: None,
452        }
453    }
454}
455
456fn current_unix_millis() -> u64 {
457    std::time::SystemTime::now()
458        .duration_since(std::time::UNIX_EPOCH)
459        .unwrap_or_default()
460        .as_millis() as u64
461}
462
463/// The thinking/reasoning effort level requested from the model.
464///
465/// Maps to provider-specific parameters via [`ThinkingConfig::to_thinking_request`].
466///
467/// Effort is fixed for a session preference (no adaptive re-scoring). Registry
468/// `reasoning_levels` drive the picker; models without levels get binary
469/// thinking on/off. Models that do not support reasoning are forced to [`Off`].
470/// Default preference is [`Max`] (highest available effort).
471#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
472#[serde(rename_all = "lowercase")]
473pub enum ThinkingConfig {
474    /// Maximum reasoning effort (default).
475    ///
476    /// Legacy config/session values `"adaptive"` / `"auto"` deserialize as Max.
477    #[serde(alias = "adaptive", alias = "auto")]
478    Max,
479    /// High reasoning effort.
480    High,
481    /// Medium reasoning effort.
482    Medium,
483    /// Low reasoning effort.
484    Low,
485    /// Thinking/reasoning disabled.
486    Off,
487}
488
489/// Normalized thinking/reasoning request produced by [`ThinkingConfig::to_thinking_request`].
490///
491/// This is a provider-agnostic representation. Each provider converts these
492/// fields into its own wire format in the stream layer.
493#[derive(Debug, Clone, PartialEq, Eq)]
494pub struct ThinkingRequest {
495    /// Whether thinking/reasoning is enabled.
496    pub enabled: bool,
497    /// Reasoning effort level for providers that use effort strings
498    /// (OpenAI, OpenRouter, Groq, etc.). Owned so registry labels
499    /// (e.g. `xhigh`, `minimal`) can pass through unchanged.
500    pub effort: Option<String>,
501    /// Token budget for providers that use budget-based thinking
502    /// (Anthropic, Gemini).
503    pub budget_tokens: Option<u32>,
504}
505
506impl ThinkingConfig {
507    /// Produces a normalized [`ThinkingRequest`] from this config.
508    ///
509    /// The caller (provider stream layer) converts the normalized fields
510    /// into the provider-specific wire format.
511    pub fn to_thinking_request(self) -> ThinkingRequest {
512        match self {
513            Self::Max => ThinkingRequest {
514                enabled: true,
515                // Prefer xhigh on wire when providers accept it; callers may
516                // remap via [`resolve_effort_label`] using registry levels.
517                effort: Some("xhigh".to_string()),
518                budget_tokens: Some(32000),
519            },
520            Self::High => ThinkingRequest {
521                enabled: true,
522                effort: Some("high".to_string()),
523                budget_tokens: Some(10000),
524            },
525            Self::Medium => ThinkingRequest {
526                enabled: true,
527                effort: Some("medium".to_string()),
528                budget_tokens: Some(4096),
529            },
530            Self::Low => ThinkingRequest {
531                enabled: true,
532                effort: Some("low".to_string()),
533                budget_tokens: Some(1024),
534            },
535            Self::Off => ThinkingRequest {
536                enabled: false,
537                effort: None,
538                budget_tokens: None,
539            },
540        }
541    }
542
543    /// Config key used in `tui.thinking_level` / UI labels.
544    pub fn as_config_str(self) -> &'static str {
545        match self {
546            Self::Max => "max",
547            Self::High => "high",
548            Self::Medium => "medium",
549            Self::Low => "low",
550            Self::Off => "off",
551        }
552    }
553
554    /// Parse a config / registry effort string into a [`ThinkingConfig`].
555    ///
556    /// Unknown values (including legacy `"adaptive"`) fall back to [`Max`].
557    pub fn from_config_str(value: &str) -> Self {
558        parse_reasoning_level(value).unwrap_or(Self::Max)
559    }
560
561    /// Clamp this level to one supported by the model (registry reasoning_levels).
562    ///
563    /// Prefers the highest remaining effort; `Off` is last. Empty `supported`
564    /// leaves the value unchanged.
565    pub fn clamp_to_supported(self, supported: &[ThinkingConfig]) -> Self {
566        if supported.is_empty() || supported.contains(&self) {
567            return self;
568        }
569        // Prefer maximum effort when the requested level is unavailable.
570        for candidate in [Self::Max, Self::High, Self::Medium, Self::Low, Self::Off] {
571            if supported.contains(&candidate) {
572                return candidate;
573            }
574        }
575        supported[0]
576    }
577}
578
579/// Parse a registry / config reasoning level string.
580///
581/// Accepts common aliases used across OpenAI, OpenRouter, Anthropic, and xAI.
582/// `"on"` / `"enabled"` / `"true"` map to [`ThinkingConfig::Medium`] (binary
583/// effort "thinking on").
584pub fn parse_reasoning_level(raw: &str) -> Option<ThinkingConfig> {
585    match raw.trim().to_ascii_lowercase().as_str() {
586        // Legacy adaptive/auto maps to Max (highest fixed effort).
587        "adaptive" | "auto" | "max" | "xhigh" | "x-high" | "ultra" | "highest" => {
588            Some(ThinkingConfig::Max)
589        }
590        "high" => Some(ThinkingConfig::High),
591        // Binary "thinking on" uses Max so the default highest effort is stable.
592        "medium" | "med" | "mid" | "default" => Some(ThinkingConfig::Medium),
593        "on" | "enabled" | "true" | "1" => Some(ThinkingConfig::Max),
594        "low" | "minimal" | "min" => Some(ThinkingConfig::Low),
595        "off" | "none" | "disabled" | "false" | "0" => Some(ThinkingConfig::Off),
596        _ => None,
597    }
598}
599
600/// User-facing effort label for a level.
601///
602/// In binary mode (no registry levels) non-off levels display as
603/// `"thinking on"` and off as `"thinking off"`.
604pub fn effort_display_label(level: ThinkingConfig, binary: bool) -> &'static str {
605    if binary {
606        match level {
607            ThinkingConfig::Off => "thinking off",
608            _ => "thinking on",
609        }
610    } else {
611        level.as_config_str()
612    }
613}
614
615/// Canonical sort order for effort levels in pickers (most → least / off last).
616pub const DEFAULT_REASONING_LEVELS: &[ThinkingConfig] = &[
617    ThinkingConfig::Max,
618    ThinkingConfig::High,
619    ThinkingConfig::Medium,
620    ThinkingConfig::Low,
621    ThinkingConfig::Off,
622];
623
624/// Binary effort options when a model has no registry `reasoning_levels`.
625///
626/// UI presents these as "thinking on" / "thinking off". Internally "on" is
627/// [`ThinkingConfig::Max`] so the highest fixed effort is the default.
628pub const BINARY_REASONING_LEVELS: &[ThinkingConfig] = &[ThinkingConfig::Max, ThinkingConfig::Off];
629
630/// Resolve the effort levels the UI / runtime should offer for a model.
631///
632/// - `supports_thinking == false` → only Off (reasoning unsupported)
633/// - empty / unparseable `reasoning_levels` + thinking supported/unknown →
634///   binary thinking on (`Max`) / thinking off
635/// - non-empty registry levels → exactly those levels
636pub fn thinking_levels_for_model(
637    supports_thinking: Option<bool>,
638    reasoning_levels: &[String],
639) -> Vec<ThinkingConfig> {
640    if supports_thinking == Some(false) {
641        return vec![ThinkingConfig::Off];
642    }
643
644    if reasoning_levels.is_empty() {
645        return BINARY_REASONING_LEVELS.to_vec();
646    }
647
648    let mut out = Vec::new();
649    for raw in reasoning_levels {
650        if let Some(level) = parse_reasoning_level(raw) {
651            if !out.contains(&level) {
652                out.push(level);
653            }
654        }
655    }
656    if out.is_empty() {
657        return BINARY_REASONING_LEVELS.to_vec();
658    }
659    // Stable UI order (max/high/medium/low/off).
660    let order = DEFAULT_REASONING_LEVELS;
661    out.sort_by_key(|l| order.iter().position(|o| o == l).unwrap_or(99));
662    out
663}
664
665/// Whether the model uses the binary off/on effort picker (no registry levels).
666pub fn is_binary_effort_model(
667    supports_thinking: Option<bool>,
668    reasoning_levels: &[String],
669) -> bool {
670    if supports_thinking == Some(false) {
671        return false;
672    }
673    if reasoning_levels.is_empty() {
674        return true;
675    }
676    // Unparseable registry levels also fall back to binary.
677    !reasoning_levels
678        .iter()
679        .any(|raw| parse_reasoning_level(raw).is_some())
680}
681
682/// Pick a thinking level for a model from registry + current preference.
683///
684/// - Models without reasoning support always resolve to [`ThinkingConfig::Off`].
685/// - Supported preference is kept when valid.
686/// - Otherwise uses registry `default_reasoning_effort`, then highest supported
687///   (typically [`ThinkingConfig::Max`]).
688pub fn resolve_model_thinking_level(
689    current: ThinkingConfig,
690    supports_thinking: Option<bool>,
691    reasoning_levels: &[String],
692    default_reasoning_effort: Option<&str>,
693) -> ThinkingConfig {
694    let supported = thinking_levels_for_model(supports_thinking, reasoning_levels);
695    if supports_thinking == Some(false) {
696        return ThinkingConfig::Off;
697    }
698    if supported.contains(&current) {
699        return current;
700    }
701    if let Some(def) = default_reasoning_effort.and_then(parse_reasoning_level) {
702        return def.clamp_to_supported(&supported);
703    }
704    // Default: maximum supported effort (stable across tool-loop iterations).
705    ThinkingConfig::Max.clamp_to_supported(&supported)
706}
707
708/// Map a [`ThinkingConfig`] to a provider effort label, preferring registry strings.
709///
710/// Returns `None` when thinking is off.
711pub fn resolve_effort_label(
712    thinking: ThinkingConfig,
713    reasoning_levels: &[String],
714    provider_id: &str,
715) -> Option<String> {
716    if matches!(thinking, ThinkingConfig::Off) {
717        return None;
718    }
719    let concrete = thinking;
720
721    // Prefer an exact registry string that maps to this level.
722    for raw in reasoning_levels {
723        if parse_reasoning_level(raw) == Some(concrete) {
724            return Some(raw.trim().to_ascii_lowercase());
725        }
726    }
727
728    // Provider-specific fallbacks when registry has no levels yet.
729    let provider = crate::ProviderId::from_config_id(provider_id);
730    if provider.as_str() == crate::ProviderId::OPENROUTER {
731        return Some(
732            match concrete {
733                ThinkingConfig::Max => "xhigh",
734                ThinkingConfig::High => "high",
735                ThinkingConfig::Medium => "medium",
736                ThinkingConfig::Low => "low",
737                ThinkingConfig::Off => "medium",
738            }
739            .to_string(),
740        );
741    }
742
743    Some(
744        match concrete {
745            ThinkingConfig::Max => {
746                // OpenAI-style: xhigh when present in levels else high.
747                if reasoning_levels
748                    .iter()
749                    .any(|l| matches!(l.trim().to_ascii_lowercase().as_str(), "xhigh" | "max"))
750                {
751                    "xhigh"
752                } else {
753                    "high"
754                }
755            }
756            ThinkingConfig::High => "high",
757            ThinkingConfig::Medium => "medium",
758            ThinkingConfig::Low => "low",
759            ThinkingConfig::Off => return None,
760        }
761        .to_string(),
762    )
763}
764
765#[cfg(test)]
766mod tests {
767    use super::*;
768
769    // ── Regression: ThinkingConfig to ThinkingRequest ──────────────────────────
770
771    #[test]
772    fn regression_thinking_request_high_produces_effort_and_budget() {
773        let request = ThinkingConfig::High.to_thinking_request();
774        assert!(request.enabled);
775        assert_eq!(request.effort.as_deref(), Some("high"));
776        assert_eq!(request.budget_tokens, Some(10000));
777    }
778
779    #[test]
780    fn regression_thinking_request_max_produces_effort_and_budget() {
781        let request = ThinkingConfig::Max.to_thinking_request();
782        assert!(request.enabled);
783        assert_eq!(request.effort.as_deref(), Some("xhigh"));
784        assert_eq!(request.budget_tokens, Some(32000));
785    }
786
787    #[test]
788    fn regression_thinking_request_off_produces_disabled() {
789        let request = ThinkingConfig::Off.to_thinking_request();
790        assert!(!request.enabled);
791        assert!(request.effort.is_none());
792        assert!(request.budget_tokens.is_none());
793    }
794
795    #[test]
796    fn regression_thinking_request_medium_produces_medium_effort() {
797        let request = ThinkingConfig::Medium.to_thinking_request();
798        assert!(request.enabled);
799        assert_eq!(request.effort.as_deref(), Some("medium"));
800        assert_eq!(request.budget_tokens, Some(4096));
801    }
802
803    #[test]
804    fn regression_thinking_request_low_produces_low_effort() {
805        let request = ThinkingConfig::Low.to_thinking_request();
806        assert!(request.enabled);
807        assert_eq!(request.effort.as_deref(), Some("low"));
808        assert_eq!(request.budget_tokens, Some(1024));
809    }
810
811    #[test]
812    fn thinking_levels_for_model_respects_registry() {
813        let levels = thinking_levels_for_model(
814            Some(true),
815            &["none".into(), "low".into(), "high".into(), "xhigh".into()],
816        );
817        // Exactly the registry levels — no Adaptive inject, no Medium fill-in.
818        assert_eq!(
819            levels,
820            vec![
821                ThinkingConfig::Max,
822                ThinkingConfig::High,
823                ThinkingConfig::Low,
824                ThinkingConfig::Off,
825            ]
826        );
827        assert!(!is_binary_effort_model(
828            Some(true),
829            &["none".into(), "low".into(), "high".into(), "xhigh".into()],
830        ));
831    }
832
833    #[test]
834    fn thinking_levels_off_only_when_no_thinking() {
835        let levels = thinking_levels_for_model(Some(false), &["high".into()]);
836        assert_eq!(levels, vec![ThinkingConfig::Off]);
837        assert!(!is_binary_effort_model(Some(false), &["high".into()]));
838    }
839
840    #[test]
841    fn thinking_levels_binary_when_registry_empty() {
842        let levels = thinking_levels_for_model(Some(true), &[]);
843        assert_eq!(levels, vec![ThinkingConfig::Max, ThinkingConfig::Off]);
844        assert!(is_binary_effort_model(Some(true), &[]));
845        assert!(is_binary_effort_model(None, &[]));
846    }
847
848    #[test]
849    fn thinking_levels_model_specific_no_extra_options() {
850        let levels = thinking_levels_for_model(Some(true), &["low".into(), "high".into()]);
851        assert_eq!(levels, vec![ThinkingConfig::High, ThinkingConfig::Low]);
852    }
853
854    #[test]
855    fn resolve_effort_prefers_registry_label() {
856        let label = resolve_effort_label(
857            ThinkingConfig::Max,
858            &["low".into(), "high".into(), "xhigh".into()],
859            "openai",
860        );
861        assert_eq!(label.as_deref(), Some("xhigh"));
862    }
863
864    #[test]
865    fn clamp_unsupported_level_to_supported() {
866        let supported = vec![ThinkingConfig::Low, ThinkingConfig::Off];
867        assert_eq!(
868            ThinkingConfig::High.clamp_to_supported(&supported),
869            ThinkingConfig::Low
870        );
871    }
872
873    // ── Regression: ModelMessage constructors ─────────────────────────────────
874
875    #[test]
876    fn regression_system_message_has_correct_role() {
877        let msg = ModelMessage::system("test".to_string());
878        assert_eq!(msg.role, ModelRole::System);
879        assert_eq!(msg.content, "test");
880    }
881
882    #[test]
883    fn regression_user_message_has_correct_role() {
884        let msg = ModelMessage::user("hello".to_string());
885        assert_eq!(msg.role, ModelRole::User);
886        assert_eq!(msg.content, "hello");
887    }
888
889    #[test]
890    fn regression_assistant_message_has_correct_role() {
891        let msg = ModelMessage::assistant("response".to_string());
892        assert_eq!(msg.role, ModelRole::Assistant);
893        assert_eq!(msg.content, "response");
894    }
895
896    #[test]
897    fn regression_tool_result_sets_call_id_and_name() {
898        let msg = ModelMessage::tool_result("call-1", "read_file", "content");
899        assert_eq!(msg.role, ModelRole::Tool);
900        assert_eq!(msg.tool_call_id.as_deref(), Some("call-1"));
901        assert_eq!(msg.tool_name.as_deref(), Some("read_file"));
902        assert_eq!(msg.content, "content");
903    }
904
905    #[test]
906    fn regression_assistant_tool_call_with_context_sets_fields() {
907        let inv = ToolInvocation {
908            id: "call-1".to_string(),
909            tool_name: "read_file".to_string(),
910            input: serde_json::json!({"path": "test.rs"}),
911        };
912        let msg = ModelMessage::assistant_tool_call_with_context(
913            inv,
914            "thinking text",
915            Some("reasoning".to_string()),
916        );
917        assert_eq!(msg.role, ModelRole::Assistant);
918        assert_eq!(msg.content, "thinking text");
919        assert_eq!(msg.thinking_content.as_deref(), Some("reasoning"));
920        assert_eq!(msg.tool_calls.len(), 1);
921        assert_eq!(msg.tool_calls[0].id, "call-1");
922    }
923
924    // ── Regression: ModelMessage serialization roundtrip ──────────────────────
925
926    #[test]
927    fn regression_model_message_serialization_roundtrip() {
928        let msg = ModelMessage {
929            role: ModelRole::Assistant,
930            content: "hello".to_string(),
931            content_parts: Vec::new(),
932            tool_call_id: None,
933            tool_name: None,
934            tool_calls: vec![],
935            thinking_content: Some("thinking".to_string()),
936            created_at: Some(12345),
937        };
938        let json = serde_json::to_string(&msg).unwrap();
939        let deserialized: ModelMessage = serde_json::from_str(&json).unwrap();
940        assert_eq!(deserialized.role, msg.role);
941        assert_eq!(deserialized.content, msg.content);
942        assert_eq!(deserialized.thinking_content, msg.thinking_content);
943        // created_at is intentionally not serialized (runtime-only field)
944        assert!(deserialized.created_at.is_none());
945    }
946
947    // ── Effort resolution (no adaptive) ────────────────────────────────────────
948
949    #[test]
950    fn legacy_adaptive_string_maps_to_max() {
951        assert_eq!(
952            ThinkingConfig::from_config_str("adaptive"),
953            ThinkingConfig::Max
954        );
955        assert_eq!(parse_reasoning_level("auto"), Some(ThinkingConfig::Max));
956        assert_eq!(parse_reasoning_level("on"), Some(ThinkingConfig::Max));
957        let deserialized: ThinkingConfig = serde_json::from_str("\"adaptive\"").unwrap();
958        assert_eq!(deserialized, ThinkingConfig::Max);
959    }
960
961    #[test]
962    fn resolve_forces_off_when_model_lacks_reasoning() {
963        let resolved =
964            resolve_model_thinking_level(ThinkingConfig::Max, Some(false), &["high".into()], None);
965        assert_eq!(resolved, ThinkingConfig::Off);
966    }
967
968    #[test]
969    fn resolve_defaults_to_max_when_preference_unsupported() {
970        let resolved = resolve_model_thinking_level(
971            ThinkingConfig::Low,
972            Some(true),
973            &["high".into(), "xhigh".into()],
974            None,
975        );
976        assert_eq!(resolved, ThinkingConfig::Max);
977    }
978
979    #[test]
980    fn binary_on_is_max() {
981        assert_eq!(
982            BINARY_REASONING_LEVELS,
983            &[ThinkingConfig::Max, ThinkingConfig::Off]
984        );
985        assert_eq!(
986            effort_display_label(ThinkingConfig::Max, true),
987            "thinking on"
988        );
989        assert_eq!(
990            effort_display_label(ThinkingConfig::Off, true),
991            "thinking off"
992        );
993    }
994}