Skip to main content

navi_core/
model.rs

1use crate::tool::{ToolDefinition, ToolInvocation};
2use anyhow::Result;
3use async_trait::async_trait;
4use futures_util::StreamExt;
5use futures_util::stream::BoxStream;
6use serde::{Deserialize, Serialize};
7
8/// A single part of a multimodal message content.
9///
10/// Models like GPT-4o, Claude, and Gemini accept messages with mixed
11/// text and attachment parts. When a [`ModelMessage`] contains non-empty
12/// [`ModelMessage::content_parts`], providers serialize each part
13/// according to their native wire format instead of using the plain
14/// [`ModelMessage::content`] string.
15#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
16#[serde(tag = "type", rename_all = "snake_case")]
17pub enum ContentPart {
18    /// A plain text content block.
19    Text {
20        /// The text content.
21        text: String,
22    },
23    /// An inline image (base64-encoded).
24    Image {
25        /// MIME type of the image (e.g. `"image/png"`, `"image/jpeg"`).
26        media_type: String,
27        /// Base64-encoded image data (no data-URL prefix, raw base64 only).
28        data: String,
29    },
30    /// An inline audio attachment (base64-encoded).
31    Audio {
32        /// MIME type of the audio (e.g. `"audio/mpeg"`, `"audio/wav"`).
33        media_type: String,
34        /// Base64-encoded audio data (no data-URL prefix, raw base64 only).
35        data: String,
36        /// Optional filename or user-facing label.
37        #[serde(default, skip_serializing_if = "Option::is_none")]
38        name: Option<String>,
39    },
40    /// An inline video attachment (base64-encoded).
41    Video {
42        /// MIME type of the video (e.g. `"video/mp4"`).
43        media_type: String,
44        /// Base64-encoded video data (no data-URL prefix, raw base64 only).
45        data: String,
46        /// Optional filename or user-facing label.
47        #[serde(default, skip_serializing_if = "Option::is_none")]
48        name: Option<String>,
49    },
50    /// An inline document attachment (base64-encoded).
51    Document {
52        /// MIME type of the document (e.g. `"application/pdf"`, `"text/plain"`).
53        media_type: String,
54        /// Base64-encoded document data (no data-URL prefix, raw base64 only).
55        data: String,
56        /// Optional filename or user-facing label.
57        #[serde(default, skip_serializing_if = "Option::is_none")]
58        name: Option<String>,
59    },
60}
61
62impl ContentPart {
63    /// Returns `true` if this is a text part.
64    pub fn is_text(&self) -> bool {
65        matches!(self, Self::Text { .. })
66    }
67
68    /// Returns `true` if this is an image part.
69    pub fn is_image(&self) -> bool {
70        matches!(self, Self::Image { .. })
71    }
72
73    /// Returns `true` if this is an audio part.
74    pub fn is_audio(&self) -> bool {
75        matches!(self, Self::Audio { .. })
76    }
77
78    /// Returns `true` if this is a video part.
79    pub fn is_video(&self) -> bool {
80        matches!(self, Self::Video { .. })
81    }
82
83    /// Returns `true` if this is a document part.
84    pub fn is_document(&self) -> bool {
85        matches!(self, Self::Document { .. })
86    }
87
88    /// Returns the attachment kind, if this part is an attachment.
89    pub fn attachment_kind(&self) -> Option<AttachmentKind> {
90        match self {
91            Self::Text { .. } => None,
92            Self::Image { .. } => Some(AttachmentKind::Image),
93            Self::Audio { .. } => Some(AttachmentKind::Audio),
94            Self::Video { .. } => Some(AttachmentKind::Video),
95            Self::Document { .. } => Some(AttachmentKind::Document),
96        }
97    }
98
99    /// Returns the MIME type for attachment parts.
100    pub fn media_type(&self) -> Option<&str> {
101        match self {
102            Self::Text { .. } => None,
103            Self::Image { media_type, .. }
104            | Self::Audio { media_type, .. }
105            | Self::Video { media_type, .. }
106            | Self::Document { media_type, .. } => Some(media_type),
107        }
108    }
109
110    /// Returns base64 data for attachment parts.
111    pub fn data(&self) -> Option<&str> {
112        match self {
113            Self::Text { .. } => None,
114            Self::Image { data, .. }
115            | Self::Audio { data, .. }
116            | Self::Video { data, .. }
117            | Self::Document { data, .. } => Some(data),
118        }
119    }
120
121    /// Optional filename/label for audio, video, or document attachments.
122    pub fn name(&self) -> Option<&str> {
123        match self {
124            Self::Audio { name, .. } | Self::Video { name, .. } | Self::Document { name, .. } => {
125                name.as_deref()
126            }
127            Self::Text { .. } | Self::Image { .. } => None,
128        }
129    }
130
131    /// Extracts the text content if this is a text part.
132    pub fn as_text(&self) -> Option<&str> {
133        match self {
134            Self::Text { text } => Some(text),
135            _ => None,
136        }
137    }
138
139    /// Returns all text content from a slice of parts, concatenated.
140    pub fn text_from_parts(parts: &[ContentPart]) -> String {
141        parts
142            .iter()
143            .filter_map(|p| p.as_text())
144            .collect::<Vec<_>>()
145            .join("")
146    }
147}
148
149/// Attachment modalities NAVI can route to specialized models.
150#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
151#[serde(rename_all = "snake_case")]
152pub enum AttachmentKind {
153    Image,
154    Audio,
155    Video,
156    Document,
157}
158
159impl AttachmentKind {
160    pub fn as_str(self) -> &'static str {
161        match self {
162            Self::Image => "image",
163            Self::Audio => "audio",
164            Self::Video => "video",
165            Self::Document => "document",
166        }
167    }
168}
169
170/// Trait for model provider backends that can stream and complete requests.
171///
172/// Implementors handle the wire protocol for a specific API (OpenAI, Anthropic,
173/// Gemini, etc.) while the engine works with the generic [`ModelRequest`] and
174/// [`ModelStreamEvent`] types.
175#[async_trait]
176pub trait ModelProvider: Send + Sync {
177    /// Starts a streaming request and returns a stream of [`ModelStreamEvent`].
178    fn stream(&self, request: ModelRequest) -> ModelStream;
179
180    /// Completes a request by consuming the full stream and returning the
181    /// accumulated text response. Default implementation calls [`Self::stream`].
182    async fn complete(&self, request: ModelRequest) -> Result<ModelResponse> {
183        let mut stream = self.stream(request);
184        let mut text = String::new();
185
186        while let Some(event) = stream.next().await {
187            match event? {
188                ModelStreamEvent::TextDelta { text: delta } => text.push_str(&delta),
189                ModelStreamEvent::Done => break,
190                ModelStreamEvent::Status { .. }
191                | ModelStreamEvent::Usage { .. }
192                | ModelStreamEvent::ThinkingDelta { .. }
193                | ModelStreamEvent::ToolCall(_)
194                | ModelStreamEvent::ToolCallProgress { .. } => {}
195            }
196        }
197
198        Ok(ModelResponse { text })
199    }
200
201    /// Lists available model identifiers from this provider.
202    ///
203    /// Returns an error if the provider does not support model listing.
204    async fn list_models(&self) -> Result<Vec<String>> {
205        anyhow::bail!("listing models is not supported by this provider")
206    }
207}
208
209/// A boxed async stream of [`ModelStreamEvent`] results from a provider.
210pub type ModelStream = BoxStream<'static, Result<ModelStreamEvent>>;
211
212/// A request to a model provider containing the conversation, model name,
213/// thinking configuration, and available tool definitions.
214#[derive(Debug, Clone, Serialize, Deserialize)]
215pub struct ModelRequest {
216    /// The model identifier to use (e.g. `"gpt-5.5"`, `"claude-sonnet-4-20250514"`).
217    pub model: String,
218    /// Stable base instructions sent in the provider's `instructions` field
219    /// (Responses API) or as the first system message (Chat Completions,
220    /// Anthropic, Gemini). Kept separate from [`Self::messages`] so that
221    /// dynamic context blocks (developer messages) don't invalidate the
222    /// provider's prompt cache for this prefix.
223    #[serde(default, skip_serializing_if = "Option::is_none")]
224    pub instructions: Option<String>,
225    /// The conversation messages to send to the model.
226    pub messages: Vec<ModelMessage>,
227    /// The thinking/reasoning effort level to request.
228    pub thinking: ThinkingConfig,
229    /// Tool definitions the model may invoke.
230    #[serde(default)]
231    pub tools: Vec<ToolDefinition>,
232    /// Stable session id for provider-side prompt-cache affinity.
233    ///
234    /// Providers such as Charm Hyper use this to set `x-session-id` and
235    /// `x-session-affinity` so consecutive turns of the same agent session
236    /// hit the same KV-cache shard. Not serialized to disk/transcripts.
237    #[serde(default, skip_serializing, skip_deserializing)]
238    pub session_id: Option<String>,
239}
240
241/// A single message in a model conversation.
242///
243/// Messages carry role, content, and optional tool-related metadata so the
244/// same type can represent system prompts, user input, assistant responses,
245/// tool calls, and tool results.
246#[derive(Debug, Clone, Serialize, Deserialize)]
247pub struct ModelMessage {
248    /// The conversational role of this message.
249    pub role: ModelRole,
250    /// The text content of the message.
251    pub content: String,
252    /// Multimodal content parts (text + images).
253    ///
254    /// When non-empty, providers use these parts instead of the plain
255    /// [`content`](Self::content) field to build the wire-format message.
256    /// This allows attaching images alongside text in user messages.
257    #[serde(default, skip_serializing_if = "Vec::is_empty")]
258    pub content_parts: Vec<ContentPart>,
259    /// For tool-result messages, the id of the tool call being answered.
260    #[serde(default)]
261    pub tool_call_id: Option<String>,
262    /// For tool-result messages, the name of the tool that produced this result.
263    #[serde(default)]
264    pub tool_name: Option<String>,
265    /// For assistant messages, the tool invocations requested by the model.
266    #[serde(default)]
267    pub tool_calls: Vec<ToolInvocation>,
268    /// Creation timestamp in milliseconds since Unix epoch (not serialized).
269    #[serde(default, skip_serializing, skip_deserializing)]
270    pub created_at: Option<u64>,
271    /// Optional thinking/reasoning content from the model.
272    #[serde(default, skip_serializing_if = "Option::is_none")]
273    pub thinking_content: Option<String>,
274}
275
276/// The conversational role of a [`ModelMessage`].
277#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
278#[serde(rename_all = "lowercase")]
279pub enum ModelRole {
280    /// System-level instructions (system prompt).
281    System,
282    /// Developer-level instructions (context blocks injected separately from
283    /// the base system prompt for provider cache efficiency).
284    Developer,
285    /// End-user input.
286    User,
287    /// Model-generated response.
288    Assistant,
289    /// A tool result returned to the model.
290    Tool,
291}
292
293/// A completed model response containing the full text output.
294#[derive(Debug, Clone, Serialize, Deserialize)]
295pub struct ModelResponse {
296    /// The full text content of the model's response.
297    pub text: String,
298}
299
300/// A single event from a model provider's streaming response.
301#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
302pub enum ModelStreamEvent {
303    /// An incremental text delta from the assistant.
304    TextDelta {
305        /// The incremental text content.
306        text: String,
307    },
308    /// An incremental thinking/reasoning delta.
309    ThinkingDelta {
310        /// The incremental thinking content.
311        text: String,
312    },
313    /// A status message from the provider (e.g. "processing", "queued").
314    Status {
315        /// Human-readable status label.
316        label: String,
317    },
318    /// Token usage information reported by the provider.
319    Usage {
320        /// Number of input/prompt tokens consumed, if reported.
321        input_tokens: Option<u64>,
322        /// Number of output/completion tokens produced, if reported.
323        output_tokens: Option<u64>,
324        /// Number of tokens written to the prompt cache (Anthropic).
325        cache_creation_tokens: Option<u64>,
326        /// Number of tokens read from the prompt cache (Anthropic).
327        cache_read_tokens: Option<u64>,
328    },
329    /// The model requested a tool invocation.
330    ToolCall(ToolInvocation),
331    /// The model is streaming a tool call (name known; arguments may still be incomplete).
332    ///
333    /// Providers emit this while native tool-call arguments are being generated
334    /// so clients can show progress instead of a false "waiting for model" idle.
335    ToolCallProgress {
336        /// Provider tool-call id when known.
337        id: Option<String>,
338        /// Tool name when known (empty only until the first name chunk arrives).
339        tool_name: String,
340        /// Total characters of arguments streamed so far for this call.
341        arguments_chars: usize,
342    },
343    /// The stream has ended.
344    Done,
345}
346
347impl ModelMessage {
348    /// Creates a system-role message.
349    pub fn system(content: impl Into<String>) -> Self {
350        Self::new(ModelRole::System, content)
351    }
352
353    /// Creates a developer-role message (context block injected separately
354    /// from the base system prompt for provider cache efficiency).
355    pub fn developer(content: impl Into<String>) -> Self {
356        Self::new(ModelRole::Developer, content)
357    }
358
359    /// Creates a user-role message.
360    pub fn user(content: impl Into<String>) -> Self {
361        Self::new(ModelRole::User, content)
362    }
363
364    /// Creates a user-role message with text and optional image attachments.
365    pub fn user_multimodal(content: impl Into<String>, parts: Vec<ContentPart>) -> Self {
366        Self {
367            role: ModelRole::User,
368            content: content.into(),
369            content_parts: parts,
370            tool_call_id: None,
371            tool_name: None,
372            tool_calls: Vec::new(),
373            created_at: Some(current_unix_millis()),
374            thinking_content: None,
375        }
376    }
377
378    /// Creates an assistant-role message without thinking content.
379    pub fn assistant(content: impl Into<String>) -> Self {
380        Self {
381            thinking_content: None,
382            ..Self::new(ModelRole::Assistant, content)
383        }
384    }
385
386    /// Creates an assistant-role message with optional thinking content.
387    pub fn assistant_with_thinking(content: impl Into<String>, thinking: Option<String>) -> Self {
388        Self {
389            thinking_content: thinking,
390            ..Self::new(ModelRole::Assistant, content)
391        }
392    }
393
394    /// Creates a tool-result message responding to a specific tool call.
395    pub fn tool_result(
396        tool_call_id: impl Into<String>,
397        tool_name: impl Into<String>,
398        content: impl Into<String>,
399    ) -> Self {
400        Self::tool_result_with_parts(tool_call_id, tool_name, content, Vec::new())
401    }
402
403    /// Creates a tool-result message with optional multimodal content parts
404    /// (e.g. images from `view_image` for vision-capable models).
405    pub fn tool_result_with_parts(
406        tool_call_id: impl Into<String>,
407        tool_name: impl Into<String>,
408        content: impl Into<String>,
409        content_parts: Vec<ContentPart>,
410    ) -> Self {
411        Self {
412            role: ModelRole::Tool,
413            content: content.into(),
414            content_parts,
415            tool_call_id: Some(tool_call_id.into()),
416            tool_name: Some(tool_name.into()),
417            tool_calls: Vec::new(),
418            created_at: Some(current_unix_millis()),
419            thinking_content: None,
420        }
421    }
422
423    /// Creates an assistant message that requests a single tool invocation.
424    pub fn assistant_tool_call(invocation: ToolInvocation) -> Self {
425        Self::assistant_tool_call_with_context(invocation, String::new(), None)
426    }
427
428    /// Creates an assistant message that requests a tool invocation with
429    /// accompanying text content and optional thinking.
430    pub fn assistant_tool_call_with_context(
431        invocation: ToolInvocation,
432        content: impl Into<String>,
433        thinking: Option<String>,
434    ) -> Self {
435        Self::assistant_tool_calls_with_context(vec![invocation], content, thinking)
436    }
437
438    pub fn assistant_tool_calls_with_context(
439        invocations: Vec<ToolInvocation>,
440        content: impl Into<String>,
441        thinking: Option<String>,
442    ) -> Self {
443        Self {
444            role: ModelRole::Assistant,
445            content: content.into(),
446            content_parts: Vec::new(),
447            tool_call_id: None,
448            tool_name: None,
449            tool_calls: invocations,
450            created_at: Some(current_unix_millis()),
451            thinking_content: thinking,
452        }
453    }
454
455    fn new(role: ModelRole, content: impl Into<String>) -> Self {
456        Self {
457            role,
458            content: content.into(),
459            content_parts: Vec::new(),
460            tool_call_id: None,
461            tool_name: None,
462            tool_calls: Vec::new(),
463            created_at: Some(current_unix_millis()),
464            thinking_content: None,
465        }
466    }
467}
468
469fn current_unix_millis() -> u64 {
470    std::time::SystemTime::now()
471        .duration_since(std::time::UNIX_EPOCH)
472        .unwrap_or_default()
473        .as_millis() as u64
474}
475
476/// The thinking/reasoning effort level requested from the model.
477///
478/// Maps to provider-specific parameters via [`ThinkingConfig::to_thinking_request`].
479///
480/// Effort is fixed for a session preference (no adaptive re-scoring). Registry
481/// `reasoning_levels` drive the picker; models without levels get binary
482/// thinking on/off. Models that do not support reasoning are forced to [`Off`].
483/// Default preference is [`Max`] (highest available effort).
484#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
485#[serde(rename_all = "lowercase")]
486pub enum ThinkingConfig {
487    /// Maximum reasoning effort (default).
488    ///
489    /// Legacy config/session values `"adaptive"` / `"auto"` deserialize as Max.
490    #[serde(alias = "adaptive", alias = "auto")]
491    Max,
492    /// High reasoning effort.
493    High,
494    /// Medium reasoning effort.
495    Medium,
496    /// Low reasoning effort.
497    Low,
498    /// Thinking/reasoning disabled.
499    Off,
500}
501
502/// Normalized thinking/reasoning request produced by [`ThinkingConfig::to_thinking_request`].
503///
504/// This is a provider-agnostic representation. Each provider converts these
505/// fields into its own wire format in the stream layer.
506#[derive(Debug, Clone, PartialEq, Eq)]
507pub struct ThinkingRequest {
508    /// Whether thinking/reasoning is enabled.
509    pub enabled: bool,
510    /// Reasoning effort level for providers that use effort strings
511    /// (OpenAI, OpenRouter, Groq, etc.). Owned so registry labels
512    /// (e.g. `xhigh`, `minimal`) can pass through unchanged.
513    pub effort: Option<String>,
514    /// Token budget for providers that use budget-based thinking
515    /// (Anthropic, Gemini).
516    pub budget_tokens: Option<u32>,
517}
518
519impl ThinkingConfig {
520    /// Produces a normalized [`ThinkingRequest`] from this config.
521    ///
522    /// The caller (provider stream layer) converts the normalized fields
523    /// into the provider-specific wire format.
524    pub fn to_thinking_request(self) -> ThinkingRequest {
525        match self {
526            Self::Max => ThinkingRequest {
527                enabled: true,
528                // Prefer xhigh on wire when providers accept it; callers may
529                // remap via [`resolve_effort_label`] using registry levels.
530                effort: Some("xhigh".to_string()),
531                budget_tokens: Some(32000),
532            },
533            Self::High => ThinkingRequest {
534                enabled: true,
535                effort: Some("high".to_string()),
536                budget_tokens: Some(10000),
537            },
538            Self::Medium => ThinkingRequest {
539                enabled: true,
540                effort: Some("medium".to_string()),
541                budget_tokens: Some(4096),
542            },
543            Self::Low => ThinkingRequest {
544                enabled: true,
545                effort: Some("low".to_string()),
546                budget_tokens: Some(1024),
547            },
548            Self::Off => ThinkingRequest {
549                enabled: false,
550                effort: None,
551                budget_tokens: None,
552            },
553        }
554    }
555
556    /// Config key used in `tui.thinking_level` / UI labels.
557    pub fn as_config_str(self) -> &'static str {
558        match self {
559            Self::Max => "max",
560            Self::High => "high",
561            Self::Medium => "medium",
562            Self::Low => "low",
563            Self::Off => "off",
564        }
565    }
566
567    /// Parse a config / registry effort string into a [`ThinkingConfig`].
568    ///
569    /// Unknown values (including legacy `"adaptive"`) fall back to [`Max`].
570    pub fn from_config_str(value: &str) -> Self {
571        parse_reasoning_level(value).unwrap_or(Self::Max)
572    }
573
574    /// Clamp this level to one supported by the model (registry reasoning_levels).
575    ///
576    /// Prefers the highest remaining effort; `Off` is last. Empty `supported`
577    /// leaves the value unchanged.
578    pub fn clamp_to_supported(self, supported: &[ThinkingConfig]) -> Self {
579        if supported.is_empty() || supported.contains(&self) {
580            return self;
581        }
582        // Prefer maximum effort when the requested level is unavailable.
583        for candidate in [Self::Max, Self::High, Self::Medium, Self::Low, Self::Off] {
584            if supported.contains(&candidate) {
585                return candidate;
586            }
587        }
588        supported[0]
589    }
590}
591
592/// Parse a registry / config reasoning level string.
593///
594/// Accepts common aliases used across OpenAI, OpenRouter, Anthropic, and xAI.
595/// `"on"` / `"enabled"` / `"true"` map to [`ThinkingConfig::Medium`] (binary
596/// effort "thinking on").
597pub fn parse_reasoning_level(raw: &str) -> Option<ThinkingConfig> {
598    match raw.trim().to_ascii_lowercase().as_str() {
599        // Legacy adaptive/auto maps to Max (highest fixed effort).
600        "adaptive" | "auto" | "max" | "xhigh" | "x-high" | "ultra" | "highest" => {
601            Some(ThinkingConfig::Max)
602        }
603        "high" => Some(ThinkingConfig::High),
604        // Binary "thinking on" uses Max so the default highest effort is stable.
605        "medium" | "med" | "mid" | "default" => Some(ThinkingConfig::Medium),
606        "on" | "enabled" | "true" | "1" => Some(ThinkingConfig::Max),
607        "low" | "minimal" | "min" => Some(ThinkingConfig::Low),
608        "off" | "none" | "disabled" | "false" | "0" => Some(ThinkingConfig::Off),
609        _ => None,
610    }
611}
612
613/// User-facing effort label for a level.
614///
615/// In binary mode (no registry levels) non-off levels display as
616/// `"thinking on"` and off as `"thinking off"`.
617pub fn effort_display_label(level: ThinkingConfig, binary: bool) -> &'static str {
618    if binary {
619        match level {
620            ThinkingConfig::Off => "thinking off",
621            _ => "thinking on",
622        }
623    } else {
624        level.as_config_str()
625    }
626}
627
628/// Canonical sort order for effort levels in pickers (most → least / off last).
629pub const DEFAULT_REASONING_LEVELS: &[ThinkingConfig] = &[
630    ThinkingConfig::Max,
631    ThinkingConfig::High,
632    ThinkingConfig::Medium,
633    ThinkingConfig::Low,
634    ThinkingConfig::Off,
635];
636
637/// Binary effort options when a model has no registry `reasoning_levels`.
638///
639/// UI presents these as "thinking on" / "thinking off". Internally "on" is
640/// [`ThinkingConfig::Max`] so the highest fixed effort is the default.
641pub const BINARY_REASONING_LEVELS: &[ThinkingConfig] = &[ThinkingConfig::Max, ThinkingConfig::Off];
642
643/// Resolve the effort levels the UI / runtime should offer for a model.
644///
645/// - `supports_thinking == false` → only Off (reasoning unsupported)
646/// - empty / unparseable `reasoning_levels` + thinking supported/unknown →
647///   binary thinking on (`Max`) / thinking off
648/// - non-empty registry levels → exactly those levels
649pub fn thinking_levels_for_model(
650    supports_thinking: Option<bool>,
651    reasoning_levels: &[String],
652) -> Vec<ThinkingConfig> {
653    if supports_thinking == Some(false) {
654        return vec![ThinkingConfig::Off];
655    }
656
657    if reasoning_levels.is_empty() {
658        return BINARY_REASONING_LEVELS.to_vec();
659    }
660
661    let mut out = Vec::new();
662    for raw in reasoning_levels {
663        if let Some(level) = parse_reasoning_level(raw) {
664            if !out.contains(&level) {
665                out.push(level);
666            }
667        }
668    }
669    if out.is_empty() {
670        return BINARY_REASONING_LEVELS.to_vec();
671    }
672    // Stable UI order (max/high/medium/low/off).
673    let order = DEFAULT_REASONING_LEVELS;
674    out.sort_by_key(|l| order.iter().position(|o| o == l).unwrap_or(99));
675    out
676}
677
678/// Whether the model uses the binary off/on effort picker (no registry levels).
679pub fn is_binary_effort_model(
680    supports_thinking: Option<bool>,
681    reasoning_levels: &[String],
682) -> bool {
683    if supports_thinking == Some(false) {
684        return false;
685    }
686    if reasoning_levels.is_empty() {
687        return true;
688    }
689    // Unparseable registry levels also fall back to binary.
690    !reasoning_levels
691        .iter()
692        .any(|raw| parse_reasoning_level(raw).is_some())
693}
694
695/// Pick a thinking level for a model from registry + current preference.
696///
697/// - Models without reasoning support always resolve to [`ThinkingConfig::Off`].
698/// - Supported preference is kept when valid.
699/// - Otherwise uses registry `default_reasoning_effort`, then highest supported
700///   (typically [`ThinkingConfig::Max`]).
701pub fn resolve_model_thinking_level(
702    current: ThinkingConfig,
703    supports_thinking: Option<bool>,
704    reasoning_levels: &[String],
705    default_reasoning_effort: Option<&str>,
706) -> ThinkingConfig {
707    let supported = thinking_levels_for_model(supports_thinking, reasoning_levels);
708    if supports_thinking == Some(false) {
709        return ThinkingConfig::Off;
710    }
711    if supported.contains(&current) {
712        return current;
713    }
714    if let Some(def) = default_reasoning_effort.and_then(parse_reasoning_level) {
715        return def.clamp_to_supported(&supported);
716    }
717    // Default: maximum supported effort (stable across tool-loop iterations).
718    ThinkingConfig::Max.clamp_to_supported(&supported)
719}
720
721/// Map a [`ThinkingConfig`] to a provider effort label, preferring registry strings.
722///
723/// Returns `None` when thinking is off.
724pub fn resolve_effort_label(
725    thinking: ThinkingConfig,
726    reasoning_levels: &[String],
727    provider_id: &str,
728) -> Option<String> {
729    if matches!(thinking, ThinkingConfig::Off) {
730        return None;
731    }
732    let concrete = thinking;
733
734    // Prefer an exact registry string that maps to this level.
735    for raw in reasoning_levels {
736        if parse_reasoning_level(raw) == Some(concrete) {
737            return Some(raw.trim().to_ascii_lowercase());
738        }
739    }
740
741    // Provider-specific fallbacks when registry has no levels yet.
742    let provider = crate::ProviderId::from_config_id(provider_id);
743    if provider.as_str() == crate::ProviderId::OPENROUTER {
744        return Some(
745            match concrete {
746                ThinkingConfig::Max => "xhigh",
747                ThinkingConfig::High => "high",
748                ThinkingConfig::Medium => "medium",
749                ThinkingConfig::Low => "low",
750                ThinkingConfig::Off => "medium",
751            }
752            .to_string(),
753        );
754    }
755
756    Some(
757        match concrete {
758            ThinkingConfig::Max => {
759                // OpenAI-style: xhigh when present in levels else high.
760                if reasoning_levels
761                    .iter()
762                    .any(|l| matches!(l.trim().to_ascii_lowercase().as_str(), "xhigh" | "max"))
763                {
764                    "xhigh"
765                } else {
766                    "high"
767                }
768            }
769            ThinkingConfig::High => "high",
770            ThinkingConfig::Medium => "medium",
771            ThinkingConfig::Low => "low",
772            ThinkingConfig::Off => return None,
773        }
774        .to_string(),
775    )
776}
777
778#[cfg(test)]
779mod tests {
780    use super::*;
781
782    // ── Regression: ThinkingConfig to ThinkingRequest ──────────────────────────
783
784    #[test]
785    fn regression_thinking_request_high_produces_effort_and_budget() {
786        let request = ThinkingConfig::High.to_thinking_request();
787        assert!(request.enabled);
788        assert_eq!(request.effort.as_deref(), Some("high"));
789        assert_eq!(request.budget_tokens, Some(10000));
790    }
791
792    #[test]
793    fn regression_thinking_request_max_produces_effort_and_budget() {
794        let request = ThinkingConfig::Max.to_thinking_request();
795        assert!(request.enabled);
796        assert_eq!(request.effort.as_deref(), Some("xhigh"));
797        assert_eq!(request.budget_tokens, Some(32000));
798    }
799
800    #[test]
801    fn regression_thinking_request_off_produces_disabled() {
802        let request = ThinkingConfig::Off.to_thinking_request();
803        assert!(!request.enabled);
804        assert!(request.effort.is_none());
805        assert!(request.budget_tokens.is_none());
806    }
807
808    #[test]
809    fn regression_thinking_request_medium_produces_medium_effort() {
810        let request = ThinkingConfig::Medium.to_thinking_request();
811        assert!(request.enabled);
812        assert_eq!(request.effort.as_deref(), Some("medium"));
813        assert_eq!(request.budget_tokens, Some(4096));
814    }
815
816    #[test]
817    fn regression_thinking_request_low_produces_low_effort() {
818        let request = ThinkingConfig::Low.to_thinking_request();
819        assert!(request.enabled);
820        assert_eq!(request.effort.as_deref(), Some("low"));
821        assert_eq!(request.budget_tokens, Some(1024));
822    }
823
824    #[test]
825    fn thinking_levels_for_model_respects_registry() {
826        let levels = thinking_levels_for_model(
827            Some(true),
828            &["none".into(), "low".into(), "high".into(), "xhigh".into()],
829        );
830        // Exactly the registry levels — no Adaptive inject, no Medium fill-in.
831        assert_eq!(
832            levels,
833            vec![
834                ThinkingConfig::Max,
835                ThinkingConfig::High,
836                ThinkingConfig::Low,
837                ThinkingConfig::Off,
838            ]
839        );
840        assert!(!is_binary_effort_model(
841            Some(true),
842            &["none".into(), "low".into(), "high".into(), "xhigh".into()],
843        ));
844    }
845
846    #[test]
847    fn thinking_levels_off_only_when_no_thinking() {
848        let levels = thinking_levels_for_model(Some(false), &["high".into()]);
849        assert_eq!(levels, vec![ThinkingConfig::Off]);
850        assert!(!is_binary_effort_model(Some(false), &["high".into()]));
851    }
852
853    #[test]
854    fn thinking_levels_binary_when_registry_empty() {
855        let levels = thinking_levels_for_model(Some(true), &[]);
856        assert_eq!(levels, vec![ThinkingConfig::Max, ThinkingConfig::Off]);
857        assert!(is_binary_effort_model(Some(true), &[]));
858        assert!(is_binary_effort_model(None, &[]));
859    }
860
861    #[test]
862    fn thinking_levels_model_specific_no_extra_options() {
863        let levels = thinking_levels_for_model(Some(true), &["low".into(), "high".into()]);
864        assert_eq!(levels, vec![ThinkingConfig::High, ThinkingConfig::Low]);
865    }
866
867    #[test]
868    fn resolve_effort_prefers_registry_label() {
869        let label = resolve_effort_label(
870            ThinkingConfig::Max,
871            &["low".into(), "high".into(), "xhigh".into()],
872            "openai",
873        );
874        assert_eq!(label.as_deref(), Some("xhigh"));
875    }
876
877    #[test]
878    fn clamp_unsupported_level_to_supported() {
879        let supported = vec![ThinkingConfig::Low, ThinkingConfig::Off];
880        assert_eq!(
881            ThinkingConfig::High.clamp_to_supported(&supported),
882            ThinkingConfig::Low
883        );
884    }
885
886    // ── Regression: ModelMessage constructors ─────────────────────────────────
887
888    #[test]
889    fn regression_system_message_has_correct_role() {
890        let msg = ModelMessage::system("test".to_string());
891        assert_eq!(msg.role, ModelRole::System);
892        assert_eq!(msg.content, "test");
893    }
894
895    #[test]
896    fn regression_user_message_has_correct_role() {
897        let msg = ModelMessage::user("hello".to_string());
898        assert_eq!(msg.role, ModelRole::User);
899        assert_eq!(msg.content, "hello");
900    }
901
902    #[test]
903    fn regression_assistant_message_has_correct_role() {
904        let msg = ModelMessage::assistant("response".to_string());
905        assert_eq!(msg.role, ModelRole::Assistant);
906        assert_eq!(msg.content, "response");
907    }
908
909    #[test]
910    fn regression_tool_result_sets_call_id_and_name() {
911        let msg = ModelMessage::tool_result("call-1", "read_file", "content");
912        assert_eq!(msg.role, ModelRole::Tool);
913        assert_eq!(msg.tool_call_id.as_deref(), Some("call-1"));
914        assert_eq!(msg.tool_name.as_deref(), Some("read_file"));
915        assert_eq!(msg.content, "content");
916    }
917
918    #[test]
919    fn regression_assistant_tool_call_with_context_sets_fields() {
920        let inv = ToolInvocation {
921            id: "call-1".to_string(),
922            tool_name: "read_file".to_string(),
923            input: serde_json::json!({"path": "test.rs"}),
924        };
925        let msg = ModelMessage::assistant_tool_call_with_context(
926            inv,
927            "thinking text",
928            Some("reasoning".to_string()),
929        );
930        assert_eq!(msg.role, ModelRole::Assistant);
931        assert_eq!(msg.content, "thinking text");
932        assert_eq!(msg.thinking_content.as_deref(), Some("reasoning"));
933        assert_eq!(msg.tool_calls.len(), 1);
934        assert_eq!(msg.tool_calls[0].id, "call-1");
935    }
936
937    // ── Regression: ModelMessage serialization roundtrip ──────────────────────
938
939    #[test]
940    fn regression_model_message_serialization_roundtrip() {
941        let msg = ModelMessage {
942            role: ModelRole::Assistant,
943            content: "hello".to_string(),
944            content_parts: Vec::new(),
945            tool_call_id: None,
946            tool_name: None,
947            tool_calls: vec![],
948            thinking_content: Some("thinking".to_string()),
949            created_at: Some(12345),
950        };
951        let json = serde_json::to_string(&msg).unwrap();
952        let deserialized: ModelMessage = serde_json::from_str(&json).unwrap();
953        assert_eq!(deserialized.role, msg.role);
954        assert_eq!(deserialized.content, msg.content);
955        assert_eq!(deserialized.thinking_content, msg.thinking_content);
956        // created_at is intentionally not serialized (runtime-only field)
957        assert!(deserialized.created_at.is_none());
958    }
959
960    // ── Effort resolution (no adaptive) ────────────────────────────────────────
961
962    #[test]
963    fn legacy_adaptive_string_maps_to_max() {
964        assert_eq!(
965            ThinkingConfig::from_config_str("adaptive"),
966            ThinkingConfig::Max
967        );
968        assert_eq!(parse_reasoning_level("auto"), Some(ThinkingConfig::Max));
969        assert_eq!(parse_reasoning_level("on"), Some(ThinkingConfig::Max));
970        let deserialized: ThinkingConfig = serde_json::from_str("\"adaptive\"").unwrap();
971        assert_eq!(deserialized, ThinkingConfig::Max);
972    }
973
974    #[test]
975    fn resolve_forces_off_when_model_lacks_reasoning() {
976        let resolved =
977            resolve_model_thinking_level(ThinkingConfig::Max, Some(false), &["high".into()], None);
978        assert_eq!(resolved, ThinkingConfig::Off);
979    }
980
981    #[test]
982    fn resolve_defaults_to_max_when_preference_unsupported() {
983        let resolved = resolve_model_thinking_level(
984            ThinkingConfig::Low,
985            Some(true),
986            &["high".into(), "xhigh".into()],
987            None,
988        );
989        assert_eq!(resolved, ThinkingConfig::Max);
990    }
991
992    #[test]
993    fn binary_on_is_max() {
994        assert_eq!(
995            BINARY_REASONING_LEVELS,
996            &[ThinkingConfig::Max, ThinkingConfig::Off]
997        );
998        assert_eq!(
999            effort_display_label(ThinkingConfig::Max, true),
1000            "thinking on"
1001        );
1002        assert_eq!(
1003            effort_display_label(ThinkingConfig::Off, true),
1004            "thinking off"
1005        );
1006    }
1007}