Skip to main content

llm_api/
policy.rs

1//! Pure invocation policy shared by deployments. No discovery or provider routing.
2use std::collections::BTreeSet;
3
4use artifact_api::ArtifactKind;
5#[cfg(test)]
6use artifact_api::ArtifactReference;
7use serde::{Deserialize, Serialize};
8
9use crate::{ContentPart, Message, ModelConstraints};
10
11/// Closed vocabulary for non-text model input. `text` is implied and must not be listed.
12pub const INPUT_IMAGE: &str = "image";
13/// Video input.
14pub const INPUT_VIDEO: &str = "video";
15/// Audio input.
16pub const INPUT_AUDIO: &str = "audio";
17/// Generic file input.
18pub const INPUT_FILE: &str = "file";
19
20/// Confirmed model capabilities. Missing information never establishes support.
21#[derive(Clone, Debug, Default, Eq, PartialEq, Serialize, Deserialize)]
22pub struct ModelCapabilities {
23    /// Closed vocabulary: `image`, `video`, `audio`, `file`. `text` is implied.
24    #[serde(default)]
25    pub input: Vec<String>,
26    pub tool_calling: bool,
27    pub structured_output: bool,
28}
29
30impl ModelConstraints {
31    /// Checks declarations only, including requirements on requests with no attachments/tools.
32    pub fn validate(&self, capabilities: &ModelCapabilities) -> Result<(), &'static str> {
33        let required = normalize_constraint_input(&self.input)?;
34        let supported = normalize_capability_input(&capabilities.input)?;
35        for token in &required {
36            if !supported.iter().any(|item| item == token) {
37                return Err(static_input_token(token));
38            }
39        }
40        if self.tool_calling && !capabilities.tool_calling {
41            return Err("tool_calling");
42        }
43        if self.structured_output && !capabilities.structured_output {
44            return Err("structured_output");
45        }
46        Ok(())
47    }
48
49    /// Payload modalities must be a subset of the caller-declared `input` list.
50    pub fn validate_payload<'a>(
51        &self,
52        messages: impl IntoIterator<Item = &'a Message>,
53    ) -> Result<(), &'static str> {
54        let declared = normalize_constraint_input(&self.input)?;
55        for token in payload_input_modalities(messages)? {
56            if !declared.iter().any(|item| item == &token) {
57                return Err(static_input_token(&token));
58            }
59        }
60        Ok(())
61    }
62}
63
64/// Reasoning intensity ordered from least to most. Turning thinking off is the `thinking`
65/// switch, never a listed intensity, so `none` is not on this ladder.
66pub const REASONING_EFFORT_LADDER: &[&str] =
67    &["minimal", "low", "medium", "high", "xhigh", "max", "ultra"];
68
69/// Internal per-response output cap. Not a user setting. Adapters omit it when
70/// the model snapshot says `maxTokens` is false.
71pub const DEFAULT_MAX_OUTPUT_TOKENS: u32 = 32_768;
72
73/// Deployment-owned generation settings, never supplied through Agent constraints.
74/// None leaves the setting unspecified. These are desired preferences. Adapters
75/// omit unsupported fields and clamp reasoning intensity onto the model's list.
76/// Syntax validation does not establish model/provider support.
77#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
78#[serde(deny_unknown_fields)]
79pub struct GenerationParameters {
80    pub reasoning_effort: Option<String>,
81    pub temperature: Option<f64>,
82    pub thinking: Option<bool>,
83    pub fast_mode: Option<bool>,
84}
85
86/// Maps a desired effort onto `supported`. Exact matches are kept. Otherwise the nearest
87/// ladder neighbor is used; ties pick the lower effort. An empty supported list, or no
88/// desired value, omits the field. Values not on the ladder are ignored.
89fn clamp_reasoning_effort(desired: Option<&str>, supported: &[impl AsRef<str>]) -> Option<String> {
90    let desired = desired?;
91    let supported: Vec<&str> = supported
92        .iter()
93        .map(AsRef::as_ref)
94        .filter(|value| !value.is_empty())
95        .collect();
96    if supported.is_empty() {
97        return None;
98    }
99    if supported.contains(&desired) {
100        return Some(desired.to_string());
101    }
102    let want = effort_rank(desired)?;
103    let mut best: Option<(&str, usize, usize)> = None;
104    for item in supported {
105        let Some(rank) = effort_rank(item) else {
106            continue;
107        };
108        let dist = rank.abs_diff(want);
109        match best {
110            None => best = Some((item, dist, rank)),
111            Some((_, best_dist, best_rank))
112                if dist < best_dist || (dist == best_dist && rank < best_rank) =>
113            {
114                best = Some((item, dist, rank));
115            }
116            _ => {}
117        }
118    }
119    best.map(|(item, _, _)| item.to_string())
120}
121
122fn effort_rank(effort: &str) -> Option<usize> {
123    REASONING_EFFORT_LADDER
124        .iter()
125        .position(|item| *item == effort)
126}
127
128impl GenerationParameters {
129    pub fn validate(&self) -> Result<(), &'static str> {
130        if let Some(effort) = self.reasoning_effort.as_deref() {
131            if effort_rank(effort).is_none() {
132                return Err("invalid reasoning_effort");
133            }
134        }
135        if self
136            .temperature
137            .is_some_and(|value| !value.is_finite() || value < 0.0)
138        {
139            return Err("temperature must be finite and nonnegative");
140        }
141        Ok(())
142    }
143
144    /// Stored preferences are wishes, not a contract. Retired `none` means thinking off;
145    /// any other illegal value is dropped so a request can still run.
146    #[must_use]
147    pub fn normalize_stored(mut self) -> Self {
148        if self.reasoning_effort.as_deref() == Some("none") {
149            self.reasoning_effort = None;
150            if self.thinking.is_none() {
151                self.thinking = Some(false);
152            }
153        }
154        if self
155            .reasoning_effort
156            .as_deref()
157            .is_some_and(|effort| effort_rank(effort).is_none())
158        {
159            self.reasoning_effort = None;
160        }
161        if self
162            .temperature
163            .is_some_and(|value| !value.is_finite() || value < 0.0)
164        {
165            self.temperature = None;
166        }
167        self
168    }
169}
170
171/// Confirmed per-model parameter support from catalog `metadata.generationSupport`.
172/// Missing information never establishes support; it is never inferred from preferences.
173/// Unknown catalog keys are ignored so a newer catalog cannot break an older client.
174#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
175#[serde(rename_all = "camelCase")]
176pub struct GenerationSupport {
177    pub temperature: Option<bool>,
178    pub max_tokens: Option<bool>,
179    /// True when the model can run with thinking turned off.
180    #[serde(default, skip_serializing_if = "Option::is_none")]
181    pub thinking: Option<bool>,
182    /// Selectable intensities. `none` is not an intensity; see `thinking`.
183    pub reasoning_efforts: Option<Vec<String>>,
184    #[serde(default, skip_serializing_if = "Option::is_none")]
185    pub reasoning_effort_default: Option<String>,
186    pub temperature_with_reasoning: Option<bool>,
187    pub temperature_max: Option<f64>,
188    pub max_output_tokens: Option<u32>,
189    #[serde(default, skip_serializing_if = "Option::is_none")]
190    pub fast_mode: Option<bool>,
191}
192
193impl GenerationSupport {
194    pub fn validate(&self) -> Result<(), &'static str> {
195        if self
196            .temperature_max
197            .is_some_and(|value| !value.is_finite() || value < 0.0)
198            || self.max_output_tokens == Some(0)
199        {
200            return Err("invalid generation support limits");
201        }
202        for effort in self.efforts() {
203            if effort_rank(effort).is_none() {
204                return Err("reasoningEfforts must list reasoning intensities");
205            }
206        }
207        if let Some(default) = self.reasoning_effort_default.as_deref() {
208            if !self.efforts().any(|effort| effort == default) {
209                return Err("reasoningEffortDefault must be listed in reasoningEfforts");
210            }
211        }
212        Ok(())
213    }
214
215    fn efforts(&self) -> impl Iterator<Item = &str> {
216        self.reasoning_efforts.iter().flatten().map(String::as_str)
217    }
218
219    fn thinking_supported(&self) -> bool {
220        self.thinking == Some(true)
221    }
222
223    fn reasons(&self) -> bool {
224        self.thinking_supported() || self.efforts().next().is_some()
225    }
226
227    fn thinking_on(&self, desired: Option<bool>) -> bool {
228        if self.thinking_supported() {
229            desired.unwrap_or(true)
230        } else {
231            self.efforts().next().is_some()
232        }
233    }
234
235    fn effective_max_output_tokens(&self) -> Option<u32> {
236        if self.max_tokens == Some(false) {
237            None
238        } else {
239            Some(
240                self.max_output_tokens
241                    .map_or(DEFAULT_MAX_OUTPUT_TOKENS, |max| {
242                        DEFAULT_MAX_OUTPUT_TOKENS.min(max)
243                    }),
244            )
245        }
246    }
247}
248
249/// What one backend protocol can express, independent of any model. These are wire facts
250/// owned by the adapter, not catalog facts: a protocol that cannot carry a field makes the
251/// field unusable even when the model supports it.
252#[derive(Clone, Copy, Debug, Eq, PartialEq)]
253pub struct BackendCapability {
254    /// Accepts a per-response output token cap.
255    pub token_cap: bool,
256    /// Accepts a sampling temperature.
257    pub temperature: bool,
258    /// Can encode the thinking switch, and therefore can turn thinking off.
259    pub thinking_switch: bool,
260    /// Can encode a reasoning intensity.
261    pub intensity: bool,
262    /// Can request the fast service tier.
263    pub fast: bool,
264}
265
266/// Generation parameters that may reach the wire. Fields left `None` are omitted from the
267/// payload, which leaves the provider default in effect.
268#[derive(Clone, Debug, Default, PartialEq, Serialize)]
269pub struct EffectiveGeneration {
270    pub temperature: Option<f64>,
271    pub max_output_tokens: Option<u32>,
272    pub thinking: Option<bool>,
273    pub reasoning_effort: Option<String>,
274    pub fast_mode: Option<bool>,
275}
276
277impl EffectiveGeneration {
278    /// Provider payloads carry temperature as f32.
279    #[must_use]
280    #[allow(clippy::cast_possible_truncation)]
281    pub fn temperature_f32(&self) -> Option<f32> {
282        self.temperature.map(|value| value as f32)
283    }
284
285    /// The internal cap, lowered by an optional per-call output budget. A budget never raises
286    /// the cap, and it stays out of the wire identity so it can vary between steps of one run.
287    #[must_use]
288    pub fn output_cap(&self, budget: Option<u32>) -> Option<u32> {
289        self.max_output_tokens
290            .map(|ceiling| budget.unwrap_or(ceiling).min(ceiling))
291    }
292}
293
294/// What a settings UI may expose after intersecting catalog facts with the backend protocol.
295/// This is the same judgment `resolve` uses; a hidden control cannot appear on the wire, and
296/// a shown control is one the current backend can actually carry.
297#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
298#[serde(rename_all = "camelCase")]
299pub struct GenerationControls {
300    pub temperature: bool,
301    #[serde(skip_serializing_if = "Option::is_none")]
302    pub temperature_max: Option<f64>,
303    pub thinking: bool,
304    #[serde(default, skip_serializing_if = "Vec::is_empty")]
305    pub reasoning_efforts: Vec<String>,
306    #[serde(skip_serializing_if = "Option::is_none")]
307    pub reasoning_effort_default: Option<String>,
308    pub fast_mode: bool,
309}
310
311/// Settings that can take effect on this backend for this model. Callers must not infer
312/// controls from catalog fields alone.
313#[must_use]
314pub fn controls(support: &GenerationSupport, backend: &BackendCapability) -> GenerationControls {
315    let thinking = backend.thinking_switch && support.thinking_supported();
316    let reasoning_efforts: Vec<String> = if backend.intensity {
317        support.efforts().map(str::to_owned).collect()
318    } else {
319        Vec::new()
320    };
321    let can_use_temperature_while_reasoning = support.temperature_with_reasoning.unwrap_or(false);
322    let temperature = backend.temperature
323        && support.temperature == Some(true)
324        && (can_use_temperature_while_reasoning || thinking || !support.reasons());
325    GenerationControls {
326        temperature_max: temperature.then_some(support.temperature_max).flatten(),
327        temperature,
328        thinking,
329        reasoning_effort_default: reasoning_efforts
330            .iter()
331            .find(|effort| Some(effort.as_str()) == support.reasoning_effort_default.as_deref())
332            .cloned(),
333        reasoning_efforts,
334        fast_mode: backend.fast && support.fast_mode == Some(true),
335    }
336}
337
338/// The single resolution of desired generation parameters against model and protocol facts.
339/// Every caller resolves exactly once, against facts frozen for the request, so replaying a
340/// frozen route always yields the same wire payload.
341///
342/// An unsupported or undeclared setting is omitted rather than guessed, except for the
343/// internal output cap, which is a product default and applies unless the protocol or the
344/// model rejects a cap. Illegal *desired* values are dropped the same way; illegal catalog
345/// support still fails, because that data is ours.
346pub fn resolve(
347    desired: &GenerationParameters,
348    support: &GenerationSupport,
349    backend: &BackendCapability,
350) -> Result<EffectiveGeneration, &'static str> {
351    support.validate()?;
352    let desired = desired.clone().normalize_stored();
353
354    let thinking_on = support.thinking_on(desired.thinking);
355    let thinking = (backend.thinking_switch && support.thinking_supported()).then_some(thinking_on);
356
357    let reasoning_effort = if thinking_on && backend.intensity {
358        let intensities: Vec<&str> = support.efforts().collect();
359        clamp_reasoning_effort(
360            desired
361                .reasoning_effort
362                .as_deref()
363                .or(support.reasoning_effort_default.as_deref()),
364            &intensities,
365        )
366    } else {
367        None
368    };
369
370    let temperature = desired
371        .temperature
372        .filter(|_| {
373            backend.temperature
374                && support.temperature == Some(true)
375                && (support.temperature_with_reasoning.unwrap_or(false) || !thinking_on)
376        })
377        .map(|value| support.temperature_max.map_or(value, |max| value.min(max)));
378
379    let max_output_tokens = backend
380        .token_cap
381        .then(|| support.effective_max_output_tokens())
382        .flatten();
383
384    let fast_mode =
385        (backend.fast && support.fast_mode == Some(true) && desired.fast_mode == Some(true))
386            .then_some(true);
387
388    Ok(EffectiveGeneration {
389        temperature,
390        max_output_tokens,
391        thinking,
392        reasoning_effort,
393        fast_mode,
394    })
395}
396
397/// Maps one catalog/constraint token. `text` is dropped. `vision` and unknown values fail.
398pub fn normalize_input_token(raw: &str) -> Result<Option<&'static str>, &'static str> {
399    let token = raw.trim().to_ascii_lowercase();
400    if token.is_empty() || token == "text" {
401        return Ok(None);
402    }
403    match token.as_str() {
404        INPUT_IMAGE => Ok(Some(INPUT_IMAGE)),
405        INPUT_VIDEO => Ok(Some(INPUT_VIDEO)),
406        INPUT_AUDIO => Ok(Some(INPUT_AUDIO)),
407        INPUT_FILE => Ok(Some(INPUT_FILE)),
408        "vision" => Err("vision"),
409        _ => Err("unknown input modality"),
410    }
411}
412
413/// Normalizes catalog capability tokens. `text` is ignored; unknown tokens fail.
414pub fn normalize_capability_input<I, S>(raw: I) -> Result<Vec<String>, &'static str>
415where
416    I: IntoIterator<Item = S>,
417    S: AsRef<str>,
418{
419    let mut set = BTreeSet::new();
420    for item in raw {
421        if let Some(token) = normalize_input_token(item.as_ref())? {
422            set.insert(token.to_string());
423        }
424    }
425    Ok(set.into_iter().collect())
426}
427
428/// Normalizes caller-declared constraint tokens. `text` is illegal here.
429pub fn normalize_constraint_input<I, S>(raw: I) -> Result<Vec<String>, &'static str>
430where
431    I: IntoIterator<Item = S>,
432    S: AsRef<str>,
433{
434    let mut set = BTreeSet::new();
435    for item in raw {
436        match normalize_input_token(item.as_ref())? {
437            None => return Err("text"),
438            Some(token) => {
439                set.insert(token.to_string());
440            }
441        }
442    }
443    Ok(set.into_iter().collect())
444}
445
446/// Closed-vocabulary token for an artifact kind.
447#[must_use]
448pub const fn input_modality_for_kind(kind: ArtifactKind) -> &'static str {
449    match kind {
450        ArtifactKind::Image => INPUT_IMAGE,
451        ArtifactKind::Video => INPUT_VIDEO,
452        ArtifactKind::Audio => INPUT_AUDIO,
453        ArtifactKind::File => INPUT_FILE,
454    }
455}
456
457/// Closed-vocabulary token for a MIME type. Non-media types map to `file`.
458#[must_use]
459pub fn input_modality_for_mime(mime_type: &str) -> &'static str {
460    let mime = mime_type.trim().to_ascii_lowercase();
461    if mime.starts_with("image/") {
462        INPUT_IMAGE
463    } else if mime.starts_with("video/") {
464        INPUT_VIDEO
465    } else if mime.starts_with("audio/") {
466        INPUT_AUDIO
467    } else {
468        INPUT_FILE
469    }
470}
471
472/// Distinct payload modalities required by artifact parts.
473pub fn payload_input_modalities<'a>(
474    messages: impl IntoIterator<Item = &'a Message>,
475) -> Result<Vec<String>, &'static str> {
476    let mut set = BTreeSet::new();
477    for message in messages {
478        for part in &message.content {
479            if let Some(token) = part_input_modality(part)? {
480                set.insert(token.to_string());
481            }
482        }
483    }
484    Ok(set.into_iter().collect())
485}
486
487fn part_input_modality(part: &ContentPart) -> Result<Option<&'static str>, &'static str> {
488    match part {
489        ContentPart::Artifact { uri } => Ok(Some(input_modality_for_kind(
490            uri.reference().metadata().kind(),
491        ))),
492        _ => Ok(None),
493    }
494}
495
496fn static_input_token(token: &str) -> &'static str {
497    match token {
498        INPUT_IMAGE => INPUT_IMAGE,
499        INPUT_VIDEO => INPUT_VIDEO,
500        INPUT_AUDIO => INPUT_AUDIO,
501        INPUT_FILE => INPUT_FILE,
502        "text" => "text",
503        "vision" => "vision",
504        "unknown input modality" => "unknown input modality",
505        _ => "undeclared input",
506    }
507}
508
509#[cfg(test)]
510mod tests {
511    use super::*;
512    use crate::{MessageRole, ModelConstraints};
513    use artifact_api::ArtifactMetadata;
514
515    fn image_uri() -> (String, String) {
516        let artifact = ArtifactReference::new(
517            "tenant-1",
518            "scope-1",
519            "a".repeat(64),
520            ArtifactMetadata::image("image/png", 100, 10, 10).expect("valid image metadata"),
521        )
522        .expect("valid image artifact");
523        (artifact.uri().expect("uri"), "image/png".to_owned())
524    }
525
526    fn audio_uri() -> (String, String) {
527        let artifact = ArtifactReference::new(
528            "tenant-1",
529            "scope-1",
530            "b".repeat(64),
531            ArtifactMetadata::audio("audio/mpeg", 100, Some(1_000)).expect("valid audio metadata"),
532        )
533        .expect("valid audio artifact");
534        (artifact.uri().expect("uri"), "audio/mpeg".to_owned())
535    }
536
537    #[test]
538    fn declarations_are_requirements_not_prohibitions() {
539        let supported = ModelCapabilities {
540            input: vec![INPUT_IMAGE.into()],
541            tool_calling: true,
542            structured_output: true,
543        };
544        assert!(ModelConstraints::default().validate(&supported).is_ok());
545        assert!(ModelConstraints::default()
546            .validate(&ModelCapabilities::default())
547            .is_ok());
548        for constraints in [
549            ModelConstraints {
550                input: vec![INPUT_IMAGE.into()],
551                ..Default::default()
552            },
553            ModelConstraints {
554                tool_calling: true,
555                ..Default::default()
556            },
557            ModelConstraints {
558                structured_output: true,
559                ..Default::default()
560            },
561        ] {
562            assert!(constraints.validate(&supported).is_ok());
563            assert!(constraints.validate(&ModelCapabilities::default()).is_err());
564        }
565    }
566
567    #[test]
568    fn input_lists_use_closed_vocabulary_and_ignore_catalog_text() {
569        assert_eq!(
570            normalize_capability_input(["text", "IMAGE", "image"]).unwrap(),
571            vec![INPUT_IMAGE.to_string()]
572        );
573        assert!(normalize_constraint_input(["text"]).is_err());
574        assert!(normalize_input_token("vision").is_err());
575        assert!(normalize_input_token("unknown").is_err());
576        let constraints = ModelConstraints {
577            input: vec![INPUT_VIDEO.into()],
578            ..Default::default()
579        };
580        assert_eq!(
581            constraints
582                .validate(&ModelCapabilities {
583                    input: vec![INPUT_IMAGE.into()],
584                    ..Default::default()
585                })
586                .unwrap_err(),
587            INPUT_VIDEO
588        );
589    }
590
591    #[test]
592    fn payload_must_be_declared_and_kind_must_match_mime() {
593        let (image_uri, _image_mime) = image_uri();
594        let (audio_uri, _audio_mime) = audio_uri();
595        let image_message = Message {
596            role: MessageRole::User,
597            content: vec![ContentPart::Artifact {
598                uri: image_uri.parse().unwrap(),
599            }],
600            continuation: None,
601        };
602        let audio_message = Message {
603            role: MessageRole::User,
604            content: vec![ContentPart::Artifact {
605                uri: audio_uri.parse().unwrap(),
606            }],
607            continuation: None,
608        };
609        assert_eq!(
610            payload_input_modalities([&image_message]).unwrap(),
611            vec![INPUT_IMAGE.to_string()]
612        );
613        let declared = ModelConstraints {
614            input: vec![INPUT_IMAGE.into()],
615            ..Default::default()
616        };
617        assert!(declared.validate_payload([&image_message]).is_ok());
618        assert_eq!(
619            declared.validate_payload([&audio_message]).unwrap_err(),
620            INPUT_AUDIO
621        );
622        assert!(serde_json::from_value::<ContentPart>(serde_json::json!({
623            "type":"artifact", "uri": audio_uri, "mime_type":"image/png"
624        }))
625        .is_err());
626        assert!(serde_json::from_value::<ContentPart>(serde_json::json!({
627            "type":"artifact", "uri":"https://example.com/a.png"
628        }))
629        .is_err());
630    }
631
632    #[test]
633    fn retired_generation_control_is_not_silently_ignored() {
634        assert!(
635            serde_json::from_value::<ModelConstraints>(serde_json::json!({
636                "vision": false, "tool_calling": false, "structured_output": false
637            }))
638            .is_err()
639        );
640        assert!(serde_json::from_value::<ModelConstraints>(serde_json::json!({
641            "input": ["image"], "tool_calling": false, "structured_output": false, "reasoning": "high"
642        }))
643        .is_err());
644    }
645
646    #[test]
647    fn clamps_desired_effort_onto_supported_list() {
648        let flash = ["max", "high", "low"];
649        assert_eq!(
650            clamp_reasoning_effort(Some("low"), &flash).as_deref(),
651            Some("low")
652        );
653        assert_eq!(
654            clamp_reasoning_effort(Some("ultra"), &flash).as_deref(),
655            Some("max")
656        );
657        assert_eq!(
658            clamp_reasoning_effort(Some("medium"), &["low", "high"]).as_deref(),
659            Some("low")
660        );
661        assert_eq!(
662            clamp_reasoning_effort(Some("minimal"), &["high", "max"]).as_deref(),
663            Some("high")
664        );
665        assert_eq!(clamp_reasoning_effort(Some("low"), &[] as &[&str]), None);
666        assert_eq!(clamp_reasoning_effort(None, &flash), None);
667    }
668
669    #[test]
670    fn validates_explicit_generation_settings() {
671        for effort in ["minimal", "low", "high", "max", "ultra"] {
672            assert!(GenerationParameters {
673                reasoning_effort: Some(effort.into()),
674                ..Default::default()
675            }
676            .validate()
677            .is_ok());
678        }
679        for effort in ["unknown", "none"] {
680            assert!(GenerationParameters {
681                reasoning_effort: Some(effort.into()),
682                ..Default::default()
683            }
684            .validate()
685            .is_err());
686        }
687        for temperature in [-1.0, f64::NAN, f64::INFINITY] {
688            assert!(GenerationParameters {
689                temperature: Some(temperature),
690                ..Default::default()
691            }
692            .validate()
693            .is_err());
694        }
695    }
696
697    const FULL: BackendCapability = BackendCapability {
698        token_cap: true,
699        temperature: true,
700        thinking_switch: true,
701        intensity: true,
702        fast: true,
703    };
704
705    fn switchable() -> GenerationSupport {
706        GenerationSupport {
707            temperature: Some(true),
708            max_tokens: Some(true),
709            thinking: Some(true),
710            reasoning_efforts: Some(vec!["low".into(), "medium".into(), "high".into()]),
711            ..Default::default()
712        }
713    }
714
715    #[test]
716    fn thinking_defaults_on_and_the_switch_turns_it_off() {
717        let support = switchable();
718        let on = resolve(&GenerationParameters::default(), &support, &FULL).unwrap();
719        assert_eq!(on.thinking, Some(true));
720        let off = resolve(
721            &GenerationParameters {
722                thinking: Some(false),
723                ..Default::default()
724            },
725            &support,
726            &FULL,
727        )
728        .unwrap();
729        assert_eq!(off.thinking, Some(false));
730        assert_eq!(off.reasoning_effort, None);
731    }
732
733    #[test]
734    fn intensity_comes_from_the_request_or_the_catalog_default_and_is_never_invented() {
735        let support = switchable();
736        assert_eq!(
737            resolve(&GenerationParameters::default(), &support, &FULL)
738                .unwrap()
739                .reasoning_effort,
740            None
741        );
742        assert_eq!(
743            resolve(
744                &GenerationParameters {
745                    reasoning_effort: Some("ultra".into()),
746                    ..Default::default()
747                },
748                &support,
749                &FULL
750            )
751            .unwrap()
752            .reasoning_effort
753            .as_deref(),
754            Some("high")
755        );
756        let defaulted = GenerationSupport {
757            reasoning_effort_default: Some("medium".into()),
758            ..switchable()
759        };
760        assert_eq!(
761            resolve(&GenerationParameters::default(), &defaulted, &FULL)
762                .unwrap()
763                .reasoning_effort
764                .as_deref(),
765            Some("medium")
766        );
767    }
768
769    #[test]
770    fn a_protocol_that_cannot_express_a_field_omits_it() {
771        let support = GenerationSupport {
772            fast_mode: Some(true),
773            ..switchable()
774        };
775        let desired = GenerationParameters {
776            temperature: Some(0.7),
777            reasoning_effort: Some("low".into()),
778            thinking: Some(false),
779            fast_mode: Some(true),
780        };
781        let full = resolve(&desired, &support, &FULL).unwrap();
782        assert_eq!(
783            (full.thinking, full.fast_mode, full.max_output_tokens),
784            (Some(false), Some(true), Some(DEFAULT_MAX_OUTPUT_TOKENS))
785        );
786        let bare = resolve(
787            &desired,
788            &support,
789            &BackendCapability {
790                token_cap: false,
791                temperature: false,
792                thinking_switch: false,
793                intensity: false,
794                fast: false,
795            },
796        )
797        .unwrap();
798        assert_eq!(bare, EffectiveGeneration::default());
799    }
800
801    #[test]
802    fn undeclared_support_omits_instead_of_guessing() {
803        let silent = GenerationSupport::default();
804        let effective = resolve(
805            &GenerationParameters {
806                temperature: Some(0.7),
807                reasoning_effort: Some("high".into()),
808                thinking: Some(false),
809                fast_mode: Some(true),
810            },
811            &silent,
812            &FULL,
813        )
814        .unwrap();
815        assert_eq!(
816            effective,
817            EffectiveGeneration {
818                max_output_tokens: Some(DEFAULT_MAX_OUTPUT_TOKENS),
819                ..Default::default()
820            }
821        );
822    }
823
824    #[test]
825    fn temperature_needs_declared_support_and_coexistence_with_thinking() {
826        let hot = GenerationSupport {
827            temperature_max: Some(1.0),
828            ..switchable()
829        };
830        let desired = GenerationParameters {
831            temperature: Some(1.5),
832            ..Default::default()
833        };
834        assert_eq!(resolve(&desired, &hot, &FULL).unwrap().temperature, None);
835        let coexists = GenerationSupport {
836            temperature_with_reasoning: Some(true),
837            ..hot.clone()
838        };
839        assert_eq!(
840            resolve(&desired, &coexists, &FULL).unwrap().temperature,
841            Some(1.0)
842        );
843        assert_eq!(
844            resolve(
845                &GenerationParameters {
846                    thinking: Some(false),
847                    ..desired
848                },
849                &hot,
850                &FULL
851            )
852            .unwrap()
853            .temperature,
854            Some(1.0)
855        );
856    }
857
858    #[test]
859    fn non_switchable_models_reason_whenever_they_declare_intensities() {
860        let always = GenerationSupport {
861            reasoning_efforts: Some(vec!["low".into(), "high".into()]),
862            max_tokens: Some(true),
863            ..Default::default()
864        };
865        let effective = resolve(
866            &GenerationParameters {
867                thinking: Some(false),
868                ..Default::default()
869            },
870            &always,
871            &FULL,
872        )
873        .unwrap();
874        assert_eq!(effective.thinking, None);
875        assert_eq!(effective.reasoning_effort, None);
876        let none = GenerationSupport::default();
877        assert_eq!(
878            resolve(&GenerationParameters::default(), &none, &FULL)
879                .unwrap()
880                .thinking,
881            None
882        );
883    }
884
885    #[test]
886    fn resolution_is_stable_when_replayed_against_the_same_facts() {
887        let support = GenerationSupport {
888            fast_mode: Some(true),
889            temperature_with_reasoning: Some(true),
890            reasoning_effort_default: Some("medium".into()),
891            ..switchable()
892        };
893        let desired = GenerationParameters {
894            temperature: Some(0.4),
895            reasoning_effort: Some("ultra".into()),
896            thinking: Some(true),
897            fast_mode: Some(true),
898        };
899        let first = resolve(&desired, &support, &FULL).unwrap();
900        let replay = resolve(
901            &GenerationParameters {
902                temperature: first.temperature,
903                reasoning_effort: first.reasoning_effort.clone(),
904                thinking: first.thinking,
905                fast_mode: first.fast_mode,
906            },
907            &support,
908            &FULL,
909        )
910        .unwrap();
911        assert_eq!(first, replay);
912    }
913
914    #[test]
915    fn support_rejects_disable_as_an_intensity_and_unlisted_defaults() {
916        assert!(GenerationSupport {
917            reasoning_efforts: Some(vec!["none".into()]),
918            ..Default::default()
919        }
920        .validate()
921        .is_err());
922        assert!(GenerationSupport {
923            reasoning_effort_default: Some("max".into()),
924            ..switchable()
925        }
926        .validate()
927        .is_err());
928        assert!(GenerationSupport {
929            max_output_tokens: Some(0),
930            ..Default::default()
931        }
932        .validate()
933        .is_err());
934    }
935
936    #[test]
937    fn the_internal_output_cap_is_a_product_default_not_a_capability() {
938        assert_eq!(
939            GenerationSupport::default().effective_max_output_tokens(),
940            Some(DEFAULT_MAX_OUTPUT_TOKENS)
941        );
942        assert_eq!(
943            GenerationSupport {
944                max_output_tokens: Some(4_096),
945                ..Default::default()
946            }
947            .effective_max_output_tokens(),
948            Some(4_096)
949        );
950        assert_eq!(
951            GenerationSupport {
952                max_tokens: Some(false),
953                ..Default::default()
954            }
955            .effective_max_output_tokens(),
956            None
957        );
958        assert_eq!(
959            EffectiveGeneration {
960                max_output_tokens: Some(4_096),
961                ..Default::default()
962            }
963            .output_cap(Some(1_024)),
964            Some(1_024)
965        );
966        assert_eq!(
967            EffectiveGeneration {
968                max_output_tokens: Some(4_096),
969                ..Default::default()
970            }
971            .output_cap(Some(100_000)),
972            Some(4_096)
973        );
974    }
975
976    #[test]
977    fn controls_match_what_resolve_can_put_on_the_wire() {
978        let switchable = GenerationSupport {
979            temperature: Some(true),
980            temperature_max: Some(1.0),
981            fast_mode: Some(true),
982            ..switchable()
983        };
984        let shown = controls(&switchable, &FULL);
985        assert!(shown.temperature);
986        assert_eq!(shown.temperature_max, Some(1.0));
987        assert!(shown.thinking);
988        assert!(shown.fast_mode);
989        assert_eq!(
990            shown.reasoning_efforts,
991            vec!["low".to_string(), "medium".to_string(), "high".to_string()]
992        );
993        let always = GenerationSupport {
994            temperature: Some(true),
995            reasoning_efforts: Some(vec!["low".into(), "high".into()]),
996            ..Default::default()
997        };
998        let hidden = controls(&always, &FULL);
999        assert!(!hidden.temperature);
1000        assert!(!hidden.thinking);
1001        assert_eq!(hidden.reasoning_efforts, vec!["low", "high"]);
1002        let protocol = controls(
1003            &switchable,
1004            &BackendCapability {
1005                token_cap: true,
1006                temperature: true,
1007                thinking_switch: false,
1008                intensity: true,
1009                fast: false,
1010            },
1011        );
1012        assert!(!protocol.thinking);
1013        assert!(!protocol.temperature);
1014        assert!(!protocol.fast_mode);
1015    }
1016
1017    #[test]
1018    fn retired_and_illegal_preferences_are_dropped_instead_of_failing_the_request() {
1019        let support = switchable();
1020        let from_none = resolve(
1021            &GenerationParameters {
1022                reasoning_effort: Some("none".into()),
1023                ..Default::default()
1024            },
1025            &support,
1026            &FULL,
1027        )
1028        .unwrap();
1029        assert_eq!(from_none.thinking, Some(false));
1030        assert_eq!(from_none.reasoning_effort, None);
1031        let garbage = resolve(
1032            &GenerationParameters {
1033                reasoning_effort: Some("not-a-ladder".into()),
1034                temperature: Some(f64::NAN),
1035                thinking: Some(true),
1036                ..Default::default()
1037            },
1038            &support,
1039            &FULL,
1040        )
1041        .unwrap();
1042        assert_eq!(garbage.thinking, Some(true));
1043        assert_eq!(garbage.reasoning_effort, None);
1044        assert_eq!(garbage.temperature, None);
1045        assert!(GenerationParameters {
1046            reasoning_effort: Some("none".into()),
1047            ..Default::default()
1048        }
1049        .validate()
1050        .is_err());
1051    }
1052
1053    #[test]
1054    fn catalog_support_ignores_unknown_keys() {
1055        let support = serde_json::from_value::<GenerationSupport>(serde_json::json!({
1056            "temperature": true,
1057            "thinking": true,
1058            "reasoningEfforts": ["low"],
1059            "futureFlag": true
1060        }))
1061        .unwrap();
1062        assert_eq!(support.temperature, Some(true));
1063        assert!(support.validate().is_ok());
1064    }
1065}