rig_core/providers/openai/wire.rs
1//! OpenAI-compatible configurations, dialect policies, and endpoint wires.
2//! A [`Dialect`](crate::providers::openai::wire::Dialect) selects request
3//! and response policies; [`OpenAIConfig`](crate::providers::openai::OpenAIConfig)
4//! holds credentials and overrides. An
5//! [`OpenAI`](crate::providers::openai::OpenAI) client puts
6//! the configuration on a transport and builds each endpoint's model.
7//!
8//! ```
9//! use rig_core::providers::openai::{OpenAIConfig, Route, wire::OpenAiWire};
10//!
11//! let openai = OpenAIConfig::new("key").with_route(Route::Chat).client();
12//! assert!(matches!(openai.completion("gpt-5.2").wire, OpenAiWire::Chat(_)));
13//! ```
14
15use serde::{Deserialize, Serialize};
16
17use crate::client::env::{self, EnvError};
18use crate::error::EncodeError;
19use crate::message::Issuer;
20use crate::wire::Secret;
21
22use super::responses_api::SystemInstructionsPlacement;
23use super::responses_api::wire::Responses;
24
25mod chat;
26mod dialects;
27/// Shared Chat Completions response shapes.
28pub(crate) mod dto;
29mod modality;
30mod route;
31
32pub use chat::{Chat, ChatDecoder, ChatEvent};
33pub use dialects::*;
34pub use dto::{ChatChoice, ChatFrame, ChatUsage, FinishReason, StreamingCompletionResponse};
35pub use modality::{
36 Embeddings, EmbeddingsDecoder, ModelEntry, Models, ModelsDecoder, ModelsReply, Rerank,
37 RerankDecoder, RerankReply, RerankResultEntry, RerankUsage, Transcriptions,
38 TranscriptionsDecoder, Verify, VerifyDecoder,
39};
40pub use route::{OpenAiDecoder, OpenAiEvent, OpenAiWire, Route};
41
42#[cfg(feature = "image")]
43pub use modality::{ImageDatum, Images, ImagesDecoder, ImagesEvent, ImagesReply};
44#[cfg(feature = "audio")]
45pub use modality::{Speech, SpeechDecoder};
46
47/// Credential header policy, including omission of empty optional tokens.
48#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)]
49pub enum Auth {
50 /// `Authorization: Bearer <key>`.
51 Bearer,
52 /// `Authorization: Bearer <key>`, omitted entirely when the key is empty.
53 OptionalBearer,
54 /// Azure's `api-key: <key>`.
55 ApiKeyHeader,
56}
57
58/// Alternative credential environment variable and its authentication policy.
59#[derive(Clone, Copy, Debug, PartialEq, Eq)]
60pub struct AuthAlternative {
61 /// The variable holding this credential.
62 pub api_key_env: &'static str,
63 /// How it is sent.
64 pub auth: Auth,
65}
66
67/// Backend selected through the Hugging Face router.
68#[derive(Clone, Debug, Default, PartialEq, Eq, Serialize, Deserialize)]
69pub enum SubRoute {
70 /// Hugging Face's own inference backend: the only one that serves
71 /// transcription and image generation.
72 #[default]
73 HFInference,
74 /// Together AI, through the router.
75 Together,
76 /// SambaNova, through the router.
77 SambaNova,
78 /// Fireworks AI, which addresses models by a qualified id.
79 Fireworks,
80 /// Hyperbolic, through the router.
81 Hyperbolic,
82 /// Nebius, through the router.
83 Nebius,
84 /// Novita, through the router.
85 Novita,
86 /// A route this build does not name.
87 Custom(String),
88}
89
90impl SubRoute {
91 /// The router's slug for this sub-provider.
92 pub fn slug(&self) -> &str {
93 match self {
94 Self::HFInference => "hf-inference/models",
95 Self::Together => "together",
96 Self::SambaNova => "sambanova",
97 Self::Fireworks => "fireworks-ai",
98 Self::Hyperbolic => "hyperbolic",
99 Self::Nebius => "nebius",
100 Self::Novita => "novita",
101 Self::Custom(route) => route,
102 }
103 }
104
105 /// Qualify Fireworks model identifiers unless already prefixed.
106 /// Return other sub-routes' identifiers unchanged.
107 pub fn model_identifier(&self, model: &str) -> String {
108 const FIREWORKS_PREFIX: &str = "accounts/fireworks/models/";
109 match self {
110 Self::Fireworks if !model.starts_with(FIREWORKS_PREFIX) => {
111 format!("{FIREWORKS_PREFIX}{model}")
112 }
113 _ => model.to_owned(),
114 }
115 }
116
117 /// Whether this sub-provider serves the endpoints that address the model
118 /// through the URL (transcription, image generation).
119 pub fn serves_model_routed_endpoints(&self) -> bool {
120 matches!(self, Self::HFInference)
121 }
122}
123
124impl std::fmt::Display for SubRoute {
125 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
126 f.write_str(self.slug())
127 }
128}
129
130impl From<&str> for SubRoute {
131 fn from(route: &str) -> Self {
132 Self::Custom(route.to_owned())
133 }
134}
135
136impl From<String> for SubRoute {
137 fn from(route: String) -> Self {
138 Self::Custom(route)
139 }
140}
141
142/// Paired request and response formats for image generation.
143#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
144pub enum ImageBody {
145 /// `{model, prompt, size}`, answered with `data[].b64_json`.
146 #[default]
147 OpenAi,
148 /// xAI: `{model, prompt, response_format, aspect_ratio}` and no `size`,
149 /// answered with `data[].b64_json` and no `created`.
150 Xai,
151 /// `{model_name, prompt, height, width}`, answered with `images[].image`.
152 Hyperbolic,
153 /// `{model, prompt, width, height}`, answered with base64 strings in `images`.
154 Venice,
155 /// `{inputs, parameters: {width, height}}`, answered with raw image bytes.
156 /// The model is addressed through the URL path.
157 HuggingFace,
158}
159
160/// Which body a speech endpoint takes.
161#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
162pub enum SpeechBody {
163 /// OpenAI: `{model, input, voice, speed}`.
164 #[default]
165 OpenAi,
166 /// xAI: `{text, voice_id, language}`, with `eve` as the default voice.
167 Xai,
168 /// Hyperbolic: `{language, speaker, text, speed}`, answered with
169 /// `{"audio": "<base64>"}` rather than the audio bytes themselves.
170 ///
171 /// It addresses this endpoint by *language*, so the identifier a caller
172 /// passes as the model is the language tag (`"EN"`).
173 Hyperbolic,
174}
175
176/// Which body a transcription endpoint takes.
177#[derive(Clone, Copy, Debug, PartialEq, Eq)]
178pub enum TranscriptionBody {
179 /// OpenAI: a `multipart/form-data` upload with the audio as a file part
180 /// beside `model`, `language`, `prompt` and `temperature`.
181 Multipart,
182 /// OpenRouter: a JSON body whose audio rides base64-encoded under
183 /// `input_audio`, with its container format beside it
184 /// (`{"input_audio": {"data": "…", "format": "mp3"}, "model": …}`).
185 /// The gateway's speech-to-text route serves only this shape, and has no
186 /// top-level `prompt` field at all.
187 InputAudioJson,
188}
189
190/// How a dialect addresses a model.
191#[derive(Clone, Copy, Debug, PartialEq, Eq)]
192pub enum Routing {
193 /// Resolve the endpoint under the base URL and send the model in the body.
194 Path,
195 /// Azure: the model is a *deployment* in the URL
196 /// (`{base}/openai/deployments/{model}/chat/completions?api-version=…`)
197 /// and the body carries no `model` field.
198 AzureDeployment,
199}
200
201/// How a dialect spells the output-token cap.
202#[derive(Clone, Copy, Debug, PartialEq, Eq)]
203pub enum OutputCap {
204 /// `max_tokens`, for every endpoint not observed to reject it.
205 Legacy,
206 /// `max_completion_tokens` for OpenAI's reasoning families, which answer
207 /// a `max_tokens` request with `Unsupported parameter`. Scoped to the
208 /// model, because this same wire reaches compatible servers that know
209 /// only the legacy field.
210 OpenAiReasoningFamilies,
211}
212
213/// Which field a dialect takes an embedding width in.
214#[derive(Clone, Copy, Debug, PartialEq, Eq)]
215pub enum DimensionsField {
216 /// The OpenAI-compatible `dimensions` field.
217 Dimensions,
218 /// Mistral's `output_dimension`.
219 OutputDimension,
220 /// The server ignores any width field, so none is sent (`llama-server`
221 /// reads no such field and would answer 200 with the native width).
222 Ignored,
223}
224
225impl DimensionsField {
226 /// The body field a requested width goes in, or `None` when the dialect
227 /// reads no width field at all.
228 ///
229 /// The encoder puts a width in this field and a refusal names it, so
230 /// both spell it from here rather than from two matching literals.
231 pub const fn name(self) -> Option<&'static str> {
232 match self {
233 Self::Dimensions => Some("dimensions"),
234 Self::OutputDimension => Some("output_dimension"),
235 Self::Ignored => None,
236 }
237 }
238}
239
240/// Which widths a request may name for one embedding model.
241#[derive(Clone, Copy, Debug, PartialEq, Eq)]
242pub enum AcceptedWidths {
243 /// The model emits one width and reads no width field, so any value but
244 /// its own native width is a request for a parameter the provider does
245 /// not accept there, and is refused as a request error.
246 Fixed,
247 /// The model truncates to any width in `min..=max`, and anything else is
248 /// refused with [`requirement`](Self::Range::requirement).
249 Range {
250 /// Narrowest width the provider honours.
251 min: usize,
252 /// Widest width the provider honours.
253 max: usize,
254 /// Static error clause describing the accepted bounds.
255 /// Must agree with `min` and `max`.
256 requirement: &'static str,
257 },
258}
259
260/// One embedding model's width contract: the width it returns unasked, and
261/// the widths it will honour when asked.
262///
263/// Stated per model rather than per dialect because a dialect serves models
264/// of different widths, and a model's default is not always its maximum.
265#[derive(Clone, Copy, Debug, PartialEq, Eq)]
266#[non_exhaustive]
267pub struct ModelWidth {
268 /// The model identifier, as the `model` field spells it.
269 pub model: &'static str,
270 /// Default width reported when no width is requested, or `None` if unknown.
271 /// Unknown widths report zero as the embedding model's `ndims`.
272 pub default: Option<usize>,
273 /// The widths a request may name.
274 pub accepted: AcceptedWidths,
275}
276
277/// Dialect-specific transformation of the serialized chat request.
278#[derive(Clone, Copy, Debug, PartialEq, Eq)]
279pub enum BodyRewrite {
280 /// Send the OpenAI-compatible body unchanged.
281 None,
282 /// Groq: fold `additional_params.tools` (its compound-system native
283 /// tools) into `compound_custom.enabled_tools` so they do not clobber
284 /// the function-tool array on serialization, and replay assistant turns
285 /// without `reasoning_content`, which Groq rejects.
286 GroqCompoundTools,
287 /// Hugging Face's router: qualify the model identifier for sub-providers
288 /// that demand one (Fireworks).
289 HuggingFaceRouter,
290 /// DeepSeek: string-flattened content, `content: ""` on tool-call-only
291 /// assistant turns, `index` on echoed tool calls, and forced tool
292 /// choices suppressed unless thinking is explicitly disabled.
293 DeepSeek,
294 /// Mira's gateway: plain `{role, content}` history, names stripped,
295 /// content-part arrays flattened.
296 Mira,
297 /// Perplexity: plain text history with strict user/assistant
298 /// alternation, text-only arrays flattened.
299 Perplexity,
300 /// Hyperbolic: tool-exchange remnants stripped, content-part arrays kept
301 /// (its vision models need them).
302 Hyperbolic,
303 /// Mistral: `any` for a forced tool choice, the choice relaxed to `auto`
304 /// beside a structured response format, `prefix` on assistant turns and
305 /// `reasoning_content` removed.
306 Mistral,
307 /// llama.cpp: refuse a specific-function tool choice, which
308 /// `llama-server` silently treats as `auto`.
309 LlamaCpp,
310 /// Moonshot: refuse a specific-function tool choice and coerce
311 /// `required` to `auto` with a steering message.
312 Moonshot,
313 /// OpenRouter: ephemeral `cache_control` on the system prompt when
314 /// prompt caching is on, and `reasoning_content` respelled `reasoning`.
315 OpenRouter,
316}
317
318/// Reranking endpoint policy. An empty [`Self::path`] disables reranking.
319#[derive(Clone, Copy, Debug, PartialEq, Eq)]
320#[non_exhaustive]
321pub struct RerankQuirks {
322 /// The rerank path, or empty when the dialect offers none.
323 pub path: &'static str,
324 /// Most documents the provider accepts in one request.
325 pub max_documents: usize,
326 /// Whether the model is a body field.
327 pub sends_model_field: bool,
328}
329
330impl RerankQuirks {
331 /// The signal for a dialect with no reranking endpoint.
332 pub const fn unsupported() -> Self {
333 Self {
334 path: "",
335 max_documents: 0,
336 sends_model_field: true,
337 }
338 }
339}
340
341/// What a dialect's embeddings endpoint accepts.
342#[derive(Clone, Copy, Debug, PartialEq, Eq)]
343#[non_exhaustive]
344pub struct EmbeddingQuirks {
345 /// Most inputs the provider embeds in one request.
346 pub max_documents: usize,
347 /// Whether a successful reply must carry usage.
348 pub requires_usage: bool,
349 /// Whether the provider accepts `encoding_format`.
350 pub supports_encoding_format: bool,
351 /// Whether the provider accepts `user`.
352 pub supports_user: bool,
353 /// Whether the model is a body field (false for Azure, which addresses a
354 /// deployment through the URL).
355 pub sends_model_field: bool,
356 /// Which field a requested width goes in.
357 pub dimensions: DimensionsField,
358 /// Model width contracts used for capability reporting and request validation.
359 /// Consulted before the shared OpenAI model-width table.
360 pub widths: &'static [ModelWidth],
361 /// The `requirement` clause refusing a declared width of zero, or
362 /// `None` for a dialect that lets zero through as rig's own "unknown"
363 /// sentinel rather than a claim.
364 pub refuse_zero_width: Option<&'static str>,
365}
366
367impl EmbeddingQuirks {
368 /// OpenAI's own embeddings contract, which most dialects inherit.
369 pub const fn openai() -> Self {
370 Self {
371 max_documents: 1024,
372 requires_usage: true,
373 supports_encoding_format: true,
374 supports_user: true,
375 sends_model_field: true,
376 dimensions: DimensionsField::Dimensions,
377 // Shared OpenAI model widths are resolved separately.
378 widths: &[],
379 // Zero represents an unknown width, not a requested dimension.
380 refuse_zero_width: None,
381 }
382 }
383}
384
385/// The caller identity a gateway requires on every request.
386#[derive(Debug, Clone, Copy, PartialEq, Eq)]
387pub struct Identity {
388 /// The `originator` header's default value.
389 pub originator: &'static str,
390 /// The environment variable overriding `originator`.
391 pub originator_env: &'static str,
392 /// The environment variable overriding `user-agent`.
393 pub user_agent_env: &'static str,
394 /// Whether every request carries a fresh `session_id` header.
395 pub session_ids: bool,
396}
397
398/// The identity a gateway requires on every request, resolved.
399#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
400#[serde(deny_unknown_fields)]
401pub struct CallerIdentity {
402 /// The `originator` header.
403 pub originator: String,
404 /// The `user-agent` header.
405 pub user_agent: String,
406}
407
408/// The user agent a gateway that asks for one is told: the crate, the host,
409/// and who is calling.
410fn default_user_agent(originator: &str) -> String {
411 format!(
412 "rig/{} ({} {}; {originator})",
413 env!("CARGO_PKG_VERSION"),
414 std::env::consts::OS,
415 std::env::consts::ARCH,
416 )
417}
418
419/// Request restrictions and response handling for a Responses dialect.
420#[derive(Debug, Clone, Copy, PartialEq, Eq)]
421pub enum ResponsesContract {
422 /// Standard Responses behavior with independently configured instruction placement.
423 OpenAi,
424 /// xAI's `/v1/responses`: it answers a success with its error envelope
425 /// as the whole body and publishes a finished tool call at
426 /// `output_item.done`, and its native structured output does not
427 /// compose with tool calls. The stream's own `error` event is not this:
428 /// that is protocol on every dialect and the decoder always reads it.
429 Xai,
430 /// Always-streamed Codex responses with optional content-type and envelope fields.
431 /// Requests omit sampling controls, storage, metadata, and structured output.
432 Codex,
433}
434
435/// What a dialect's Responses endpoint is: where it lives, where the system
436/// preamble goes, and which contract it speaks.
437#[derive(Debug, Clone, Copy, PartialEq, Eq)]
438#[non_exhaustive]
439pub struct ResponsesQuirks {
440 /// The endpoint path, appended to the base URL.
441 pub path: &'static str,
442 /// Where Rig's system instructions go in the request. Not part of
443 /// [`Self::contract`]: OpenAI's contract is served with all three
444 /// placements (OpenAI's own `instructions`, Copilot's and OpenRouter's
445 /// `system` items in `input`).
446 pub system_instructions: SystemInstructionsPlacement,
447 /// Which contract this dialect's endpoint speaks.
448 pub contract: ResponsesContract,
449 /// Whether newly constructed Responses wires normalize tools for strict validation.
450 pub strict_tools_by_default: bool,
451}
452
453impl ResponsesQuirks {
454 /// OpenAI's own Responses contract.
455 pub const fn openai() -> Self {
456 Self {
457 path: "/responses",
458 system_instructions: SystemInstructionsPlacement::Instructions,
459 contract: ResponsesContract::OpenAi,
460 strict_tools_by_default: false,
461 }
462 }
463}
464
465/// Optional executable extensions to the shared OpenAI dialect.
466///
467/// Store this value in a `static`: equality means the same extension definition,
468/// not equality of function addresses (which code generation can merge or duplicate).
469/// This lets named dialect persistence reject replaced hooks without interpreting
470/// their behavior or pretending arbitrary callbacks can be serialized.
471#[derive(Debug)]
472pub struct DialectHooks {
473 /// Derive a default endpoint from a credential. `None` uses the dialect's
474 /// static URL. Called only at construction, never on credential replacement.
475 pub default_endpoint: Option<fn(&str) -> Option<String>>,
476 /// Select the default route for a model, unless configuration chose a route.
477 pub model_route: Option<fn(&str) -> Route>,
478 /// Apply the completion envelope after shared authentication and identity.
479 /// Called once by either completion encoder; builder errors remain attached
480 /// and are returned when the encoder finishes the request.
481 pub completion_envelope: Option<CompletionEnvelope>,
482 /// Stamp the envelope every modality request (embeddings, listing,
483 /// verification, transcription, images, speech) carries, on the finished
484 /// request. Runs after shared authentication, so a hook may replace the
485 /// credential header rather than add a second one.
486 pub modality_envelope: Option<ModalityEnvelope>,
487}
488
489/// A dialect's modality-request headers, applied to the built request.
490pub type ModalityEnvelope =
491 fn(&OpenAIConfig, &mut http::Request<crate::wire::Body>) -> Result<(), http::Error>;
492
493/// A dialect's completion headers, applied to the authenticated request builder.
494pub type CompletionEnvelope = fn(
495 &OpenAIConfig,
496 &crate::completion::CompletionRequest,
497 http::request::Builder,
498) -> http::request::Builder;
499
500impl PartialEq for DialectHooks {
501 fn eq(&self, other: &Self) -> bool {
502 std::ptr::eq(self, other)
503 }
504}
505
506impl Eq for DialectHooks {}
507
508/// Everything about a dialect that is not its identity: paths, capability
509/// flags, and the one body rewrite it needs.
510///
511/// `#[non_exhaustive]` because the constants live in this crate and a new
512/// quirk must not be a breaking change for a host that stored a wire.
513#[derive(Clone, Copy, Debug, PartialEq, Eq)]
514#[non_exhaustive]
515pub struct Quirks {
516 /// Provider-owned extensions for defaults and completion headers.
517 pub hooks: Option<&'static DialectHooks>,
518 /// How the dialect authenticates.
519 pub auth: Auth,
520 /// How the dialect addresses a model.
521 pub routing: Routing,
522 /// Which completion endpoint
523 /// [`OpenAI::completion`](crate::providers::openai::OpenAI::completion) builds: the
524 /// dialect's flagship. Chat Completions is the one endpoint every
525 /// dialect serves, so it is the baseline; OpenAI itself, xAI and ChatGPT
526 /// serve `/responses` as their primary API and say so.
527 pub completion_route: Route,
528 /// The chat-completions path, relative to the base URL.
529 pub completion_path: &'static str,
530 /// The embeddings path.
531 pub embeddings_path: &'static str,
532 /// The model-listing path.
533 pub models_path: &'static str,
534 /// The path a credential check hits. Empty means the dialect offers no
535 /// check that does not consume tokens (Azure, Perplexity, Copilot).
536 pub verify_path: &'static str,
537 /// The transcription path.
538 pub transcription_path: &'static str,
539 /// The image-generation path.
540 pub image_generation_path: &'static str,
541 /// The speech path.
542 pub audio_generation_path: &'static str,
543 /// Whether `tools`/`tool_choice` reach the provider at all.
544 pub supports_tools: bool,
545 /// Whether `output_schema` maps to `response_format`.
546 pub supports_response_format: bool,
547 /// Whether to send `response_format` with tools before any tool result.
548 /// When false, defer the format until a tool result exists to avoid suppressing calls.
549 pub response_format_with_tools: bool,
550 /// Whether this server honours an image inside a `role:"tool"` message.
551 pub supports_image_tool_results: bool,
552 /// Whether a streaming request asks for the usage chunk through
553 /// `stream_options`.
554 pub stream_include_usage: bool,
555 /// Whether the backend can emit a whole tool call in one chunk.
556 pub emits_complete_single_chunk_tool_calls: bool,
557 /// How the dialect spells the output-token cap.
558 pub output_cap: OutputCap,
559 /// Whether to consult upstream-native finish reasons when normalized ones are absent.
560 pub native_finish_reason: bool,
561 /// Whether the dialect emits `reasoning_details` entries (OpenRouter's
562 /// encrypted reasoning blobs and replay signatures).
563 pub reasoning_details: bool,
564 /// Whether reasoning belongs to the upstream model's family rather than
565 /// to this dialect: a gateway relays each upstream's own reasoning state,
566 /// valid only for that family ([`upstream_reasoning_issuer`]).
567 pub upstream_reasoning_issuer: bool,
568 /// Whether `completion_tokens_details.reasoning_tokens` can be trusted as
569 /// a part of `completion_tokens`, as OpenAI documents it. A dialect whose
570 /// replies report more reasoning than completion leaves the count
571 /// unreported, so [`Usage`](crate::completion::Usage) never reports more
572 /// reasoning than output.
573 pub reliable_reasoning_count: bool,
574 /// Whether a bare JSON string is accepted as a text-only completion reply.
575 pub accepts_bare_string_reply: bool,
576 /// Whether document and file inputs may use provider file IDs.
577 pub accepts_file_ids: bool,
578 /// The rewrite this dialect applies to the serialized chat body.
579 pub rewrite: BodyRewrite,
580 /// Paths that strip a trailing `/v1` from the configured base URL.
581 pub root_relative_routes: &'static [&'static str],
582 /// Whether the model is the modality endpoint's *path* rather than a
583 /// body field. Hugging Face's router addresses transcription and image
584 /// generation as `/{model}`; everyone else uses a fixed path.
585 pub model_is_modality_path: bool,
586 /// Which body the image endpoint takes.
587 pub image_body: ImageBody,
588 /// Which body the transcription endpoint takes.
589 pub transcription_body: TranscriptionBody,
590 /// Which body the speech endpoint takes.
591 pub speech_body: SpeechBody,
592 /// What the embeddings endpoint accepts.
593 pub embedding: EmbeddingQuirks,
594 /// What the rerank endpoint accepts.
595 pub rerank: RerankQuirks,
596 /// A second environment variable naming the base URL, kept because the
597 /// provider documents both spellings.
598 pub base_url_env_alias: Option<&'static str>,
599 /// The environment variable naming the account a credential belongs to,
600 /// sent as `ChatGPT-Account-Id`.
601 pub account_id_env: Option<&'static str>,
602 /// Instructions this gateway expects every turn to carry, merged ahead
603 /// of the caller's preamble.
604 pub default_instructions: Option<&'static str>,
605 /// The environment variable overriding [`Self::default_instructions`].
606 pub instructions_env: Option<&'static str>,
607 /// The caller identity this gateway requires on every request.
608 pub identity: Option<Identity>,
609 /// What the Responses endpoint accepts.
610 pub responses: ResponsesQuirks,
611}
612
613impl Quirks {
614 /// Baseline compatible endpoint policies with Chat Completions routing and
615 /// the legacy `max_tokens` cap. Dialects override supported differences.
616 pub const fn openai() -> Self {
617 Self {
618 hooks: None,
619 auth: Auth::Bearer,
620 routing: Routing::Path,
621 completion_route: Route::Chat,
622 completion_path: "/chat/completions",
623 embeddings_path: "/embeddings",
624 models_path: "/models",
625 verify_path: "/models",
626 transcription_path: "/audio/transcriptions",
627 image_generation_path: "/images/generations",
628 audio_generation_path: "/audio/speech",
629 supports_tools: true,
630 supports_response_format: true,
631 response_format_with_tools: false,
632 supports_image_tool_results: false,
633 stream_include_usage: true,
634 emits_complete_single_chunk_tool_calls: false,
635 output_cap: OutputCap::Legacy,
636 native_finish_reason: false,
637 reasoning_details: false,
638 upstream_reasoning_issuer: false,
639 reliable_reasoning_count: true,
640 accepts_bare_string_reply: false,
641 accepts_file_ids: true,
642 rewrite: BodyRewrite::None,
643 embedding: EmbeddingQuirks::openai(),
644 // OpenAI has no reranking endpoint, and neither does any dialect
645 // on this wire but llama.cpp.
646 rerank: RerankQuirks::unsupported(),
647 root_relative_routes: &[],
648 model_is_modality_path: false,
649 image_body: ImageBody::OpenAi,
650 speech_body: SpeechBody::OpenAi,
651 transcription_body: TranscriptionBody::Multipart,
652 base_url_env_alias: None,
653 account_id_env: None,
654 default_instructions: None,
655 instructions_env: None,
656 identity: None,
657 responses: ResponsesQuirks::openai(),
658 }
659 }
660
661 /// These quirks without the streamed usage chunk: a streaming request
662 /// sends no `stream_options`, so streamed usage reports `None`.
663 pub const fn without_stream_usage(mut self) -> Self {
664 self.stream_include_usage = false;
665 self
666 }
667
668 /// These quirks without structured output: `output_schema` does not map
669 /// to `response_format`.
670 pub const fn without_response_format(mut self) -> Self {
671 self.supports_response_format = false;
672 self
673 }
674}
675
676/// Provider identity, endpoint defaults, and shared-wire policies.
677#[derive(Clone, Copy, Debug, PartialEq, Eq)]
678pub struct Dialect {
679 /// The provider descriptor name, as records and telemetry name it.
680 pub name: &'static str,
681 /// The default base URL.
682 pub base_url: &'static str,
683 /// The environment variable holding the credential.
684 pub api_key_env: &'static str,
685 /// The environment variable overriding the base URL, when the provider
686 /// has one.
687 pub base_url_env: Option<&'static str>,
688 /// The reply header carrying the provider's transport request id.
689 pub request_id_header: Option<&'static str>,
690 /// A second credential this dialect accepts, with its own variable and
691 /// header. `None` for every dialect but Azure.
692 pub alternate_auth: Option<AuthAlternative>,
693 /// Everything that is not identity.
694 pub quirks: Quirks,
695}
696
697impl Dialect {
698 /// Create a dialect with [`Quirks::openai`] and the supplied identity.
699 /// URL overrides, request-ID headers, and alternative credentials are unset.
700 pub const fn gateway(
701 name: &'static str,
702 base_url: &'static str,
703 api_key_env: &'static str,
704 ) -> Self {
705 Self {
706 name,
707 base_url,
708 api_key_env,
709 base_url_env: None,
710 request_id_header: None,
711 alternate_auth: None,
712 quirks: Quirks::openai(),
713 }
714 }
715
716 /// This dialect with `quirks`.
717 pub const fn with_quirks(mut self, quirks: Quirks) -> Self {
718 self.quirks = quirks;
719 self
720 }
721}
722
723/// Serialize the registered dialect name, rejecting unregistered or modified definitions.
724/// Deserialization resolves that name from this build's registry.
725impl Serialize for Dialect {
726 fn serialize<S: serde::Serializer>(&self, serializer: S) -> Result<S::Ok, S::Error> {
727 let registered = dialects::by_name(self.name) == Some(self);
728 crate::providers::internal::named_dialect::serialize(
729 serializer, "OpenAI", self.name, registered,
730 )
731 }
732}
733
734impl<'de> Deserialize<'de> for Dialect {
735 fn deserialize<D: serde::Deserializer<'de>>(deserializer: D) -> Result<Self, D::Error> {
736 crate::providers::internal::named_dialect::deserialize(deserializer, "OpenAI", |name| {
737 dialects::by_name(name).copied()
738 })
739 }
740}
741
742/// The settings of an OpenAI-shaped provider: serializable, and the
743/// credential is never serialized. [`connect`](Self::connect) puts it on a
744/// transport as an [`OpenAI`](super::OpenAI) client.
745#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
746#[serde(deny_unknown_fields)]
747pub struct OpenAIConfig {
748 /// The credential. Never serialized (see [`Secret`]).
749 pub api_key: Secret,
750 /// The base URL every path resolves against.
751 pub base_url: String,
752 /// Which OpenAI-shaped provider this is.
753 pub dialect: Dialect,
754 /// The completion endpoint this configuration uses when asked for "a
755 /// completion", when it differs from the dialect's flagship
756 /// ([`Quirks::completion_route`]). Set by [`with_route`](Self::with_route).
757 #[serde(default, skip_serializing_if = "Option::is_none")]
758 pub route: Option<Route>,
759 /// Azure's `api-version` query parameter, which every Azure route
760 /// requires. `None` for every other dialect.
761 #[serde(default, skip_serializing_if = "Option::is_none")]
762 pub api_version: Option<String>,
763 /// Azure versions its speech endpoint separately from the rest, so a
764 /// speech request carries this `api-version` instead of
765 /// [`Self::api_version`]. `None` falls back to `api_version`.
766 #[serde(default, skip_serializing_if = "Option::is_none")]
767 pub audio_api_version: Option<String>,
768 /// How this configuration's credential is sent. Taken from the dialect,
769 /// except when the credential came from the dialect's
770 /// [`alternate_auth`](Dialect::alternate_auth) variable, which has its
771 /// own header.
772 pub auth: Auth,
773 /// Which sub-provider the Hugging Face router forwards to. `None` behaves
774 /// as [`SubRoute::HFInference`], the router's own default. `None` for
775 /// every other dialect, which routes nothing.
776 #[serde(default, skip_serializing_if = "Option::is_none")]
777 pub sub_route: Option<SubRoute>,
778 /// The account the credential belongs to, when the gateway asks which
779 /// (`ChatGPT-Account-Id`).
780 #[serde(default, skip_serializing_if = "Option::is_none")]
781 pub account_id: Option<String>,
782 /// Instructions merged ahead of every Responses turn's preamble, when
783 /// the gateway expects some.
784 #[serde(default, skip_serializing_if = "Option::is_none")]
785 pub instructions: Option<String>,
786 /// The caller identity, when the gateway requires one.
787 #[serde(default, skip_serializing_if = "Option::is_none")]
788 pub identity: Option<CallerIdentity>,
789 /// Responses instruction placement override. `None` uses the dialect default.
790 #[serde(default, skip_serializing_if = "Option::is_none")]
791 pub system_instructions: Option<SystemInstructionsPlacement>,
792}
793
794impl OpenAIConfig {
795 /// Official OpenAI, with `api_key`.
796 pub fn new(api_key: impl Into<Secret>) -> Self {
797 Self::with_key(&OPENAI, api_key)
798 }
799
800 /// `dialect` with `api_key`, at the dialect's default base URL and with
801 /// the instructions and caller identity its gateway expects, if any.
802 pub fn with_key(dialect: &Dialect, api_key: impl Into<Secret>) -> Self {
803 let quirks = &dialect.quirks;
804 let api_key = api_key.into();
805 let base_url = quirks
806 .hooks
807 .and_then(|hooks| hooks.default_endpoint)
808 .and_then(|endpoint| endpoint(api_key.expose()))
809 .unwrap_or_else(|| dialect.base_url.to_owned());
810 Self {
811 api_key,
812 base_url,
813 dialect: *dialect,
814 route: None,
815 // Azure deployment URLs require an explicit API version.
816 api_version: match quirks.routing {
817 Routing::AzureDeployment => Some(dialects::AZURE_DEFAULT_API_VERSION.to_owned()),
818 Routing::Path => None,
819 },
820 audio_api_version: None,
821 auth: quirks.auth,
822 sub_route: None,
823 account_id: None,
824 instructions: quirks.default_instructions.map(str::to_owned),
825 identity: quirks.identity.map(|identity| CallerIdentity {
826 originator: identity.originator.to_owned(),
827 user_agent: default_user_agent(identity.originator),
828 }),
829 system_instructions: None,
830 }
831 }
832
833 /// `dialect` with the credential it accepts through its
834 /// [`alternate_auth`](Dialect::alternate_auth) variable, sent with that
835 /// alternative's header.
836 ///
837 /// Azure's account key and its Entra bearer token are both credentials
838 /// for the same account but go out under different headers, so which one
839 /// is held has to be recorded rather than guessed from the value.
840 pub fn with_alternate_key(dialect: &Dialect, api_key: impl Into<Secret>) -> Self {
841 let auth = dialect
842 .alternate_auth
843 .map_or(dialect.quirks.auth, |alternative| alternative.auth);
844 Self {
845 auth,
846 ..Self::with_key(dialect, api_key)
847 }
848 }
849
850 /// Read `OPENAI_API_KEY` and the optional `OPENAI_BASE_URL` override.
851 /// Return an environment error for missing credentials or invalid values.
852 pub fn from_env() -> Result<Self, EnvError> {
853 Self::from_env_with(&OPENAI)
854 }
855
856 /// `dialect` from its own `api_key_env` and `base_url_env` (or the
857 /// alias its quirks name), plus whatever else its gateway reads: the
858 /// account id, the default instructions and the caller identity.
859 ///
860 /// Azure additionally reads `AZURE_API_VERSION`, because every Azure
861 /// route carries it and there is no default that would not silently
862 /// address the wrong API.
863 pub fn from_env_with(dialect: &Dialect) -> Result<Self, EnvError> {
864 let (api_key, auth) = Self::credential_from_env(dialect)?;
865 Self::from_env_with_credential(dialect, api_key, auth)
866 }
867
868 /// The credential `dialect` reads from the environment, and how it is
869 /// sent. Alternative credentials require their own header policy; the
870 /// primary is preferred.
871 pub(crate) fn credential_from_env(dialect: &Dialect) -> Result<(String, Auth), EnvError> {
872 let quirks = &dialect.quirks;
873 Ok(match dialect.alternate_auth {
874 Some(alternative) => match env::optional(dialect.api_key_env)? {
875 Some(api_key) => (api_key, quirks.auth),
876 None => match env::optional(alternative.api_key_env)? {
877 Some(api_key) => (api_key, alternative.auth),
878 None => {
879 return Err(EnvError::Invalid {
880 name: dialect.api_key_env,
881 detail: format!(
882 "either `{}` or `{}` must be set",
883 dialect.api_key_env, alternative.api_key_env
884 ),
885 });
886 }
887 },
888 },
889 None => (env::required(dialect.api_key_env)?, quirks.auth),
890 })
891 }
892
893 /// [`Self::from_env_with`] with the credential already read.
894 pub(crate) fn from_env_with_credential(
895 dialect: &Dialect,
896 api_key: String,
897 auth: Auth,
898 ) -> Result<Self, EnvError> {
899 let quirks = &dialect.quirks;
900 let mut provider = Self::with_key(dialect, api_key);
901 provider.auth = auth;
902 for name in [dialect.base_url_env, quirks.base_url_env_alias]
903 .into_iter()
904 .flatten()
905 {
906 if let Some(base_url) = env::optional(name)? {
907 provider.base_url = base_url;
908 break;
909 }
910 }
911 // Azure speech uses an independently versioned endpoint.
912 if let Routing::AzureDeployment = quirks.routing {
913 provider.api_version = Some(env::required(dialects::AZURE_API_VERSION_ENV)?);
914 provider.audio_api_version = env::optional(dialects::AZURE_AUDIO_API_VERSION_ENV)?
915 .or_else(|| Some(dialects::AZURE_DEFAULT_AUDIO_API_VERSION.to_owned()));
916 }
917 if let Some(name) = quirks.account_id_env {
918 provider.account_id = env::optional(name)?;
919 }
920 if let Some(name) = quirks.instructions_env
921 && let Some(instructions) = env::optional(name)?
922 && !instructions.trim().is_empty()
923 {
924 provider.instructions = Some(instructions);
925 }
926 if let (Some(identity), Some(resolved)) = (quirks.identity, provider.identity.as_mut()) {
927 if let Some(originator) =
928 env::optional(identity.originator_env)?.filter(|value| !value.is_empty())
929 {
930 resolved.originator = originator;
931 resolved.user_agent = default_user_agent(&resolved.originator);
932 }
933 if let Some(user_agent) =
934 env::optional(identity.user_agent_env)?.filter(|value| !value.is_empty())
935 {
936 resolved.user_agent = user_agent;
937 }
938 }
939 Ok(provider)
940 }
941
942 /// Point this configuration at another dialect: the same credential,
943 /// with everything else at that dialect's defaults.
944 pub fn with_dialect(self, dialect: &Dialect) -> Self {
945 Self::with_key(dialect, self.api_key)
946 }
947
948 /// Route through a Hugging Face sub-provider.
949 pub fn with_sub_route(mut self, sub_route: SubRoute) -> Self {
950 self.sub_route = Some(sub_route);
951 self
952 }
953
954 /// Send the credential with `auth`'s header.
955 pub fn with_auth(mut self, auth: Auth) -> Self {
956 self.auth = auth;
957 self
958 }
959
960 /// Override the base URL.
961 pub fn with_base_url(mut self, base_url: impl Into<String>) -> Self {
962 self.base_url = base_url.into();
963 self
964 }
965
966 /// Set Azure's `api-version`.
967 pub fn with_api_version(mut self, api_version: impl Into<String>) -> Self {
968 self.api_version = Some(api_version.into());
969 self
970 }
971
972 /// Set the `api-version` Azure's speech endpoint is versioned by.
973 pub fn with_audio_api_version(mut self, api_version: impl Into<String>) -> Self {
974 self.audio_api_version = Some(api_version.into());
975 self
976 }
977
978 /// Name the account the credential belongs to (`ChatGPT-Account-Id`).
979 pub fn with_account_id(mut self, account_id: impl Into<String>) -> Self {
980 self.account_id = Some(account_id.into());
981 self
982 }
983
984 /// Merge these instructions ahead of every Responses turn's preamble.
985 pub fn with_instructions(mut self, instructions: impl Into<String>) -> Self {
986 self.instructions = Some(instructions.into());
987 self
988 }
989
990 /// Put Rig's system instructions somewhere other than the dialect's
991 /// default placement, for every Responses wire this configuration
992 /// builds.
993 pub fn with_system_instructions_placement(
994 mut self,
995 placement: SystemInstructionsPlacement,
996 ) -> Self {
997 self.system_instructions = Some(placement);
998 self
999 }
1000
1001 /// Send Rig's system instructions as `system` messages in `input`, for a
1002 /// backend that rejects or ignores top-level `instructions`.
1003 pub fn with_system_instructions_as_messages(self) -> Self {
1004 self.with_system_instructions_placement(SystemInstructionsPlacement::InputSystemMessages)
1005 }
1006
1007 /// Where a Responses wire built from this configuration puts Rig's
1008 /// system instructions: the dialect's placement unless
1009 /// [`with_system_instructions_placement`](Self::with_system_instructions_placement)
1010 /// chose another.
1011 pub fn system_instructions_placement(&self) -> SystemInstructionsPlacement {
1012 self.system_instructions
1013 .unwrap_or(self.dialect.quirks.responses.system_instructions)
1014 }
1015
1016 /// Override dialect and model-specific routing for the client's
1017 /// [`completion`](crate::providers::openai::OpenAI::completion).
1018 pub fn with_route(mut self, route: Route) -> Self {
1019 self.route = Some(route);
1020 self
1021 }
1022
1023 /// The configured route or dialect's static default. A model-route hook
1024 /// may refine the default when a completion wire is constructed.
1025 pub fn completion_route(&self) -> Route {
1026 self.route.unwrap_or(self.dialect.quirks.completion_route)
1027 }
1028
1029 /// The completion wire for `model` on this configuration's
1030 /// [`completion_route`](Self::completion_route): Responses for OpenAI,
1031 /// xAI and ChatGPT, model-dependent routing when a dialect supplies it, and
1032 /// Chat Completions for other compatible gateways, unless
1033 /// [`with_route`](Self::with_route) chose the other one.
1034 pub(crate) fn completion(&self, model: impl Into<String>) -> OpenAiWire {
1035 OpenAiWire::new(self.clone(), model)
1036 }
1037
1038 /// The Responses wire for `model`: `POST /responses`.
1039 pub(crate) fn responses(&self, model: impl Into<String>) -> Responses {
1040 Responses::new(self.clone(), model)
1041 }
1042
1043 /// The chat-completions wire for `model`, whatever the dialect's
1044 /// [`completion_route`](Quirks::completion_route).
1045 pub fn chat(&self, model: impl Into<String>) -> Chat {
1046 Chat::new(self.clone(), model)
1047 }
1048
1049 /// The embeddings wire for `model`.
1050 pub(crate) fn embedding(&self, model: impl Into<String>, ndims: Option<usize>) -> Embeddings {
1051 Embeddings::new(self.clone(), model, ndims)
1052 }
1053
1054 /// The rerank wire for `model`.
1055 pub(crate) fn rerank(&self, model: impl Into<String>) -> Rerank {
1056 Rerank::new(self.clone(), model)
1057 }
1058
1059 /// The transcription wire for `model`.
1060 pub(crate) fn transcription(&self, model: impl Into<String>) -> Transcriptions {
1061 Transcriptions::new(self.clone(), model)
1062 }
1063
1064 /// The model-listing wire.
1065 pub(crate) fn models(&self) -> Models {
1066 Models::new(self.clone())
1067 }
1068
1069 /// The credential-check wire.
1070 pub(crate) fn verify(&self) -> Verify {
1071 Verify::new(self.clone())
1072 }
1073
1074 /// The image-generation wire for `model`.
1075 #[cfg(feature = "image")]
1076 pub(crate) fn image_generation(&self, model: impl Into<String>) -> Images {
1077 Images::new(self.clone(), model)
1078 }
1079
1080 /// The speech wire for `model`.
1081 #[cfg(feature = "audio")]
1082 pub(crate) fn audio_generation(&self, model: impl Into<String>) -> Speech {
1083 Speech::new(self.clone(), model)
1084 }
1085
1086 pub(crate) fn completion_headers(
1087 &self,
1088 request: &crate::completion::CompletionRequest,
1089 builder: http::request::Builder,
1090 ) -> http::request::Builder {
1091 let builder = self.headers(builder);
1092 match self
1093 .dialect
1094 .quirks
1095 .hooks
1096 .and_then(|hooks| hooks.completion_envelope)
1097 {
1098 Some(envelope) => envelope(self, request, builder),
1099 None => builder,
1100 }
1101 }
1102
1103 /// Resolve `path` against the base URL, applying Azure's
1104 /// deployment-in-URL routing when the dialect uses it.
1105 pub(crate) fn uri(&self, path: &str, model: Option<&str>) -> String {
1106 self.uri_versioned(path, model, self.api_version.as_deref())
1107 }
1108
1109 /// [`Self::uri`] with an explicit `api-version`, for the one endpoint
1110 /// Azure versions separately (speech).
1111 pub(crate) fn uri_versioned(
1112 &self,
1113 path: &str,
1114 model: Option<&str>,
1115 api_version: Option<&str>,
1116 ) -> String {
1117 match (self.dialect.quirks.routing, model) {
1118 (Routing::AzureDeployment, Some(model)) => format!(
1119 "{}/openai/deployments/{}{}?api-version={}",
1120 self.base_url.trim_end_matches('/'),
1121 model.trim_start_matches('/'),
1122 path,
1123 api_version.unwrap_or_default(),
1124 ),
1125 _ => format!("{}{}", self.base(path), path),
1126 }
1127 }
1128
1129 /// The base URL `path` resolves against: the configured one, with the
1130 /// version segment dropped for a route the dialect serves at the root.
1131 fn base(&self, path: &str) -> &str {
1132 let base = self.base_url.trim_end_matches('/');
1133 if self.dialect.quirks.root_relative_routes.contains(&path) {
1134 return base.strip_suffix("/v1").unwrap_or(base);
1135 }
1136 base
1137 }
1138
1139 /// The `api-version` a speech request carries.
1140 #[cfg(feature = "audio")]
1141 pub(crate) fn speech_api_version(&self) -> Option<&str> {
1142 self.audio_api_version
1143 .as_deref()
1144 .or(self.api_version.as_deref())
1145 }
1146
1147 /// The sub-provider the Hugging Face router forwards to. `None` on the
1148 /// configuration means the router's own default.
1149 pub(crate) fn route(&self) -> std::borrow::Cow<'_, SubRoute> {
1150 match &self.sub_route {
1151 Some(route) => std::borrow::Cow::Borrowed(route),
1152 None => std::borrow::Cow::Owned(SubRoute::default()),
1153 }
1154 }
1155
1156 /// Return `model` for Azure deployment routing, otherwise `None`.
1157 pub(crate) fn deployment<'a>(&self, model: &'a str) -> Option<&'a str> {
1158 match self.dialect.quirks.routing {
1159 Routing::AzureDeployment => Some(model),
1160 Routing::Path => None,
1161 }
1162 }
1163
1164 /// Resolve a fixed or model-addressed modality URL.
1165 /// Return an error if the selected sub-route does not serve model-routed endpoints.
1166 pub(crate) fn modality_uri(
1167 &self,
1168 endpoint: &str,
1169 fixed: &'static str,
1170 model: &str,
1171 ) -> Result<String, String> {
1172 if !self.dialect.quirks.model_is_modality_path {
1173 return Ok(self.uri(fixed, self.deployment(model)));
1174 }
1175 let route = self.route();
1176 if !route.serves_model_routed_endpoints() {
1177 return Err(format!(
1178 "{endpoint} endpoint is not supported yet for {route}"
1179 ));
1180 }
1181 Ok(format!(
1182 "{}/{}",
1183 self.base_url.trim_end_matches('/'),
1184 model.trim_start_matches('/')
1185 ))
1186 }
1187
1188 /// Apply the dialect's authentication to a request builder.
1189 pub(crate) fn authenticate(&self, builder: http::request::Builder) -> http::request::Builder {
1190 match self.auth {
1191 Auth::Bearer => {
1192 builder.header("Authorization", format!("Bearer {}", self.api_key.expose()))
1193 }
1194 Auth::OptionalBearer if self.api_key.is_empty() => builder,
1195 Auth::OptionalBearer => {
1196 builder.header("Authorization", format!("Bearer {}", self.api_key.expose()))
1197 }
1198 Auth::ApiKeyHeader => builder.header("api-key", self.api_key.expose()),
1199 }
1200 }
1201
1202 /// Apply authentication, configured identity, account, and per-request session headers.
1203 pub(crate) fn headers(&self, builder: http::request::Builder) -> http::request::Builder {
1204 let mut builder = self.authenticate(builder);
1205 if let Some(identity) = &self.identity {
1206 builder = builder
1207 .header("originator", &identity.originator)
1208 .header(http::header::USER_AGENT, &identity.user_agent);
1209 }
1210 if self
1211 .dialect
1212 .quirks
1213 .identity
1214 .is_some_and(|identity| identity.session_ids)
1215 {
1216 // Session identity must be fresh for each request.
1217 builder = builder.header("session_id", crate::providers::chatgpt::session_id());
1218 }
1219 if let Some(account_id) = &self.account_id {
1220 builder = builder.header("ChatGPT-Account-Id", account_id);
1221 }
1222 builder
1223 }
1224}
1225
1226/// The issuer of reasoning a gateway relays from `model` (`vendor/name`):
1227/// `anthropic` for Claude, whose thinking signatures Anthropic documents as
1228/// valid across its platforms and which verified as valid between OpenRouter
1229/// and the Claude API in both directions; `<gateway>/<vendor>` for every
1230/// other family, whose signatures and ciphertext are bound to the gateway's
1231/// upstream account.
1232pub fn upstream_reasoning_issuer(gateway: &str, model: &str) -> String {
1233 let vendor = model_vendor(model);
1234 if vendor == "anthropic" {
1235 vendor.to_owned()
1236 } else {
1237 format!("{gateway}/{vendor}")
1238 }
1239}
1240
1241fn model_vendor(model: &str) -> &str {
1242 let vendor = model.split_once('/').map_or(model, |(vendor, _)| vendor);
1243 vendor.trim_start_matches('~')
1244}
1245
1246/// The issuers whose reasoning a request to its model over `dialect`
1247/// replays, and the request as they read it
1248/// ([`CompletionRequest::replayable_to`](crate::completion::CompletionRequest::replayable_to)).
1249/// The request's model override, when it names one, is the model replayed
1250/// for; `model` otherwise.
1251pub(crate) fn scope_reasoning(
1252 dialect: &Dialect,
1253 model: &str,
1254 request: crate::completion::CompletionRequest,
1255) -> Result<(crate::completion::CompletionRequest, Vec<Issuer>), EncodeError> {
1256 let issuers: Vec<Issuer> = replay_issuers(dialect, request.model.as_deref().unwrap_or(model))
1257 .into_iter()
1258 .map(Issuer::from)
1259 .collect();
1260 Ok((request.replayable_to(&issuers)?, issuers))
1261}
1262
1263/// The reasoning issuers a request to `model` over `dialect` replays.
1264///
1265/// A dialect without upstream issuers replays its own reasoning. A gateway
1266/// replays the requested model's family, plus reasoning stamped with the
1267/// gateway alone, which predates upstream issuers. A model under the
1268/// gateway's own vendor (`openrouter/auto`) or a preset (`@preset/name`)
1269/// names no family, so the turn may be served by any of them: it replays
1270/// every family the gateway relayed and Claude reasoning from any surface,
1271/// rather than dropping reasoning a Claude tool loop must send back. An
1272/// unrecognised vendor is its own family.
1273pub(crate) fn replay_issuers(dialect: &Dialect, model: &str) -> Vec<String> {
1274 let gateway = dialect.name;
1275 if !dialect.quirks.upstream_reasoning_issuer {
1276 return vec![gateway.to_owned()];
1277 }
1278 if model.starts_with('@') || model_vendor(model) == gateway {
1279 return vec![
1280 "anthropic".to_owned(),
1281 format!("{gateway}/"),
1282 gateway.to_owned(),
1283 ];
1284 }
1285 vec![
1286 upstream_reasoning_issuer(gateway, model),
1287 gateway.to_owned(),
1288 ]
1289}
1290
1291#[cfg(test)]
1292mod tests;