rig_core/providers/openai/wire.rs
1//! OpenAI-compatible configurations, dialect policies, and endpoint wires.
2//! A [`Dialect`](crate::providers::openai::wire::Dialect) selects request
3//! and response policies; [`OpenAIConfig`](crate::providers::openai::OpenAIConfig)
4//! holds credentials and overrides. An
5//! [`OpenAI`](crate::providers::openai::OpenAI) client puts
6//! the configuration on a transport and builds each endpoint's model.
7//!
8//! ```
9//! use rig_core::providers::openai::{OpenAIConfig, Route, wire::OpenAiWire};
10//!
11//! let openai = OpenAIConfig::new("key").with_route(Route::Chat).client();
12//! assert!(matches!(openai.completion("gpt-5.2").wire, OpenAiWire::Chat(_)));
13//! ```
14
15use serde::{Deserialize, Serialize};
16
17use crate::client::env::{self, EnvError};
18use crate::wire::Secret;
19
20use super::responses_api::SystemInstructionsPlacement;
21use super::responses_api::wire::Responses;
22
23mod auth;
24pub(crate) mod chat;
25mod dialects;
26/// The merge that assembles a streamed provider object from its fragments.
27pub(crate) mod dto;
28mod modality;
29mod route;
30
31use auth::default_user_agent;
32pub use auth::{Auth, AuthAlternative, CallerIdentity, Identity};
33pub use chat::Chat;
34pub use dialects::*;
35pub use modality::{
36 AcceptedWidths, DimensionsField, EmbeddingQuirks, Embeddings, EmbeddingsDecoder, ImageBody,
37 ModelEntry, ModelWidth, Models, ModelsDecoder, ModelsReply, Rerank, RerankDecoder,
38 RerankQuirks, RerankReply, RerankResultEntry, RerankUsage, SpeechBody, TranscriptionBody,
39 Transcriptions, TranscriptionsDecoder, Verify, VerifyDecoder,
40};
41pub use route::{OpenAiDecoder, OpenAiEvent, OpenAiReassembler, OpenAiWire, Route};
42
43#[cfg(feature = "image")]
44pub use modality::{ImageDatum, Images, ImagesDecoder, ImagesEvent, ImagesReply};
45#[cfg(feature = "audio")]
46pub use modality::{Speech, SpeechDecoder};
47
48/// Backend selected through the Hugging Face router.
49#[derive(Clone, Debug, Default, PartialEq, Eq, Serialize, Deserialize)]
50pub enum SubRoute {
51 /// Hugging Face's own inference backend: the only one that serves
52 /// transcription and image generation.
53 #[default]
54 HFInference,
55 /// Together AI, through the router.
56 Together,
57 /// SambaNova, through the router.
58 SambaNova,
59 /// Fireworks AI, which addresses models by a qualified id.
60 Fireworks,
61 /// Hyperbolic, through the router.
62 Hyperbolic,
63 /// Nebius, through the router.
64 Nebius,
65 /// Novita, through the router.
66 Novita,
67 /// A route this build does not name.
68 Custom(String),
69}
70
71impl SubRoute {
72 /// The router's slug for this sub-provider.
73 pub fn slug(&self) -> &str {
74 match self {
75 Self::HFInference => "hf-inference/models",
76 Self::Together => "together",
77 Self::SambaNova => "sambanova",
78 Self::Fireworks => "fireworks-ai",
79 Self::Hyperbolic => "hyperbolic",
80 Self::Nebius => "nebius",
81 Self::Novita => "novita",
82 Self::Custom(route) => route,
83 }
84 }
85
86 /// Qualify Fireworks model identifiers unless already prefixed.
87 /// Return other sub-routes' identifiers unchanged.
88 pub fn model_identifier(&self, model: &str) -> String {
89 const FIREWORKS_PREFIX: &str = "accounts/fireworks/models/";
90 match self {
91 Self::Fireworks if !model.starts_with(FIREWORKS_PREFIX) => {
92 format!("{FIREWORKS_PREFIX}{model}")
93 }
94 _ => model.to_owned(),
95 }
96 }
97}
98
99impl std::fmt::Display for SubRoute {
100 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
101 f.write_str(self.slug())
102 }
103}
104
105impl From<&str> for SubRoute {
106 fn from(route: &str) -> Self {
107 Self::Custom(route.to_owned())
108 }
109}
110
111impl From<String> for SubRoute {
112 fn from(route: String) -> Self {
113 Self::Custom(route)
114 }
115}
116
117/// How a dialect addresses a model.
118#[derive(Clone, Copy, Debug, PartialEq, Eq)]
119pub enum Routing {
120 /// Resolve the endpoint under the base URL and send the model in the body.
121 Path,
122 /// Azure: the model is a *deployment* in the URL
123 /// (`{base}/openai/deployments/{model}/chat/completions?api-version=…`)
124 /// and the body carries no `model` field.
125 AzureDeployment,
126}
127
128/// How a dialect spells the output-token cap.
129#[derive(Clone, Copy, Debug, PartialEq, Eq)]
130pub enum OutputCap {
131 /// `max_tokens`, for every endpoint not observed to reject it.
132 Legacy,
133 /// `max_completion_tokens` for OpenAI's reasoning families, which answer
134 /// a `max_tokens` request with `Unsupported parameter`. Scoped to the
135 /// model, because this same wire reaches compatible servers that know
136 /// only the legacy field.
137 OpenAiReasoningFamilies,
138}
139
140/// Dialect-specific transformation of the serialized chat request.
141#[derive(Clone, Copy, Debug, PartialEq, Eq)]
142#[non_exhaustive]
143pub enum BodyRewrite {
144 /// Send the OpenAI-compatible body unchanged.
145 None,
146 /// Hugging Face's router: qualify the model identifier for sub-providers
147 /// that demand one (Fireworks).
148 HuggingFaceRouter,
149 /// DeepSeek: forced tool choices dropped while the model thinks. Its
150 /// reasoning field is [`Quirks::reasoning_field`].
151 DeepSeek,
152 /// Mira's gateway: content-part arrays flattened to strings.
153 Mira,
154 /// Perplexity: text-only content arrays sent as one string.
155 Perplexity,
156 /// Mistral: the forced tool choice relaxed to `auto` beside a structured
157 /// response format, and its own content chunks.
158 Mistral,
159 /// llama.cpp: refuse a specific-function tool choice, which
160 /// `llama-server` silently treats as `auto`.
161 LlamaCpp,
162 /// Moonshot: refuse a specific-function tool choice and coerce
163 /// `required` to `auto` with a steering message.
164 Moonshot,
165 /// OpenRouter: model ids name the upstream vendor (`anthropic/...`),
166 /// which decides what a request carries.
167 OpenRouter,
168 /// Ollama's OpenAI-compatible API: a refusal of `num_ctx` and
169 /// `options`, which only the native route can send.
170 Ollama,
171}
172
173/// Request restrictions and response handling for a Responses dialect.
174#[derive(Debug, Clone, Copy, PartialEq, Eq)]
175pub enum ResponsesContract {
176 /// Standard Responses behavior with independently configured instruction placement.
177 OpenAi,
178 /// xAI's `/v1/responses`: it answers a success with its error envelope
179 /// as the whole body and publishes a finished tool call at
180 /// `output_item.done`, and its native structured output does not
181 /// compose with tool calls. The stream's own `error` event is not this:
182 /// that is protocol on every dialect and the decoder always reads it.
183 Xai,
184 /// Always-streamed Codex responses with optional content-type and envelope fields.
185 /// Requests omit sampling controls, storage, metadata, and structured output.
186 Codex,
187}
188
189/// What a dialect's Responses endpoint is: where it lives, where the system
190/// preamble goes, and which contract it speaks.
191#[derive(Debug, Clone, Copy, PartialEq, Eq)]
192#[non_exhaustive]
193pub struct ResponsesQuirks {
194 /// The endpoint path, appended to the base URL.
195 pub path: &'static str,
196 /// Where Rig's system instructions go in the request. Not part of
197 /// [`Self::contract`]: OpenAI's contract is served with all three
198 /// placements (OpenAI's own `instructions`, Copilot's and OpenRouter's
199 /// `system` items in `input`).
200 pub system_instructions: SystemInstructionsPlacement,
201 /// Which contract this dialect's endpoint speaks.
202 pub contract: ResponsesContract,
203 /// Whether newly constructed Responses wires normalize tools for strict validation.
204 pub strict_tools_by_default: bool,
205}
206
207impl ResponsesQuirks {
208 /// OpenAI's own Responses contract.
209 pub const fn openai() -> Self {
210 Self {
211 path: "/responses",
212 system_instructions: SystemInstructionsPlacement::Instructions,
213 contract: ResponsesContract::OpenAi,
214 strict_tools_by_default: false,
215 }
216 }
217}
218
219/// Optional executable extensions to the shared OpenAI dialect.
220///
221/// Store this value in a `static`: equality means the same extension definition,
222/// not equality of function addresses (which code generation can merge or duplicate).
223/// This lets named dialect persistence reject replaced hooks without interpreting
224/// their behavior or pretending arbitrary callbacks can be serialized.
225#[derive(Debug)]
226pub struct DialectHooks {
227 /// Derive a default endpoint from a credential. `None` uses the dialect's
228 /// static URL. Called only at construction, never on credential replacement.
229 pub default_endpoint: Option<fn(&str) -> Option<String>>,
230 /// Select the default route for a model, unless configuration chose a route.
231 pub model_route: Option<fn(&str) -> Route>,
232 /// Apply the completion envelope after shared authentication and identity.
233 /// Called once by either completion encoder; builder errors remain attached
234 /// and are returned when the encoder finishes the request.
235 pub completion_envelope: Option<CompletionEnvelope>,
236 /// Stamp the envelope every modality request (embeddings, listing,
237 /// verification, transcription, images, speech) carries, on the finished
238 /// request. Runs after shared authentication, so a hook may replace the
239 /// credential header rather than add a second one.
240 pub modality_envelope: Option<ModalityEnvelope>,
241}
242
243/// A dialect's modality-request headers, applied to the built request.
244pub type ModalityEnvelope =
245 fn(&OpenAIConfig, &mut http::Request<crate::wire::Body>) -> Result<(), http::Error>;
246
247/// A dialect's completion headers, applied to the authenticated request builder.
248pub type CompletionEnvelope = fn(
249 &OpenAIConfig,
250 &crate::completion::CompletionRequest,
251 http::request::Builder,
252) -> http::request::Builder;
253
254impl PartialEq for DialectHooks {
255 fn eq(&self, other: &Self) -> bool {
256 std::ptr::eq(self, other)
257 }
258}
259
260impl Eq for DialectHooks {}
261
262/// Everything about a dialect that is not its identity: paths, capability
263/// flags, and the one body rewrite it needs.
264///
265/// `#[non_exhaustive]` because the constants live in this crate and a new
266/// quirk must not be a breaking change for a host that stored a wire.
267#[derive(Clone, Copy, Debug, PartialEq, Eq)]
268#[non_exhaustive]
269pub struct Quirks {
270 /// Provider-owned extensions for defaults and completion headers.
271 pub hooks: Option<&'static DialectHooks>,
272 /// How the dialect authenticates.
273 pub auth: Auth,
274 /// How the dialect addresses a model.
275 pub routing: Routing,
276 /// Which completion endpoint
277 /// [`OpenAI::completion`](crate::providers::openai::OpenAI::completion) builds: the
278 /// dialect's flagship. Chat Completions is the one endpoint every
279 /// dialect serves, so it is the baseline; OpenAI itself, xAI and ChatGPT
280 /// serve `/responses` as their primary API and say so.
281 pub completion_route: Route,
282 /// The chat-completions path, relative to the base URL.
283 pub completion_path: &'static str,
284 /// The embeddings path.
285 pub embeddings_path: &'static str,
286 /// The model-listing path.
287 pub models_path: &'static str,
288 /// The path a credential check hits. Empty means the dialect offers no
289 /// check that does not consume tokens (Azure, Perplexity, Copilot).
290 pub verify_path: &'static str,
291 /// The transcription path.
292 pub transcription_path: &'static str,
293 /// The image-generation path.
294 pub image_generation_path: &'static str,
295 /// The speech path.
296 pub audio_generation_path: &'static str,
297 /// Whether `tools`/`tool_choice` reach the provider at all.
298 pub supports_tools: bool,
299 /// Whether `output_schema` maps to `response_format`.
300 pub supports_response_format: bool,
301 /// Whether to send `response_format` with tools before any tool result.
302 /// When false, defer the format until a tool result exists to avoid suppressing calls.
303 pub response_format_with_tools: bool,
304 /// Whether this server honours an image inside a `role:"tool"` message.
305 pub supports_image_tool_results: bool,
306 /// Where the server takes `system` messages that come after the
307 /// conversation begins.
308 pub later_system: crate::completion::LaterSystem,
309 /// Whether a streaming request asks for the usage chunk through
310 /// `stream_options`.
311 pub stream_include_usage: bool,
312 /// Whether a stream's `[DONE]` ends the turn when no finish reason
313 /// arrived, as a tool call when the reply holds one and a stop otherwise.
314 /// Off, such a stream is truncated.
315 pub done_without_finish_reason: bool,
316 /// How the dialect spells the output-token cap.
317 pub output_cap: OutputCap,
318 /// Whether to consult upstream-native finish reasons when normalized ones are absent.
319 pub native_finish_reason: bool,
320 /// The finish reasons the dialect documents beyond the ones every Chat
321 /// dialect shares (`stop`, `length`, `tool_calls`, `content_filter` and
322 /// their compatible spellings). Any other reason fails the turn.
323 pub finishes: &'static [(&'static str, crate::completion::FinishReason)],
324 /// The field rebuilt reasoning goes under, which every assistant message
325 /// then carries, empty when the turn has none (pi's
326 /// `requiresReasoningContentOnAssistantMessages`). `None` sends
327 /// reasoning only under the field it arrived in.
328 pub reasoning_field: Option<&'static str>,
329 /// Whether `completion_tokens_details.reasoning_tokens` can be trusted as
330 /// a part of `completion_tokens`, as OpenAI documents it. A dialect whose
331 /// replies report more reasoning than completion leaves the count
332 /// unreported, so [`Usage`](crate::completion::Usage) never reports more
333 /// reasoning than output.
334 pub reliable_reasoning_count: bool,
335 /// Whether a bare JSON string is accepted as a text-only completion reply.
336 pub accepts_bare_string_reply: bool,
337 /// Whether document and file inputs may use provider file IDs.
338 pub accepts_file_ids: bool,
339 /// The rewrite this dialect applies to the serialized chat body.
340 pub rewrite: BodyRewrite,
341 /// Paths that strip a trailing `/v1` from the configured base URL.
342 pub root_relative_routes: &'static [&'static str],
343 /// Whether the model is the modality endpoint's *path* rather than a
344 /// body field. Hugging Face's router addresses transcription and image
345 /// generation as `/{model}`; everyone else uses a fixed path.
346 pub model_is_modality_path: bool,
347 /// Which body the image endpoint takes.
348 pub image_body: ImageBody,
349 /// Which body the transcription endpoint takes.
350 pub transcription_body: TranscriptionBody,
351 /// Which body the speech endpoint takes.
352 pub speech_body: SpeechBody,
353 /// What the embeddings endpoint accepts.
354 pub embedding: EmbeddingQuirks,
355 /// What the rerank endpoint accepts.
356 pub rerank: RerankQuirks,
357 /// A second environment variable naming the base URL, kept because the
358 /// provider documents both spellings.
359 pub base_url_env_alias: Option<&'static str>,
360 /// The environment variable naming the account a credential belongs to,
361 /// sent as `ChatGPT-Account-Id`.
362 pub account_id_env: Option<&'static str>,
363 /// Instructions this gateway expects every turn to carry, merged ahead
364 /// of the caller's preamble.
365 pub default_instructions: Option<&'static str>,
366 /// The environment variable overriding [`Self::default_instructions`].
367 pub instructions_env: Option<&'static str>,
368 /// The caller identity this gateway requires on every request.
369 pub identity: Option<Identity>,
370 /// What the Responses endpoint accepts.
371 pub responses: ResponsesQuirks,
372}
373
374impl Quirks {
375 /// Baseline compatible endpoint policies with Chat Completions routing and
376 /// the legacy `max_tokens` cap. Dialects override supported differences.
377 pub const fn openai() -> Self {
378 Self {
379 hooks: None,
380 auth: Auth::Bearer,
381 routing: Routing::Path,
382 completion_route: Route::Chat,
383 completion_path: "/chat/completions",
384 embeddings_path: "/embeddings",
385 models_path: "/models",
386 verify_path: "/models",
387 transcription_path: "/audio/transcriptions",
388 image_generation_path: "/images/generations",
389 audio_generation_path: "/audio/speech",
390 supports_tools: true,
391 supports_response_format: true,
392 response_format_with_tools: false,
393 supports_image_tool_results: false,
394 later_system: crate::completion::LaterSystem::InPlace,
395 stream_include_usage: true,
396 done_without_finish_reason: false,
397 output_cap: OutputCap::Legacy,
398 native_finish_reason: false,
399 finishes: &[],
400 reasoning_field: None,
401 reliable_reasoning_count: true,
402 accepts_bare_string_reply: false,
403 accepts_file_ids: true,
404 rewrite: BodyRewrite::None,
405 embedding: EmbeddingQuirks::openai(),
406 // OpenAI has no reranking endpoint, and neither does any dialect
407 // on this wire but llama.cpp.
408 rerank: RerankQuirks::unsupported(),
409 root_relative_routes: &[],
410 model_is_modality_path: false,
411 image_body: ImageBody::OpenAi,
412 speech_body: SpeechBody::OpenAi,
413 transcription_body: TranscriptionBody::Multipart,
414 base_url_env_alias: None,
415 account_id_env: None,
416 default_instructions: None,
417 instructions_env: None,
418 identity: None,
419 responses: ResponsesQuirks::openai(),
420 }
421 }
422
423 /// These quirks without the streamed usage chunk: a streaming request
424 /// sends no `stream_options`, so streamed usage reports `None`.
425 pub const fn without_stream_usage(mut self) -> Self {
426 self.stream_include_usage = false;
427 self
428 }
429
430 /// These quirks for a gateway whose streams omit `finish_reason`: a
431 /// `[DONE]` after at least one chunk ends the turn as
432 /// [`FinishReason::ToolCalls`](crate::completion::FinishReason::ToolCalls)
433 /// when the reply holds a tool call, and
434 /// [`FinishReason::Stop`](crate::completion::FinishReason::Stop)
435 /// otherwise. A cut-off stream that still sends `[DONE]` then reads as
436 /// complete, so set this only for a gateway known to omit the field.
437 pub const fn done_without_finish_reason(mut self) -> Self {
438 self.done_without_finish_reason = true;
439 self
440 }
441
442 /// These quirks without structured output: `output_schema` does not map
443 /// to `response_format`.
444 pub const fn without_response_format(mut self) -> Self {
445 self.supports_response_format = false;
446 self
447 }
448}
449
450impl Default for Quirks {
451 /// [`Quirks::openai`].
452 fn default() -> Self {
453 Self::openai()
454 }
455}
456
457/// Provider identity, endpoint defaults, and shared-wire policies.
458#[derive(Clone, Copy, Debug, PartialEq, Eq)]
459pub struct Dialect {
460 /// The provider descriptor name, as records and telemetry name it.
461 pub name: &'static str,
462 /// The default base URL.
463 pub base_url: &'static str,
464 /// The environment variable holding the credential.
465 pub api_key_env: &'static str,
466 /// The environment variable overriding the base URL, when the provider
467 /// has one.
468 pub base_url_env: Option<&'static str>,
469 /// The reply header carrying the provider's transport request id.
470 pub request_id_header: Option<&'static str>,
471 /// A second credential this dialect accepts, with its own variable and
472 /// header. `None` for every dialect but Azure.
473 pub alternate_auth: Option<AuthAlternative>,
474 /// Everything that is not identity.
475 pub quirks: Quirks,
476}
477
478impl Dialect {
479 /// Create a dialect with [`Quirks::openai`] and the supplied identity.
480 /// URL overrides, request-ID headers, and alternative credentials are unset.
481 pub const fn gateway(
482 name: &'static str,
483 base_url: &'static str,
484 api_key_env: &'static str,
485 ) -> Self {
486 Self {
487 name,
488 base_url,
489 api_key_env,
490 base_url_env: None,
491 request_id_header: None,
492 alternate_auth: None,
493 quirks: Quirks::openai(),
494 }
495 }
496
497 /// This dialect with `quirks`.
498 pub const fn with_quirks(mut self, quirks: Quirks) -> Self {
499 self.quirks = quirks;
500 self
501 }
502}
503
504/// Serialize the registered dialect name, rejecting unregistered or modified definitions.
505/// Deserialization resolves that name from this build's registry.
506impl Serialize for Dialect {
507 fn serialize<S: serde::Serializer>(&self, serializer: S) -> Result<S::Ok, S::Error> {
508 let registered = dialects::by_name(self.name) == Some(self);
509 crate::providers::internal::named_dialect::serialize(
510 serializer, "OpenAI", self.name, registered,
511 )
512 }
513}
514
515impl<'de> Deserialize<'de> for Dialect {
516 fn deserialize<D: serde::Deserializer<'de>>(deserializer: D) -> Result<Self, D::Error> {
517 crate::providers::internal::named_dialect::deserialize(deserializer, "OpenAI", |name| {
518 dialects::by_name(name).copied()
519 })
520 }
521}
522
523/// The settings of an OpenAI-shaped provider: serializable, and the
524/// credential is never serialized. [`connect`](Self::connect) puts it on a
525/// transport as an [`OpenAI`](super::OpenAI) client.
526#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
527#[serde(deny_unknown_fields)]
528pub struct OpenAIConfig {
529 /// The credential. Never serialized (see [`Secret`]).
530 pub api_key: Secret,
531 /// The base URL every path resolves against.
532 pub base_url: String,
533 /// Which OpenAI-shaped provider this is.
534 pub dialect: Dialect,
535 /// The completion endpoint this configuration uses when asked for "a
536 /// completion", when it differs from the dialect's flagship
537 /// ([`Quirks::completion_route`]). Set by [`with_route`](Self::with_route).
538 #[serde(default, skip_serializing_if = "Option::is_none")]
539 pub route: Option<Route>,
540 /// Azure's `api-version` query parameter, which every Azure route
541 /// requires. `None` for every other dialect.
542 #[serde(default, skip_serializing_if = "Option::is_none")]
543 pub api_version: Option<String>,
544 /// Azure versions its speech endpoint separately from the rest, so a
545 /// speech request carries this `api-version` instead of
546 /// [`Self::api_version`]. `None` falls back to `api_version`.
547 #[serde(default, skip_serializing_if = "Option::is_none")]
548 pub audio_api_version: Option<String>,
549 /// How this configuration's credential is sent. Taken from the dialect,
550 /// except when the credential came from the dialect's
551 /// [`alternate_auth`](Dialect::alternate_auth) variable, which has its
552 /// own header.
553 pub auth: Auth,
554 /// Which sub-provider the Hugging Face router forwards to. `None` behaves
555 /// as [`SubRoute::HFInference`], the router's own default. `None` for
556 /// every other dialect, which routes nothing.
557 #[serde(default, skip_serializing_if = "Option::is_none")]
558 pub sub_route: Option<SubRoute>,
559 /// The account the credential belongs to, when the gateway asks which
560 /// (`ChatGPT-Account-Id`).
561 #[serde(default, skip_serializing_if = "Option::is_none")]
562 pub account_id: Option<String>,
563 /// Instructions merged ahead of every Responses turn's preamble, when
564 /// the gateway expects some.
565 #[serde(default, skip_serializing_if = "Option::is_none")]
566 pub instructions: Option<String>,
567 /// The caller identity, when the gateway requires one.
568 #[serde(default, skip_serializing_if = "Option::is_none")]
569 pub identity: Option<CallerIdentity>,
570 /// Responses instruction placement override. `None` uses the dialect default.
571 #[serde(default, skip_serializing_if = "Option::is_none")]
572 pub system_instructions: Option<SystemInstructionsPlacement>,
573}
574
575impl OpenAIConfig {
576 /// Official OpenAI, with `api_key`.
577 pub fn new(api_key: impl Into<Secret>) -> Self {
578 Self::with_key(&OPENAI, api_key)
579 }
580
581 /// `dialect` with `api_key`, at the dialect's default base URL and with
582 /// the instructions and caller identity its gateway expects, if any.
583 pub fn with_key(dialect: &Dialect, api_key: impl Into<Secret>) -> Self {
584 let quirks = &dialect.quirks;
585 let api_key = api_key.into();
586 let base_url = quirks
587 .hooks
588 .and_then(|hooks| hooks.default_endpoint)
589 .and_then(|endpoint| endpoint(api_key.expose()))
590 .unwrap_or_else(|| dialect.base_url.to_owned());
591 Self {
592 api_key,
593 base_url,
594 dialect: *dialect,
595 route: None,
596 // Azure deployment URLs require an explicit API version.
597 api_version: match quirks.routing {
598 Routing::AzureDeployment => Some(dialects::AZURE_DEFAULT_API_VERSION.to_owned()),
599 Routing::Path => None,
600 },
601 audio_api_version: None,
602 auth: quirks.auth,
603 sub_route: None,
604 account_id: None,
605 instructions: quirks.default_instructions.map(str::to_owned),
606 identity: quirks.identity.map(|identity| CallerIdentity {
607 originator: identity.originator.to_owned(),
608 user_agent: default_user_agent(identity.originator),
609 }),
610 system_instructions: None,
611 }
612 }
613
614 /// Read `OPENAI_API_KEY` and the optional `OPENAI_BASE_URL` override.
615 /// Return an environment error for missing credentials or invalid values.
616 pub fn from_env() -> Result<Self, EnvError> {
617 Self::from_env_with(&OPENAI)
618 }
619
620 /// `dialect` from its own `api_key_env` and `base_url_env` (or the
621 /// alias its quirks name), plus whatever else its gateway reads: the
622 /// account id, the default instructions and the caller identity.
623 ///
624 /// Azure additionally reads `AZURE_API_VERSION`, because every Azure
625 /// route carries it and there is no default that would not silently
626 /// address the wrong API.
627 pub fn from_env_with(dialect: &Dialect) -> Result<Self, EnvError> {
628 let (api_key, auth) = Self::credential_from_env(dialect)?;
629 Self::from_env_with_credential(dialect, api_key, auth)
630 }
631
632 /// [`Self::from_env_with`] with the credential already read.
633 pub(crate) fn from_env_with_credential(
634 dialect: &Dialect,
635 api_key: String,
636 auth: Auth,
637 ) -> Result<Self, EnvError> {
638 let quirks = &dialect.quirks;
639 let mut provider = Self::with_key(dialect, api_key);
640 provider.auth = auth;
641 for name in [dialect.base_url_env, quirks.base_url_env_alias]
642 .into_iter()
643 .flatten()
644 {
645 if let Some(base_url) = env::optional(name)? {
646 provider.base_url = base_url;
647 break;
648 }
649 }
650 // Azure speech uses an independently versioned endpoint.
651 if let Routing::AzureDeployment = quirks.routing {
652 provider.api_version = Some(env::required(dialects::AZURE_API_VERSION_ENV)?);
653 provider.audio_api_version = env::optional(dialects::AZURE_AUDIO_API_VERSION_ENV)?
654 .or_else(|| Some(dialects::AZURE_DEFAULT_AUDIO_API_VERSION.to_owned()));
655 }
656 if let Some(name) = quirks.account_id_env {
657 provider.account_id = env::optional(name)?;
658 }
659 if let Some(name) = quirks.instructions_env
660 && let Some(instructions) = env::optional(name)?
661 && !instructions.trim().is_empty()
662 {
663 provider.instructions = Some(instructions);
664 }
665 if let (Some(identity), Some(resolved)) = (quirks.identity, provider.identity.as_mut()) {
666 if let Some(originator) =
667 env::optional(identity.originator_env)?.filter(|value| !value.is_empty())
668 {
669 resolved.originator = originator;
670 resolved.user_agent = default_user_agent(&resolved.originator);
671 }
672 if let Some(user_agent) =
673 env::optional(identity.user_agent_env)?.filter(|value| !value.is_empty())
674 {
675 resolved.user_agent = user_agent;
676 }
677 }
678 Ok(provider)
679 }
680
681 /// Point this configuration at another dialect: the same credential,
682 /// with everything else at that dialect's defaults.
683 pub fn with_dialect(self, dialect: &Dialect) -> Self {
684 Self::with_key(dialect, self.api_key)
685 }
686
687 /// Route through a Hugging Face sub-provider.
688 pub fn with_sub_route(mut self, sub_route: SubRoute) -> Self {
689 self.sub_route = Some(sub_route);
690 self
691 }
692
693 /// Override the base URL.
694 pub fn with_base_url(mut self, base_url: impl Into<String>) -> Self {
695 self.base_url = base_url.into();
696 self
697 }
698
699 /// Set Azure's `api-version`.
700 pub fn with_api_version(mut self, api_version: impl Into<String>) -> Self {
701 self.api_version = Some(api_version.into());
702 self
703 }
704
705 /// Merge these instructions ahead of every Responses turn's preamble.
706 pub fn with_instructions(mut self, instructions: impl Into<String>) -> Self {
707 self.instructions = Some(instructions.into());
708 self
709 }
710
711 /// Put Rig's system instructions somewhere other than the dialect's
712 /// default placement, for every Responses wire this configuration
713 /// builds.
714 pub fn with_system_instructions_placement(
715 mut self,
716 placement: SystemInstructionsPlacement,
717 ) -> Self {
718 self.system_instructions = Some(placement);
719 self
720 }
721
722 /// Send Rig's system instructions as `system` messages in `input`, for a
723 /// backend that rejects or ignores top-level `instructions`.
724 pub fn with_system_instructions_as_messages(self) -> Self {
725 self.with_system_instructions_placement(SystemInstructionsPlacement::InputSystemMessages)
726 }
727
728 /// Where a Responses wire built from this configuration puts Rig's
729 /// system instructions: the dialect's placement unless
730 /// [`with_system_instructions_placement`](Self::with_system_instructions_placement)
731 /// chose another.
732 pub fn system_instructions_placement(&self) -> SystemInstructionsPlacement {
733 self.system_instructions
734 .unwrap_or(self.dialect.quirks.responses.system_instructions)
735 }
736
737 /// Override dialect and model-specific routing for the client's
738 /// [`completion`](crate::providers::openai::OpenAI::completion).
739 pub fn with_route(mut self, route: Route) -> Self {
740 self.route = Some(route);
741 self
742 }
743
744 /// The configured route or dialect's static default. A model-route hook
745 /// may refine the default when a completion wire is constructed.
746 pub fn completion_route(&self) -> Route {
747 self.route.unwrap_or(self.dialect.quirks.completion_route)
748 }
749
750 /// The completion wire for `model` on this configuration's
751 /// [`completion_route`](Self::completion_route): Responses for OpenAI,
752 /// xAI and ChatGPT, model-dependent routing when a dialect supplies it, and
753 /// Chat Completions for other compatible gateways, unless
754 /// [`with_route`](Self::with_route) chose the other one.
755 pub(crate) fn completion(&self, model: impl Into<String>) -> OpenAiWire {
756 OpenAiWire::new(self.clone(), model)
757 }
758
759 /// The Responses wire for `model`: `POST /responses`.
760 pub(crate) fn responses(&self, model: impl Into<String>) -> Responses {
761 Responses::new(self.clone(), model)
762 }
763
764 /// The chat-completions wire for `model`, whatever the dialect's
765 /// [`completion_route`](Quirks::completion_route).
766 pub fn chat(&self, model: impl Into<String>) -> Chat {
767 Chat::new(self.clone(), model)
768 }
769
770 pub(crate) fn completion_headers(
771 &self,
772 request: &crate::completion::CompletionRequest,
773 builder: http::request::Builder,
774 ) -> http::request::Builder {
775 let builder = self.headers(builder);
776 match self
777 .dialect
778 .quirks
779 .hooks
780 .and_then(|hooks| hooks.completion_envelope)
781 {
782 Some(envelope) => envelope(self, request, builder),
783 None => builder,
784 }
785 }
786
787 /// Resolve `path` against the base URL, applying Azure's
788 /// deployment-in-URL routing when the dialect uses it.
789 pub(crate) fn uri(&self, path: &str, model: Option<&str>) -> String {
790 self.uri_versioned(path, model, self.api_version.as_deref())
791 }
792
793 /// [`Self::uri`] with an explicit `api-version`, for the one endpoint
794 /// Azure versions separately (speech).
795 pub(crate) fn uri_versioned(
796 &self,
797 path: &str,
798 model: Option<&str>,
799 api_version: Option<&str>,
800 ) -> String {
801 match (self.dialect.quirks.routing, model) {
802 (Routing::AzureDeployment, Some(model)) => format!(
803 "{}/openai/deployments/{}{}?api-version={}",
804 self.base_url.trim_end_matches('/'),
805 model.trim_start_matches('/'),
806 path,
807 api_version.unwrap_or_default(),
808 ),
809 _ => format!("{}{}", self.base(path), path),
810 }
811 }
812
813 /// The base URL `path` resolves against: the configured one, with the
814 /// version segment dropped for a route the dialect serves at the root.
815 fn base(&self, path: &str) -> &str {
816 let base = self.base_url.trim_end_matches('/');
817 if self.dialect.quirks.root_relative_routes.contains(&path) {
818 return base.strip_suffix("/v1").unwrap_or(base);
819 }
820 base
821 }
822
823 /// The sub-provider the Hugging Face router forwards to. `None` on the
824 /// configuration means the router's own default.
825 pub(crate) fn route(&self) -> std::borrow::Cow<'_, SubRoute> {
826 match &self.sub_route {
827 Some(route) => std::borrow::Cow::Borrowed(route),
828 None => std::borrow::Cow::Owned(SubRoute::default()),
829 }
830 }
831
832 /// Return `model` for Azure deployment routing, otherwise `None`.
833 pub(crate) fn deployment<'a>(&self, model: &'a str) -> Option<&'a str> {
834 match self.dialect.quirks.routing {
835 Routing::AzureDeployment => Some(model),
836 Routing::Path => None,
837 }
838 }
839}
840
841#[cfg(test)]
842mod tests;