1use crate::completion::Usage;
19use crate::error::EncodeError;
20use crate::error::ProviderError;
21use crate::message::DocumentSourceKind;
22use crate::message::{CallId, ToolName};
23use crate::model::ModelInfo;
24use crate::operation::{Completion, Finish, TextPart};
25use crate::providers::internal;
26use crate::providers::internal::thoughts::Thoughts;
27use crate::wire::{Flow, Out};
28use crate::{
29 completion::{self, CompletionRequest},
30 json_utils, message,
31};
32use serde::{Deserialize, Serialize};
33use serde_json::{Value, json};
34
35pub mod wire;
36
37pub use crate::client::ollama::Ollama;
38pub use wire::{Chat, Embeddings, Models, OllamaConfig};
39
40const OLLAMA_API_BASE_URL: &str = "http://localhost:11434";
42
43const PROVIDER_NAME: &str = "ollama";
46
47const ISSUER: crate::message::Issuer = crate::message::Issuer::from_static(PROVIDER_NAME);
49
50pub const ALL_MINILM: &str = "all-minilm";
52pub const NOMIC_EMBED_TEXT: &str = "nomic-embed-text";
54pub const MXBAI_EMBED_LARGE: &str = "mxbai-embed-large";
56pub const BGE_M3: &str = "bge-m3";
58pub const EMBEDDINGGEMMA: &str = "embeddinggemma";
60pub const QWEN3_EMBEDDING: &str = "qwen3-embedding";
62
63fn model_dimensions_from_identifier(identifier: &str) -> Option<usize> {
64 match identifier {
65 ALL_MINILM => Some(384),
66 NOMIC_EMBED_TEXT => Some(768),
67 MXBAI_EMBED_LARGE => Some(1024),
68 BGE_M3 => Some(1024),
69 EMBEDDINGGEMMA => Some(768),
70 _ => None,
71 }
72}
73
74#[derive(Debug, Clone, Serialize, Deserialize)]
75pub struct EmbeddingResponse {
76 pub model: String,
77 pub embeddings: Vec<Vec<f64>>,
78 #[serde(default)]
79 pub total_duration: Option<u64>,
80 #[serde(default)]
81 pub load_duration: Option<u64>,
82 #[serde(default)]
83 pub prompt_eval_count: Option<u64>,
84}
85
86pub const LLAMA3_2: &str = "llama3.2";
88pub const LLAMA3_1: &str = "llama3.1";
90pub const LLAMA3_3: &str = "llama3.3";
92pub const LLAMA4: &str = "llama4";
94pub const LLAVA: &str = "llava";
96pub const MISTRAL: &str = "mistral";
98pub const MISTRAL_SMALL3_2: &str = "mistral-small3.2";
100pub const GEMMA3: &str = "gemma3";
102pub const GEMMA4: &str = "gemma4";
104pub const QWEN3: &str = "qwen3";
106pub const QWEN3_5: &str = "qwen3.5";
108pub const QWEN3_6: &str = "qwen3.6";
110pub const QWEN3_8: &str = "qwen3.8";
112pub const QWEN3_CODER: &str = "qwen3-coder";
114pub const DEEPSEEK_R1: &str = "deepseek-r1";
116pub const DEEPSEEK_V3_1: &str = "deepseek-v3.1";
118pub const GPT_OSS: &str = "gpt-oss";
120pub const PHI4: &str = "phi4";
122
123#[derive(Debug, Serialize, Deserialize)]
124pub struct CompletionResponse {
125 pub model: String,
126 pub created_at: String,
127 pub message: Message,
128 pub done: bool,
129 #[serde(default)]
130 pub done_reason: Option<String>,
131 #[serde(default)]
132 pub total_duration: Option<u64>,
133 #[serde(default)]
134 pub load_duration: Option<u64>,
135 #[serde(default)]
136 pub prompt_eval_count: Option<u64>,
137 #[serde(default)]
138 pub prompt_eval_duration: Option<u64>,
139 #[serde(default)]
140 pub eval_count: Option<u64>,
141 #[serde(default)]
142 pub eval_duration: Option<u64>,
143}
144pub(crate) fn map_done_reason(reason: &str) -> completion::FinishReason {
150 match reason {
151 "stop" => completion::FinishReason::Stop,
152 "length" => completion::FinishReason::Length,
153 other => completion::FinishReason::Other(other.to_owned()),
154 }
155}
156
157fn ollama_usage(prompt_eval_count: Option<u64>, eval_count: Option<u64>) -> Usage {
160 Usage {
161 input_tokens: prompt_eval_count,
162 output_tokens: eval_count,
163 total_tokens: prompt_eval_count
164 .zip(eval_count)
165 .map(|(input, output)| input + output),
166 ..Default::default()
167 }
168}
169
170fn split_legacy_thinking(content: &str, permits_omitted_start: bool) -> (Option<&str>, &str) {
173 let trimmed = content.trim_start();
174 let split = if let Some(reasoning_start) = trimmed.strip_prefix("<think>") {
175 reasoning_start.split_once("</think>")
176 } else if permits_omitted_start {
177 trimmed.split_once("\n</think>\n\n")
181 } else {
182 None
183 };
184 let Some((reasoning, visible)) = split else {
185 return (None, content);
186 };
187
188 let reasoning = reasoning.trim();
189 if reasoning.is_empty() {
190 return (None, visible.trim_start());
191 }
192
193 (Some(reasoning), visible.trim_start())
194}
195
196#[derive(Debug, Serialize, Deserialize)]
197pub(super) struct OllamaCompletionRequest {
198 model: String,
199 pub messages: Vec<Message>,
200 #[serde(skip_serializing_if = "Vec::is_empty")]
201 tools: Vec<ToolDefinition>,
202 pub stream: bool,
203 #[serde(skip_serializing_if = "Option::is_none")]
204 think: Option<Think>,
205 #[serde(skip_serializing_if = "Option::is_none")]
206 keep_alive: Option<String>,
207 #[serde(skip_serializing_if = "Option::is_none")]
208 format: Option<schemars::Schema>,
209 options: serde_json::Value,
210}
211
212impl TryFrom<(&str, CompletionRequest)> for OllamaCompletionRequest {
213 type Error = EncodeError;
214
215 fn try_from((model, req): (&str, CompletionRequest)) -> Result<Self, Self::Error> {
216 let chat_history = req.chat_history_with_documents();
217 let model = req.model.clone().unwrap_or_else(|| model.to_string());
218 if req.tool_choice.is_some() {
219 tracing::warn!("WARNING: `tool_choice` not supported for Ollama");
220 }
221 let mut partial_history = vec![];
222 partial_history.extend(chat_history);
223
224 let mut full_history: Vec<Message> = Vec::new();
225 full_history.extend(
226 partial_history
227 .into_iter()
228 .map(message::Message::try_into)
229 .collect::<Result<Vec<Vec<Message>>, _>>()?
230 .into_iter()
231 .flatten(),
232 );
233
234 let mut think: Option<Think> = None;
235 let mut keep_alive: Option<String> = None;
236
237 let mut base_options = serde_json::Map::new();
241 if let Some(temperature) = req.temperature {
242 base_options.insert("temperature".to_string(), json!(temperature));
243 }
244 if let Some(max_tokens) = req.max_tokens {
245 base_options.insert("num_predict".to_string(), json!(max_tokens));
246 }
247 let base_options = Value::Object(base_options);
248
249 let options = if let Some(mut extra) = req.additional_params {
250 if let Some(obj) = extra.as_object_mut() {
252 if let Some(think_val) = obj.remove("think") {
253 think = Some(match think_val {
254 Value::Bool(think) => Think::Bool(think),
255 Value::String(think) => Think::Level(match think.to_lowercase().as_str() {
256 "low" => Level::Low,
257 "medium" => Level::Medium,
258 "high" => Level::High,
259 "max" => Level::Max,
260 _ => {
261 return Err(EncodeError::request(
262 "`think` must be a 'low', 'medium', 'high', 'max' or bool",
263 ));
264 }
265 }),
266 _ => {
267 return Err(EncodeError::request(
268 "`think` must be a 'low', 'medium', 'high', 'max' or bool",
269 ));
270 }
271 });
272 }
273
274 if let Some(keep_alive_val) = obj.remove("keep_alive") {
275 keep_alive = Some(
276 keep_alive_val
277 .as_str()
278 .ok_or_else(|| EncodeError::request("`keep_alive` must be a string"))?
279 .to_string(),
280 );
281 }
282 }
283
284 json_utils::merge(base_options, extra)
285 } else {
286 base_options
287 };
288
289 Ok(Self {
290 model,
291 messages: full_history,
292 stream: false,
293 think,
294 keep_alive,
295 format: req.output_schema,
296 tools: req
297 .tools
298 .clone()
299 .into_iter()
300 .map(ToolDefinition::from)
301 .collect::<Vec<_>>(),
302 options,
303 })
304 }
305}
306
307#[derive(Debug, Clone, Serialize, Deserialize)]
308#[serde(untagged)]
309enum Think {
310 Bool(bool),
311 Level(Level),
312}
313
314#[derive(Debug, Clone, Serialize, Deserialize)]
315#[serde(rename_all = "lowercase")]
316enum Level {
317 Low,
318 Medium,
319 High,
320 Max,
321}
322
323#[derive(Clone, Serialize, Deserialize, Debug)]
326pub struct StreamingCompletionResponse {
327 pub model: String,
329 pub done_reason: Option<String>,
330 pub total_duration: Option<u64>,
331 pub load_duration: Option<u64>,
332 pub prompt_eval_count: Option<u64>,
333 pub prompt_eval_duration: Option<u64>,
334 pub eval_count: Option<u64>,
335 pub eval_duration: Option<u64>,
336}
337
338impl From<&StreamingCompletionResponse> for Usage {
339 fn from(response: &StreamingCompletionResponse) -> Usage {
340 ollama_usage(response.prompt_eval_count, response.eval_count)
341 }
342}
343
344fn finish_of(response: StreamingCompletionResponse) -> Finish {
346 Finish {
349 usage: Usage::from(&response),
350 reason: response.done_reason.as_deref().map(map_done_reason),
351 model: Some(response.model),
352 ..Finish::default()
353 }
354}
355
356#[derive(Default)]
359pub struct OllamaDecoder<'id> {
360 thoughts: Thoughts<'id>,
362 text: Option<TextPart<'id>>,
363}
364
365impl<'id> OllamaDecoder<'id> {
366 fn close_text(&mut self, out: &mut Out<'id, Completion>) {
367 if let Some(part) = self.text.take() {
368 out.close_text(part);
369 }
370 }
371
372 fn interpret_record(
375 &mut self,
376 response: CompletionResponse,
377 mut out: Out<'id, Completion>,
378 ) -> Result<Flow, ProviderError> {
379 let done = response.done;
380 let model = response.model;
381 if let Message::Assistant {
382 content,
383 thinking,
384 tool_calls,
385 ..
386 } = response.message
387 {
388 let (reasoning, text) = match thinking.as_deref() {
391 None | Some("") if done => {
392 let permits_omitted_think_start = model.to_ascii_lowercase().contains("qwen3");
393 let (legacy, visible) =
394 split_legacy_thinking(&content, permits_omitted_think_start);
395 (legacy.map(str::to_owned), visible.to_owned())
396 }
397 _ => (thinking, content),
398 };
399 if let Some(reasoning) = reasoning.filter(|reasoning| !reasoning.is_empty()) {
400 self.close_text(&mut out);
401 self.thoughts.fragment(&mut out, &reasoning);
402 }
403 if !text.is_empty() || !tool_calls.is_empty() {
404 self.thoughts.boundary();
405 }
406 if !text.is_empty() {
407 let part = self.text.get_or_insert_with(|| out.text());
408 out.push_text(part, &text);
409 }
410 for tool_call in tool_calls {
413 self.close_text(&mut out);
414 let Ok(name) = ToolName::new(tool_call.function.name) else {
415 continue;
416 };
417 out.tool_call(crate::message::ToolCall {
418 id: CallId::from_wire(tool_call.id.unwrap_or_default()),
419 function: crate::message::ToolFunction {
420 name,
421 arguments: tool_call.function.arguments,
422 },
423 signature: None,
424 additional_params: None,
425 })?;
426 }
427 }
428
429 if !done {
431 return Ok(Flow::More);
432 }
433 let native = StreamingCompletionResponse {
434 model,
435 total_duration: response.total_duration,
436 load_duration: response.load_duration,
437 prompt_eval_count: response.prompt_eval_count,
438 prompt_eval_duration: response.prompt_eval_duration,
439 eval_count: response.eval_count,
440 eval_duration: response.eval_duration,
441 done_reason: response.done_reason,
442 };
443 self.close_text(&mut out);
444 self.thoughts.close(&mut out, None);
445 out.raw(serde_json::to_value(&native)?);
446 Ok(out.end(finish_of(native)))
447 }
448
449 fn classify_line(frame: crate::wire::WireFrame) -> crate::wire::WireEvent<CompletionResponse> {
452 match frame {
453 crate::wire::WireFrame::Bytes(line) => internal::wire::classify_untyped_line(&line),
454 crate::wire::WireFrame::Text(line) => {
455 internal::wire::classify_untyped_line(line.as_bytes())
456 }
457 }
458 }
459}
460
461impl<'id> crate::wire::Decoder<'id, Completion> for OllamaDecoder<'id> {
463 type Event = CompletionResponse;
464
465 fn classify(
466 &self,
467 frame: crate::wire::WireFrame,
468 ) -> crate::wire::WireEvent<CompletionResponse> {
469 OllamaDecoder::classify_line(frame)
470 }
471
472 fn decode(
473 &mut self,
474 response: CompletionResponse,
475 out: Out<'id, Completion>,
476 ) -> Result<Flow, ProviderError> {
477 self.interpret_record(response, out)
478 }
479}
480
481#[derive(Debug, Deserialize)]
483pub struct ListModelsResponse {
484 pub models: Vec<ListModelEntry>,
486}
487
488#[derive(Debug, Deserialize)]
490pub struct ListModelEntry {
491 pub name: String,
493 pub model: String,
495}
496
497impl From<ListModelEntry> for ModelInfo {
498 fn from(value: ListModelEntry) -> Self {
499 ModelInfo::new(value.model, value.name)
500 }
501}
502
503#[derive(Clone, Debug, Deserialize, Serialize)]
505pub struct ToolDefinition {
506 #[serde(rename = "type")]
507 pub type_field: String,
508 pub function: completion::ToolDefinition,
509}
510
511impl From<crate::completion::ToolDefinition> for ToolDefinition {
513 fn from(tool: crate::completion::ToolDefinition) -> Self {
514 ToolDefinition {
515 type_field: "function".to_owned(),
516 function: completion::ToolDefinition {
517 name: tool.name,
518 description: tool.description,
519 parameters: tool.parameters,
520 },
521 }
522 }
523}
524
525#[derive(Debug, Serialize, Deserialize, PartialEq, Clone)]
526pub struct ToolCall {
527 #[serde(default, skip_serializing_if = "Option::is_none")]
532 pub id: Option<String>,
533 #[serde(default, rename = "type")]
534 pub r#type: ToolType,
535 pub function: Function,
536}
537#[derive(Default, Debug, Serialize, Deserialize, PartialEq, Clone)]
538#[serde(rename_all = "lowercase")]
539pub enum ToolType {
540 #[default]
541 Function,
542}
543#[derive(Debug, Serialize, Deserialize, PartialEq, Clone)]
544pub struct Function {
545 pub name: String,
546 pub arguments: Value,
547}
548
549#[derive(Debug, Serialize, Deserialize, PartialEq, Clone)]
550#[serde(tag = "role", rename_all = "lowercase")]
551pub enum Message {
552 User {
553 content: String,
554 #[serde(skip_serializing_if = "Option::is_none")]
555 images: Option<Vec<String>>,
556 #[serde(skip_serializing_if = "Option::is_none")]
557 name: Option<String>,
558 },
559 Assistant {
560 #[serde(default)]
561 content: String,
562 #[serde(skip_serializing_if = "Option::is_none")]
563 thinking: Option<String>,
564 #[serde(skip_serializing_if = "Option::is_none")]
565 images: Option<Vec<String>>,
566 #[serde(skip_serializing_if = "Option::is_none")]
567 name: Option<String>,
568 #[serde(default, deserialize_with = "json_utils::null_or_default")]
569 tool_calls: Vec<ToolCall>,
570 },
571 System {
572 content: String,
573 #[serde(skip_serializing_if = "Option::is_none")]
574 images: Option<Vec<String>>,
575 #[serde(skip_serializing_if = "Option::is_none")]
576 name: Option<String>,
577 },
578 #[serde(rename = "tool")]
579 ToolResult {
580 #[serde(rename = "tool_name")]
581 name: String,
582 content: String,
583 #[serde(
586 rename = "tool_call_id",
587 default,
588 skip_serializing_if = "Option::is_none"
589 )]
590 call_id: Option<String>,
591 },
592}
593
594fn user_message_from_content(
597 content: Vec<crate::message::UserContent>,
598) -> Result<Message, crate::message::MessageError> {
599 let mut texts = Vec::new();
600 let mut images = Vec::new();
601
602 for content in content {
603 match content {
604 crate::message::UserContent::Text(crate::message::Text { text, .. }) => {
605 texts.push(text);
606 }
607 crate::message::UserContent::Image(crate::message::Image {
608 data: DocumentSourceKind::Base64(data),
609 ..
610 }) => images.push(data),
611 crate::message::UserContent::Image(_) => {
612 return Err(crate::message::MessageError::ConversionError(
613 "Ollama images must be base64 encoded data".into(),
614 ));
615 }
616 crate::message::UserContent::Document(crate::message::Document {
617 data: DocumentSourceKind::Base64(data) | DocumentSourceKind::String(data),
618 ..
619 }) => texts.push(data),
620 crate::message::UserContent::Document(_) => {
621 return Err(crate::message::MessageError::ConversionError(
622 "Ollama documents must be string or base64 encoded data".into(),
623 ));
624 }
625 crate::message::UserContent::Audio(_) => {
626 return Err(crate::message::MessageError::ConversionError(
627 "Ollama does not support audio user content".into(),
628 ));
629 }
630 crate::message::UserContent::Video(_) => {
631 return Err(crate::message::MessageError::ConversionError(
632 "Ollama does not support video user content".into(),
633 ));
634 }
635 crate::message::UserContent::ToolResult(_) => {
636 return Err(crate::message::MessageError::ConversionError(
637 "tool results must be converted to a separate Ollama message".into(),
638 ));
639 }
640 }
641 }
642
643 Ok(Message::User {
644 content: texts.join(" "),
645 images: (!images.is_empty()).then_some(images),
646 name: None,
647 })
648}
649
650impl TryFrom<crate::message::Message> for Vec<Message> {
653 type Error = crate::message::MessageError;
654 fn try_from(internal_msg: crate::message::Message) -> Result<Self, Self::Error> {
655 use crate::message::Message as InternalMessage;
656 match internal_msg {
657 InternalMessage::System { content } => Ok(vec![Message::System {
658 content,
659 images: None,
660 name: None,
661 }]),
662 InternalMessage::User { content, .. } => {
663 let mut messages = Vec::new();
664 let mut pending_user_content = Vec::new();
665
666 for content in content {
667 match content {
668 crate::message::UserContent::ToolResult(crate::message::ToolResult {
669 call,
670 name,
671 content,
672 }) => {
673 let function_name = name;
674 if !pending_user_content.is_empty() {
675 messages.push(user_message_from_content(std::mem::take(
676 &mut pending_user_content,
677 ))?);
678 }
679
680 let content = content
681 .into_iter()
682 .map(|content| match content {
683 crate::message::ToolResultContent::Text(text) => Ok(text.text),
684 crate::message::ToolResultContent::Json { value } => {
685 Ok(value.to_string())
686 }
687 crate::message::ToolResultContent::Image(_) => {
688 Err(crate::message::MessageError::ConversionError(
689 "Ollama does not support images in tool results".into(),
690 ))
691 }
692 })
693 .collect::<Result<Vec<_>, _>>()?
694 .join("\n");
695 messages.push(Message::ToolResult {
696 name: function_name.into(),
697 content,
698 call_id: call.provider().map(|provider| provider.call_id.clone()),
699 });
700 }
701 content => pending_user_content.push(content),
702 }
703 }
704
705 if !pending_user_content.is_empty() {
706 messages.push(user_message_from_content(pending_user_content)?);
707 }
708
709 Ok(messages)
710 }
711 InternalMessage::Assistant { content, .. } => {
712 let mut thinking: Option<String> = None;
713 let mut text_content = Vec::new();
714 let mut tool_calls = Vec::new();
715
716 for content in content.into_iter() {
717 match content {
718 crate::message::AssistantContent::Text(text) => {
719 text_content.push(text.text);
720 }
721 crate::message::AssistantContent::ToolCall(tool_call) => {
722 tool_calls.push(tool_call);
723 }
724 crate::message::AssistantContent::Reasoning(reasoning) => {
725 let Some(reasoning) = reasoning.open(&ISSUER) else {
727 continue;
728 };
729 let display = reasoning.display_text();
730 if !display.is_empty() {
731 thinking = Some(display);
732 }
733 }
734 crate::message::AssistantContent::Image(_) => {
735 return Err(crate::message::MessageError::ConversionError(
736 "Ollama currently doesn't support images.".into(),
737 ));
738 }
739 }
740 }
741
742 Ok(vec![Message::Assistant {
743 content: text_content.join(" "),
744 thinking,
745 images: None,
746 name: None,
747 tool_calls: tool_calls
748 .into_iter()
749 .map(std::convert::Into::into)
750 .collect::<Vec<_>>(),
751 }])
752 }
753 }
754 }
755}
756
757impl Message {
758 pub fn system(content: &str) -> Self {
760 Message::System {
761 content: content.to_owned(),
762 images: None,
763 name: None,
764 }
765 }
766}
767
768impl From<crate::message::ToolCall> for ToolCall {
769 fn from(tool_call: crate::message::ToolCall) -> Self {
770 Self {
771 id: tool_call
772 .id
773 .provider()
774 .map(|provider| provider.call_id.clone()),
775 r#type: ToolType::Function,
776 function: Function {
777 name: tool_call.function.name.into(),
778 arguments: tool_call.function.arguments,
779 },
780 }
781 }
782}
783
784#[cfg(test)]
785mod tests;