1use anyhow::{Context, Result};
11use serde_json::Value as JsonValue;
12use std::fmt::Write;
13
14use super::common::{
15 NormalizeNonText, REASONING_EFFORT_HIGH, REASONING_EFFORT_MAX, RESPONSE_FORMAT_TEMPLATE,
16 TOOL_CALLS_BLOCK_NAME, TOOLS_TEMPLATE, drop_thinking_messages, encode_arguments_to_dsml,
17 find_last_user_index, merge_tool_messages, normalize_message_contents, render_tools,
18 sort_tool_results_by_call_order, task_token, to_json,
19};
20pub use super::common::{ReasoningEffort, ThinkingMode, tokens};
21
22#[derive(Clone, Copy)]
23pub(super) enum Encoding {
24 V4(Option<ReasoningEffort>),
25 V41(u8),
26}
27
28impl Encoding {
29 fn is_v41(self) -> bool {
30 matches!(self, Self::V41(_))
31 }
32
33 fn tag(self, v4: &'static str, v41: &'static str) -> &'static str {
34 if self.is_v41() { v41 } else { v4 }
35 }
36
37 fn reasoning_prefix(self) -> String {
38 match self {
39 Self::V4(Some(ReasoningEffort::High)) => REASONING_EFFORT_HIGH.to_string(),
40 Self::V4(Some(ReasoningEffort::Max)) => REASONING_EFFORT_MAX.to_string(),
41 Self::V4(None) => String::new(),
42 Self::V41(effort) => format!(
43 "Reasoning Effort: {effort} (range 1-100, the higher the value, the more thorough the reasoning)\n\n"
44 ),
45 }
46 }
47
48 fn render_tools(self, tools: &[JsonValue]) -> String {
49 let template = if self.is_v41() {
50 TOOLS_TEMPLATE
51 .replace("{dsml_token}tool_calls", "{dsml_token} calls")
52 .replace("{dsml_token}invoke", "{dsml_token} invoke")
53 .replace("{dsml_token}parameter", "{dsml_token} parameter")
54 } else {
55 TOOLS_TEMPLATE.to_string()
56 };
57 render_tools(&template, tools)
58 }
59}
60
61fn render_message(
63 prompt: &mut String,
64 index: usize,
65 messages: &[JsonValue],
66 thinking_mode: ThinkingMode,
67 drop_thinking: bool,
68 encoding: Encoding,
69 last_user_idx: Option<usize>,
70) -> Result<()> {
71 let msg = &messages[index];
72
73 let role = msg
74 .get("role")
75 .and_then(|r| r.as_str())
76 .context("Missing 'role' field")?;
77
78 if encoding.is_v41()
79 && (role == "system" || (index == 0 && thinking_mode == ThinkingMode::Thinking))
80 {
81 prompt.push_str("<|System|>");
82 }
83 if index == 0 && thinking_mode == ThinkingMode::Thinking {
84 prompt.push_str(&encoding.reasoning_prefix());
85 }
86
87 match role {
88 "system" => {
89 let content = msg.get("content").and_then(|c| c.as_str()).unwrap_or("");
90 prompt.push_str(content);
91 if let Some(tools) = msg.get("tools").and_then(|t| t.as_array()) {
92 prompt.push_str("\n\n");
93 prompt.push_str(&encoding.render_tools(tools));
94 }
95 if let Some(response_format) = msg.get("response_format") {
96 prompt.push_str("\n\n");
97 prompt.push_str(
98 &RESPONSE_FORMAT_TEMPLATE.replace("{schema}", &to_json(response_format)),
99 );
100 }
101 }
102
103 "developer" => {
104 let content = msg
105 .get("content")
106 .and_then(|c| c.as_str())
107 .filter(|s| !s.is_empty())
108 .context("Developer role requires content")?;
109
110 prompt.push_str(tokens::USER_START);
111 prompt.push_str(content);
112
113 if let Some(tools) = msg.get("tools").and_then(|t| t.as_array()) {
114 prompt.push_str("\n\n");
115 prompt.push_str(&encoding.render_tools(tools));
116 }
117 if let Some(response_format) = msg.get("response_format") {
118 prompt.push_str("\n\n");
119 prompt.push_str(
120 &RESPONSE_FORMAT_TEMPLATE.replace("{schema}", &to_json(response_format)),
121 );
122 }
123 }
124
125 "user" => {
126 prompt.push_str(tokens::USER_START);
127 if let Some(blocks) = msg.get("content_blocks").and_then(|b| b.as_array()) {
128 for (block_idx, block) in blocks.iter().enumerate() {
129 if block_idx > 0 {
130 prompt.push_str("\n\n");
131 }
132 let block_type = block.get("type").and_then(|v| v.as_str()).unwrap_or("");
133 match block_type {
134 "text" => {
135 let text = block.get("text").and_then(|v| v.as_str()).unwrap_or("");
136 prompt.push_str(text);
137 }
138 "tool_result" => {
139 prompt.push_str("<tool_result>");
140 render_tool_result_content(
141 prompt,
142 block.get("content").unwrap_or(&JsonValue::Null),
143 )?;
144 prompt.push_str("</tool_result>");
145 }
146 other => {
147 write!(prompt, "[Unsupported {}]", other)?;
148 }
149 }
150 }
151 }
152 }
153
154 "latest_reminder" => {
155 let content = msg.get("content").and_then(|c| c.as_str()).unwrap_or("");
156 prompt.push_str(tokens::LATEST_REMINDER);
157 prompt.push_str(content);
158 }
159
160 "tool" => {
161 anyhow::bail!(
162 "deepseek_v4 merges tool messages into user; preprocess with merge_tool_messages()"
163 );
164 }
165
166 "assistant" => {
167 let content = msg.get("content").and_then(|c| c.as_str()).unwrap_or("");
168 let reasoning = msg
169 .get("reasoning_content")
170 .and_then(|c| c.as_str())
171 .unwrap_or("");
172 let wo_eos = msg.get("wo_eos").and_then(|v| v.as_bool()).unwrap_or(false);
173
174 let prev_has_task = index > 0
175 && messages[index - 1]
176 .get("task")
177 .map(|v| !v.is_null())
178 .unwrap_or(false);
179
180 if thinking_mode == ThinkingMode::Thinking && !prev_has_task {
181 let render_thinking = !drop_thinking || last_user_idx.is_none_or(|u| index > u);
182 if render_thinking {
183 prompt.push_str(reasoning);
184 prompt.push_str(tokens::THINKING_END);
185 }
186 }
187
188 prompt.push_str(content);
189
190 if let Some(tool_calls) = msg.get("tool_calls").and_then(|t| t.as_array())
191 && !tool_calls.is_empty()
192 {
193 prompt.push_str("\n\n");
194 writeln!(
195 prompt,
196 "<{}{}>",
197 tokens::DSML_TOKEN,
198 encoding.tag(TOOL_CALLS_BLOCK_NAME, " calls")
199 )?;
200
201 for (call_idx, tc) in tool_calls.iter().enumerate() {
202 if call_idx > 0 {
203 prompt.push('\n');
204 }
205 let fn_obj = tc.get("function").unwrap_or(tc);
208 let name = fn_obj
209 .get("name")
210 .and_then(|n| n.as_str())
211 .context("Missing tool call name")?;
212 let arguments = if encoding.is_v41() {
213 super::v41::encode_arguments(fn_obj)?
214 } else {
215 encode_arguments_to_dsml(fn_obj)?
216 };
217 write!(
218 prompt,
219 "<{}{} name=\"{}\">\n{}\n</{}{}>",
220 tokens::DSML_TOKEN,
221 encoding.tag("invoke", " invoke"),
222 name,
223 arguments,
224 tokens::DSML_TOKEN,
225 encoding.tag("invoke", " invoke")
226 )?;
227 }
228 write!(
229 prompt,
230 "\n</{}{}>",
231 tokens::DSML_TOKEN,
232 encoding.tag(TOOL_CALLS_BLOCK_NAME, " calls")
233 )?;
234 }
235
236 if !wo_eos {
237 prompt.push_str(tokens::EOS);
238 }
239 }
240
241 other => anyhow::bail!("Unknown role: {}", other),
242 }
243
244 if index + 1 < messages.len() {
246 let next_role = messages[index + 1].get("role").and_then(|r| r.as_str());
247 if !matches!(next_role, Some("assistant") | Some("latest_reminder")) {
248 return Ok(());
249 }
250 }
251
252 let task = msg.get("task").and_then(|v| v.as_str());
254 if let Some(task) = task {
255 let sp = task_token(task).with_context(|| format!("Invalid task: '{}'", task))?;
256 if task != "action" {
257 prompt.push_str(sp);
258 } else {
259 prompt.push_str(tokens::ASSISTANT_START);
260 prompt.push_str(if thinking_mode != ThinkingMode::Thinking {
261 tokens::THINKING_END
262 } else {
263 tokens::THINKING_START
264 });
265 prompt.push_str(sp);
266 }
267 } else if matches!(role, "user" | "developer")
268 || (encoding.is_v41() && role == "system" && index > 0)
269 {
270 prompt.push_str(tokens::ASSISTANT_START);
271 let seed_thinking = thinking_mode == ThinkingMode::Thinking
272 && (!drop_thinking || last_user_idx.is_none_or(|u| index >= u));
273 prompt.push_str(if seed_thinking {
274 tokens::THINKING_START
275 } else {
276 tokens::THINKING_END
277 });
278 }
279
280 Ok(())
281}
282
283fn render_tool_result_content(prompt: &mut String, content: &JsonValue) -> Result<()> {
285 match content {
286 JsonValue::String(s) => prompt.push_str(s),
287 JsonValue::Array(items) => {
288 for (index, item) in items.iter().enumerate() {
289 if index > 0 {
290 prompt.push_str("\n\n");
291 }
292 let item_type = item.get("type").and_then(|v| v.as_str()).unwrap_or("");
293 if item_type == "text" {
294 prompt.push_str(item.get("text").and_then(|v| v.as_str()).unwrap_or(""));
295 } else {
296 write!(prompt, "[Unsupported {}]", item_type)?;
297 }
298 }
299 }
300 JsonValue::Null => {}
301 _ => prompt.push_str(&to_json(content)),
302 }
303 Ok(())
304}
305
306pub fn encode_messages(
310 messages: &[JsonValue],
311 thinking_mode: ThinkingMode,
312 add_bos_token: bool,
313) -> Result<String> {
314 encode_messages_with_options(messages, thinking_mode, add_bos_token, true, None)
315}
316
317pub fn encode_messages_with_options(
326 messages: &[JsonValue],
327 thinking_mode: ThinkingMode,
328 add_bos_token: bool,
329 drop_thinking: bool,
330 reasoning_effort: Option<ReasoningEffort>,
331) -> Result<String> {
332 encode_owned_messages(
333 messages.to_vec(),
334 thinking_mode,
335 add_bos_token,
336 drop_thinking,
337 Encoding::V4(reasoning_effort),
338 )
339}
340
341pub(super) fn encode_owned_messages(
342 messages: Vec<JsonValue>,
343 thinking_mode: ThinkingMode,
344 add_bos_token: bool,
345 drop_thinking: bool,
346 encoding: Encoding,
347) -> Result<String> {
348 let merged = merge_tool_messages(messages);
349 let mut full = if encoding.is_v41() {
351 merged
352 } else {
353 sort_tool_results_by_call_order(merged)
354 };
355
356 let mut prompt = String::new();
357 if add_bos_token {
358 prompt.push_str(tokens::BOS);
359 }
360
361 let has_tools = full.iter().any(|m| {
363 m.get("tools")
364 .map(|v| match v {
365 JsonValue::Array(a) => !a.is_empty(),
366 JsonValue::Null => false,
367 _ => true,
368 })
369 .unwrap_or(false)
370 });
371 let effective_drop_thinking = drop_thinking && !has_tools;
372
373 if thinking_mode == ThinkingMode::Thinking && effective_drop_thinking {
374 full = if encoding.is_v41() {
375 super::v41::drop_thinking_messages(full)
376 } else {
377 drop_thinking_messages(full)
378 };
379 }
380
381 let last_user_idx = if encoding.is_v41() {
382 super::v41::find_last_user_index(&full)
383 } else {
384 find_last_user_index(&full)
385 };
386 for idx in 0..full.len() {
387 render_message(
388 &mut prompt,
389 idx,
390 &full,
391 thinking_mode,
392 effective_drop_thinking,
393 encoding,
394 last_user_idx,
395 )?;
396 }
397
398 Ok(prompt)
399}
400
401#[derive(Debug)]
403pub struct DeepSeekV4Formatter {
404 thinking_mode: ThinkingMode,
405}
406
407impl DeepSeekV4Formatter {
408 pub fn new(thinking_mode: ThinkingMode) -> Self {
409 Self { thinking_mode }
410 }
411
412 pub fn new_thinking() -> Self {
414 Self::new(ThinkingMode::Thinking)
415 }
416
417 pub fn new_chat() -> Self {
419 Self::new(ThinkingMode::Chat)
420 }
421
422 fn resolve_reasoning_effort(v: Option<&JsonValue>) -> (bool, Option<ReasoningEffort>) {
423 match v.and_then(JsonValue::as_str) {
424 Some("none") => (true, None),
425 Some("max") => (false, Some(ReasoningEffort::Max)),
426 Some("high") | Some("medium") | Some("xhigh") => (false, Some(ReasoningEffort::High)),
427 Some("low") | Some("minimal") => (false, None),
428 None if v.is_none() => (false, Some(ReasoningEffort::High)),
429 _ => {
430 tracing::warn!(
431 value = ?v,
432 "reasoning_effort must be one of \"none\", \"minimal\", \"low\", \"medium\", \"high\", \"xhigh\", \"max\"; ignoring and using API default (high)"
433 );
434 (false, Some(ReasoningEffort::High))
435 }
436 }
437 }
438
439 fn resolve_drop_thinking(
440 args: Option<&std::collections::HashMap<String, serde_json::Value>>,
441 ) -> bool {
442 let Some(args) = args else { return true };
443 let Some(v) = args.get("drop_thinking") else {
444 return true;
445 };
446 if let Some(b) = v.as_bool() {
447 return b;
448 }
449 tracing::warn!(
450 value = ?v,
451 "chat_template_args.drop_thinking must be a bool; ignoring and using default (true)"
452 );
453 true
454 }
455}
456
457impl crate::OAIPromptFormatter for DeepSeekV4Formatter {
458 fn supports_add_generation_prompt(&self) -> bool {
459 true
460 }
461
462 fn render(&self, req: &dyn crate::OAIChatLikeRequest) -> Result<String> {
463 let args = req.chat_template_args();
464 let effort_value = req
465 .reasoning_effort()
466 .map(|value| serde_json::to_value(value).context("serialize reasoning_effort"))
467 .transpose()?
468 .or_else(|| args.and_then(|args| args.get("reasoning_effort").cloned()));
469 let (disable_thinking, reasoning_effort) =
470 Self::resolve_reasoning_effort(effort_value.as_ref());
471 let mut thinking_mode = super::common::resolve_thinking_mode(args, self.thinking_mode);
472 if disable_thinking {
473 thinking_mode = ThinkingMode::Chat;
474 }
475 let drop_thinking = Self::resolve_drop_thinking(args);
476
477 let messages_json = crate::messages_to_json(req)?;
478 crate::reject_unsupported_partial_assistant(&messages_json)?;
479 crate::reject_unsupported_message_tools(&messages_json, &["developer"])?;
480
481 let JsonValue::Array(mut messages_array) = messages_json else {
482 anyhow::bail!("Messages is not an array");
483 };
484
485 normalize_message_contents(&mut messages_array, NormalizeNonText::LeaveUntouched);
486
487 super::common::inject_tools_and_response_format(&mut messages_array, req)?;
488
489 encode_owned_messages(
490 messages_array,
491 thinking_mode,
492 true,
493 drop_thinking,
494 Encoding::V4(reasoning_effort),
495 )
496 }
497}
498
499#[cfg(test)]
500mod tests {
501 use super::*;
502 use serde_json::json;
503
504 #[test]
505 fn test_simple_conversation() {
506 let messages = json!([
507 {"role": "system", "content": "You are a helpful assistant."},
508 {"role": "user", "content": "Hello"},
509 {"role": "assistant", "reasoning_content": "greet", "content": "Hi!"},
510 {"role": "user", "content": "What is 2+2?"}
511 ]);
512 let out =
513 encode_messages(messages.as_array().unwrap(), ThinkingMode::Thinking, true).unwrap();
514 assert!(out.starts_with(tokens::BOS));
515 assert!(out.ends_with(&format!(
516 "{}{}",
517 tokens::ASSISTANT_START,
518 tokens::THINKING_START
519 )));
520 assert!(!out.contains("greet"));
522 }
523
524 #[test]
525 fn test_reasoning_effort_prefixes() {
526 let messages = json!([
527 {"role": "system", "content": "hi"},
528 {"role": "user", "content": "hello"}
529 ]);
530
531 let high = encode_messages_with_options(
532 messages.as_array().unwrap(),
533 ThinkingMode::Thinking,
534 true,
535 true,
536 Some(ReasoningEffort::High),
537 )
538 .unwrap();
539 let max = encode_messages_with_options(
540 messages.as_array().unwrap(),
541 ThinkingMode::Thinking,
542 true,
543 true,
544 Some(ReasoningEffort::Max),
545 )
546 .unwrap();
547 let low = encode_messages_with_options(
548 messages.as_array().unwrap(),
549 ThinkingMode::Thinking,
550 true,
551 true,
552 None,
553 )
554 .unwrap();
555
556 assert_eq!(
557 high,
558 concat!(
559 "<|begin▁of▁sentence|>Reasoning Effort: Absolute maximum with no shortcuts permitted.\n",
560 "You MUST be very thorough in your thinking and comprehensively decompose the problem to resolve the root cause, rigorously stress-testing your logic against all potential paths, edge cases, and adversarial scenarios.\n",
561 "Explicitly write out your entire deliberation process, documenting every intermediate step, considered alternative, and rejected hypothesis to ensure absolutely no assumption is left unchecked.\n\n",
562 "hi<|User|>hello<|Assistant|><think>"
563 )
564 );
565 assert_eq!(
566 max,
567 concat!(
568 "<|begin▁of▁sentence|>Reasoning Effort: Beyond maximum — exhaustive, relentless, and uncompromising.\n",
569 "You MUST reason with the utmost depth and rigor, leaving absolutely nothing to chance: exhaustively decompose the problem into its most fundamental components, trace every causal chain to its root, and resolve the underlying cause rather than any surface symptom.\n",
570 "Do not stop reasoning until you have independently verified the solution from multiple angles and are certain that no assumption remains unchecked and no error remains undiscovered.\n\n",
571 "hi<|User|>hello<|Assistant|><think>"
572 )
573 );
574 assert_eq!(
575 low,
576 "<|begin▁of▁sentence|>hi<|User|>hello<|Assistant|><think>"
577 );
578 }
579
580 #[test]
581 fn test_content_blocks_with_tool_result() {
582 let messages = json!([
588 {"role": "user", "content": "call tool"},
589 {"role": "assistant", "content": "", "tool_calls": [{
590 "id": "c1", "type": "function",
591 "function": {"name": "f", "arguments": "{}"}
592 }]},
593 {"role": "tool", "tool_call_id": "c1", "content": "RESULT"},
594 {"role": "user", "content": "thanks"}
595 ]);
596 let out = encode_messages(messages.as_array().unwrap(), ThinkingMode::Chat, true).unwrap();
597 assert!(
598 out.contains("<tool_result>RESULT</tool_result>\n\nthanks"),
599 "expected tool_result block followed by 'thanks' in the merged user turn, got:\n{}",
600 out
601 );
602 }
603
604 #[test]
605 fn test_user_task_preserved_when_merged_after_tool_result() {
606 let messages = json!([
607 {"role": "assistant", "content": "", "tool_calls": [{
608 "id": "c1", "type": "function",
609 "function": {"name": "search", "arguments": "{}"}
610 }]},
611 {"role": "tool", "tool_call_id": "c1", "content": "RESULT"},
612 {"role": "user", "content": "Search", "task": "action"},
613 {"role": "assistant", "content": "OK"}
614 ]);
615
616 let out = encode_messages(messages.as_array().unwrap(), ThinkingMode::Chat, true).unwrap();
617 assert!(
618 out.contains(&format!(
619 "{}Search{}{}{}OK",
620 "<tool_result>RESULT</tool_result>\n\n",
621 tokens::ASSISTANT_START,
622 tokens::THINKING_END,
623 tokens::TASK_ACTION
624 )),
625 "expected merged user text to keep the action task transition, got:\n{}",
626 out
627 );
628 }
629
630 #[test]
631 fn test_drop_thinking_auto_disable_when_tools_present() {
632 let messages = json!([
633 {"role": "system", "content": "s", "tools": [{
634 "type": "function",
635 "function": {"name": "f", "description": "", "parameters": {"type": "object", "properties": {}}}
636 }]},
637 {"role": "user", "content": "hi"},
638 {"role": "assistant", "reasoning_content": "PRIOR_REASONING", "content": "reply"},
639 {"role": "user", "content": "again"}
640 ]);
641 let out =
642 encode_messages(messages.as_array().unwrap(), ThinkingMode::Thinking, true).unwrap();
643 assert!(out.contains("PRIOR_REASONING"));
645 }
646
647 #[test]
658 fn test_assistant_reasoning_preserved_when_no_user_in_history() {
659 let messages = json!([
660 {"role": "system", "content": "sys"},
661 {"role": "assistant", "content": "hello", "reasoning_content": "REASONING_BLOCK"}
662 ]);
663 let out =
664 encode_messages(messages.as_array().unwrap(), ThinkingMode::Thinking, true).unwrap();
665 assert_eq!(
666 out, "<|begin▁of▁sentence|>sysREASONING_BLOCK</think>hello<|end▁of▁sentence|>",
667 "Output must match Python reference byte-for-byte when no user/developer in history"
668 );
669 }
670
671 #[test]
679 fn test_to_json_preserves_spacing_past_escaped_backslash() {
680 let v = json!({"path": "\\", "count": 5});
681 let got = to_json(&v);
682 assert_eq!(
683 got, r#"{"path": "\\", "count": 5}"#,
684 "to_json must match Python's json.dumps formatting past an escaped backslash"
685 );
686 }
687
688 #[test]
689 fn test_resolve_drop_thinking_warns_on_malformed_value() {
690 use std::collections::HashMap;
691 let mut args = HashMap::new();
693 args.insert(
694 "drop_thinking".to_string(),
695 serde_json::Value::String("false".to_string()),
696 );
697 assert!(DeepSeekV4Formatter::resolve_drop_thinking(Some(&args)));
698 let malformed = serde_json::Value::String("HIGH".to_string());
700 assert_eq!(
701 DeepSeekV4Formatter::resolve_reasoning_effort(Some(&malformed)),
702 (false, Some(ReasoningEffort::High))
703 );
704 }
705
706 #[test]
707 fn test_resolve_thinking_mode_honors_enable_thinking() {
708 use std::collections::HashMap;
709 let mut args = HashMap::new();
710 args.insert(
711 "enable_thinking".to_string(),
712 serde_json::Value::Bool(false),
713 );
714 assert_eq!(
715 super::super::common::resolve_thinking_mode(Some(&args), ThinkingMode::Thinking),
716 ThinkingMode::Chat
717 );
718 args.insert("enable_thinking".to_string(), serde_json::Value::Bool(true));
719 assert_eq!(
720 super::super::common::resolve_thinking_mode(Some(&args), ThinkingMode::Thinking),
721 ThinkingMode::Thinking
722 );
723 }
724
725 struct MockRequest {
726 messages: JsonValue,
727 typed: Option<Vec<dynamo_protocols::types::ChatCompletionRequestMessage>>,
728 chat_template_args: Option<std::collections::HashMap<String, JsonValue>>,
729 reasoning_effort: Option<JsonValue>,
730 tools: Option<JsonValue>,
731 tool_choice: Option<JsonValue>,
732 response_format: Option<JsonValue>,
733 }
734
735 impl MockRequest {
736 fn new(messages: JsonValue) -> Self {
737 Self {
738 messages,
739 typed: None,
740 chat_template_args: None,
741 reasoning_effort: None,
742 tools: None,
743 tool_choice: None,
744 response_format: None,
745 }
746 }
747
748 fn with_chat_template_args(
749 mut self,
750 args: std::collections::HashMap<String, JsonValue>,
751 ) -> Self {
752 self.chat_template_args = Some(args);
753 self
754 }
755
756 fn with_reasoning_effort(mut self, reasoning_effort: JsonValue) -> Self {
757 self.reasoning_effort = Some(reasoning_effort);
758 self
759 }
760
761 fn with_tools(mut self, tools: JsonValue) -> Self {
762 self.tools = Some(tools);
763 self
764 }
765
766 fn with_tool_choice(mut self, tool_choice: JsonValue) -> Self {
767 self.tool_choice = Some(tool_choice);
768 self
769 }
770
771 fn with_response_format(mut self, response_format: JsonValue) -> Self {
772 self.response_format = Some(response_format);
773 self
774 }
775 }
776
777 impl crate::OAIChatLikeRequest for MockRequest {
778 fn model(&self) -> String {
779 "deepseek-v4".to_string()
780 }
781
782 fn messages(&self) -> minijinja::value::Value {
783 assert!(
784 self.typed.is_none(),
785 "typed requests must skip MiniJinja conversion"
786 );
787 minijinja::value::Value::from_serialize(&self.messages)
788 }
789
790 fn typed_messages(
791 &self,
792 ) -> Option<&[dynamo_protocols::types::ChatCompletionRequestMessage]> {
793 self.typed.as_deref()
794 }
795
796 fn should_add_generation_prompt(&self) -> bool {
797 true
798 }
799
800 fn chat_template_args(
801 &self,
802 ) -> Option<&std::collections::HashMap<String, serde_json::Value>> {
803 self.chat_template_args.as_ref()
804 }
805
806 fn reasoning_effort(&self) -> Option<minijinja::value::Value> {
807 self.reasoning_effort
808 .as_ref()
809 .map(minijinja::value::Value::from_serialize)
810 }
811
812 fn tools(&self) -> Option<minijinja::value::Value> {
813 self.tools
814 .as_ref()
815 .map(minijinja::value::Value::from_serialize)
816 }
817
818 fn tool_choice(&self) -> Option<minijinja::value::Value> {
819 self.tool_choice
820 .as_ref()
821 .map(minijinja::value::Value::from_serialize)
822 }
823
824 fn response_format(&self) -> Option<minijinja::value::Value> {
825 self.response_format
826 .as_ref()
827 .map(minijinja::value::Value::from_serialize)
828 }
829 }
830
831 #[test]
832 fn typed_messages_match_value_messages() {
833 use crate::OAIPromptFormatter;
834 let messages = json!([
835 {"role": "system", "content": "Use tools. 中文 🦀"},
836 {"role": "user", "content": [{"type": "text", "text": "weather?"}]},
837 {"role": "assistant", "content": null, "reasoning_content": "check",
838 "tool_calls": [{"id": "call_1", "type": "function",
839 "function": {"name": "weather", "arguments": "{\"city\":\"東京\"}"}}]},
840 {"role": "tool", "tool_call_id": "call_1", "content": "sunny"},
841 {"role": "user", "content": "explain"}
842 ]);
843 let mut typed = MockRequest::new(messages.clone());
844 typed.typed = Some(serde_json::from_value(messages.clone()).unwrap());
845 let value = MockRequest::new(messages);
846 for formatter in [
847 DeepSeekV4Formatter::new_thinking(),
848 DeepSeekV4Formatter::new_chat(),
849 ] {
850 assert_eq!(
851 formatter.render(&typed).unwrap(),
852 formatter.render(&value).unwrap()
853 );
854 }
855 }
856
857 fn weather_tool() -> JsonValue {
858 json!([{
859 "type": "function",
860 "function": {
861 "name": "get_current_weather",
862 "description": "Get the current weather in a given location",
863 "parameters": {
864 "type": "object",
865 "properties": {"location": {"type": "string"}},
866 "required": ["location"]
867 }
868 }
869 }])
870 }
871
872 #[test]
873 fn test_formatter_rejects_unsupported_partial_assistant() {
874 use crate::OAIPromptFormatter;
875
876 let request = MockRequest::new(json!([
877 {"role": "user", "content": "Continue"},
878 {"role": "assistant", "content": "prefix", "partial": true}
879 ]));
880 let error = DeepSeekV4Formatter::new_thinking()
881 .render(&request)
882 .unwrap_err();
883
884 assert!(matches!(
885 error.downcast_ref::<crate::PromptRenderError>(),
886 Some(crate::PromptRenderError::InvalidRequest(message))
887 if message.contains("`partial: true` is not supported")
888 ));
889 }
890
891 #[test]
892 fn test_formatter_rejects_system_tools_before_injection() {
893 use crate::OAIPromptFormatter;
894
895 let request = MockRequest::new(json!([
896 {"role": "system", "tools": [
897 {"type": "function", "function": {"name": "dynamic_tool"}}
898 ]},
899 {"role": "user", "content": "Use a tool"}
900 ]))
901 .with_tools(weather_tool());
902 let error = DeepSeekV4Formatter::new_thinking()
903 .render(&request)
904 .unwrap_err();
905
906 assert!(matches!(
907 error.downcast_ref::<crate::PromptRenderError>(),
908 Some(crate::PromptRenderError::InvalidRequest(message))
909 if message.contains("message-level `tools`") && message.contains("system")
910 ));
911 }
912
913 #[test]
914 fn test_formatter_preserves_developer_tools_with_top_level_tools() {
915 use crate::OAIPromptFormatter;
916
917 let request = MockRequest::new(json!([
918 {"role": "developer", "content": "Use a tool", "tools": [
919 {"type": "function", "function": {"name": "developer_tool"}}
920 ]}
921 ]))
922 .with_tools(weather_tool());
923 let rendered = DeepSeekV4Formatter::new_thinking()
924 .render(&request)
925 .unwrap();
926
927 assert!(rendered.contains("developer_tool"));
928 assert!(rendered.contains("get_current_weather"));
929 }
930
931 #[test]
932 fn test_render_tool_choice_none_strips_tools_keeps_response_format() {
933 use crate::OAIPromptFormatter;
934
935 let req = MockRequest::new(json!([
936 {"role": "system", "content": "sys"},
937 {"role": "user", "content": "weather in Boston?"}
938 ]))
939 .with_tools(weather_tool())
940 .with_tool_choice(json!("none"))
941 .with_response_format(json!({"type": "json_object"}));
942
943 let formatter = DeepSeekV4Formatter::new_chat();
944 let out = formatter.render(&req).unwrap();
945
946 assert!(
947 !out.contains("## Tools"),
948 "tool_choice=none must strip the tools block, got: {out}"
949 );
950 assert!(
951 !out.contains("get_current_weather"),
952 "tool schema leaked into prompt despite tool_choice=none: {out}"
953 );
954 assert!(
955 out.contains("## Response Format"),
956 "response_format must survive tool_choice=none: {out}"
957 );
958 }
959
960 #[test]
961 fn test_render_tool_choice_auto_keeps_tools() {
962 use crate::OAIPromptFormatter;
963
964 let req = MockRequest::new(json!([
965 {"role": "system", "content": "sys"},
966 {"role": "user", "content": "weather in Boston?"}
967 ]))
968 .with_tools(weather_tool())
969 .with_tool_choice(json!("auto"));
970
971 let formatter = DeepSeekV4Formatter::new_chat();
972 let out = formatter.render(&req).unwrap();
973
974 assert!(out.contains("## Tools"));
975 assert!(out.contains("get_current_weather"));
976 }
977
978 #[test]
979 fn test_render_absent_tool_choice_keeps_tools() {
980 use crate::OAIPromptFormatter;
981
982 let req = MockRequest::new(json!([
983 {"role": "system", "content": "sys"},
984 {"role": "user", "content": "weather in Boston?"}
985 ]))
986 .with_tools(weather_tool());
987
988 let formatter = DeepSeekV4Formatter::new_chat();
989 let out = formatter.render(&req).unwrap();
990
991 assert!(out.contains("## Tools"));
992 assert!(out.contains("get_current_weather"));
993 }
994
995 #[test]
996 fn test_resolve_reasoning_effort_accepts_full_range() {
997 let effort = |v: &str| {
998 let value = json!(v);
999 DeepSeekV4Formatter::resolve_reasoning_effort(Some(&value))
1000 };
1001
1002 assert_eq!(effort("max"), (false, Some(ReasoningEffort::Max)));
1003 assert_eq!(effort("xhigh"), (false, Some(ReasoningEffort::High)));
1004 assert_eq!(effort("high"), (false, Some(ReasoningEffort::High)));
1005 assert_eq!(effort("minimal"), (false, None));
1006 assert_eq!(effort("low"), (false, None));
1007 assert_eq!(effort("medium"), (false, Some(ReasoningEffort::High)));
1008 assert_eq!(effort("none"), (true, None));
1009 assert_eq!(effort("bogus"), (false, Some(ReasoningEffort::High)));
1010 assert_eq!(
1011 DeepSeekV4Formatter::resolve_reasoning_effort(None),
1012 (false, Some(ReasoningEffort::High))
1013 );
1014 }
1015
1016 #[test]
1017 fn test_render_leaves_null_assistant_tool_content_empty() {
1018 use crate::OAIPromptFormatter;
1019
1020 let req = MockRequest::new(json!([
1021 {"role": "user", "content": "call tool"},
1022 {"role": "assistant", "content": null, "tool_calls": [{
1023 "id": "c1", "type": "function",
1024 "function": {"name": "f", "arguments": "{}"}
1025 }]}
1026 ]));
1027
1028 let formatter = DeepSeekV4Formatter::new_chat();
1029 let out = formatter.render(&req).unwrap();
1030
1031 assert!(out.contains(&format!(
1032 "<{}{}>",
1033 tokens::DSML_TOKEN,
1034 TOOL_CALLS_BLOCK_NAME
1035 )));
1036 assert!(!out.contains("null"));
1037 }
1038
1039 #[test]
1040 fn test_render_wires_reasoning_effort_from_chat_template_args() {
1041 use crate::OAIPromptFormatter;
1042 use std::collections::HashMap;
1043
1044 for (effort, expected) in [
1045 ("high", REASONING_EFFORT_HIGH),
1046 ("max", REASONING_EFFORT_MAX),
1047 ] {
1048 let mut args = HashMap::new();
1049 args.insert("reasoning_effort".to_string(), json!(effort));
1050
1051 let req = MockRequest::new(json!([
1052 {"role": "system", "content": "sys"},
1053 {"role": "user", "content": "hi"}
1054 ]))
1055 .with_chat_template_args(args);
1056
1057 let formatter = DeepSeekV4Formatter::new_thinking();
1058 let out = formatter.render(&req).unwrap();
1059
1060 assert!(out.starts_with(tokens::BOS));
1061 assert!(
1062 out[tokens::BOS.len()..].starts_with(expected),
1063 "{effort} preamble should appear after BOS, got:\n{out}"
1064 );
1065 }
1066 }
1067
1068 #[test]
1069 fn test_render_wires_top_level_reasoning_effort_and_none_disables_thinking() {
1070 use crate::OAIPromptFormatter;
1071
1072 let formatter = DeepSeekV4Formatter::new_thinking();
1073 for (effort, expected_prefix) in [
1074 ("high", "Reasoning Effort: Absolute maximum"),
1075 ("max", "Reasoning Effort: Beyond maximum"),
1076 ] {
1077 let req: dynamo_protocols::types::CreateChatCompletionRequest =
1078 serde_json::from_value(json!({
1079 "model": "deepseek-v4",
1080 "messages": [{"role": "user", "content": "hi"}],
1081 "reasoning_effort": effort
1082 }))
1083 .unwrap();
1084 let out = formatter.render(&req).unwrap();
1085
1086 assert!(
1087 out[tokens::BOS.len()..].starts_with(expected_prefix),
1088 "top-level {effort} did not select its prefix: {out}"
1089 );
1090 assert!(out.ends_with(tokens::THINKING_START));
1091 }
1092
1093 let req: dynamo_protocols::types::CreateChatCompletionRequest =
1094 serde_json::from_value(json!({
1095 "model": "deepseek-v4",
1096 "messages": [{"role": "user", "content": "hi"}],
1097 "reasoning_effort": "none"
1098 }))
1099 .unwrap();
1100 let out = formatter.render(&req).unwrap();
1101
1102 assert_eq!(
1103 out,
1104 "<|begin▁of▁sentence|><|User|>hi<|Assistant|></think>"
1105 );
1106 }
1107
1108 #[test]
1109 fn test_top_level_reasoning_effort_precedes_template_argument() {
1110 use crate::OAIPromptFormatter;
1111 use std::collections::HashMap;
1112
1113 let mut args = HashMap::new();
1114 args.insert("reasoning_effort".to_string(), json!("max"));
1115 let req = MockRequest::new(json!([{"role": "user", "content": "hi"}]))
1116 .with_chat_template_args(args)
1117 .with_reasoning_effort(json!("low"));
1118
1119 let out = DeepSeekV4Formatter::new_thinking().render(&req).unwrap();
1120
1121 assert_eq!(
1122 out,
1123 "<|begin▁of▁sentence|><|User|>hi<|Assistant|><think>"
1124 );
1125 }
1126
1127 #[test]
1128 fn test_render_drop_thinking_override_from_chat_template_args() {
1129 use crate::OAIPromptFormatter;
1130 use std::collections::HashMap;
1131
1132 let messages = json!([
1133 {"role": "user", "content": "first"},
1134 {"role": "assistant", "reasoning_content": "PRIOR", "content": "reply"},
1135 {"role": "user", "content": "again"}
1136 ]);
1137
1138 let req_default = MockRequest::new(messages.clone());
1140 let formatter = DeepSeekV4Formatter::new_thinking();
1141 let out_default = formatter.render(&req_default).unwrap();
1142 assert!(
1143 !out_default.contains("PRIOR"),
1144 "default drop_thinking=true should strip prior reasoning, got:\n{}",
1145 out_default
1146 );
1147
1148 let mut args = HashMap::new();
1150 args.insert("drop_thinking".to_string(), json!(false));
1151 let req_keep = MockRequest::new(messages).with_chat_template_args(args);
1152 let out_keep = formatter.render(&req_keep).unwrap();
1153 assert!(
1154 out_keep.contains("PRIOR"),
1155 "drop_thinking=false override should preserve prior reasoning, got:\n{}",
1156 out_keep
1157 );
1158 }
1159
1160 #[test]
1166 fn test_developer_only_conversation_renders_developer_content() {
1167 let messages = json!([
1168 {"role": "system", "content": "sys"},
1169 {"role": "developer", "content": "x"},
1170 {"role": "assistant", "reasoning_content": "R", "content": "ok"}
1171 ]);
1172 let out =
1173 encode_messages(messages.as_array().unwrap(), ThinkingMode::Thinking, true).unwrap();
1174 assert!(
1175 out.contains("x"),
1176 "developer content should appear in output, got:\n{}",
1177 out
1178 );
1179 }
1180
1181 #[test]
1182 fn test_developer_as_last_user_index_controls_reasoning_cutoff() {
1183 let messages = json!([
1188 {"role": "user", "content": "a"},
1189 {"role": "assistant", "reasoning_content": "FIRST", "content": "r1"},
1190 {"role": "developer", "content": "y"},
1191 {"role": "assistant", "reasoning_content": "SECOND", "content": "r2"}
1192 ]);
1193 let out =
1194 encode_messages(messages.as_array().unwrap(), ThinkingMode::Thinking, true).unwrap();
1195 assert!(
1196 !out.contains("FIRST"),
1197 "reasoning before last user/developer (idx 1 < 2) should be stripped, got:\n{}",
1198 out
1199 );
1200 assert!(
1201 out.contains("SECOND"),
1202 "reasoning at/after last user/developer (idx 3 > 2) should survive, got:\n{}",
1203 out
1204 );
1205 }
1206}