Skip to main content

dynamo_renderer/deepseek/
v4.rs

1// SPDX-FileCopyrightText: Copyright (c) 2024-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
2// SPDX-License-Identifier: Apache-2.0
3
4//! DeepSeek V4 native prompt formatting
5//!
6//! Native Rust port of DeepSeek V4's chat encoding (encoding_dsv4.py).
7//!
8//! Reference: DeepSeek-V4-Pro/encoding/encoding_dsv4.py
9
10use anyhow::{Context, Result};
11use serde_json::Value as JsonValue;
12
13use super::common::{
14    NormalizeNonText, REASONING_EFFORT_HIGH, REASONING_EFFORT_MAX, RESPONSE_FORMAT_TEMPLATE,
15    TOOL_CALLS_BLOCK_NAME, TOOLS_TEMPLATE, drop_thinking_messages, encode_arguments_to_dsml,
16    find_last_user_index, merge_tool_messages, normalize_message_contents, render_tools,
17    sort_tool_results_by_call_order, task_token, to_json,
18};
19pub use super::common::{ReasoningEffort, ThinkingMode, tokens};
20
21#[derive(Clone, Copy)]
22pub(super) enum Encoding {
23    V4(Option<ReasoningEffort>),
24    V41(u8),
25}
26
27impl Encoding {
28    fn is_v41(self) -> bool {
29        matches!(self, Self::V41(_))
30    }
31
32    fn tag(self, v4: &'static str, v41: &'static str) -> &'static str {
33        if self.is_v41() { v41 } else { v4 }
34    }
35
36    fn reasoning_prefix(self) -> String {
37        match self {
38            Self::V4(Some(ReasoningEffort::High)) => REASONING_EFFORT_HIGH.to_string(),
39            Self::V4(Some(ReasoningEffort::Max)) => REASONING_EFFORT_MAX.to_string(),
40            Self::V4(None) => String::new(),
41            Self::V41(effort) => format!(
42                "Reasoning Effort: {effort} (range 1-100, the higher the value, the more thorough the reasoning)\n\n"
43            ),
44        }
45    }
46
47    fn render_tools(self, tools: &[JsonValue]) -> String {
48        let template = if self.is_v41() {
49            TOOLS_TEMPLATE
50                .replace("{dsml_token}tool_calls", "{dsml_token} calls")
51                .replace("{dsml_token}invoke", "{dsml_token} invoke")
52                .replace("{dsml_token}parameter", "{dsml_token} parameter")
53        } else {
54            TOOLS_TEMPLATE.to_string()
55        };
56        render_tools(&template, tools)
57    }
58}
59
60/// Render a single message at the given index.
61fn render_message(
62    index: usize,
63    messages: &[JsonValue],
64    thinking_mode: ThinkingMode,
65    drop_thinking: bool,
66    encoding: Encoding,
67    last_user_idx: Option<usize>,
68) -> Result<String> {
69    let msg = &messages[index];
70
71    let role = msg
72        .get("role")
73        .and_then(|r| r.as_str())
74        .context("Missing 'role' field")?;
75
76    let mut prompt = String::new();
77
78    if encoding.is_v41()
79        && (role == "system" || (index == 0 && thinking_mode == ThinkingMode::Thinking))
80    {
81        prompt.push_str("<|System|>");
82    }
83    if index == 0 && thinking_mode == ThinkingMode::Thinking {
84        prompt.push_str(&encoding.reasoning_prefix());
85    }
86
87    match role {
88        "system" => {
89            let content = msg.get("content").and_then(|c| c.as_str()).unwrap_or("");
90            prompt.push_str(content);
91            if let Some(tools) = msg.get("tools").and_then(|t| t.as_array()) {
92                prompt.push_str("\n\n");
93                prompt.push_str(&encoding.render_tools(tools));
94            }
95            if let Some(response_format) = msg.get("response_format") {
96                prompt.push_str("\n\n");
97                prompt.push_str(
98                    &RESPONSE_FORMAT_TEMPLATE.replace("{schema}", &to_json(response_format)),
99                );
100            }
101        }
102
103        "developer" => {
104            let content = msg
105                .get("content")
106                .and_then(|c| c.as_str())
107                .filter(|s| !s.is_empty())
108                .context("Developer role requires content")?;
109
110            let mut content_developer = String::from(tokens::USER_START);
111            content_developer.push_str(content);
112
113            if let Some(tools) = msg.get("tools").and_then(|t| t.as_array()) {
114                content_developer.push_str("\n\n");
115                content_developer.push_str(&encoding.render_tools(tools));
116            }
117            if let Some(response_format) = msg.get("response_format") {
118                content_developer.push_str("\n\n");
119                content_developer.push_str(
120                    &RESPONSE_FORMAT_TEMPLATE.replace("{schema}", &to_json(response_format)),
121                );
122            }
123            prompt.push_str(&content_developer);
124        }
125
126        "user" => {
127            prompt.push_str(tokens::USER_START);
128            if let Some(blocks) = msg.get("content_blocks").and_then(|b| b.as_array()) {
129                let mut parts: Vec<String> = Vec::with_capacity(blocks.len());
130                for block in blocks {
131                    let block_type = block.get("type").and_then(|v| v.as_str()).unwrap_or("");
132                    match block_type {
133                        "text" => {
134                            let text = block.get("text").and_then(|v| v.as_str()).unwrap_or("");
135                            parts.push(text.to_string());
136                        }
137                        "tool_result" => {
138                            let rendered = render_tool_result_content(
139                                block.get("content").unwrap_or(&JsonValue::Null),
140                            );
141                            parts.push(format!("<tool_result>{}</tool_result>", rendered));
142                        }
143                        other => {
144                            parts.push(format!("[Unsupported {}]", other));
145                        }
146                    }
147                }
148                prompt.push_str(&parts.join("\n\n"));
149            } else {
150                let content = msg.get("content").and_then(|c| c.as_str()).unwrap_or("");
151                prompt.push_str(content);
152            }
153        }
154
155        "latest_reminder" => {
156            let content = msg.get("content").and_then(|c| c.as_str()).unwrap_or("");
157            prompt.push_str(tokens::LATEST_REMINDER);
158            prompt.push_str(content);
159        }
160
161        "tool" => {
162            anyhow::bail!(
163                "deepseek_v4 merges tool messages into user; preprocess with merge_tool_messages()"
164            );
165        }
166
167        "assistant" => {
168            let content = msg.get("content").and_then(|c| c.as_str()).unwrap_or("");
169            let reasoning = msg
170                .get("reasoning_content")
171                .and_then(|c| c.as_str())
172                .unwrap_or("");
173            let wo_eos = msg.get("wo_eos").and_then(|v| v.as_bool()).unwrap_or(false);
174
175            let prev_has_task = index > 0
176                && messages[index - 1]
177                    .get("task")
178                    .map(|v| !v.is_null())
179                    .unwrap_or(false);
180
181            let mut thinking_part = String::new();
182            if thinking_mode == ThinkingMode::Thinking && !prev_has_task {
183                let render_thinking = !drop_thinking || last_user_idx.is_none_or(|u| index > u);
184                if render_thinking {
185                    thinking_part.push_str(reasoning);
186                    thinking_part.push_str(tokens::THINKING_END);
187                }
188            }
189
190            prompt.push_str(&thinking_part);
191            prompt.push_str(content);
192
193            if let Some(tool_calls) = msg.get("tool_calls").and_then(|t| t.as_array())
194                && !tool_calls.is_empty()
195            {
196                prompt.push_str("\n\n");
197                prompt.push_str(&format!(
198                    "<{}{}>\n",
199                    tokens::DSML_TOKEN,
200                    encoding.tag(TOOL_CALLS_BLOCK_NAME, " calls")
201                ));
202
203                let mut invocations = Vec::with_capacity(tool_calls.len());
204                for tc in tool_calls {
205                    // Accept both OpenAI-format (nested `function`) and internal
206                    // `{name, arguments}` shape, matching Python's `tool_calls_from_openai_format`.
207                    let fn_obj = tc.get("function").unwrap_or(tc);
208                    let name = fn_obj
209                        .get("name")
210                        .and_then(|n| n.as_str())
211                        .context("Missing tool call name")?;
212                    let arguments = if encoding.is_v41() {
213                        super::v41::encode_arguments(fn_obj)?
214                    } else {
215                        encode_arguments_to_dsml(fn_obj)?
216                    };
217                    invocations.push(format!(
218                        "<{}{} name=\"{}\">\n{}\n</{}{}>",
219                        tokens::DSML_TOKEN,
220                        encoding.tag("invoke", " invoke"),
221                        name,
222                        arguments,
223                        tokens::DSML_TOKEN,
224                        encoding.tag("invoke", " invoke")
225                    ));
226                }
227                prompt.push_str(&invocations.join("\n"));
228                prompt.push_str(&format!(
229                    "\n</{}{}>",
230                    tokens::DSML_TOKEN,
231                    encoding.tag(TOOL_CALLS_BLOCK_NAME, " calls")
232                ));
233            }
234
235            if !wo_eos {
236                prompt.push_str(tokens::EOS);
237            }
238        }
239
240        other => anyhow::bail!("Unknown role: {}", other),
241    }
242
243    // Early return if the next message is not assistant/latest_reminder — no transition appended.
244    if index + 1 < messages.len() {
245        let next_role = messages[index + 1].get("role").and_then(|r| r.as_str());
246        if !matches!(next_role, Some("assistant") | Some("latest_reminder")) {
247            return Ok(prompt);
248        }
249    }
250
251    // Transition tokens based on task field and role.
252    let task = msg.get("task").and_then(|v| v.as_str());
253    if let Some(task) = task {
254        let sp = task_token(task).with_context(|| format!("Invalid task: '{}'", task))?;
255        if task != "action" {
256            prompt.push_str(sp);
257        } else {
258            prompt.push_str(tokens::ASSISTANT_START);
259            prompt.push_str(if thinking_mode != ThinkingMode::Thinking {
260                tokens::THINKING_END
261            } else {
262                tokens::THINKING_START
263            });
264            prompt.push_str(sp);
265        }
266    } else if matches!(role, "user" | "developer")
267        || (encoding.is_v41() && role == "system" && index > 0)
268    {
269        prompt.push_str(tokens::ASSISTANT_START);
270        let seed_thinking = thinking_mode == ThinkingMode::Thinking
271            && (!drop_thinking || last_user_idx.is_none_or(|u| index >= u));
272        prompt.push_str(if seed_thinking {
273            tokens::THINKING_START
274        } else {
275            tokens::THINKING_END
276        });
277    }
278
279    Ok(prompt)
280}
281
282/// Render a tool_result `content` payload (string or content-block list).
283fn render_tool_result_content(content: &JsonValue) -> String {
284    match content {
285        JsonValue::String(s) => s.clone(),
286        JsonValue::Array(items) => {
287            let mut parts: Vec<String> = Vec::with_capacity(items.len());
288            for item in items {
289                let item_type = item.get("type").and_then(|v| v.as_str()).unwrap_or("");
290                if item_type == "text" {
291                    parts.push(
292                        item.get("text")
293                            .and_then(|v| v.as_str())
294                            .unwrap_or("")
295                            .to_string(),
296                    );
297                } else {
298                    parts.push(format!("[Unsupported {}]", item_type));
299                }
300            }
301            parts.join("\n\n")
302        }
303        JsonValue::Null => String::new(),
304        _ => to_json(content),
305    }
306}
307
308/// Encode messages to prompt string with default options.
309///
310/// Equivalent to `encode_messages_with_options(.., drop_thinking=true, reasoning_effort=None)`.
311pub fn encode_messages(
312    messages: &[JsonValue],
313    thinking_mode: ThinkingMode,
314    add_bos_token: bool,
315) -> Result<String> {
316    encode_messages_with_options(messages, thinking_mode, add_bos_token, true, None)
317}
318
319/// Encode messages to prompt string.
320///
321/// # Arguments
322/// * `messages` - Array of messages in OpenAI format
323/// * `thinking_mode` - Chat or Thinking
324/// * `add_bos_token` - Whether to prepend BOS token
325/// * `drop_thinking` - Drop reasoning_content from earlier turns (auto-disabled if tools present)
326/// * `reasoning_effort` - Optional reasoning effort level (High and Max prepend distinct verbatim blocks)
327pub fn encode_messages_with_options(
328    messages: &[JsonValue],
329    thinking_mode: ThinkingMode,
330    add_bos_token: bool,
331    drop_thinking: bool,
332    reasoning_effort: Option<ReasoningEffort>,
333) -> Result<String> {
334    encode_messages_with_encoding(
335        messages,
336        thinking_mode,
337        add_bos_token,
338        drop_thinking,
339        Encoding::V4(reasoning_effort),
340    )
341}
342
343pub(super) fn encode_messages_with_encoding(
344    messages: &[JsonValue],
345    thinking_mode: ThinkingMode,
346    add_bos_token: bool,
347    drop_thinking: bool,
348    encoding: Encoding,
349) -> Result<String> {
350    let merged = merge_tool_messages(messages);
351    // V4.1 orders source messages before merging, using the same routine as its media hook.
352    let mut full = if encoding.is_v41() {
353        merged
354    } else {
355        sort_tool_results_by_call_order(merged)
356    };
357
358    let mut prompt = String::new();
359    if add_bos_token {
360        prompt.push_str(tokens::BOS);
361    }
362
363    // Auto-disable drop_thinking when any message carries a `tools` field.
364    let has_tools = full.iter().any(|m| {
365        m.get("tools")
366            .map(|v| match v {
367                JsonValue::Array(a) => !a.is_empty(),
368                JsonValue::Null => false,
369                _ => true,
370            })
371            .unwrap_or(false)
372    });
373    let effective_drop_thinking = drop_thinking && !has_tools;
374
375    if thinking_mode == ThinkingMode::Thinking && effective_drop_thinking {
376        full = if encoding.is_v41() {
377            super::v41::drop_thinking_messages(full)
378        } else {
379            drop_thinking_messages(full)
380        };
381    }
382
383    let last_user_idx = if encoding.is_v41() {
384        super::v41::find_last_user_index(&full)
385    } else {
386        find_last_user_index(&full)
387    };
388    for idx in 0..full.len() {
389        let part = render_message(
390            idx,
391            &full,
392            thinking_mode,
393            effective_drop_thinking,
394            encoding,
395            last_user_idx,
396        )?;
397        prompt.push_str(&part);
398    }
399
400    Ok(prompt)
401}
402
403/// DeepSeek V4 Prompt Formatter
404#[derive(Debug)]
405pub struct DeepSeekV4Formatter {
406    thinking_mode: ThinkingMode,
407}
408
409impl DeepSeekV4Formatter {
410    pub fn new(thinking_mode: ThinkingMode) -> Self {
411        Self { thinking_mode }
412    }
413
414    /// Create formatter with thinking mode enabled (default for DSV4)
415    pub fn new_thinking() -> Self {
416        Self::new(ThinkingMode::Thinking)
417    }
418
419    /// Create formatter with chat mode
420    pub fn new_chat() -> Self {
421        Self::new(ThinkingMode::Chat)
422    }
423
424    fn resolve_reasoning_effort(v: Option<&JsonValue>) -> (bool, Option<ReasoningEffort>) {
425        match v.and_then(JsonValue::as_str) {
426            Some("none") => (true, None),
427            Some("max") => (false, Some(ReasoningEffort::Max)),
428            Some("high") | Some("medium") | Some("xhigh") => (false, Some(ReasoningEffort::High)),
429            Some("low") | Some("minimal") => (false, None),
430            None if v.is_none() => (false, Some(ReasoningEffort::High)),
431            _ => {
432                tracing::warn!(
433                    value = ?v,
434                    "reasoning_effort must be one of \"none\", \"minimal\", \"low\", \"medium\", \"high\", \"xhigh\", \"max\"; ignoring and using API default (high)"
435                );
436                (false, Some(ReasoningEffort::High))
437            }
438        }
439    }
440
441    fn resolve_drop_thinking(
442        args: Option<&std::collections::HashMap<String, serde_json::Value>>,
443    ) -> bool {
444        let Some(args) = args else { return true };
445        let Some(v) = args.get("drop_thinking") else {
446            return true;
447        };
448        if let Some(b) = v.as_bool() {
449            return b;
450        }
451        tracing::warn!(
452            value = ?v,
453            "chat_template_args.drop_thinking must be a bool; ignoring and using default (true)"
454        );
455        true
456    }
457}
458
459impl crate::OAIPromptFormatter for DeepSeekV4Formatter {
460    fn supports_add_generation_prompt(&self) -> bool {
461        true
462    }
463
464    fn render(&self, req: &dyn crate::OAIChatLikeRequest) -> Result<String> {
465        let args = req.chat_template_args();
466        let effort_value = req
467            .reasoning_effort()
468            .map(|value| serde_json::to_value(value).context("serialize reasoning_effort"))
469            .transpose()?
470            .or_else(|| args.and_then(|args| args.get("reasoning_effort").cloned()));
471        let (disable_thinking, reasoning_effort) =
472            Self::resolve_reasoning_effort(effort_value.as_ref());
473        let mut thinking_mode = super::common::resolve_thinking_mode(args, self.thinking_mode);
474        if disable_thinking {
475            thinking_mode = ThinkingMode::Chat;
476        }
477        let drop_thinking = Self::resolve_drop_thinking(args);
478
479        let messages_value = req.messages();
480        let messages_json =
481            serde_json::to_value(&messages_value).context("Failed to convert messages to JSON")?;
482        crate::reject_unsupported_partial_assistant(&messages_json)?;
483        crate::reject_unsupported_message_tools(&messages_json, &["developer"])?;
484
485        let mut messages_array = messages_json
486            .as_array()
487            .context("Messages is not an array")?
488            .clone();
489
490        normalize_message_contents(&mut messages_array, NormalizeNonText::LeaveUntouched);
491
492        super::common::inject_tools_and_response_format(&mut messages_array, req)?;
493
494        encode_messages_with_options(
495            &messages_array,
496            thinking_mode,
497            true,
498            drop_thinking,
499            reasoning_effort,
500        )
501    }
502}
503
504#[cfg(test)]
505mod tests {
506    use super::*;
507    use serde_json::json;
508
509    #[test]
510    fn test_simple_conversation() {
511        let messages = json!([
512            {"role": "system", "content": "You are a helpful assistant."},
513            {"role": "user", "content": "Hello"},
514            {"role": "assistant", "reasoning_content": "greet", "content": "Hi!"},
515            {"role": "user", "content": "What is 2+2?"}
516        ]);
517        let out =
518            encode_messages(messages.as_array().unwrap(), ThinkingMode::Thinking, true).unwrap();
519        assert!(out.starts_with(tokens::BOS));
520        assert!(out.ends_with(&format!(
521            "{}{}",
522            tokens::ASSISTANT_START,
523            tokens::THINKING_START
524        )));
525        // drop_thinking default true → earlier reasoning stripped
526        assert!(!out.contains("greet"));
527    }
528
529    #[test]
530    fn test_reasoning_effort_prefixes() {
531        let messages = json!([
532            {"role": "system", "content": "hi"},
533            {"role": "user", "content": "hello"}
534        ]);
535
536        let high = encode_messages_with_options(
537            messages.as_array().unwrap(),
538            ThinkingMode::Thinking,
539            true,
540            true,
541            Some(ReasoningEffort::High),
542        )
543        .unwrap();
544        let max = encode_messages_with_options(
545            messages.as_array().unwrap(),
546            ThinkingMode::Thinking,
547            true,
548            true,
549            Some(ReasoningEffort::Max),
550        )
551        .unwrap();
552        let low = encode_messages_with_options(
553            messages.as_array().unwrap(),
554            ThinkingMode::Thinking,
555            true,
556            true,
557            None,
558        )
559        .unwrap();
560
561        assert_eq!(
562            high,
563            concat!(
564                "<|begin▁of▁sentence|>Reasoning Effort: Absolute maximum with no shortcuts permitted.\n",
565                "You MUST be very thorough in your thinking and comprehensively decompose the problem to resolve the root cause, rigorously stress-testing your logic against all potential paths, edge cases, and adversarial scenarios.\n",
566                "Explicitly write out your entire deliberation process, documenting every intermediate step, considered alternative, and rejected hypothesis to ensure absolutely no assumption is left unchecked.\n\n",
567                "hi<|User|>hello<|Assistant|><think>"
568            )
569        );
570        assert_eq!(
571            max,
572            concat!(
573                "<|begin▁of▁sentence|>Reasoning Effort: Beyond maximum — exhaustive, relentless, and uncompromising.\n",
574                "You MUST reason with the utmost depth and rigor, leaving absolutely nothing to chance: exhaustively decompose the problem into its most fundamental components, trace every causal chain to its root, and resolve the underlying cause rather than any surface symptom.\n",
575                "Do not stop reasoning until you have independently verified the solution from multiple angles and are certain that no assumption remains unchecked and no error remains undiscovered.\n\n",
576                "hi<|User|>hello<|Assistant|><think>"
577            )
578        );
579        assert_eq!(
580            low,
581            "<|begin▁of▁sentence|>hi<|User|>hello<|Assistant|><think>"
582        );
583    }
584
585    #[test]
586    fn test_content_blocks_with_tool_result() {
587        // `merge_tool_messages` turns a `tool` role followed by a plain user text
588        // into a single user turn whose `content_blocks` interleave the tool result
589        // with the text, joined by "\n\n" at render time. Users don't construct
590        // `content_blocks` directly — both the Python reference and this port
591        // overwrite any user-supplied `content_blocks` with a single text block.
592        let messages = json!([
593            {"role": "user", "content": "call tool"},
594            {"role": "assistant", "content": "", "tool_calls": [{
595                "id": "c1", "type": "function",
596                "function": {"name": "f", "arguments": "{}"}
597            }]},
598            {"role": "tool", "tool_call_id": "c1", "content": "RESULT"},
599            {"role": "user", "content": "thanks"}
600        ]);
601        let out = encode_messages(messages.as_array().unwrap(), ThinkingMode::Chat, true).unwrap();
602        assert!(
603            out.contains("<tool_result>RESULT</tool_result>\n\nthanks"),
604            "expected tool_result block followed by 'thanks' in the merged user turn, got:\n{}",
605            out
606        );
607    }
608
609    #[test]
610    fn test_user_task_preserved_when_merged_after_tool_result() {
611        let messages = json!([
612            {"role": "assistant", "content": "", "tool_calls": [{
613                "id": "c1", "type": "function",
614                "function": {"name": "search", "arguments": "{}"}
615            }]},
616            {"role": "tool", "tool_call_id": "c1", "content": "RESULT"},
617            {"role": "user", "content": "Search", "task": "action"},
618            {"role": "assistant", "content": "OK"}
619        ]);
620
621        let out = encode_messages(messages.as_array().unwrap(), ThinkingMode::Chat, true).unwrap();
622        assert!(
623            out.contains(&format!(
624                "{}Search{}{}{}OK",
625                "<tool_result>RESULT</tool_result>\n\n",
626                tokens::ASSISTANT_START,
627                tokens::THINKING_END,
628                tokens::TASK_ACTION
629            )),
630            "expected merged user text to keep the action task transition, got:\n{}",
631            out
632        );
633    }
634
635    #[test]
636    fn test_drop_thinking_auto_disable_when_tools_present() {
637        let messages = json!([
638            {"role": "system", "content": "s", "tools": [{
639                "type": "function",
640                "function": {"name": "f", "description": "", "parameters": {"type": "object", "properties": {}}}
641            }]},
642            {"role": "user", "content": "hi"},
643            {"role": "assistant", "reasoning_content": "PRIOR_REASONING", "content": "reply"},
644            {"role": "user", "content": "again"}
645        ]);
646        let out =
647            encode_messages(messages.as_array().unwrap(), ThinkingMode::Thinking, true).unwrap();
648        // Tools present → drop_thinking auto-disabled → earlier reasoning preserved.
649        assert!(out.contains("PRIOR_REASONING"));
650    }
651
652    // ---- Regression tests for known divergences from the Python reference ----
653
654    /// Bug: `last_user_idx = None` (no user/developer in history) should behave
655    /// like Python's `-1` sentinel — `index >= -1` / `idx >= -1` always true, so
656    /// earlier reasoning is preserved and the assistant's reasoning block is
657    /// rendered. Rust defaulting `None` to `usize::MAX` / `is_some_and` silently
658    /// stripped reasoning instead.
659    ///
660    /// Byte-equivalent to Python reference with the same input:
661    /// `<BOS>sysREASONING_BLOCK</think>hello<EOS>`
662    #[test]
663    fn test_assistant_reasoning_preserved_when_no_user_in_history() {
664        let messages = json!([
665            {"role": "system", "content": "sys"},
666            {"role": "assistant", "content": "hello", "reasoning_content": "REASONING_BLOCK"}
667        ]);
668        let out =
669            encode_messages(messages.as_array().unwrap(), ThinkingMode::Thinking, true).unwrap();
670        assert_eq!(
671            out, "<|begin▁of▁sentence|>sysREASONING_BLOCK</think>hello<|end▁of▁sentence|>",
672            "Output must match Python reference byte-for-byte when no user/developer in history"
673        );
674    }
675
676    /// Bug: `to_json` tracks in-string state via `prev_char != '\\'` which
677    /// mis-handles consecutive backslashes. A value containing `\\` (one literal
678    /// backslash in JSON) makes the helper think the closing `"` is escaped,
679    /// so it stops inserting Python-compatible spaces after subsequent `:`/`,`.
680    ///
681    /// Python `json.dumps({"path": "\\", "count": 5}, ensure_ascii=False)`
682    /// emits `{"path": "\\", "count": 5}` — space after every `:` and `,`.
683    #[test]
684    fn test_to_json_preserves_spacing_past_escaped_backslash() {
685        let v = json!({"path": "\\", "count": 5});
686        let got = to_json(&v);
687        assert_eq!(
688            got, r#"{"path": "\\", "count": 5}"#,
689            "to_json must match Python's json.dumps formatting past an escaped backslash"
690        );
691    }
692
693    #[test]
694    fn test_resolve_drop_thinking_warns_on_malformed_value() {
695        use std::collections::HashMap;
696        // String "false" where a bool is expected → fall back to default (true) and warn.
697        let mut args = HashMap::new();
698        args.insert(
699            "drop_thinking".to_string(),
700            serde_json::Value::String("false".to_string()),
701        );
702        assert!(DeepSeekV4Formatter::resolve_drop_thinking(Some(&args)));
703        // Malformed reasoning_effort falls back to the API default (high).
704        let malformed = serde_json::Value::String("HIGH".to_string());
705        assert_eq!(
706            DeepSeekV4Formatter::resolve_reasoning_effort(Some(&malformed)),
707            (false, Some(ReasoningEffort::High))
708        );
709    }
710
711    #[test]
712    fn test_resolve_thinking_mode_honors_enable_thinking() {
713        use std::collections::HashMap;
714        let mut args = HashMap::new();
715        args.insert(
716            "enable_thinking".to_string(),
717            serde_json::Value::Bool(false),
718        );
719        assert_eq!(
720            super::super::common::resolve_thinking_mode(Some(&args), ThinkingMode::Thinking),
721            ThinkingMode::Chat
722        );
723        args.insert("enable_thinking".to_string(), serde_json::Value::Bool(true));
724        assert_eq!(
725            super::super::common::resolve_thinking_mode(Some(&args), ThinkingMode::Thinking),
726            ThinkingMode::Thinking
727        );
728    }
729
730    struct MockRequest {
731        messages: JsonValue,
732        chat_template_args: Option<std::collections::HashMap<String, JsonValue>>,
733        reasoning_effort: Option<JsonValue>,
734        tools: Option<JsonValue>,
735        tool_choice: Option<JsonValue>,
736        response_format: Option<JsonValue>,
737    }
738
739    impl MockRequest {
740        fn new(messages: JsonValue) -> Self {
741            Self {
742                messages,
743                chat_template_args: None,
744                reasoning_effort: None,
745                tools: None,
746                tool_choice: None,
747                response_format: None,
748            }
749        }
750
751        fn with_chat_template_args(
752            mut self,
753            args: std::collections::HashMap<String, JsonValue>,
754        ) -> Self {
755            self.chat_template_args = Some(args);
756            self
757        }
758
759        fn with_reasoning_effort(mut self, reasoning_effort: JsonValue) -> Self {
760            self.reasoning_effort = Some(reasoning_effort);
761            self
762        }
763
764        fn with_tools(mut self, tools: JsonValue) -> Self {
765            self.tools = Some(tools);
766            self
767        }
768
769        fn with_tool_choice(mut self, tool_choice: JsonValue) -> Self {
770            self.tool_choice = Some(tool_choice);
771            self
772        }
773
774        fn with_response_format(mut self, response_format: JsonValue) -> Self {
775            self.response_format = Some(response_format);
776            self
777        }
778    }
779
780    impl crate::OAIChatLikeRequest for MockRequest {
781        fn model(&self) -> String {
782            "deepseek-v4".to_string()
783        }
784
785        fn messages(&self) -> minijinja::value::Value {
786            minijinja::value::Value::from_serialize(&self.messages)
787        }
788
789        fn should_add_generation_prompt(&self) -> bool {
790            true
791        }
792
793        fn chat_template_args(
794            &self,
795        ) -> Option<&std::collections::HashMap<String, serde_json::Value>> {
796            self.chat_template_args.as_ref()
797        }
798
799        fn reasoning_effort(&self) -> Option<minijinja::value::Value> {
800            self.reasoning_effort
801                .as_ref()
802                .map(minijinja::value::Value::from_serialize)
803        }
804
805        fn tools(&self) -> Option<minijinja::value::Value> {
806            self.tools
807                .as_ref()
808                .map(minijinja::value::Value::from_serialize)
809        }
810
811        fn tool_choice(&self) -> Option<minijinja::value::Value> {
812            self.tool_choice
813                .as_ref()
814                .map(minijinja::value::Value::from_serialize)
815        }
816
817        fn response_format(&self) -> Option<minijinja::value::Value> {
818            self.response_format
819                .as_ref()
820                .map(minijinja::value::Value::from_serialize)
821        }
822    }
823
824    fn weather_tool() -> JsonValue {
825        json!([{
826            "type": "function",
827            "function": {
828                "name": "get_current_weather",
829                "description": "Get the current weather in a given location",
830                "parameters": {
831                    "type": "object",
832                    "properties": {"location": {"type": "string"}},
833                    "required": ["location"]
834                }
835            }
836        }])
837    }
838
839    #[test]
840    fn test_formatter_rejects_unsupported_partial_assistant() {
841        use crate::OAIPromptFormatter;
842
843        let request = MockRequest::new(json!([
844            {"role": "user", "content": "Continue"},
845            {"role": "assistant", "content": "prefix", "partial": true}
846        ]));
847        let error = DeepSeekV4Formatter::new_thinking()
848            .render(&request)
849            .unwrap_err();
850
851        assert!(matches!(
852            error.downcast_ref::<crate::PromptRenderError>(),
853            Some(crate::PromptRenderError::InvalidRequest(message))
854                if message.contains("`partial: true` is not supported")
855        ));
856    }
857
858    #[test]
859    fn test_formatter_rejects_system_tools_before_injection() {
860        use crate::OAIPromptFormatter;
861
862        let request = MockRequest::new(json!([
863            {"role": "system", "tools": [
864                {"type": "function", "function": {"name": "dynamic_tool"}}
865            ]},
866            {"role": "user", "content": "Use a tool"}
867        ]))
868        .with_tools(weather_tool());
869        let error = DeepSeekV4Formatter::new_thinking()
870            .render(&request)
871            .unwrap_err();
872
873        assert!(matches!(
874            error.downcast_ref::<crate::PromptRenderError>(),
875            Some(crate::PromptRenderError::InvalidRequest(message))
876                if message.contains("message-level `tools`") && message.contains("system")
877        ));
878    }
879
880    #[test]
881    fn test_formatter_preserves_developer_tools_with_top_level_tools() {
882        use crate::OAIPromptFormatter;
883
884        let request = MockRequest::new(json!([
885            {"role": "developer", "content": "Use a tool", "tools": [
886                {"type": "function", "function": {"name": "developer_tool"}}
887            ]}
888        ]))
889        .with_tools(weather_tool());
890        let rendered = DeepSeekV4Formatter::new_thinking()
891            .render(&request)
892            .unwrap();
893
894        assert!(rendered.contains("developer_tool"));
895        assert!(rendered.contains("get_current_weather"));
896    }
897
898    #[test]
899    fn test_render_tool_choice_none_strips_tools_keeps_response_format() {
900        use crate::OAIPromptFormatter;
901
902        let req = MockRequest::new(json!([
903            {"role": "system", "content": "sys"},
904            {"role": "user", "content": "weather in Boston?"}
905        ]))
906        .with_tools(weather_tool())
907        .with_tool_choice(json!("none"))
908        .with_response_format(json!({"type": "json_object"}));
909
910        let formatter = DeepSeekV4Formatter::new_chat();
911        let out = formatter.render(&req).unwrap();
912
913        assert!(
914            !out.contains("## Tools"),
915            "tool_choice=none must strip the tools block, got: {out}"
916        );
917        assert!(
918            !out.contains("get_current_weather"),
919            "tool schema leaked into prompt despite tool_choice=none: {out}"
920        );
921        assert!(
922            out.contains("## Response Format"),
923            "response_format must survive tool_choice=none: {out}"
924        );
925    }
926
927    #[test]
928    fn test_render_tool_choice_auto_keeps_tools() {
929        use crate::OAIPromptFormatter;
930
931        let req = MockRequest::new(json!([
932            {"role": "system", "content": "sys"},
933            {"role": "user", "content": "weather in Boston?"}
934        ]))
935        .with_tools(weather_tool())
936        .with_tool_choice(json!("auto"));
937
938        let formatter = DeepSeekV4Formatter::new_chat();
939        let out = formatter.render(&req).unwrap();
940
941        assert!(out.contains("## Tools"));
942        assert!(out.contains("get_current_weather"));
943    }
944
945    #[test]
946    fn test_render_absent_tool_choice_keeps_tools() {
947        use crate::OAIPromptFormatter;
948
949        let req = MockRequest::new(json!([
950            {"role": "system", "content": "sys"},
951            {"role": "user", "content": "weather in Boston?"}
952        ]))
953        .with_tools(weather_tool());
954
955        let formatter = DeepSeekV4Formatter::new_chat();
956        let out = formatter.render(&req).unwrap();
957
958        assert!(out.contains("## Tools"));
959        assert!(out.contains("get_current_weather"));
960    }
961
962    #[test]
963    fn test_resolve_reasoning_effort_accepts_full_range() {
964        let effort = |v: &str| {
965            let value = json!(v);
966            DeepSeekV4Formatter::resolve_reasoning_effort(Some(&value))
967        };
968
969        assert_eq!(effort("max"), (false, Some(ReasoningEffort::Max)));
970        assert_eq!(effort("xhigh"), (false, Some(ReasoningEffort::High)));
971        assert_eq!(effort("high"), (false, Some(ReasoningEffort::High)));
972        assert_eq!(effort("minimal"), (false, None));
973        assert_eq!(effort("low"), (false, None));
974        assert_eq!(effort("medium"), (false, Some(ReasoningEffort::High)));
975        assert_eq!(effort("none"), (true, None));
976        assert_eq!(effort("bogus"), (false, Some(ReasoningEffort::High)));
977        assert_eq!(
978            DeepSeekV4Formatter::resolve_reasoning_effort(None),
979            (false, Some(ReasoningEffort::High))
980        );
981    }
982
983    #[test]
984    fn test_render_leaves_null_assistant_tool_content_empty() {
985        use crate::OAIPromptFormatter;
986
987        let req = MockRequest::new(json!([
988            {"role": "user", "content": "call tool"},
989            {"role": "assistant", "content": null, "tool_calls": [{
990                "id": "c1", "type": "function",
991                "function": {"name": "f", "arguments": "{}"}
992            }]}
993        ]));
994
995        let formatter = DeepSeekV4Formatter::new_chat();
996        let out = formatter.render(&req).unwrap();
997
998        assert!(out.contains(&format!(
999            "<{}{}>",
1000            tokens::DSML_TOKEN,
1001            TOOL_CALLS_BLOCK_NAME
1002        )));
1003        assert!(!out.contains("null"));
1004    }
1005
1006    #[test]
1007    fn test_render_wires_reasoning_effort_from_chat_template_args() {
1008        use crate::OAIPromptFormatter;
1009        use std::collections::HashMap;
1010
1011        for (effort, expected) in [
1012            ("high", REASONING_EFFORT_HIGH),
1013            ("max", REASONING_EFFORT_MAX),
1014        ] {
1015            let mut args = HashMap::new();
1016            args.insert("reasoning_effort".to_string(), json!(effort));
1017
1018            let req = MockRequest::new(json!([
1019                {"role": "system", "content": "sys"},
1020                {"role": "user", "content": "hi"}
1021            ]))
1022            .with_chat_template_args(args);
1023
1024            let formatter = DeepSeekV4Formatter::new_thinking();
1025            let out = formatter.render(&req).unwrap();
1026
1027            assert!(out.starts_with(tokens::BOS));
1028            assert!(
1029                out[tokens::BOS.len()..].starts_with(expected),
1030                "{effort} preamble should appear after BOS, got:\n{out}"
1031            );
1032        }
1033    }
1034
1035    #[test]
1036    fn test_render_wires_top_level_reasoning_effort_and_none_disables_thinking() {
1037        use crate::OAIPromptFormatter;
1038
1039        let formatter = DeepSeekV4Formatter::new_thinking();
1040        for (effort, expected_prefix) in [
1041            ("high", "Reasoning Effort: Absolute maximum"),
1042            ("max", "Reasoning Effort: Beyond maximum"),
1043        ] {
1044            let req: dynamo_protocols::types::CreateChatCompletionRequest =
1045                serde_json::from_value(json!({
1046                    "model": "deepseek-v4",
1047                    "messages": [{"role": "user", "content": "hi"}],
1048                    "reasoning_effort": effort
1049                }))
1050                .unwrap();
1051            let out = formatter.render(&req).unwrap();
1052
1053            assert!(
1054                out[tokens::BOS.len()..].starts_with(expected_prefix),
1055                "top-level {effort} did not select its prefix: {out}"
1056            );
1057            assert!(out.ends_with(tokens::THINKING_START));
1058        }
1059
1060        let req: dynamo_protocols::types::CreateChatCompletionRequest =
1061            serde_json::from_value(json!({
1062                "model": "deepseek-v4",
1063                "messages": [{"role": "user", "content": "hi"}],
1064                "reasoning_effort": "none"
1065            }))
1066            .unwrap();
1067        let out = formatter.render(&req).unwrap();
1068
1069        assert_eq!(
1070            out,
1071            "<|begin▁of▁sentence|><|User|>hi<|Assistant|></think>"
1072        );
1073    }
1074
1075    #[test]
1076    fn test_top_level_reasoning_effort_precedes_template_argument() {
1077        use crate::OAIPromptFormatter;
1078        use std::collections::HashMap;
1079
1080        let mut args = HashMap::new();
1081        args.insert("reasoning_effort".to_string(), json!("max"));
1082        let req = MockRequest::new(json!([{"role": "user", "content": "hi"}]))
1083            .with_chat_template_args(args)
1084            .with_reasoning_effort(json!("low"));
1085
1086        let out = DeepSeekV4Formatter::new_thinking().render(&req).unwrap();
1087
1088        assert_eq!(
1089            out,
1090            "<|begin▁of▁sentence|><|User|>hi<|Assistant|><think>"
1091        );
1092    }
1093
1094    #[test]
1095    fn test_render_drop_thinking_override_from_chat_template_args() {
1096        use crate::OAIPromptFormatter;
1097        use std::collections::HashMap;
1098
1099        let messages = json!([
1100            {"role": "user", "content": "first"},
1101            {"role": "assistant", "reasoning_content": "PRIOR", "content": "reply"},
1102            {"role": "user", "content": "again"}
1103        ]);
1104
1105        // Default (drop_thinking=true): prior reasoning stripped.
1106        let req_default = MockRequest::new(messages.clone());
1107        let formatter = DeepSeekV4Formatter::new_thinking();
1108        let out_default = formatter.render(&req_default).unwrap();
1109        assert!(
1110            !out_default.contains("PRIOR"),
1111            "default drop_thinking=true should strip prior reasoning, got:\n{}",
1112            out_default
1113        );
1114
1115        // drop_thinking=false override: prior reasoning survives.
1116        let mut args = HashMap::new();
1117        args.insert("drop_thinking".to_string(), json!(false));
1118        let req_keep = MockRequest::new(messages).with_chat_template_args(args);
1119        let out_keep = formatter.render(&req_keep).unwrap();
1120        assert!(
1121            out_keep.contains("PRIOR"),
1122            "drop_thinking=false override should preserve prior reasoning, got:\n{}",
1123            out_keep
1124        );
1125    }
1126
1127    // N4: developer-role interactions with drop_thinking.
1128    // find_last_user_index returns the index of user OR developer messages; the
1129    // drop_thinking reasoning cutoff and the thinking-seed insertion treat
1130    // user and developer identically.
1131
1132    #[test]
1133    fn test_developer_only_conversation_renders_developer_content() {
1134        let messages = json!([
1135            {"role": "system", "content": "sys"},
1136            {"role": "developer", "content": "x"},
1137            {"role": "assistant", "reasoning_content": "R", "content": "ok"}
1138        ]);
1139        let out =
1140            encode_messages(messages.as_array().unwrap(), ThinkingMode::Thinking, true).unwrap();
1141        assert!(
1142            out.contains("x"),
1143            "developer content should appear in output, got:\n{}",
1144            out
1145        );
1146    }
1147
1148    #[test]
1149    fn test_developer_as_last_user_index_controls_reasoning_cutoff() {
1150        // Indices: 0=user, 1=assistant(FIRST), 2=developer(y), 3=assistant(SECOND).
1151        // find_last_user_index = 2 (developer). With drop_thinking=true:
1152        //   - assistant idx=1 < 2  → reasoning_content stripped.
1153        //   - assistant idx=3 >= 2 → reasoning_content preserved.
1154        let messages = json!([
1155            {"role": "user", "content": "a"},
1156            {"role": "assistant", "reasoning_content": "FIRST", "content": "r1"},
1157            {"role": "developer", "content": "y"},
1158            {"role": "assistant", "reasoning_content": "SECOND", "content": "r2"}
1159        ]);
1160        let out =
1161            encode_messages(messages.as_array().unwrap(), ThinkingMode::Thinking, true).unwrap();
1162        assert!(
1163            !out.contains("FIRST"),
1164            "reasoning before last user/developer (idx 1 < 2) should be stripped, got:\n{}",
1165            out
1166        );
1167        assert!(
1168            out.contains("SECOND"),
1169            "reasoning at/after last user/developer (idx 3 > 2) should survive, got:\n{}",
1170            out
1171        );
1172    }
1173}