Skip to main content

dynamo_renderer/deepseek/
v4.rs

1// SPDX-FileCopyrightText: Copyright (c) 2024-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
2// SPDX-License-Identifier: Apache-2.0
3
4//! DeepSeek V4 native prompt formatting
5//!
6//! Native Rust port of DeepSeek V4's chat encoding (encoding_dsv4.py).
7//!
8//! Reference: DeepSeek-V4-Pro/encoding/encoding_dsv4.py
9
10use anyhow::{Context, Result};
11use serde_json::Value as JsonValue;
12use std::fmt::Write;
13
14use super::common::{
15    NormalizeNonText, REASONING_EFFORT_HIGH, REASONING_EFFORT_MAX, RESPONSE_FORMAT_TEMPLATE,
16    TOOL_CALLS_BLOCK_NAME, TOOLS_TEMPLATE, drop_thinking_messages, encode_arguments_to_dsml,
17    find_last_user_index, merge_tool_messages, normalize_message_contents, render_tools,
18    sort_tool_results_by_call_order, task_token, to_json,
19};
20pub use super::common::{ReasoningEffort, ThinkingMode, tokens};
21
22#[derive(Clone, Copy)]
23pub(super) enum Encoding {
24    V4(Option<ReasoningEffort>),
25    V41(u8),
26}
27
28impl Encoding {
29    fn is_v41(self) -> bool {
30        matches!(self, Self::V41(_))
31    }
32
33    fn tag(self, v4: &'static str, v41: &'static str) -> &'static str {
34        if self.is_v41() { v41 } else { v4 }
35    }
36
37    fn reasoning_prefix(self) -> String {
38        match self {
39            Self::V4(Some(ReasoningEffort::High)) => REASONING_EFFORT_HIGH.to_string(),
40            Self::V4(Some(ReasoningEffort::Max)) => REASONING_EFFORT_MAX.to_string(),
41            Self::V4(None) => String::new(),
42            Self::V41(effort) => format!(
43                "Reasoning Effort: {effort} (range 1-100, the higher the value, the more thorough the reasoning)\n\n"
44            ),
45        }
46    }
47
48    fn render_tools(self, tools: &[JsonValue]) -> String {
49        let template = if self.is_v41() {
50            TOOLS_TEMPLATE
51                .replace("{dsml_token}tool_calls", "{dsml_token} calls")
52                .replace("{dsml_token}invoke", "{dsml_token} invoke")
53                .replace("{dsml_token}parameter", "{dsml_token} parameter")
54        } else {
55            TOOLS_TEMPLATE.to_string()
56        };
57        render_tools(&template, tools)
58    }
59}
60
61/// Render a single message at the given index.
62fn render_message(
63    prompt: &mut String,
64    index: usize,
65    messages: &[JsonValue],
66    thinking_mode: ThinkingMode,
67    drop_thinking: bool,
68    encoding: Encoding,
69    last_user_idx: Option<usize>,
70) -> Result<()> {
71    let msg = &messages[index];
72
73    let role = msg
74        .get("role")
75        .and_then(|r| r.as_str())
76        .context("Missing 'role' field")?;
77
78    if encoding.is_v41()
79        && (role == "system" || (index == 0 && thinking_mode == ThinkingMode::Thinking))
80    {
81        prompt.push_str("<|System|>");
82    }
83    if index == 0 && thinking_mode == ThinkingMode::Thinking {
84        prompt.push_str(&encoding.reasoning_prefix());
85    }
86
87    match role {
88        "system" => {
89            let content = msg.get("content").and_then(|c| c.as_str()).unwrap_or("");
90            prompt.push_str(content);
91            if let Some(tools) = msg.get("tools").and_then(|t| t.as_array()) {
92                prompt.push_str("\n\n");
93                prompt.push_str(&encoding.render_tools(tools));
94            }
95            if let Some(response_format) = msg.get("response_format") {
96                prompt.push_str("\n\n");
97                prompt.push_str(
98                    &RESPONSE_FORMAT_TEMPLATE.replace("{schema}", &to_json(response_format)),
99                );
100            }
101        }
102
103        "developer" => {
104            let content = msg
105                .get("content")
106                .and_then(|c| c.as_str())
107                .filter(|s| !s.is_empty())
108                .context("Developer role requires content")?;
109
110            prompt.push_str(tokens::USER_START);
111            prompt.push_str(content);
112
113            if let Some(tools) = msg.get("tools").and_then(|t| t.as_array()) {
114                prompt.push_str("\n\n");
115                prompt.push_str(&encoding.render_tools(tools));
116            }
117            if let Some(response_format) = msg.get("response_format") {
118                prompt.push_str("\n\n");
119                prompt.push_str(
120                    &RESPONSE_FORMAT_TEMPLATE.replace("{schema}", &to_json(response_format)),
121                );
122            }
123        }
124
125        "user" => {
126            prompt.push_str(tokens::USER_START);
127            if let Some(blocks) = msg.get("content_blocks").and_then(|b| b.as_array()) {
128                for (block_idx, block) in blocks.iter().enumerate() {
129                    if block_idx > 0 {
130                        prompt.push_str("\n\n");
131                    }
132                    let block_type = block.get("type").and_then(|v| v.as_str()).unwrap_or("");
133                    match block_type {
134                        "text" => {
135                            let text = block.get("text").and_then(|v| v.as_str()).unwrap_or("");
136                            prompt.push_str(text);
137                        }
138                        "tool_result" => {
139                            prompt.push_str("<tool_result>");
140                            render_tool_result_content(
141                                prompt,
142                                block.get("content").unwrap_or(&JsonValue::Null),
143                            )?;
144                            prompt.push_str("</tool_result>");
145                        }
146                        other => {
147                            write!(prompt, "[Unsupported {}]", other)?;
148                        }
149                    }
150                }
151            }
152        }
153
154        "latest_reminder" => {
155            let content = msg.get("content").and_then(|c| c.as_str()).unwrap_or("");
156            prompt.push_str(tokens::LATEST_REMINDER);
157            prompt.push_str(content);
158        }
159
160        "tool" => {
161            anyhow::bail!(
162                "deepseek_v4 merges tool messages into user; preprocess with merge_tool_messages()"
163            );
164        }
165
166        "assistant" => {
167            let content = msg.get("content").and_then(|c| c.as_str()).unwrap_or("");
168            let reasoning = msg
169                .get("reasoning_content")
170                .and_then(|c| c.as_str())
171                .unwrap_or("");
172            let wo_eos = msg.get("wo_eos").and_then(|v| v.as_bool()).unwrap_or(false);
173
174            let prev_has_task = index > 0
175                && messages[index - 1]
176                    .get("task")
177                    .map(|v| !v.is_null())
178                    .unwrap_or(false);
179
180            if thinking_mode == ThinkingMode::Thinking && !prev_has_task {
181                let render_thinking = !drop_thinking || last_user_idx.is_none_or(|u| index > u);
182                if render_thinking {
183                    prompt.push_str(reasoning);
184                    prompt.push_str(tokens::THINKING_END);
185                }
186            }
187
188            prompt.push_str(content);
189
190            if let Some(tool_calls) = msg.get("tool_calls").and_then(|t| t.as_array())
191                && !tool_calls.is_empty()
192            {
193                prompt.push_str("\n\n");
194                writeln!(
195                    prompt,
196                    "<{}{}>",
197                    tokens::DSML_TOKEN,
198                    encoding.tag(TOOL_CALLS_BLOCK_NAME, " calls")
199                )?;
200
201                for (call_idx, tc) in tool_calls.iter().enumerate() {
202                    if call_idx > 0 {
203                        prompt.push('\n');
204                    }
205                    // Accept both OpenAI-format (nested `function`) and internal
206                    // `{name, arguments}` shape, matching Python's `tool_calls_from_openai_format`.
207                    let fn_obj = tc.get("function").unwrap_or(tc);
208                    let name = fn_obj
209                        .get("name")
210                        .and_then(|n| n.as_str())
211                        .context("Missing tool call name")?;
212                    let arguments = if encoding.is_v41() {
213                        super::v41::encode_arguments(fn_obj)?
214                    } else {
215                        encode_arguments_to_dsml(fn_obj)?
216                    };
217                    write!(
218                        prompt,
219                        "<{}{} name=\"{}\">\n{}\n</{}{}>",
220                        tokens::DSML_TOKEN,
221                        encoding.tag("invoke", " invoke"),
222                        name,
223                        arguments,
224                        tokens::DSML_TOKEN,
225                        encoding.tag("invoke", " invoke")
226                    )?;
227                }
228                write!(
229                    prompt,
230                    "\n</{}{}>",
231                    tokens::DSML_TOKEN,
232                    encoding.tag(TOOL_CALLS_BLOCK_NAME, " calls")
233                )?;
234            }
235
236            if !wo_eos {
237                prompt.push_str(tokens::EOS);
238            }
239        }
240
241        other => anyhow::bail!("Unknown role: {}", other),
242    }
243
244    // Early return if the next message is not assistant/latest_reminder — no transition appended.
245    if index + 1 < messages.len() {
246        let next_role = messages[index + 1].get("role").and_then(|r| r.as_str());
247        if !matches!(next_role, Some("assistant") | Some("latest_reminder")) {
248            return Ok(());
249        }
250    }
251
252    // Transition tokens based on task field and role.
253    let task = msg.get("task").and_then(|v| v.as_str());
254    if let Some(task) = task {
255        let sp = task_token(task).with_context(|| format!("Invalid task: '{}'", task))?;
256        if task != "action" {
257            prompt.push_str(sp);
258        } else {
259            prompt.push_str(tokens::ASSISTANT_START);
260            prompt.push_str(if thinking_mode != ThinkingMode::Thinking {
261                tokens::THINKING_END
262            } else {
263                tokens::THINKING_START
264            });
265            prompt.push_str(sp);
266        }
267    } else if matches!(role, "user" | "developer")
268        || (encoding.is_v41() && role == "system" && index > 0)
269    {
270        prompt.push_str(tokens::ASSISTANT_START);
271        let seed_thinking = thinking_mode == ThinkingMode::Thinking
272            && (!drop_thinking || last_user_idx.is_none_or(|u| index >= u));
273        prompt.push_str(if seed_thinking {
274            tokens::THINKING_START
275        } else {
276            tokens::THINKING_END
277        });
278    }
279
280    Ok(())
281}
282
283/// Render a tool_result `content` payload (string or content-block list).
284fn render_tool_result_content(prompt: &mut String, content: &JsonValue) -> Result<()> {
285    match content {
286        JsonValue::String(s) => prompt.push_str(s),
287        JsonValue::Array(items) => {
288            for (index, item) in items.iter().enumerate() {
289                if index > 0 {
290                    prompt.push_str("\n\n");
291                }
292                let item_type = item.get("type").and_then(|v| v.as_str()).unwrap_or("");
293                if item_type == "text" {
294                    prompt.push_str(item.get("text").and_then(|v| v.as_str()).unwrap_or(""));
295                } else {
296                    write!(prompt, "[Unsupported {}]", item_type)?;
297                }
298            }
299        }
300        JsonValue::Null => {}
301        _ => prompt.push_str(&to_json(content)),
302    }
303    Ok(())
304}
305
306/// Encode messages to prompt string with default options.
307///
308/// Equivalent to `encode_messages_with_options(.., drop_thinking=true, reasoning_effort=None)`.
309pub fn encode_messages(
310    messages: &[JsonValue],
311    thinking_mode: ThinkingMode,
312    add_bos_token: bool,
313) -> Result<String> {
314    encode_messages_with_options(messages, thinking_mode, add_bos_token, true, None)
315}
316
317/// Encode messages to prompt string.
318///
319/// # Arguments
320/// * `messages` - Array of messages in OpenAI format
321/// * `thinking_mode` - Chat or Thinking
322/// * `add_bos_token` - Whether to prepend BOS token
323/// * `drop_thinking` - Drop reasoning_content from earlier turns (auto-disabled if tools present)
324/// * `reasoning_effort` - Optional reasoning effort level (High and Max prepend distinct verbatim blocks)
325pub fn encode_messages_with_options(
326    messages: &[JsonValue],
327    thinking_mode: ThinkingMode,
328    add_bos_token: bool,
329    drop_thinking: bool,
330    reasoning_effort: Option<ReasoningEffort>,
331) -> Result<String> {
332    encode_owned_messages(
333        messages.to_vec(),
334        thinking_mode,
335        add_bos_token,
336        drop_thinking,
337        Encoding::V4(reasoning_effort),
338    )
339}
340
341pub(super) fn encode_owned_messages(
342    messages: Vec<JsonValue>,
343    thinking_mode: ThinkingMode,
344    add_bos_token: bool,
345    drop_thinking: bool,
346    encoding: Encoding,
347) -> Result<String> {
348    let merged = merge_tool_messages(messages);
349    // V4.1 orders source messages before merging, using the same routine as its media hook.
350    let mut full = if encoding.is_v41() {
351        merged
352    } else {
353        sort_tool_results_by_call_order(merged)
354    };
355
356    let mut prompt = String::new();
357    if add_bos_token {
358        prompt.push_str(tokens::BOS);
359    }
360
361    // Auto-disable drop_thinking when any message carries a `tools` field.
362    let has_tools = full.iter().any(|m| {
363        m.get("tools")
364            .map(|v| match v {
365                JsonValue::Array(a) => !a.is_empty(),
366                JsonValue::Null => false,
367                _ => true,
368            })
369            .unwrap_or(false)
370    });
371    let effective_drop_thinking = drop_thinking && !has_tools;
372
373    if thinking_mode == ThinkingMode::Thinking && effective_drop_thinking {
374        full = if encoding.is_v41() {
375            super::v41::drop_thinking_messages(full)
376        } else {
377            drop_thinking_messages(full)
378        };
379    }
380
381    let last_user_idx = if encoding.is_v41() {
382        super::v41::find_last_user_index(&full)
383    } else {
384        find_last_user_index(&full)
385    };
386    for idx in 0..full.len() {
387        render_message(
388            &mut prompt,
389            idx,
390            &full,
391            thinking_mode,
392            effective_drop_thinking,
393            encoding,
394            last_user_idx,
395        )?;
396    }
397
398    Ok(prompt)
399}
400
401/// DeepSeek V4 Prompt Formatter
402#[derive(Debug)]
403pub struct DeepSeekV4Formatter {
404    thinking_mode: ThinkingMode,
405}
406
407impl DeepSeekV4Formatter {
408    pub fn new(thinking_mode: ThinkingMode) -> Self {
409        Self { thinking_mode }
410    }
411
412    /// Create formatter with thinking mode enabled (default for DSV4)
413    pub fn new_thinking() -> Self {
414        Self::new(ThinkingMode::Thinking)
415    }
416
417    /// Create formatter with chat mode
418    pub fn new_chat() -> Self {
419        Self::new(ThinkingMode::Chat)
420    }
421
422    fn resolve_reasoning_effort(v: Option<&JsonValue>) -> (bool, Option<ReasoningEffort>) {
423        match v.and_then(JsonValue::as_str) {
424            Some("none") => (true, None),
425            Some("max") => (false, Some(ReasoningEffort::Max)),
426            Some("high") | Some("medium") | Some("xhigh") => (false, Some(ReasoningEffort::High)),
427            Some("low") | Some("minimal") => (false, None),
428            None if v.is_none() => (false, Some(ReasoningEffort::High)),
429            _ => {
430                tracing::warn!(
431                    value = ?v,
432                    "reasoning_effort must be one of \"none\", \"minimal\", \"low\", \"medium\", \"high\", \"xhigh\", \"max\"; ignoring and using API default (high)"
433                );
434                (false, Some(ReasoningEffort::High))
435            }
436        }
437    }
438
439    fn resolve_drop_thinking(
440        args: Option<&std::collections::HashMap<String, serde_json::Value>>,
441    ) -> bool {
442        let Some(args) = args else { return true };
443        let Some(v) = args.get("drop_thinking") else {
444            return true;
445        };
446        if let Some(b) = v.as_bool() {
447            return b;
448        }
449        tracing::warn!(
450            value = ?v,
451            "chat_template_args.drop_thinking must be a bool; ignoring and using default (true)"
452        );
453        true
454    }
455}
456
457impl crate::OAIPromptFormatter for DeepSeekV4Formatter {
458    fn supports_add_generation_prompt(&self) -> bool {
459        true
460    }
461
462    fn render(&self, req: &dyn crate::OAIChatLikeRequest) -> Result<String> {
463        let args = req.chat_template_args();
464        let effort_value = req
465            .reasoning_effort()
466            .map(|value| serde_json::to_value(value).context("serialize reasoning_effort"))
467            .transpose()?
468            .or_else(|| args.and_then(|args| args.get("reasoning_effort").cloned()));
469        let (disable_thinking, reasoning_effort) =
470            Self::resolve_reasoning_effort(effort_value.as_ref());
471        let mut thinking_mode = super::common::resolve_thinking_mode(args, self.thinking_mode);
472        if disable_thinking {
473            thinking_mode = ThinkingMode::Chat;
474        }
475        let drop_thinking = Self::resolve_drop_thinking(args);
476
477        let messages_json = crate::messages_to_json(req)?;
478        crate::reject_unsupported_partial_assistant(&messages_json)?;
479        crate::reject_unsupported_message_tools(&messages_json, &["developer"])?;
480
481        let JsonValue::Array(mut messages_array) = messages_json else {
482            anyhow::bail!("Messages is not an array");
483        };
484
485        normalize_message_contents(&mut messages_array, NormalizeNonText::LeaveUntouched);
486
487        super::common::inject_tools_and_response_format(&mut messages_array, req)?;
488
489        encode_owned_messages(
490            messages_array,
491            thinking_mode,
492            true,
493            drop_thinking,
494            Encoding::V4(reasoning_effort),
495        )
496    }
497}
498
499#[cfg(test)]
500mod tests {
501    use super::*;
502    use serde_json::json;
503
504    #[test]
505    fn test_simple_conversation() {
506        let messages = json!([
507            {"role": "system", "content": "You are a helpful assistant."},
508            {"role": "user", "content": "Hello"},
509            {"role": "assistant", "reasoning_content": "greet", "content": "Hi!"},
510            {"role": "user", "content": "What is 2+2?"}
511        ]);
512        let out =
513            encode_messages(messages.as_array().unwrap(), ThinkingMode::Thinking, true).unwrap();
514        assert!(out.starts_with(tokens::BOS));
515        assert!(out.ends_with(&format!(
516            "{}{}",
517            tokens::ASSISTANT_START,
518            tokens::THINKING_START
519        )));
520        // drop_thinking default true → earlier reasoning stripped
521        assert!(!out.contains("greet"));
522    }
523
524    #[test]
525    fn test_reasoning_effort_prefixes() {
526        let messages = json!([
527            {"role": "system", "content": "hi"},
528            {"role": "user", "content": "hello"}
529        ]);
530
531        let high = encode_messages_with_options(
532            messages.as_array().unwrap(),
533            ThinkingMode::Thinking,
534            true,
535            true,
536            Some(ReasoningEffort::High),
537        )
538        .unwrap();
539        let max = encode_messages_with_options(
540            messages.as_array().unwrap(),
541            ThinkingMode::Thinking,
542            true,
543            true,
544            Some(ReasoningEffort::Max),
545        )
546        .unwrap();
547        let low = encode_messages_with_options(
548            messages.as_array().unwrap(),
549            ThinkingMode::Thinking,
550            true,
551            true,
552            None,
553        )
554        .unwrap();
555
556        assert_eq!(
557            high,
558            concat!(
559                "<|begin▁of▁sentence|>Reasoning Effort: Absolute maximum with no shortcuts permitted.\n",
560                "You MUST be very thorough in your thinking and comprehensively decompose the problem to resolve the root cause, rigorously stress-testing your logic against all potential paths, edge cases, and adversarial scenarios.\n",
561                "Explicitly write out your entire deliberation process, documenting every intermediate step, considered alternative, and rejected hypothesis to ensure absolutely no assumption is left unchecked.\n\n",
562                "hi<|User|>hello<|Assistant|><think>"
563            )
564        );
565        assert_eq!(
566            max,
567            concat!(
568                "<|begin▁of▁sentence|>Reasoning Effort: Beyond maximum — exhaustive, relentless, and uncompromising.\n",
569                "You MUST reason with the utmost depth and rigor, leaving absolutely nothing to chance: exhaustively decompose the problem into its most fundamental components, trace every causal chain to its root, and resolve the underlying cause rather than any surface symptom.\n",
570                "Do not stop reasoning until you have independently verified the solution from multiple angles and are certain that no assumption remains unchecked and no error remains undiscovered.\n\n",
571                "hi<|User|>hello<|Assistant|><think>"
572            )
573        );
574        assert_eq!(
575            low,
576            "<|begin▁of▁sentence|>hi<|User|>hello<|Assistant|><think>"
577        );
578    }
579
580    #[test]
581    fn test_content_blocks_with_tool_result() {
582        // `merge_tool_messages` turns a `tool` role followed by a plain user text
583        // into a single user turn whose `content_blocks` interleave the tool result
584        // with the text, joined by "\n\n" at render time. Users don't construct
585        // `content_blocks` directly — both the Python reference and this port
586        // overwrite any user-supplied `content_blocks` with a single text block.
587        let messages = json!([
588            {"role": "user", "content": "call tool"},
589            {"role": "assistant", "content": "", "tool_calls": [{
590                "id": "c1", "type": "function",
591                "function": {"name": "f", "arguments": "{}"}
592            }]},
593            {"role": "tool", "tool_call_id": "c1", "content": "RESULT"},
594            {"role": "user", "content": "thanks"}
595        ]);
596        let out = encode_messages(messages.as_array().unwrap(), ThinkingMode::Chat, true).unwrap();
597        assert!(
598            out.contains("<tool_result>RESULT</tool_result>\n\nthanks"),
599            "expected tool_result block followed by 'thanks' in the merged user turn, got:\n{}",
600            out
601        );
602    }
603
604    #[test]
605    fn test_user_task_preserved_when_merged_after_tool_result() {
606        let messages = json!([
607            {"role": "assistant", "content": "", "tool_calls": [{
608                "id": "c1", "type": "function",
609                "function": {"name": "search", "arguments": "{}"}
610            }]},
611            {"role": "tool", "tool_call_id": "c1", "content": "RESULT"},
612            {"role": "user", "content": "Search", "task": "action"},
613            {"role": "assistant", "content": "OK"}
614        ]);
615
616        let out = encode_messages(messages.as_array().unwrap(), ThinkingMode::Chat, true).unwrap();
617        assert!(
618            out.contains(&format!(
619                "{}Search{}{}{}OK",
620                "<tool_result>RESULT</tool_result>\n\n",
621                tokens::ASSISTANT_START,
622                tokens::THINKING_END,
623                tokens::TASK_ACTION
624            )),
625            "expected merged user text to keep the action task transition, got:\n{}",
626            out
627        );
628    }
629
630    #[test]
631    fn test_drop_thinking_auto_disable_when_tools_present() {
632        let messages = json!([
633            {"role": "system", "content": "s", "tools": [{
634                "type": "function",
635                "function": {"name": "f", "description": "", "parameters": {"type": "object", "properties": {}}}
636            }]},
637            {"role": "user", "content": "hi"},
638            {"role": "assistant", "reasoning_content": "PRIOR_REASONING", "content": "reply"},
639            {"role": "user", "content": "again"}
640        ]);
641        let out =
642            encode_messages(messages.as_array().unwrap(), ThinkingMode::Thinking, true).unwrap();
643        // Tools present → drop_thinking auto-disabled → earlier reasoning preserved.
644        assert!(out.contains("PRIOR_REASONING"));
645    }
646
647    // ---- Regression tests for known divergences from the Python reference ----
648
649    /// Bug: `last_user_idx = None` (no user/developer in history) should behave
650    /// like Python's `-1` sentinel — `index >= -1` / `idx >= -1` always true, so
651    /// earlier reasoning is preserved and the assistant's reasoning block is
652    /// rendered. Rust defaulting `None` to `usize::MAX` / `is_some_and` silently
653    /// stripped reasoning instead.
654    ///
655    /// Byte-equivalent to Python reference with the same input:
656    /// `<BOS>sysREASONING_BLOCK</think>hello<EOS>`
657    #[test]
658    fn test_assistant_reasoning_preserved_when_no_user_in_history() {
659        let messages = json!([
660            {"role": "system", "content": "sys"},
661            {"role": "assistant", "content": "hello", "reasoning_content": "REASONING_BLOCK"}
662        ]);
663        let out =
664            encode_messages(messages.as_array().unwrap(), ThinkingMode::Thinking, true).unwrap();
665        assert_eq!(
666            out, "<|begin▁of▁sentence|>sysREASONING_BLOCK</think>hello<|end▁of▁sentence|>",
667            "Output must match Python reference byte-for-byte when no user/developer in history"
668        );
669    }
670
671    /// Bug: `to_json` tracks in-string state via `prev_char != '\\'` which
672    /// mis-handles consecutive backslashes. A value containing `\\` (one literal
673    /// backslash in JSON) makes the helper think the closing `"` is escaped,
674    /// so it stops inserting Python-compatible spaces after subsequent `:`/`,`.
675    ///
676    /// Python `json.dumps({"path": "\\", "count": 5}, ensure_ascii=False)`
677    /// emits `{"path": "\\", "count": 5}` — space after every `:` and `,`.
678    #[test]
679    fn test_to_json_preserves_spacing_past_escaped_backslash() {
680        let v = json!({"path": "\\", "count": 5});
681        let got = to_json(&v);
682        assert_eq!(
683            got, r#"{"path": "\\", "count": 5}"#,
684            "to_json must match Python's json.dumps formatting past an escaped backslash"
685        );
686    }
687
688    #[test]
689    fn test_resolve_drop_thinking_warns_on_malformed_value() {
690        use std::collections::HashMap;
691        // String "false" where a bool is expected → fall back to default (true) and warn.
692        let mut args = HashMap::new();
693        args.insert(
694            "drop_thinking".to_string(),
695            serde_json::Value::String("false".to_string()),
696        );
697        assert!(DeepSeekV4Formatter::resolve_drop_thinking(Some(&args)));
698        // Malformed reasoning_effort falls back to the API default (high).
699        let malformed = serde_json::Value::String("HIGH".to_string());
700        assert_eq!(
701            DeepSeekV4Formatter::resolve_reasoning_effort(Some(&malformed)),
702            (false, Some(ReasoningEffort::High))
703        );
704    }
705
706    #[test]
707    fn test_resolve_thinking_mode_honors_enable_thinking() {
708        use std::collections::HashMap;
709        let mut args = HashMap::new();
710        args.insert(
711            "enable_thinking".to_string(),
712            serde_json::Value::Bool(false),
713        );
714        assert_eq!(
715            super::super::common::resolve_thinking_mode(Some(&args), ThinkingMode::Thinking),
716            ThinkingMode::Chat
717        );
718        args.insert("enable_thinking".to_string(), serde_json::Value::Bool(true));
719        assert_eq!(
720            super::super::common::resolve_thinking_mode(Some(&args), ThinkingMode::Thinking),
721            ThinkingMode::Thinking
722        );
723    }
724
725    struct MockRequest {
726        messages: JsonValue,
727        typed: Option<Vec<dynamo_protocols::types::ChatCompletionRequestMessage>>,
728        chat_template_args: Option<std::collections::HashMap<String, JsonValue>>,
729        reasoning_effort: Option<JsonValue>,
730        tools: Option<JsonValue>,
731        tool_choice: Option<JsonValue>,
732        response_format: Option<JsonValue>,
733    }
734
735    impl MockRequest {
736        fn new(messages: JsonValue) -> Self {
737            Self {
738                messages,
739                typed: None,
740                chat_template_args: None,
741                reasoning_effort: None,
742                tools: None,
743                tool_choice: None,
744                response_format: None,
745            }
746        }
747
748        fn with_chat_template_args(
749            mut self,
750            args: std::collections::HashMap<String, JsonValue>,
751        ) -> Self {
752            self.chat_template_args = Some(args);
753            self
754        }
755
756        fn with_reasoning_effort(mut self, reasoning_effort: JsonValue) -> Self {
757            self.reasoning_effort = Some(reasoning_effort);
758            self
759        }
760
761        fn with_tools(mut self, tools: JsonValue) -> Self {
762            self.tools = Some(tools);
763            self
764        }
765
766        fn with_tool_choice(mut self, tool_choice: JsonValue) -> Self {
767            self.tool_choice = Some(tool_choice);
768            self
769        }
770
771        fn with_response_format(mut self, response_format: JsonValue) -> Self {
772            self.response_format = Some(response_format);
773            self
774        }
775    }
776
777    impl crate::OAIChatLikeRequest for MockRequest {
778        fn model(&self) -> String {
779            "deepseek-v4".to_string()
780        }
781
782        fn messages(&self) -> minijinja::value::Value {
783            assert!(
784                self.typed.is_none(),
785                "typed requests must skip MiniJinja conversion"
786            );
787            minijinja::value::Value::from_serialize(&self.messages)
788        }
789
790        fn typed_messages(
791            &self,
792        ) -> Option<&[dynamo_protocols::types::ChatCompletionRequestMessage]> {
793            self.typed.as_deref()
794        }
795
796        fn should_add_generation_prompt(&self) -> bool {
797            true
798        }
799
800        fn chat_template_args(
801            &self,
802        ) -> Option<&std::collections::HashMap<String, serde_json::Value>> {
803            self.chat_template_args.as_ref()
804        }
805
806        fn reasoning_effort(&self) -> Option<minijinja::value::Value> {
807            self.reasoning_effort
808                .as_ref()
809                .map(minijinja::value::Value::from_serialize)
810        }
811
812        fn tools(&self) -> Option<minijinja::value::Value> {
813            self.tools
814                .as_ref()
815                .map(minijinja::value::Value::from_serialize)
816        }
817
818        fn tool_choice(&self) -> Option<minijinja::value::Value> {
819            self.tool_choice
820                .as_ref()
821                .map(minijinja::value::Value::from_serialize)
822        }
823
824        fn response_format(&self) -> Option<minijinja::value::Value> {
825            self.response_format
826                .as_ref()
827                .map(minijinja::value::Value::from_serialize)
828        }
829    }
830
831    #[test]
832    fn typed_messages_match_value_messages() {
833        use crate::OAIPromptFormatter;
834        let messages = json!([
835            {"role": "system", "content": "Use tools. 中文 🦀"},
836            {"role": "user", "content": [{"type": "text", "text": "weather?"}]},
837            {"role": "assistant", "content": null, "reasoning_content": "check",
838             "tool_calls": [{"id": "call_1", "type": "function",
839                 "function": {"name": "weather", "arguments": "{\"city\":\"東京\"}"}}]},
840            {"role": "tool", "tool_call_id": "call_1", "content": "sunny"},
841            {"role": "user", "content": "explain"}
842        ]);
843        let mut typed = MockRequest::new(messages.clone());
844        typed.typed = Some(serde_json::from_value(messages.clone()).unwrap());
845        let value = MockRequest::new(messages);
846        for formatter in [
847            DeepSeekV4Formatter::new_thinking(),
848            DeepSeekV4Formatter::new_chat(),
849        ] {
850            assert_eq!(
851                formatter.render(&typed).unwrap(),
852                formatter.render(&value).unwrap()
853            );
854        }
855    }
856
857    fn weather_tool() -> JsonValue {
858        json!([{
859            "type": "function",
860            "function": {
861                "name": "get_current_weather",
862                "description": "Get the current weather in a given location",
863                "parameters": {
864                    "type": "object",
865                    "properties": {"location": {"type": "string"}},
866                    "required": ["location"]
867                }
868            }
869        }])
870    }
871
872    #[test]
873    fn test_formatter_rejects_unsupported_partial_assistant() {
874        use crate::OAIPromptFormatter;
875
876        let request = MockRequest::new(json!([
877            {"role": "user", "content": "Continue"},
878            {"role": "assistant", "content": "prefix", "partial": true}
879        ]));
880        let error = DeepSeekV4Formatter::new_thinking()
881            .render(&request)
882            .unwrap_err();
883
884        assert!(matches!(
885            error.downcast_ref::<crate::PromptRenderError>(),
886            Some(crate::PromptRenderError::InvalidRequest(message))
887                if message.contains("`partial: true` is not supported")
888        ));
889    }
890
891    #[test]
892    fn test_formatter_rejects_system_tools_before_injection() {
893        use crate::OAIPromptFormatter;
894
895        let request = MockRequest::new(json!([
896            {"role": "system", "tools": [
897                {"type": "function", "function": {"name": "dynamic_tool"}}
898            ]},
899            {"role": "user", "content": "Use a tool"}
900        ]))
901        .with_tools(weather_tool());
902        let error = DeepSeekV4Formatter::new_thinking()
903            .render(&request)
904            .unwrap_err();
905
906        assert!(matches!(
907            error.downcast_ref::<crate::PromptRenderError>(),
908            Some(crate::PromptRenderError::InvalidRequest(message))
909                if message.contains("message-level `tools`") && message.contains("system")
910        ));
911    }
912
913    #[test]
914    fn test_formatter_preserves_developer_tools_with_top_level_tools() {
915        use crate::OAIPromptFormatter;
916
917        let request = MockRequest::new(json!([
918            {"role": "developer", "content": "Use a tool", "tools": [
919                {"type": "function", "function": {"name": "developer_tool"}}
920            ]}
921        ]))
922        .with_tools(weather_tool());
923        let rendered = DeepSeekV4Formatter::new_thinking()
924            .render(&request)
925            .unwrap();
926
927        assert!(rendered.contains("developer_tool"));
928        assert!(rendered.contains("get_current_weather"));
929    }
930
931    #[test]
932    fn test_render_tool_choice_none_strips_tools_keeps_response_format() {
933        use crate::OAIPromptFormatter;
934
935        let req = MockRequest::new(json!([
936            {"role": "system", "content": "sys"},
937            {"role": "user", "content": "weather in Boston?"}
938        ]))
939        .with_tools(weather_tool())
940        .with_tool_choice(json!("none"))
941        .with_response_format(json!({"type": "json_object"}));
942
943        let formatter = DeepSeekV4Formatter::new_chat();
944        let out = formatter.render(&req).unwrap();
945
946        assert!(
947            !out.contains("## Tools"),
948            "tool_choice=none must strip the tools block, got: {out}"
949        );
950        assert!(
951            !out.contains("get_current_weather"),
952            "tool schema leaked into prompt despite tool_choice=none: {out}"
953        );
954        assert!(
955            out.contains("## Response Format"),
956            "response_format must survive tool_choice=none: {out}"
957        );
958    }
959
960    #[test]
961    fn test_render_tool_choice_auto_keeps_tools() {
962        use crate::OAIPromptFormatter;
963
964        let req = MockRequest::new(json!([
965            {"role": "system", "content": "sys"},
966            {"role": "user", "content": "weather in Boston?"}
967        ]))
968        .with_tools(weather_tool())
969        .with_tool_choice(json!("auto"));
970
971        let formatter = DeepSeekV4Formatter::new_chat();
972        let out = formatter.render(&req).unwrap();
973
974        assert!(out.contains("## Tools"));
975        assert!(out.contains("get_current_weather"));
976    }
977
978    #[test]
979    fn test_render_absent_tool_choice_keeps_tools() {
980        use crate::OAIPromptFormatter;
981
982        let req = MockRequest::new(json!([
983            {"role": "system", "content": "sys"},
984            {"role": "user", "content": "weather in Boston?"}
985        ]))
986        .with_tools(weather_tool());
987
988        let formatter = DeepSeekV4Formatter::new_chat();
989        let out = formatter.render(&req).unwrap();
990
991        assert!(out.contains("## Tools"));
992        assert!(out.contains("get_current_weather"));
993    }
994
995    #[test]
996    fn test_resolve_reasoning_effort_accepts_full_range() {
997        let effort = |v: &str| {
998            let value = json!(v);
999            DeepSeekV4Formatter::resolve_reasoning_effort(Some(&value))
1000        };
1001
1002        assert_eq!(effort("max"), (false, Some(ReasoningEffort::Max)));
1003        assert_eq!(effort("xhigh"), (false, Some(ReasoningEffort::High)));
1004        assert_eq!(effort("high"), (false, Some(ReasoningEffort::High)));
1005        assert_eq!(effort("minimal"), (false, None));
1006        assert_eq!(effort("low"), (false, None));
1007        assert_eq!(effort("medium"), (false, Some(ReasoningEffort::High)));
1008        assert_eq!(effort("none"), (true, None));
1009        assert_eq!(effort("bogus"), (false, Some(ReasoningEffort::High)));
1010        assert_eq!(
1011            DeepSeekV4Formatter::resolve_reasoning_effort(None),
1012            (false, Some(ReasoningEffort::High))
1013        );
1014    }
1015
1016    #[test]
1017    fn test_render_leaves_null_assistant_tool_content_empty() {
1018        use crate::OAIPromptFormatter;
1019
1020        let req = MockRequest::new(json!([
1021            {"role": "user", "content": "call tool"},
1022            {"role": "assistant", "content": null, "tool_calls": [{
1023                "id": "c1", "type": "function",
1024                "function": {"name": "f", "arguments": "{}"}
1025            }]}
1026        ]));
1027
1028        let formatter = DeepSeekV4Formatter::new_chat();
1029        let out = formatter.render(&req).unwrap();
1030
1031        assert!(out.contains(&format!(
1032            "<{}{}>",
1033            tokens::DSML_TOKEN,
1034            TOOL_CALLS_BLOCK_NAME
1035        )));
1036        assert!(!out.contains("null"));
1037    }
1038
1039    #[test]
1040    fn test_render_wires_reasoning_effort_from_chat_template_args() {
1041        use crate::OAIPromptFormatter;
1042        use std::collections::HashMap;
1043
1044        for (effort, expected) in [
1045            ("high", REASONING_EFFORT_HIGH),
1046            ("max", REASONING_EFFORT_MAX),
1047        ] {
1048            let mut args = HashMap::new();
1049            args.insert("reasoning_effort".to_string(), json!(effort));
1050
1051            let req = MockRequest::new(json!([
1052                {"role": "system", "content": "sys"},
1053                {"role": "user", "content": "hi"}
1054            ]))
1055            .with_chat_template_args(args);
1056
1057            let formatter = DeepSeekV4Formatter::new_thinking();
1058            let out = formatter.render(&req).unwrap();
1059
1060            assert!(out.starts_with(tokens::BOS));
1061            assert!(
1062                out[tokens::BOS.len()..].starts_with(expected),
1063                "{effort} preamble should appear after BOS, got:\n{out}"
1064            );
1065        }
1066    }
1067
1068    #[test]
1069    fn test_render_wires_top_level_reasoning_effort_and_none_disables_thinking() {
1070        use crate::OAIPromptFormatter;
1071
1072        let formatter = DeepSeekV4Formatter::new_thinking();
1073        for (effort, expected_prefix) in [
1074            ("high", "Reasoning Effort: Absolute maximum"),
1075            ("max", "Reasoning Effort: Beyond maximum"),
1076        ] {
1077            let req: dynamo_protocols::types::CreateChatCompletionRequest =
1078                serde_json::from_value(json!({
1079                    "model": "deepseek-v4",
1080                    "messages": [{"role": "user", "content": "hi"}],
1081                    "reasoning_effort": effort
1082                }))
1083                .unwrap();
1084            let out = formatter.render(&req).unwrap();
1085
1086            assert!(
1087                out[tokens::BOS.len()..].starts_with(expected_prefix),
1088                "top-level {effort} did not select its prefix: {out}"
1089            );
1090            assert!(out.ends_with(tokens::THINKING_START));
1091        }
1092
1093        let req: dynamo_protocols::types::CreateChatCompletionRequest =
1094            serde_json::from_value(json!({
1095                "model": "deepseek-v4",
1096                "messages": [{"role": "user", "content": "hi"}],
1097                "reasoning_effort": "none"
1098            }))
1099            .unwrap();
1100        let out = formatter.render(&req).unwrap();
1101
1102        assert_eq!(
1103            out,
1104            "<|begin▁of▁sentence|><|User|>hi<|Assistant|></think>"
1105        );
1106    }
1107
1108    #[test]
1109    fn test_top_level_reasoning_effort_precedes_template_argument() {
1110        use crate::OAIPromptFormatter;
1111        use std::collections::HashMap;
1112
1113        let mut args = HashMap::new();
1114        args.insert("reasoning_effort".to_string(), json!("max"));
1115        let req = MockRequest::new(json!([{"role": "user", "content": "hi"}]))
1116            .with_chat_template_args(args)
1117            .with_reasoning_effort(json!("low"));
1118
1119        let out = DeepSeekV4Formatter::new_thinking().render(&req).unwrap();
1120
1121        assert_eq!(
1122            out,
1123            "<|begin▁of▁sentence|><|User|>hi<|Assistant|><think>"
1124        );
1125    }
1126
1127    #[test]
1128    fn test_render_drop_thinking_override_from_chat_template_args() {
1129        use crate::OAIPromptFormatter;
1130        use std::collections::HashMap;
1131
1132        let messages = json!([
1133            {"role": "user", "content": "first"},
1134            {"role": "assistant", "reasoning_content": "PRIOR", "content": "reply"},
1135            {"role": "user", "content": "again"}
1136        ]);
1137
1138        // Default (drop_thinking=true): prior reasoning stripped.
1139        let req_default = MockRequest::new(messages.clone());
1140        let formatter = DeepSeekV4Formatter::new_thinking();
1141        let out_default = formatter.render(&req_default).unwrap();
1142        assert!(
1143            !out_default.contains("PRIOR"),
1144            "default drop_thinking=true should strip prior reasoning, got:\n{}",
1145            out_default
1146        );
1147
1148        // drop_thinking=false override: prior reasoning survives.
1149        let mut args = HashMap::new();
1150        args.insert("drop_thinking".to_string(), json!(false));
1151        let req_keep = MockRequest::new(messages).with_chat_template_args(args);
1152        let out_keep = formatter.render(&req_keep).unwrap();
1153        assert!(
1154            out_keep.contains("PRIOR"),
1155            "drop_thinking=false override should preserve prior reasoning, got:\n{}",
1156            out_keep
1157        );
1158    }
1159
1160    // N4: developer-role interactions with drop_thinking.
1161    // find_last_user_index returns the index of user OR developer messages; the
1162    // drop_thinking reasoning cutoff and the thinking-seed insertion treat
1163    // user and developer identically.
1164
1165    #[test]
1166    fn test_developer_only_conversation_renders_developer_content() {
1167        let messages = json!([
1168            {"role": "system", "content": "sys"},
1169            {"role": "developer", "content": "x"},
1170            {"role": "assistant", "reasoning_content": "R", "content": "ok"}
1171        ]);
1172        let out =
1173            encode_messages(messages.as_array().unwrap(), ThinkingMode::Thinking, true).unwrap();
1174        assert!(
1175            out.contains("x"),
1176            "developer content should appear in output, got:\n{}",
1177            out
1178        );
1179    }
1180
1181    #[test]
1182    fn test_developer_as_last_user_index_controls_reasoning_cutoff() {
1183        // Indices: 0=user, 1=assistant(FIRST), 2=developer(y), 3=assistant(SECOND).
1184        // find_last_user_index = 2 (developer). With drop_thinking=true:
1185        //   - assistant idx=1 < 2  → reasoning_content stripped.
1186        //   - assistant idx=3 >= 2 → reasoning_content preserved.
1187        let messages = json!([
1188            {"role": "user", "content": "a"},
1189            {"role": "assistant", "reasoning_content": "FIRST", "content": "r1"},
1190            {"role": "developer", "content": "y"},
1191            {"role": "assistant", "reasoning_content": "SECOND", "content": "r2"}
1192        ]);
1193        let out =
1194            encode_messages(messages.as_array().unwrap(), ThinkingMode::Thinking, true).unwrap();
1195        assert!(
1196            !out.contains("FIRST"),
1197            "reasoning before last user/developer (idx 1 < 2) should be stripped, got:\n{}",
1198            out
1199        );
1200        assert!(
1201            out.contains("SECOND"),
1202            "reasoning at/after last user/developer (idx 3 > 2) should survive, got:\n{}",
1203            out
1204        );
1205    }
1206}