Skip to main content

dynamo_renderer/
kimi_k3.rs

1// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
2// SPDX-License-Identifier: Apache-2.0
3
4//! Native Kimi K3 XTML prompt rendering.
5//!
6//! K3 does not ship a Jinja chat template. Its model-side `encoding_k3.py`
7//! emits a sequence of segments where protocol markers are encoded with
8//! tiktoken special IDs and message/tool data is encoded as ordinary text.
9//! Keeping that distinction is required both for model parity and to prevent a
10//! literal marker in user content from becoming prompt structure.
11
12use std::collections::HashMap;
13
14use anyhow::{Context, Result, bail};
15use serde_json::{Map, Value};
16
17use crate::{
18    OAIChatLikeRequest, OAIPromptFormatter, PromptRenderError, RenderedPrompt, RenderedSegment,
19    thinking_bool_from_args,
20};
21
22const OPEN_TOKEN: &str = "<|open|>";
23const CLOSE_TOKEN: &str = "<|close|>";
24const SEP_TOKEN: &str = "<|sep|>";
25const END_OF_MSG_TOKEN: &str = "<|end_of_msg|>";
26/// The one token this renderer emits per image.
27///
28/// This is the canonical frontend contract: exactly one `<|media_pad|>` per
29/// image, for every engine. It is a registered special token in the K3
30/// tokenizer (`config.json`'s `media_placeholder_token_id`), so it encodes to
31/// a single id and stays one id no matter what surrounds it.
32///
33/// The checkpoint's other spelling, `<|kimi_image_placeholder|>`, is a plain
34/// string that is *not* in the vocabulary — it BPE-shatters into several ids
35/// whose boundaries depend on neighbouring text. Engines that want that form
36/// (vLLM) convert from the pad on the worker side, where a single known id is
37/// a reliable thing to substitute; matching a shattered string is not.
38///
39/// Equivalent to calling the checkpoint's own
40/// `encoding_k3.build_chat_segments(image_prompts=["<|media_pad|>"] * n)` —
41/// `image_prompts` is the model author's hook for exactly this choice, and
42/// `<|kimi_image_placeholder|>` is only its `None` fallback.
43const MEDIA_PAD: &str = "<|media_pad|>";
44const VALID_THINKING_EFFORTS: &[&str] = &["low", "high", "max"];
45
46#[derive(Debug, Clone)]
47pub struct KimiK3Formatter {
48    exclude_tools_when_tool_choice_none: bool,
49}
50
51impl KimiK3Formatter {
52    pub fn new(exclude_tools_when_tool_choice_none: bool) -> Self {
53        Self {
54            exclude_tools_when_tool_choice_none,
55        }
56    }
57
58    /// Render the conversation into segments, also returning how many trailing
59    /// segments form the assistant generation stub (`<|open|>think<|sep|>` or
60    /// `<|open|>response<|sep|>`). Moonshot's K3 API feeds that stub to the
61    /// model but excludes it from reported `usage.prompt_tokens`.
62    fn build_segments(&self, req: &dyn OAIChatLikeRequest) -> Result<ChatSegments> {
63        let messages = crate::messages_to_json(req).context("Failed to convert K3 messages")?;
64        let Value::Array(messages) = messages else {
65            anyhow::bail!("Kimi K3 messages must be an array");
66        };
67        let messages = normalize_tool_result_messages(messages)?;
68
69        let tool_choice = req.tool_choice().map(json_value).transpose()?;
70        let (tool_choice_kind, named_tool) = resolve_tool_choice(tool_choice.as_ref())?;
71        let mut tools = req.tools().map(json_value).transpose()?;
72        // A named tool_choice may target a message-level declaration that
73        // never appears in the top-level list.
74        if let Some(named_tool) = named_tool
75            && !tools
76                .as_ref()
77                .is_some_and(|tools| contains_tool(tools, named_tool))
78            && !messages
79                .iter()
80                .any(|message| message_declares_tool(message, named_tool))
81        {
82            return Err(PromptRenderError::invalid_request(format!(
83                "tool named {named_tool:?} in tool_choice is not present in tools"
84            ))
85            .into());
86        }
87        if self.exclude_tools_when_tool_choice_none && tool_choice_kind == Some("none") {
88            tools = None;
89        }
90        let tools = tools.map(deep_sort);
91
92        let args = req.chat_template_args();
93        // Moonshot's K3 API defines named tool choice as incompatible with
94        // thinking. Make the public function-object form work without requiring
95        // clients to know K3-specific chat-template arguments.
96        let thinking = named_tool.is_none() && thinking_bool_from_args(args).unwrap_or(true);
97        let thinking_effort = resolve_thinking_effort(args);
98        if thinking && !VALID_THINKING_EFFORTS.contains(&thinking_effort.as_str()) {
99            return Err(PromptRenderError::invalid_request(format!(
100                "Unsupported Kimi K3 thinking_effort={thinking_effort:?}; supported values are low, high, and max"
101            ))
102            .into());
103        }
104
105        let response_format = req.response_format().map(json_value).transpose()?;
106        build_chat_segments(
107            &messages,
108            tools.as_ref(),
109            tool_choice_kind,
110            named_tool,
111            response_format.as_ref(),
112            req.should_add_generation_prompt(),
113            thinking,
114            thinking_effort.as_str(),
115        )
116    }
117}
118
119impl OAIPromptFormatter for KimiK3Formatter {
120    fn supports_add_generation_prompt(&self) -> bool {
121        true
122    }
123
124    fn render(&self, req: &dyn OAIChatLikeRequest) -> Result<String> {
125        Ok(self.build_segments(req)?.into_prompt().into_text())
126    }
127
128    fn render_prompt(&self, req: &dyn OAIChatLikeRequest) -> Result<RenderedPrompt> {
129        Ok(self.build_segments(req)?.into_prompt())
130    }
131}
132
133struct ChatSegments {
134    segments: Vec<RenderedSegment>,
135    /// Trailing segments that open the assistant channel for generation.
136    /// Zero when no generation prompt was added or a partial assistant
137    /// message is being continued.
138    pending_segments: usize,
139}
140
141impl ChatSegments {
142    fn into_prompt(self) -> RenderedPrompt {
143        RenderedPrompt::segmented_with_pending(self.segments, self.pending_segments)
144    }
145}
146
147fn json_value(value: minijinja::value::Value) -> Result<Value> {
148    serde_json::to_value(&value).context("Failed to convert template value to JSON")
149}
150
151fn resolve_tool_choice(tool_choice: Option<&Value>) -> Result<(Option<&str>, Option<&str>)> {
152    match tool_choice {
153        Some(Value::String(kind)) => Ok((Some(kind.as_str()), None)),
154        Some(Value::Object(choice)) => {
155            if choice.get("type").and_then(Value::as_str) != Some("function") {
156                return Err(PromptRenderError::invalid_request(
157                    "Kimi K3 named tool_choice must have type=\"function\"",
158                )
159                .into());
160            }
161            // Chat Completions uses function.name. Responses API uses a
162            // top-level name and is normalized to the same internal request in
163            // Dynamo, but accepting both shapes keeps this renderer reusable.
164            let name = choice
165                .get("function")
166                .and_then(Value::as_object)
167                .and_then(|function| function.get("name"))
168                .or_else(|| choice.get("name"))
169                .and_then(Value::as_str)
170                .filter(|name| !name.is_empty())
171                .ok_or_else(|| {
172                    PromptRenderError::invalid_request(
173                        "Kimi K3 named tool_choice requires a non-empty function name",
174                    )
175                })?;
176            Ok((Some("specified"), Some(name)))
177        }
178        Some(Value::Null) | None => Ok((None, None)),
179        Some(other) => Err(anyhow::anyhow!(
180            "Unsupported Kimi K3 tool_choice value: {other}"
181        )),
182    }
183}
184
185fn contains_tool(tools: &Value, name: &str) -> bool {
186    tools.as_array().is_some_and(|tools| {
187        tools.iter().any(|tool| {
188            tool.get("function")
189                .and_then(Value::as_object)
190                .and_then(|function| function.get("name"))
191                .or_else(|| tool.get("name"))
192                .and_then(Value::as_str)
193                == Some(name)
194        })
195    })
196}
197
198/// Longest tool name Moonshot's vendor verifier accepts.
199const MAX_TOOL_NAME_LEN: usize = 256;
200
201/// The dynamic tool declaration carried by a system or developer message.
202///
203/// `Ok(Some(..))` for a non-empty `tools` array; `Ok(None)` when `tools` is
204/// missing, `null`, or an empty array (an empty list declares nothing, so the
205/// message is an ordinary turn); `Err` for any other JSON type.
206fn dynamic_tools_of(message: &Value) -> Result<Option<&Vec<Value>>> {
207    match message.get("tools") {
208        None | Some(Value::Null) => Ok(None),
209        Some(Value::Array(tools)) if tools.is_empty() => Ok(None),
210        Some(Value::Array(tools)) => Ok(Some(tools)),
211        Some(_) => Err(PromptRenderError::invalid_request(
212            "Kimi K3 dynamic tool messages need `tools` to be an array",
213        )
214        .into()),
215    }
216}
217
218/// Accepts both OpenAI-wrapped and bare Kimi tool declarations; rejects mixed
219/// shapes so every entry has one unambiguous name.
220fn dynamic_tool_entry_name(tool: &Value) -> Result<&str> {
221    let object = tool.as_object().ok_or_else(|| {
222        PromptRenderError::invalid_request("Kimi K3 dynamic tool entries must be JSON objects")
223    })?;
224    let name = match (object.get("type"), object.get("function")) {
225        (Some(kind), function) => {
226            if kind.as_str() != Some("function") {
227                return Err(PromptRenderError::invalid_request(format!(
228                    "Kimi K3 dynamic tool entries must have type=\"function\", got {kind}"
229                ))
230                .into());
231            }
232            let function = function.and_then(Value::as_object).ok_or_else(|| {
233                PromptRenderError::invalid_request(
234                    "Kimi K3 dynamic tool entries with type=\"function\" need a `function` object",
235                )
236            })?;
237            function.get("name")
238        }
239        (None, Some(_)) => {
240            return Err(PromptRenderError::invalid_request(
241                "Kimi K3 dynamic tool entries with a `function` object need type=\"function\"",
242            )
243            .into());
244        }
245        (None, None) => object.get("name"),
246    };
247    let name = name.and_then(Value::as_str).ok_or_else(|| {
248        PromptRenderError::invalid_request("Kimi K3 dynamic tool entries need a string `name`")
249    })?;
250    validate_tool_name(name)?;
251    Ok(name)
252}
253
254/// Tool names must match `[A-Za-z_][A-Za-z0-9_-]*` and be at most
255/// [`MAX_TOOL_NAME_LEN`] characters (Moonshot vendor verifier rules).
256fn validate_tool_name(name: &str) -> Result<()> {
257    let mut chars = name.chars();
258    let valid_start = chars
259        .next()
260        .is_some_and(|c| c.is_ascii_alphabetic() || c == '_');
261    let valid_rest = chars.all(|c| c.is_ascii_alphanumeric() || c == '_' || c == '-');
262    if !valid_start || !valid_rest {
263        return Err(PromptRenderError::invalid_request(format!(
264            "Kimi K3 tool name {name:?} must match [A-Za-z_][A-Za-z0-9_-]*"
265        ))
266        .into());
267    }
268    if name.len() > MAX_TOOL_NAME_LEN {
269        return Err(PromptRenderError::invalid_request(format!(
270            "Kimi K3 tool name is {} characters; the maximum is {MAX_TOOL_NAME_LEN}",
271            name.len()
272        ))
273        .into());
274    }
275    Ok(())
276}
277
278/// Validate top-level and message-level tools as one namespace: entries must
279/// be well-formed and names unique. Raw requests retain the renderer's native
280/// developer-tool support alongside Kimi's system-tool declarations.
281fn validate_tool_declarations(top_level: Option<&Value>, messages: &[Value]) -> Result<()> {
282    let mut seen = std::collections::HashSet::new();
283    // The OpenAI schema does not enforce Kimi's tool-name rules.
284    for tool in top_level.and_then(Value::as_array).into_iter().flatten() {
285        let name = dynamic_tool_entry_name(tool)?;
286        if !seen.insert(name) {
287            return Err(PromptRenderError::invalid_request(format!(
288                "tool {name:?} is declared more than once in `tools`"
289            ))
290            .into());
291        }
292    }
293    for message in messages {
294        let role = message.get("role").and_then(Value::as_str);
295        if !matches!(role, Some("system" | "developer")) {
296            if message.get("tools").is_some_and(|tools| !tools.is_null()) {
297                return Err(PromptRenderError::invalid_request(format!(
298                    "`tools` is only accepted on system or developer messages, not on role {}",
299                    role.unwrap_or("<missing>")
300                ))
301                .into());
302            }
303            continue;
304        }
305        for tool in dynamic_tools_of(message)?.into_iter().flatten() {
306            let name = dynamic_tool_entry_name(tool)?;
307            if !seen.insert(name) {
308                return Err(PromptRenderError::invalid_request(format!(
309                    "tool {name:?} is declared more than once across `tools` and dynamic message tools"
310                ))
311                .into());
312            }
313        }
314    }
315    Ok(())
316}
317
318/// Whether `content` carries text: missing, `null`, `""`, and `[]` all count
319/// as empty, matching Moonshot's "omit `content`" contract for dynamic tools.
320fn content_is_non_empty(content: Option<&Value>) -> bool {
321    match content {
322        None | Some(Value::Null) => false,
323        Some(Value::String(text)) => !text.is_empty(),
324        Some(Value::Array(parts)) => !parts.is_empty(),
325        Some(_) => true,
326    }
327}
328
329fn message_declares_tool(message: &Value, name: &str) -> bool {
330    matches!(
331        message.get("role").and_then(Value::as_str),
332        Some("system" | "developer")
333    ) && message
334        .get("tools")
335        .is_some_and(|tools| contains_tool(tools, name))
336}
337
338fn resolve_thinking_effort(args: Option<&HashMap<String, Value>>) -> String {
339    args.and_then(|args| {
340        args.get("thinking_effort")
341            .or_else(|| args.get("reasoning_effort"))
342            .and_then(Value::as_str)
343    })
344    .unwrap_or("max")
345    .to_string()
346}
347
348fn push_segment(segments: &mut Vec<RenderedSegment>, text: impl Into<String>, allow_special: bool) {
349    let text = text.into();
350    if !text.is_empty() {
351        segments.push(RenderedSegment {
352            text,
353            allow_special,
354        });
355    }
356}
357
358fn control(segments: &mut Vec<RenderedSegment>, text: impl Into<String>) {
359    push_segment(segments, text, true);
360}
361
362fn text(segments: &mut Vec<RenderedSegment>, text: impl Into<String>) {
363    push_segment(segments, text, false);
364}
365
366fn escape_attr_value(value: impl std::fmt::Display) -> String {
367    value
368        .to_string()
369        .replace('&', "&amp;")
370        .replace('"', "&quot;")
371}
372
373fn open_tag(
374    segments: &mut Vec<RenderedSegment>,
375    tag: &str,
376    attrs: impl IntoIterator<Item = (String, String)>,
377) {
378    control(segments, OPEN_TOKEN);
379    text(segments, tag);
380    for (key, value) in attrs {
381        text(segments, format!(" {key}"));
382        text(segments, "=\"");
383        text(segments, escape_attr_value(value));
384        text(segments, "\"");
385    }
386    control(segments, SEP_TOKEN);
387}
388
389fn close_tag(segments: &mut Vec<RenderedSegment>, tag: &str) {
390    control(segments, CLOSE_TOKEN);
391    text(segments, tag);
392    control(segments, SEP_TOKEN);
393}
394
395fn end_of_msg(segments: &mut Vec<RenderedSegment>) {
396    control(segments, END_OF_MSG_TOKEN);
397}
398
399fn internal_system_message(segments: &mut Vec<RenderedSegment>, message_type: &str, body: &str) {
400    open_tag(
401        segments,
402        "message",
403        [
404            ("role".to_string(), "system".to_string()),
405            ("type".to_string(), message_type.to_string()),
406        ],
407    );
408    text(segments, body.trim());
409    close_tag(segments, "message");
410    end_of_msg(segments);
411}
412
413fn deep_sort(value: Value) -> Value {
414    match value {
415        Value::Object(map) => {
416            let mut entries: Vec<_> = map.into_iter().collect();
417            entries.sort_by(|(left, _), (right, _)| left.cmp(right));
418            Value::Object(
419                entries
420                    .into_iter()
421                    .map(|(key, value)| (key, deep_sort(value)))
422                    .collect(),
423            )
424        }
425        Value::Array(items) => Value::Array(items.into_iter().map(deep_sort).collect()),
426        other => other,
427    }
428}
429
430fn compact_json(value: &Value) -> Result<String> {
431    serde_json::to_string(value).context("Failed to serialize K3 JSON")
432}
433
434fn response_schema(response_format: &Value) -> Option<Value> {
435    let json_schema = response_format.get("json_schema")?;
436    if let Some(schema) = json_schema.get("schema") {
437        return Some(schema.clone());
438    }
439    if let Some(schema) = json_schema.get("json_schema") {
440        return Some(schema.clone());
441    }
442    Some(json_schema.clone())
443}
444
445fn value_as_body_text(value: &Value) -> Result<String> {
446    match value {
447        Value::String(value) => Ok(value.clone()),
448        Value::Array(values) if values.iter().all(Value::is_string) => Ok(values
449            .iter()
450            .filter_map(Value::as_str)
451            .filter(|value| !value.is_empty())
452            .collect::<Vec<_>>()
453            .join("\n")),
454        other => compact_json(other),
455    }
456}
457
458fn render_content_segments(
459    segments: &mut Vec<RenderedSegment>,
460    content: Option<&Value>,
461) -> Result<()> {
462    let Some(content) = content else {
463        return Ok(());
464    };
465    match content {
466        Value::Null => {}
467        Value::String(value) => text(segments, value),
468        Value::Array(parts) => {
469            for part in parts {
470                match part.get("type").and_then(Value::as_str) {
471                    Some("image" | "image_url") => control(segments, MEDIA_PAD),
472                    _ => {
473                        if let Some(part_text) = part.get("text") {
474                            text(segments, value_as_body_text(part_text)?);
475                        }
476                    }
477                }
478            }
479        }
480        other => text(segments, value_as_body_text(other)?),
481    }
482    Ok(())
483}
484
485fn render_role_message(
486    segments: &mut Vec<RenderedSegment>,
487    message: &Value,
488    role: &str,
489) -> Result<()> {
490    let mut attrs = vec![("role".to_string(), role.to_string())];
491    if let Some(name) = message
492        .get("name")
493        .and_then(Value::as_str)
494        .filter(|name| !name.is_empty())
495    {
496        attrs.push(("name".to_string(), name.to_string()));
497    }
498    open_tag(segments, "message", attrs);
499    render_content_segments(segments, message.get("content"))?;
500    close_tag(segments, "message");
501    end_of_msg(segments);
502    Ok(())
503}
504
505fn render_tool_declare(
506    segments: &mut Vec<RenderedSegment>,
507    tools: &Value,
508    dynamic: bool,
509) -> Result<()> {
510    let tools = compact_json(tools)?;
511    let body = if dynamic {
512        format!(
513            "## New Tools Available\n\
514             The system dynamically extends the toolset via lazy-loading.\n\
515             You have access to all existing and extended tools.\n\
516             Here are the specs for the extended tools.\n\n\
517             ```json\n{tools}\n```"
518        )
519    } else {
520        format!(
521            "# Tools\n\
522             Here are the available tools, described in JSONSchema.\n\n\
523             ```json\n{tools}\n```"
524        )
525    };
526    open_tag(
527        segments,
528        "message",
529        [
530            ("role".to_string(), "system".to_string()),
531            ("type".to_string(), "tool-declare".to_string()),
532        ],
533    );
534    text(segments, body);
535    close_tag(segments, "message");
536    end_of_msg(segments);
537    Ok(())
538}
539
540fn xtml_type(value: &Value) -> &'static str {
541    match value {
542        Value::Bool(_) => "boolean",
543        Value::Null => "null",
544        Value::Number(_) => "number",
545        Value::String(_) => "string",
546        Value::Object(_) => "object",
547        Value::Array(_) => "array",
548    }
549}
550
551fn xtml_value(value: &Value) -> Result<String> {
552    match value {
553        Value::String(value) => Ok(value.clone()),
554        // Python's `json.dumps(..., ensure_ascii=False)` uses `", "` and
555        // `": "` separators by default. Preserve that byte shape in prompt
556        // history; the compact form is used only for schemas/tool declarations.
557        other => python_default_json(other),
558    }
559}
560
561fn python_default_json(value: &Value) -> Result<String> {
562    let compact = compact_json(value)?;
563    let mut output = String::with_capacity(compact.len());
564    let mut in_string = false;
565    let mut escaped = false;
566    for ch in compact.chars() {
567        output.push(ch);
568        if in_string {
569            if escaped {
570                escaped = false;
571            } else if ch == '\\' {
572                escaped = true;
573            } else if ch == '"' {
574                in_string = false;
575            }
576        } else if ch == '"' {
577            in_string = true;
578        } else if matches!(ch, ',' | ':') {
579            output.push(' ');
580        }
581    }
582    Ok(output)
583}
584
585enum NormalizedArguments {
586    Object(Map<String, Value>),
587    JsonBlock(String),
588}
589
590fn normalize_arguments(arguments: Option<&Value>) -> Result<NormalizedArguments> {
591    let Some(arguments) = arguments else {
592        return Ok(NormalizedArguments::Object(Map::new()));
593    };
594    match arguments {
595        Value::Null => Ok(NormalizedArguments::Object(Map::new())),
596        Value::Object(arguments) => Ok(NormalizedArguments::Object(arguments.clone())),
597        Value::String(arguments) if arguments.trim().is_empty() => {
598            Ok(NormalizedArguments::Object(Map::new()))
599        }
600        Value::String(arguments) => match serde_json::from_str::<Value>(arguments) {
601            Ok(Value::Object(arguments)) => Ok(NormalizedArguments::Object(arguments)),
602            Ok(_) => bail!("Kimi K3 tool call arguments must be a JSON object"),
603            Err(_) => Ok(NormalizedArguments::JsonBlock(arguments.clone())),
604        },
605        _ => bail!("Kimi K3 tool call arguments must be an object or JSON object string"),
606    }
607}
608
609/// Renders an assistant message's think channel.
610///
611/// The think channel is structural in the latest K3 model encoding. Every
612/// historical assistant message carries it in thinking mode, even if its body
613/// is empty. Non-thinking mode drops both the channel and preserved reasoning
614/// content.
615fn render_think_channel(
616    segments: &mut Vec<RenderedSegment>,
617    message: &Value,
618    thinking: bool,
619) -> Result<()> {
620    if !thinking {
621        return Ok(());
622    }
623    // Match encoding_k3.py: `reasoning_content` wins when truthy, otherwise
624    // fall back to the Responses-style `reasoning` alias.
625    let reasoning = message
626        .get("reasoning_content")
627        .filter(|value| match value {
628            Value::Null => false,
629            Value::Bool(value) => *value,
630            Value::Number(value) => value.as_f64().is_some_and(|value| value != 0.0),
631            Value::String(value) => !value.is_empty(),
632            Value::Array(value) => !value.is_empty(),
633            Value::Object(value) => !value.is_empty(),
634        })
635        .or_else(|| message.get("reasoning"))
636        .map(value_as_body_text)
637        .transpose()?;
638
639    open_tag(segments, "think", []);
640    if let Some(reasoning) = reasoning.filter(|reasoning| !reasoning.trim().is_empty()) {
641        text(segments, reasoning);
642    }
643    close_tag(segments, "think");
644    Ok(())
645}
646
647fn assistant_message_attrs(message: &Value) -> Vec<(String, String)> {
648    let mut attrs = vec![("role".to_string(), "assistant".to_string())];
649    if let Some(name) = message
650        .get("name")
651        .and_then(Value::as_str)
652        .filter(|name| !name.is_empty())
653    {
654        attrs.push(("name".to_string(), name.to_string()));
655    }
656    attrs
657}
658
659fn is_partial(message: &Value) -> bool {
660    message.get("partial").and_then(Value::as_bool) == Some(true)
661}
662
663/// Leaves the assistant response and message open for prefix continuation,
664/// replacing the ordinary generation prompt. In thinking mode, the think
665/// channel is rendered and closed before the response opens.
666fn render_partial_assistant_segments(
667    segments: &mut Vec<RenderedSegment>,
668    message: &Value,
669    thinking: bool,
670) -> Result<()> {
671    if message
672        .get("tool_calls")
673        .is_some_and(|calls| !calls.is_null() && !calls.as_array().is_some_and(Vec::is_empty))
674    {
675        return Err(PromptRenderError::invalid_request(
676            "Kimi K3 partial assistant messages cannot carry tool_calls",
677        )
678        .into());
679    }
680    open_tag(segments, "message", assistant_message_attrs(message));
681    render_think_channel(segments, message, thinking)?;
682    open_tag(segments, "response", []);
683    render_content_segments(segments, message.get("content"))?;
684    Ok(())
685}
686
687fn render_assistant_segments(
688    segments: &mut Vec<RenderedSegment>,
689    message: &Value,
690    thinking: bool,
691) -> Result<()> {
692    render_think_channel(segments, message, thinking)?;
693
694    open_tag(segments, "response", []);
695    render_content_segments(segments, message.get("content"))?;
696    close_tag(segments, "response");
697
698    let Some(tool_calls) = message.get("tool_calls").and_then(Value::as_array) else {
699        return Ok(());
700    };
701    if tool_calls.is_empty() {
702        return Ok(());
703    }
704
705    open_tag(segments, "tools", []);
706    for (position, tool_call) in tool_calls.iter().enumerate() {
707        let function = tool_call.get("function").unwrap_or(tool_call);
708        let name = function
709            .get("name")
710            .and_then(Value::as_str)
711            .context("Kimi K3 tool call is missing function.name")?;
712        open_tag(
713            segments,
714            "call",
715            [
716                ("tool".to_string(), name.to_string()),
717                ("index".to_string(), (position + 1).to_string()),
718            ],
719        );
720
721        match normalize_arguments(function.get("arguments"))? {
722            NormalizedArguments::JsonBlock(raw) => {
723                open_tag(
724                    segments,
725                    "json",
726                    [("type".to_string(), "object".to_string())],
727                );
728                text(segments, raw);
729                close_tag(segments, "json");
730            }
731            NormalizedArguments::Object(arguments) => {
732                for (key, value) in arguments {
733                    open_tag(
734                        segments,
735                        "argument",
736                        [
737                            ("key".to_string(), key),
738                            ("type".to_string(), xtml_type(&value).to_string()),
739                        ],
740                    );
741                    text(segments, xtml_value(&value)?);
742                    close_tag(segments, "argument");
743                }
744            }
745        }
746        close_tag(segments, "call");
747    }
748    close_tag(segments, "tools");
749    Ok(())
750}
751
752fn tool_call_index(tool_calls: Option<&Value>) -> HashMap<String, (usize, Option<String>)> {
753    let mut index = HashMap::new();
754    let Some(tool_calls) = tool_calls.and_then(Value::as_array) else {
755        return index;
756    };
757    for (position, tool_call) in tool_calls.iter().enumerate() {
758        let Some(id) = tool_call.get("id").and_then(Value::as_str) else {
759            continue;
760        };
761        let function = tool_call.get("function").unwrap_or(tool_call);
762        let name = function
763            .get("name")
764            .and_then(Value::as_str)
765            .map(str::to_string);
766        index.entry(id.to_string()).or_insert((position + 1, name));
767    }
768    index
769}
770
771fn normalize_tool_result_messages(messages: Vec<Value>) -> Result<Vec<Value>> {
772    let mut output = Vec::with_capacity(messages.len());
773    let mut current_index = HashMap::new();
774    let mut messages = messages.into_iter().peekable();
775
776    while let Some(message) = messages.peek() {
777        let role = message.get("role").and_then(Value::as_str);
778        if role == Some("assistant") {
779            current_index = tool_call_index(message.get("tool_calls"));
780            output.push(messages.next().expect("peeked message"));
781            continue;
782        }
783        if role != Some("tool") {
784            output.push(messages.next().expect("peeked message"));
785            continue;
786        }
787
788        let mut run: Vec<(Option<usize>, usize, Value, Option<String>)> = Vec::new();
789        let mut unresolved = false;
790        let mut offset = 0;
791        while messages
792            .peek()
793            .is_some_and(|message| message.get("role").and_then(Value::as_str) == Some("tool"))
794        {
795            let tool_message = messages.next().expect("peeked tool message");
796            let call_id = tool_message
797                .get("tool_call_id")
798                .or_else(|| tool_message.get("id"))
799                .and_then(Value::as_str);
800            let matched = call_id.and_then(|id| current_index.get(id));
801            if let Some((tool_position, name)) = matched {
802                run.push((Some(*tool_position), offset, tool_message, name.clone()));
803            } else {
804                unresolved = true;
805                run.push((None, offset, tool_message, None));
806            }
807            offset += 1;
808        }
809
810        if unresolved {
811            output.extend(run.into_iter().map(|(_, _, message, _)| message));
812            continue;
813        }
814        run.sort_by_key(|(tool_position, offset, _, _)| (*tool_position, *offset));
815        for (_, _, mut message, name) in run {
816            if let (Some(name), Some(message)) = (name, message.as_object_mut()) {
817                message.insert("tool".to_string(), Value::String(name.clone()));
818                if message.contains_key("name") {
819                    message.insert("name".to_string(), Value::String(name));
820                }
821            }
822            output.push(message);
823        }
824    }
825    Ok(output)
826}
827
828#[allow(clippy::too_many_arguments)]
829fn build_chat_segments(
830    messages: &[Value],
831    tools: Option<&Value>,
832    tool_choice: Option<&str>,
833    named_tool: Option<&str>,
834    response_format: Option<&Value>,
835    add_generation_prompt: bool,
836    thinking: bool,
837    thinking_effort: &str,
838) -> Result<ChatSegments> {
839    let mut segments = Vec::new();
840    let mut pending_segments = 0usize;
841    let mut previous_tool_calls: Option<&Value> = None;
842    let mut tool_index = 0usize;
843
844    for message in messages {
845        let Some(partial) = message.get("partial").filter(|value| !value.is_null()) else {
846            continue;
847        };
848        if message.get("role").and_then(Value::as_str) != Some("assistant") {
849            return Err(PromptRenderError::invalid_request(
850                "Kimi K3 `partial` is only supported on an assistant message",
851            )
852            .into());
853        }
854        if !partial.is_boolean() {
855            return Err(
856                PromptRenderError::invalid_request("Kimi K3 `partial` must be a boolean").into(),
857            );
858        }
859    }
860
861    // Kimi Partial Mode: only the final message may be partial, and it must be
862    // an assistant turn. Split it off so the history loop renders everything
863    // before it normally and the partial turn takes the generation prompt's
864    // place at the very end (after any internal system messages).
865    let (history, partial_tail) = match messages.split_last() {
866        Some((last, history)) if is_partial(last) => (history, Some(last)),
867        _ => (messages, None),
868    };
869
870    // Validate the complete raw message list, including a split-off Partial
871    // Mode tail. Otherwise `tools` on the final partial assistant message
872    // bypasses the supported-role check below.
873    validate_tool_declarations(tools, messages)?;
874    if history.iter().any(is_partial) {
875        return Err(PromptRenderError::invalid_request(
876            "Kimi K3 `partial` is only supported on the final message",
877        )
878        .into());
879    }
880
881    if let Some(tools) = tools.filter(|tools| !tools.as_array().is_some_and(Vec::is_empty)) {
882        render_tool_declare(&mut segments, tools, false)?;
883    }
884
885    if thinking {
886        internal_system_message(
887            &mut segments,
888            "thinking-effort",
889            &format!(
890                "`thinking_effort` guides on how much to think in your thinking channel \
891                 (not including the response channel), supported values include `low`, \
892                 `medium`, `high`, and `max`.\nNow the system is invoked with \
893                 `thinking_effort={thinking_effort}`."
894            ),
895        );
896    }
897
898    for message in history {
899        let role = message.get("role").and_then(Value::as_str).ok_or_else(|| {
900            PromptRenderError::invalid_request("Kimi K3 messages must contain a string role")
901        })?;
902        // An empty `tools` list is not a dynamic-tool declaration.
903        let dynamic_tools = dynamic_tools_of(message)?;
904        match role {
905            "system" | "developer" if dynamic_tools.is_some() => {
906                let dynamic_tools = dynamic_tools.expect("guarded by the match arm");
907                // Moonshot's contract: a dynamic-tool system message omits
908                // `content` (an empty string counts as omitted; the official
909                // verifier sends `"content": ""`). Rejecting non-empty text
910                // keeps it from being silently lost.
911                if role == "system" && content_is_non_empty(message.get("content")) {
912                    return Err(PromptRenderError::invalid_request(
913                        "Kimi K3 system messages carry either `content` or `tools`, not both",
914                    )
915                    .into());
916                }
917                let dynamic_tools = deep_sort(Value::Array(dynamic_tools.clone()));
918                render_tool_declare(&mut segments, &dynamic_tools, true)?;
919                if role == "developer"
920                    && message
921                        .get("content")
922                        .is_some_and(|content| !content.is_null())
923                {
924                    render_role_message(&mut segments, message, "system")?;
925                }
926            }
927            "system" | "developer"
928                if message
929                    .get("content")
930                    .is_none_or(|content| content.is_null()) =>
931            {
932                return Err(PromptRenderError::invalid_request(format!(
933                    "Kimi K3 {role} messages need `content` or `tools`"
934                ))
935                .into());
936            }
937            "user" | "system" | "developer" => {
938                let rendered_role = if role == "developer" { "system" } else { role };
939                render_role_message(&mut segments, message, rendered_role)?;
940            }
941            "assistant" => {
942                previous_tool_calls = message.get("tool_calls");
943                tool_index = 0;
944                open_tag(&mut segments, "message", assistant_message_attrs(message));
945                render_assistant_segments(&mut segments, message, thinking)?;
946                close_tag(&mut segments, "message");
947                end_of_msg(&mut segments);
948            }
949            "tool" => {
950                tool_index += 1;
951                let fallback_name = previous_tool_calls
952                    .and_then(Value::as_array)
953                    .and_then(|calls| calls.get(tool_index - 1))
954                    .map(|call| call.get("function").unwrap_or(call))
955                    .and_then(|function| function.get("name"))
956                    .and_then(Value::as_str);
957                let tool_name = message
958                    .get("tool")
959                    .or_else(|| message.get("name"))
960                    .and_then(Value::as_str)
961                    .or(fallback_name)
962                    .context(
963                        "Kimi K3 tool messages need a tool/name or a preceding assistant tool call",
964                    )?;
965                open_tag(
966                    &mut segments,
967                    "message",
968                    [
969                        ("role".to_string(), "tool".to_string()),
970                        ("tool".to_string(), tool_name.to_string()),
971                        ("index".to_string(), tool_index.to_string()),
972                    ],
973                );
974                render_content_segments(&mut segments, message.get("content"))?;
975                close_tag(&mut segments, "message");
976                end_of_msg(&mut segments);
977            }
978            unsupported => {
979                return Err(PromptRenderError::invalid_request(format!(
980                    "Kimi K3 does not support message role {unsupported:?}"
981                ))
982                .into());
983            }
984        }
985    }
986
987    match tool_choice {
988        Some("required") => internal_system_message(
989            &mut segments,
990            "tool-choice",
991            "The system is invoked with `tool_choice=required`.\n\
992             You MUST call tools in the next message.",
993        ),
994        Some("none") => internal_system_message(
995            &mut segments,
996            "tool-choice",
997            "The system is invoked with `tool_choice=none`.\n\
998             You MUST NOT call any tools in the next message.",
999        ),
1000        Some("specified") => internal_system_message(
1001            &mut segments,
1002            "tool-choice",
1003            &format!(
1004                "The system is invoked with `tool_choice=specified`.\n\
1005                 You MUST call the tool `{}` in the next message.",
1006                named_tool.expect("specified tool_choice has a function name")
1007            ),
1008        ),
1009        _ => {}
1010    }
1011
1012    if let Some(response_format) = response_format {
1013        match response_format.get("type").and_then(Value::as_str) {
1014            Some("json_object") => internal_system_message(
1015                &mut segments,
1016                "response-format",
1017                "The system is invoked with `response_format=json_object`.\n\
1018                 Your response must be raw JSON data without markdown code blocks \
1019                 (```json) or any additional formatting.",
1020            ),
1021            Some("json_schema") => {
1022                let schema = response_schema(response_format)
1023                    .map(deep_sort)
1024                    .unwrap_or(Value::Null);
1025                internal_system_message(
1026                    &mut segments,
1027                    "response-format",
1028                    &format!(
1029                        "The system is invoked with `response_format=json_schema`.\n\
1030                         Your response must be raw JSON data without markdown code blocks \
1031                         (```json) or any additional formatting.\n\
1032                         The JSON data must match the following schema:\n\
1033                         ```json\n{}\n```",
1034                        compact_json(&schema)?
1035                    ),
1036                );
1037            }
1038            _ => {}
1039        }
1040    }
1041
1042    // A partial assistant turn *is* the generation prompt: it is left open so
1043    // the model continues from its prefix, so the generic prompt is skipped
1044    // regardless of `add_generation_prompt`.
1045    if let Some(partial) = partial_tail {
1046        render_partial_assistant_segments(&mut segments, partial, thinking)?;
1047    } else if add_generation_prompt {
1048        open_tag(
1049            &mut segments,
1050            "message",
1051            [("role".to_string(), "assistant".to_string())],
1052        );
1053        // Only the channel-opening tag is the "pending" stub under Moonshot's
1054        // accounting; the assistant message header above still counts as
1055        // prompt. Measure it rather than assuming its segment count.
1056        let before_channel_tag = segments.len();
1057        open_tag(
1058            &mut segments,
1059            if thinking { "think" } else { "response" },
1060            [],
1061        );
1062        pending_segments = segments.len() - before_channel_tag;
1063    }
1064
1065    Ok(ChatSegments {
1066        segments,
1067        pending_segments,
1068    })
1069}
1070
1071#[cfg(test)]
1072mod tests {
1073    use super::*;
1074    use minijinja::value::Value as MiniValue;
1075    use serde_json::json;
1076
1077    struct Request {
1078        messages: Value,
1079        tools: Option<Value>,
1080        tool_choice: Option<Value>,
1081        response_format: Option<Value>,
1082        args: HashMap<String, Value>,
1083        add_generation_prompt: bool,
1084    }
1085
1086    impl Request {
1087        fn new(messages: Value) -> Self {
1088            Self {
1089                messages,
1090                tools: None,
1091                tool_choice: None,
1092                response_format: None,
1093                args: HashMap::new(),
1094                add_generation_prompt: true,
1095            }
1096        }
1097    }
1098
1099    impl OAIChatLikeRequest for Request {
1100        fn model(&self) -> String {
1101            "kimi-k3".to_string()
1102        }
1103
1104        fn messages(&self) -> MiniValue {
1105            MiniValue::from_serialize(&self.messages)
1106        }
1107
1108        fn tools(&self) -> Option<MiniValue> {
1109            self.tools.as_ref().map(MiniValue::from_serialize)
1110        }
1111
1112        fn tool_choice(&self) -> Option<MiniValue> {
1113            self.tool_choice.as_ref().map(MiniValue::from_serialize)
1114        }
1115
1116        fn response_format(&self) -> Option<MiniValue> {
1117            self.response_format.as_ref().map(MiniValue::from_serialize)
1118        }
1119
1120        fn should_add_generation_prompt(&self) -> bool {
1121            self.add_generation_prompt
1122        }
1123
1124        fn chat_template_args(&self) -> Option<&HashMap<String, Value>> {
1125            Some(&self.args)
1126        }
1127    }
1128
1129    /// Default formatter: no worker declaration, so the checkpoint token.
1130    fn fmt() -> KimiK3Formatter {
1131        KimiK3Formatter::new(true)
1132    }
1133
1134    /// One user message carrying a single image part.
1135    fn image_request() -> Request {
1136        let mut request = Request::new(json!([{
1137            "role": "user",
1138            "content": [{"type": "image_url", "image_url": {"url": "http://example.com/a.png"}}]
1139        }]));
1140        request
1141            .args
1142            .insert("thinking".to_string(), Value::Bool(false));
1143        request
1144    }
1145
1146    fn image_segments(formatter: &KimiK3Formatter, request: &Request) -> Vec<RenderedSegment> {
1147        formatter
1148            .render_prompt(request)
1149            .unwrap()
1150            .segments()
1151            .expect("K3 always renders segmented prompts")
1152            .to_vec()
1153    }
1154
1155    #[test]
1156    fn renders_one_media_pad_per_image() {
1157        let segments = image_segments(&fmt(), &image_request());
1158
1159        let matches: Vec<_> = segments
1160            .iter()
1161            .filter(|segment| segment.text == MEDIA_PAD)
1162            .collect();
1163        assert_eq!(matches.len(), 1, "exactly one pad per image");
1164        // The pad MUST stay special: it is a registered token, and only the
1165        // special-aware encode path yields its single id.
1166        assert!(matches[0].allow_special);
1167        // The checkpoint's non-vocabulary spelling must never be emitted --
1168        // the vLLM worker converts from the pad instead.
1169        assert!(
1170            !segments
1171                .iter()
1172                .any(|segment| segment.text.contains("kimi_image_placeholder")),
1173        );
1174    }
1175
1176    #[test]
1177    fn image_token_cardinality_is_one_per_image() {
1178        let mut request = Request::new(json!([{
1179            "role": "user",
1180            "content": [
1181                {"type": "image_url", "image_url": {"url": "http://example.com/a.png"}},
1182                {"type": "text", "text": "and"},
1183                {"type": "image_url", "image_url": {"url": "http://example.com/b.png"}},
1184                {"type": "text", "text": "compare them"}
1185            ]
1186        }]));
1187        request
1188            .args
1189            .insert("thinking".to_string(), Value::Bool(false));
1190
1191        let segments = image_segments(&fmt(), &request);
1192
1193        assert_eq!(
1194            segments
1195                .iter()
1196                .filter(|segment| segment.text == MEDIA_PAD)
1197                .count(),
1198            2
1199        );
1200        // Interleaved prose must stay ordinary text.
1201        for body in ["and", "compare them"] {
1202            assert!(
1203                segments
1204                    .iter()
1205                    .any(|segment| segment.text == body && !segment.allow_special)
1206            );
1207        }
1208    }
1209
1210    #[test]
1211    fn user_text_spelling_the_pad_stays_ordinary() {
1212        let body = "please describe <|media_pad|>";
1213        let mut request = Request::new(json!([{"role": "user", "content": body}]));
1214        request
1215            .args
1216            .insert("thinking".to_string(), Value::Bool(false));
1217
1218        let segments = image_segments(&fmt(), &request);
1219
1220        assert!(
1221            segments
1222                .iter()
1223                .any(|segment| segment.text == body && !segment.allow_special),
1224            "user content must never be promoted into prompt structure"
1225        );
1226    }
1227
1228    #[test]
1229    fn renders_off_mode_like_model_encoding() {
1230        let mut request = Request::new(json!([{"role": "user", "content": "Hello"}]));
1231        request
1232            .args
1233            .insert("thinking".to_string(), Value::Bool(false));
1234        let rendered = fmt().render(&request).unwrap();
1235        assert_eq!(
1236            rendered,
1237            concat!(
1238                "<|open|>message role=\"user\"<|sep|>Hello",
1239                "<|close|>message<|sep|><|end_of_msg|>",
1240                "<|open|>message role=\"assistant\"<|sep|>",
1241                "<|open|>response<|sep|>"
1242            )
1243        );
1244    }
1245
1246    #[test]
1247    fn renders_developer_messages_as_system() {
1248        let mut request = Request::new(json!([
1249            {"role": "developer", "content": "Follow this policy", "name": "policy"},
1250            {"role": "user", "content": "Hello"}
1251        ]));
1252        request
1253            .args
1254            .insert("thinking".to_string(), Value::Bool(false));
1255
1256        let rendered = fmt().render(&request).unwrap();
1257
1258        assert!(
1259            rendered.contains(
1260                "<|open|>message role=\"system\" name=\"policy\"<|sep|>Follow this policy"
1261            )
1262        );
1263        assert!(!rendered.contains("role=\"developer\""));
1264        assert!(
1265            rendered.find("Follow this policy").unwrap() < rendered.find("Hello").unwrap(),
1266            "developer instructions must retain their position"
1267        );
1268    }
1269
1270    #[test]
1271    fn renders_developer_tools_and_content_in_place_with_named_tool_choice() {
1272        let mut request = Request::new(json!([
1273            {"role": "user", "content": "Start"},
1274            {
1275                "role": "developer",
1276                "name": "policy",
1277                "content": "Use the lookup tool",
1278                "tools": [{"type": "function", "function": {"name": "lookup"}}]
1279            },
1280            {"role": "user", "content": "Look this up"}
1281        ]));
1282        request.tool_choice = Some(json!({
1283            "type": "function",
1284            "function": {"name": "lookup"}
1285        }));
1286        let rendered = fmt().render(&request).unwrap();
1287        let developer_turn = concat!(
1288            "<|open|>message role=\"system\" name=\"policy\"<|sep|>Use the lookup tool",
1289            "<|close|>message<|sep|><|end_of_msg|>"
1290        );
1291        let declaration = rendered.find("## New Tools Available").unwrap();
1292        let content = rendered.find(developer_turn).unwrap();
1293        assert!(rendered.find("Start").unwrap() < declaration);
1294        assert!(declaration < content);
1295        assert!(content < rendered.find("Look this up").unwrap());
1296        assert!(rendered.contains("\"name\":\"lookup\""));
1297        assert!(rendered.contains("MUST call the tool `lookup`"));
1298
1299        request.messages[1]
1300            .as_object_mut()
1301            .unwrap()
1302            .remove("content");
1303        assert_eq!(
1304            fmt().render(&request).unwrap(),
1305            rendered.replace(developer_turn, "")
1306        );
1307    }
1308
1309    #[test]
1310    fn rejects_tools_on_unsupported_message_roles() {
1311        let tools = json!([{"type": "function", "function": {"name": "lookup"}}]);
1312        for (role, extra) in [
1313            ("user", json!({"content": "Look this up"})),
1314            ("assistant", json!({"content": "ok"})),
1315        ] {
1316            let mut message = extra;
1317            message["role"] = json!(role);
1318            message["tools"] = tools.clone();
1319            let request = Request::new(json!([message, {"role": "user", "content": "Go"}]));
1320
1321            let error = fmt().render(&request).unwrap_err();
1322            assert_eq!(
1323                invalid_request_message(&error),
1324                format!(
1325                    "`tools` is only accepted on system or developer messages, not on role {role}"
1326                ),
1327                "role={role}"
1328            );
1329        }
1330    }
1331
1332    #[test]
1333    fn rejects_unsupported_message_roles() {
1334        for role in ["function", "unknown"] {
1335            let request = Request::new(json!([{"role": role, "content": "ignored before"}]));
1336
1337            let error = fmt().render(&request).unwrap_err();
1338
1339            assert!(matches!(
1340                error.downcast_ref::<PromptRenderError>(),
1341                Some(PromptRenderError::InvalidRequest(message))
1342                    if message == &format!("Kimi K3 does not support message role {role:?}")
1343            ));
1344        }
1345    }
1346
1347    #[test]
1348    fn rejects_messages_without_a_string_role() {
1349        for messages in [json!([{"content": "missing"}]), json!([{"role": 7}])] {
1350            let request = Request::new(messages);
1351
1352            let error = fmt().render(&request).unwrap_err();
1353
1354            assert!(matches!(
1355                error.downcast_ref::<PromptRenderError>(),
1356                Some(PromptRenderError::InvalidRequest(message))
1357                    if message == "Kimi K3 messages must contain a string role"
1358            ));
1359        }
1360    }
1361
1362    #[test]
1363    fn rejects_unsupported_thinking_effort_as_invalid_request() {
1364        let mut request = Request::new(json!([{"role": "user", "content": "Hello"}]));
1365        request.args.insert(
1366            "thinking_effort".to_string(),
1367            Value::String("medium".to_string()),
1368        );
1369
1370        let error = fmt().render(&request).unwrap_err();
1371        assert!(matches!(
1372            error.downcast_ref::<PromptRenderError>(),
1373            Some(PromptRenderError::InvalidRequest(message))
1374                if message.contains("thinking_effort=\"medium\"")
1375        ));
1376    }
1377
1378    // -- Kimi Partial Mode (prefix continuation) --
1379
1380    #[test]
1381    fn partial_assistant_renders_open_turn_in_place_of_generation_prompt() {
1382        let mut request = Request::new(json!([
1383            {"role": "user", "content": "Greet the customer"},
1384            {"role": "assistant", "content": "Dear customer, hello", "partial": true}
1385        ]));
1386        request
1387            .args
1388            .insert("thinking".to_string(), Value::Bool(false));
1389
1390        let rendered = fmt().render(&request).unwrap();
1391
1392        assert_eq!(
1393            rendered,
1394            concat!(
1395                "<|open|>message role=\"user\"<|sep|>Greet the customer",
1396                "<|close|>message<|sep|><|end_of_msg|>",
1397                "<|open|>message role=\"assistant\"<|sep|>",
1398                "<|open|>response<|sep|>Dear customer, hello"
1399            ),
1400            "the partial turn must stay open: no <|close|>response / <|close|>message / <|end_of_msg|>, \
1401             and no extra generation prompt after it"
1402        );
1403        for tool_calls in [Value::Null, json!([])] {
1404            request.messages[1]["tool_calls"] = tool_calls;
1405            assert_eq!(fmt().render(&request).unwrap(), rendered);
1406        }
1407    }
1408
1409    #[test]
1410    fn partial_assistant_ignores_add_generation_prompt_flag() {
1411        let mut request = Request::new(json!([
1412            {"role": "user", "content": "Go"},
1413            {"role": "assistant", "content": "prefix", "partial": true}
1414        ]));
1415        request
1416            .args
1417            .insert("thinking".to_string(), Value::Bool(false));
1418        request.add_generation_prompt = false;
1419
1420        let rendered = fmt().render(&request).unwrap();
1421
1422        assert!(rendered.ends_with("<|open|>response<|sep|>prefix"));
1423        assert_eq!(rendered.matches("role=\"assistant\"").count(), 1);
1424    }
1425
1426    #[test]
1427    fn partial_assistant_in_thinking_mode_closes_think_then_opens_response() {
1428        let mut request = Request::new(json!([
1429            {"role": "user", "content": "Go"},
1430            {
1431                "role": "assistant",
1432                "reasoning_content": "carried over reasoning",
1433                "content": "prefix",
1434                "partial": true
1435            }
1436        ]));
1437        request
1438            .args
1439            .insert("thinking".to_string(), Value::Bool(true));
1440
1441        let rendered = fmt().render(&request).unwrap();
1442
1443        assert!(rendered.ends_with(concat!(
1444            "<|open|>message role=\"assistant\"<|sep|>",
1445            "<|open|>think<|sep|>carried over reasoning<|close|>think<|sep|>",
1446            "<|open|>response<|sep|>prefix"
1447        )));
1448    }
1449
1450    #[test]
1451    fn partial_assistant_keeps_name_as_part_of_the_prefix() {
1452        let mut request = Request::new(json!([
1453            {"role": "user", "content": "Who are you?"},
1454            {"role": "assistant", "name": "Sherlock", "content": "Elementary", "partial": true}
1455        ]));
1456        request
1457            .args
1458            .insert("thinking".to_string(), Value::Bool(false));
1459
1460        let rendered = fmt().render(&request).unwrap();
1461
1462        assert!(rendered.ends_with(concat!(
1463            "<|open|>message role=\"assistant\" name=\"Sherlock\"<|sep|>",
1464            "<|open|>response<|sep|>Elementary"
1465        )));
1466    }
1467
1468    #[test]
1469    fn partial_assistant_follows_internal_system_messages() {
1470        // tool_choice / response_format hints are injected after history and
1471        // before the generation turn; a partial turn must not be split by them.
1472        let mut request = Request::new(json!([
1473            {"role": "user", "content": "Go"},
1474            {"role": "assistant", "content": "prefix", "partial": true}
1475        ]));
1476        request.tools = Some(json!([{
1477            "type": "function",
1478            "function": {"name": "lookup", "parameters": {"type": "object"}}
1479        }]));
1480        request.tool_choice = Some(json!("none"));
1481        request
1482            .args
1483            .insert("thinking".to_string(), Value::Bool(false));
1484
1485        let rendered = fmt().render(&request).unwrap();
1486
1487        let hint = rendered
1488            .find("tool_choice=none")
1489            .expect("tool-choice hint rendered");
1490        let turn = rendered
1491            .rfind("<|open|>message role=\"assistant\"<|sep|>")
1492            .expect("partial turn rendered");
1493        assert!(
1494            hint < turn,
1495            "internal system messages must precede the open partial turn"
1496        );
1497        assert!(rendered.ends_with("<|open|>response<|sep|>prefix"));
1498    }
1499
1500    #[test]
1501    fn partial_false_is_an_ordinary_assistant_turn() {
1502        let mut request = Request::new(json!([
1503            {"role": "user", "content": "Go"},
1504            {"role": "assistant", "content": "done", "partial": false}
1505        ]));
1506        request
1507            .args
1508            .insert("thinking".to_string(), Value::Bool(false));
1509
1510        let rendered = fmt().render(&request).unwrap();
1511
1512        assert!(rendered.contains(
1513            "<|open|>response<|sep|>done<|close|>response<|sep|><|close|>message<|sep|><|end_of_msg|>"
1514        ));
1515        assert!(
1516            rendered.ends_with("<|open|>message role=\"assistant\"<|sep|><|open|>response<|sep|>")
1517        );
1518    }
1519
1520    #[test]
1521    fn rejects_partial_on_a_non_final_message() {
1522        let request = Request::new(json!([
1523            {"role": "assistant", "content": "early", "partial": true},
1524            {"role": "user", "content": "Go"}
1525        ]));
1526
1527        let error = fmt().render(&request).unwrap_err();
1528
1529        assert!(matches!(
1530            error.downcast_ref::<PromptRenderError>(),
1531            Some(PromptRenderError::InvalidRequest(message))
1532                if message == "Kimi K3 `partial` is only supported on the final message"
1533        ));
1534    }
1535
1536    #[test]
1537    fn rejects_partial_on_a_non_assistant_message() {
1538        let request = Request::new(json!([
1539            {"role": "user", "content": "Go", "partial": false}
1540        ]));
1541        let error = fmt().render(&request).unwrap_err();
1542        assert_eq!(
1543            invalid_request_message(&error),
1544            "Kimi K3 `partial` is only supported on an assistant message"
1545        );
1546    }
1547
1548    #[test]
1549    fn rejects_non_boolean_partial() {
1550        let request = Request::new(json!([
1551            {"role": "assistant", "content": "done", "partial": "true"}
1552        ]));
1553        let error = fmt().render(&request).unwrap_err();
1554        assert_eq!(
1555            invalid_request_message(&error),
1556            "Kimi K3 `partial` must be a boolean"
1557        );
1558    }
1559
1560    #[test]
1561    fn null_partial_is_equivalent_to_absent() {
1562        let mut request = Request::new(json!([
1563            {"role": "user", "content": "Go"}
1564        ]));
1565        let expected = fmt().render(&request).unwrap();
1566        request.messages[0]["partial"] = Value::Null;
1567        assert_eq!(fmt().render(&request).unwrap(), expected);
1568    }
1569
1570    // -- Dynamic tool system messages: content XOR tools --
1571
1572    fn invalid_request_message(error: &anyhow::Error) -> &str {
1573        match error.downcast_ref::<PromptRenderError>() {
1574            Some(PromptRenderError::InvalidRequest(message)) => message,
1575            other => panic!("expected InvalidRequest, got {other:?}"),
1576        }
1577    }
1578
1579    #[test]
1580    fn rejects_system_message_with_both_content_and_tools() {
1581        let request = Request::new(json!([
1582            {
1583                "role": "system",
1584                "content": "You are helpful",
1585                "tools": [{"type": "function", "function": {"name": "lookup"}}]
1586            },
1587            {"role": "user", "content": "Go"}
1588        ]));
1589
1590        let error = fmt().render(&request).unwrap_err();
1591        assert_eq!(
1592            invalid_request_message(&error),
1593            "Kimi K3 system messages carry either `content` or `tools`, not both"
1594        );
1595    }
1596
1597    #[test]
1598    fn rejects_system_message_tools_that_are_not_an_array() {
1599        let request = Request::new(json!([
1600            {"role": "system", "tools": {"name": "lookup"}},
1601            {"role": "user", "content": "Go"}
1602        ]));
1603
1604        let error = fmt().render(&request).unwrap_err();
1605        assert_eq!(
1606            invalid_request_message(&error),
1607            "Kimi K3 dynamic tool messages need `tools` to be an array"
1608        );
1609    }
1610
1611    /// Moonshot's official dynamic-tools verifier sends `"content": ""`
1612    /// alongside `tools` and expects success. Empty content is "omitted".
1613    #[test]
1614    fn accepts_dynamic_tools_with_empty_string_content() {
1615        for empty in [json!(""), json!([]), Value::Null] {
1616            let mut request = Request::new(json!([
1617                {"role": "user", "content": "Start"},
1618                {
1619                    "role": "system",
1620                    "content": empty,
1621                    "tools": [{"type": "function", "function": {"name": "lookup"}}]
1622                },
1623                {"role": "user", "content": "Go"}
1624            ]));
1625            request
1626                .args
1627                .insert("thinking".to_string(), Value::Bool(false));
1628
1629            let rendered = fmt()
1630                .render(&request)
1631                .unwrap_or_else(|e| panic!("content={empty}: {e}"));
1632            assert!(
1633                rendered.contains("## New Tools Available"),
1634                "content={empty}"
1635            );
1636            assert!(
1637                !rendered.contains("<|open|>message role=\"system\"<|sep|><|close|>message"),
1638                "content={empty}: must not emit an empty system turn"
1639            );
1640        }
1641    }
1642
1643    #[test]
1644    fn rejects_malformed_dynamic_tool_entries() {
1645        let long_name = "a".repeat(257);
1646        for (entry, needle) in [
1647            (json!("lookup"), "must be JSON objects"),
1648            (
1649                json!({"parameters": {"type": "object"}}),
1650                "need a string `name`",
1651            ),
1652            (
1653                json!({"type": "web_search", "function": {"name": "lookup"}}),
1654                "type=\"function\"",
1655            ),
1656            (
1657                json!({"function": {"name": "lookup"}}),
1658                "need type=\"function\"",
1659            ),
1660            (
1661                json!({"type": "function", "name": "lookup"}),
1662                "need a `function` object",
1663            ),
1664            (json!({"name": ""}), "must match"),
1665            (json!({"name": "1bad_name"}), "must match"),
1666            (json!({"name": "bad@name"}), "must match"),
1667            (json!({"name": long_name}), "maximum is 256"),
1668        ] {
1669            let request = Request::new(json!([
1670                {"role": "system", "tools": [entry]},
1671                {"role": "user", "content": "Go"}
1672            ]));
1673            let error = fmt().render(&request).unwrap_err();
1674            assert!(
1675                invalid_request_message(&error).contains(needle),
1676                "entry={entry}: {}",
1677                invalid_request_message(&error)
1678            );
1679        }
1680    }
1681
1682    #[test]
1683    fn rejects_duplicate_tool_names_across_declarations() {
1684        let mut request = Request::new(json!([
1685            {"role": "system", "tools": [{"name": "lookup"}]},
1686            {"role": "user", "content": "Go"}
1687        ]));
1688        request.tools = Some(json!([{
1689            "type": "function",
1690            "function": {"name": "lookup", "parameters": {"type": "object"}}
1691        }]));
1692        let error = fmt().render(&request).unwrap_err();
1693        assert!(invalid_request_message(&error).contains("declared more than once"));
1694
1695        request.messages[0]["role"] = json!("developer");
1696        let error = fmt().render(&request).unwrap_err();
1697        assert!(invalid_request_message(&error).contains("declared more than once"));
1698
1699        let mut request = Request::new(json!([{"role": "user", "content": "Go"}]));
1700        request.tools = Some(json!([
1701            {"type": "function", "function": {"name": "lookup"}},
1702            {"type": "function", "function": {"name": "lookup"}}
1703        ]));
1704        let error = fmt().render(&request).unwrap_err();
1705        assert!(invalid_request_message(&error).contains("more than once in `tools`"));
1706
1707        let max_name = "a".repeat(256);
1708        let request = Request::new(json!([
1709            {"role": "system", "tools": [
1710                {"name": "_private-tool_2"},
1711                {"type": "function", "function": {"name": max_name}}
1712            ]},
1713            {"role": "user", "content": "Go"}
1714        ]));
1715        fmt().render(&request).unwrap();
1716
1717        let mut request = Request::new(json!([
1718            {"role": "system", "tools": [{"name": "lookup"}, {"type": "function", "function": {"name": "search"}}]},
1719            {"role": "user", "content": "Go"}
1720        ]));
1721        request.tools = Some(json!([{
1722            "type": "function",
1723            "function": {"name": "add", "parameters": {"type": "object"}}
1724        }]));
1725        fmt().render(&request).unwrap();
1726    }
1727
1728    #[test]
1729    fn empty_tools_list_is_an_ordinary_system_message() {
1730        let mut request = Request::new(json!([
1731            {"role": "system", "content": "You are helpful", "tools": []},
1732            {"role": "user", "content": "Go"}
1733        ]));
1734        request
1735            .args
1736            .insert("thinking".to_string(), Value::Bool(false));
1737
1738        let rendered = fmt().render(&request).unwrap();
1739        assert!(rendered.contains("<|open|>message role=\"system\"<|sep|>You are helpful"));
1740        assert!(!rendered.contains("## New Tools Available"));
1741
1742        let request = Request::new(json!([
1743            {"role": "system", "tools": []},
1744            {"role": "user", "content": "Go"}
1745        ]));
1746        let error = fmt().render(&request).unwrap_err();
1747        assert_eq!(
1748            invalid_request_message(&error),
1749            "Kimi K3 system messages need `content` or `tools`"
1750        );
1751    }
1752
1753    #[test]
1754    fn rejects_system_message_with_neither_content_nor_tools() {
1755        let request = Request::new(json!([
1756            {"role": "system"},
1757            {"role": "user", "content": "Go"}
1758        ]));
1759
1760        let error = fmt().render(&request).unwrap_err();
1761        assert_eq!(
1762            invalid_request_message(&error),
1763            "Kimi K3 system messages need `content` or `tools`"
1764        );
1765    }
1766
1767    // -- Typed request path: JSON -> CreateChatCompletionRequest -> renderer --
1768    //
1769    // The raw-JSON `Request` above bypasses protocol deserialization. These
1770    // tests go through `dynamo_protocols::types::CreateChatCompletionRequest`
1771    // and its default `OAIChatLikeRequest` impl, which is what an HTTP frontend
1772    // actually hands to the formatter.
1773
1774    fn typed(body: Value) -> dynamo_protocols::types::CreateChatCompletionRequest {
1775        serde_json::from_value(body).expect("request deserializes")
1776    }
1777
1778    #[test]
1779    fn typed_request_rejects_invalid_top_level_tool_name() {
1780        let request = typed(json!({
1781            "model": "kimi-k3",
1782            "messages": [{"role": "user", "content": "Look it up"}],
1783            "tools": [{
1784                "type": "function",
1785                "function": {"name": "bad@name", "parameters": {"type": "object"}}
1786            }]
1787        }));
1788
1789        let error = fmt().render(&request).unwrap_err();
1790        assert_eq!(
1791            invalid_request_message(&error),
1792            "Kimi K3 tool name \"bad@name\" must match [A-Za-z_][A-Za-z0-9_-]*"
1793        );
1794    }
1795
1796    #[test]
1797    fn typed_request_renders_dynamic_tools_and_final_partial_end_to_end() {
1798        let request = typed(json!({
1799            "model": "kimi-k3",
1800            "messages": [
1801                {"role": "system", "tools": [{
1802                    "type": "function",
1803                    "function": {"name": "lookup", "parameters": {"type": "object"}}
1804                }]},
1805                {"role": "user", "content": "Look it up"},
1806                {"role": "assistant", "content": "Looking", "partial": true}
1807            ]
1808        }));
1809
1810        let rendered = fmt().render(&request).unwrap();
1811
1812        assert!(rendered.contains("## New Tools Available"));
1813        assert!(rendered.contains("\"lookup\""));
1814        assert!(
1815            rendered.ends_with("<|open|>response<|sep|>Looking"),
1816            "partial turn must stay open, got {rendered:?}"
1817        );
1818    }
1819
1820    #[test]
1821    fn typed_request_preserves_content_and_tools_for_renderer_conflict_check() {
1822        let request = typed(json!({
1823            "model": "kimi-k3",
1824            "messages": [
1825                {
1826                    "role": "system",
1827                    "content": "You are helpful",
1828                    "tools": [{"type": "function", "function": {"name": "lookup"}}]
1829                },
1830                {"role": "user", "content": "Go"}
1831            ]
1832        }));
1833
1834        let system = serde_json::to_value(&request.messages[0]).unwrap();
1835        assert_eq!(system["content"], json!("You are helpful"));
1836        assert_eq!(system["tools"][0]["function"]["name"], json!("lookup"));
1837
1838        let error = fmt().render(&request).unwrap_err();
1839        assert_eq!(
1840            invalid_request_message(&error),
1841            "Kimi K3 system messages carry either `content` or `tools`, not both"
1842        );
1843    }
1844
1845    #[test]
1846    fn rejects_partial_assistant_with_tool_calls() {
1847        let tool_call = json!({
1848            "id": "call_1",
1849            "type": "function",
1850            "function": {"name": "lookup", "arguments": "{}"}
1851        });
1852        for tool_calls in [json!([tool_call.clone()]), tool_call] {
1853            let request = Request::new(json!([
1854                {"role": "user", "content": "Go"},
1855                {
1856                    "role": "assistant",
1857                    "content": "prefix",
1858                    "partial": true,
1859                    "tool_calls": tool_calls
1860                }
1861            ]));
1862            let error = fmt().render(&request).unwrap_err();
1863            assert_eq!(
1864                invalid_request_message(&error),
1865                "Kimi K3 partial assistant messages cannot carry tool_calls",
1866                "tool_calls={tool_calls}"
1867            );
1868        }
1869    }
1870
1871    #[test]
1872    fn rejects_tools_on_final_partial_assistant_raw_path() {
1873        let request = Request::new(json!([
1874            {"role": "user", "content": "Go"},
1875            {
1876                "role": "assistant",
1877                "content": "prefix",
1878                "partial": true,
1879                "tools": [{"type": "function", "function": {"name": "lookup"}}]
1880            }
1881        ]));
1882
1883        let error = fmt().render(&request).unwrap_err();
1884
1885        assert!(matches!(
1886            error.downcast_ref::<PromptRenderError>(),
1887            Some(PromptRenderError::InvalidRequest(message))
1888                if message == "`tools` is only accepted on system or developer messages, not on role assistant"
1889        ));
1890    }
1891
1892    #[test]
1893    fn named_tool_choice_forces_tool_and_disables_thinking() {
1894        let mut request = Request::new(json!([
1895            {"role": "user", "content": "What did you do before?"},
1896            {
1897                "role": "assistant",
1898                "reasoning_content": "historical hidden reasoning",
1899                "content": "I answered the earlier question."
1900            },
1901            {"role": "user", "content": "Calculate"}
1902        ]));
1903        request.tools = Some(json!([{
1904            "type": "function",
1905            "function": {
1906                "name": "add_numbers",
1907                "parameters": {
1908                    "type": "object",
1909                    "properties": {
1910                        "a": {"type": "integer"},
1911                        "b": {"type": "integer"}
1912                    },
1913                    "required": ["a", "b"]
1914                }
1915            }
1916        }]));
1917        request.tool_choice = Some(json!({
1918            "type": "function",
1919            "function": {"name": "add_numbers"}
1920        }));
1921        request
1922            .args
1923            .insert("thinking".to_string(), Value::Bool(true));
1924
1925        let rendered = fmt().render(&request).unwrap();
1926        assert!(rendered.contains("The system is invoked with `tool_choice=specified`."));
1927        assert!(rendered.contains("MUST call the tool `add_numbers`"));
1928        assert!(
1929            rendered.ends_with("<|open|>message role=\"assistant\"<|sep|><|open|>response<|sep|>"),
1930            "named tool choice must use K3's non-thinking generation prefix"
1931        );
1932        assert!(
1933            !rendered.contains("<|open|>think<|sep|>"),
1934            "named tool choice must override thinking=true"
1935        );
1936        assert!(
1937            !rendered.contains("historical hidden reasoning"),
1938            "named tool choice must also suppress preserved thinking history"
1939        );
1940    }
1941
1942    #[test]
1943    fn named_tool_choice_accepts_a_dynamic_system_tool() {
1944        for lookup in [
1945            json!({"type": "function", "function": {"name": "lookup", "parameters": {"type": "object"}}}),
1946            json!({"name": "lookup", "parameters": {"type": "object"}}),
1947        ] {
1948            let mut request = Request::new(json!([
1949                {"role": "user", "content": "Start"},
1950                {"role": "system", "tools": [lookup]},
1951                {"role": "user", "content": "Look this up"}
1952            ]));
1953            request.tool_choice = Some(json!({
1954                "type": "function",
1955                "function": {"name": "lookup"}
1956            }));
1957
1958            let rendered = fmt().render(&request).unwrap();
1959            assert!(rendered.contains("## New Tools Available"));
1960            assert!(rendered.contains("MUST call the tool `lookup`"));
1961            assert!(
1962                request.tools.is_none(),
1963                "dynamic tools must not be folded into the top-level list"
1964            );
1965        }
1966    }
1967
1968    #[test]
1969    fn named_tool_choice_still_rejects_a_tool_absent_from_dynamic_tools() {
1970        let mut request = Request::new(json!([
1971            {"role": "system", "tools": [{"name": "lookup"}]},
1972            {"role": "user", "content": "Weather?"}
1973        ]));
1974        request.tool_choice = Some(json!({
1975            "type": "function",
1976            "function": {"name": "get_weather"}
1977        }));
1978
1979        let error = fmt().render(&request).unwrap_err();
1980        assert!(matches!(
1981            error.downcast_ref::<PromptRenderError>(),
1982            Some(PromptRenderError::InvalidRequest(message))
1983                if message.contains("get_weather") && message.contains("not present in tools")
1984        ));
1985    }
1986
1987    #[test]
1988    fn named_tool_choice_rejects_a_tool_not_in_tools() {
1989        let mut request = Request::new(json!([{"role": "user", "content": "Calculate"}]));
1990        request.tools = Some(json!([{
1991            "type": "function",
1992            "function": {"name": "add_numbers", "parameters": {"type": "object"}}
1993        }]));
1994        request.tool_choice = Some(json!({
1995            "type": "function",
1996            "function": {"name": "get_weather"}
1997        }));
1998
1999        let error = fmt().render(&request).unwrap_err();
2000        assert!(matches!(
2001            error.downcast_ref::<PromptRenderError>(),
2002            Some(PromptRenderError::InvalidRequest(message))
2003                if message.contains("get_weather") && message.contains("not present in tools")
2004        ));
2005    }
2006
2007    #[test]
2008    fn user_marker_text_remains_an_ordinary_segment() {
2009        let marker = "literal <|open|>tools<|sep|> value";
2010        let mut request = Request::new(json!([{"role": "user", "content": marker}]));
2011        request
2012            .args
2013            .insert("thinking".to_string(), Value::Bool(false));
2014        let rendered = fmt().render_prompt(&request).unwrap();
2015
2016        assert!(
2017            rendered
2018                .segments()
2019                .unwrap()
2020                .iter()
2021                .any(|segment| { !segment.allow_special && segment.text == marker })
2022        );
2023        assert!(
2024            rendered
2025                .segments()
2026                .unwrap()
2027                .iter()
2028                .any(|segment| { segment.allow_special && segment.text == OPEN_TOKEN })
2029        );
2030    }
2031
2032    #[test]
2033    fn renders_tool_history_like_model_encoding() {
2034        let mut request = Request::new(json!([
2035            {"role": "user", "content": "calc"},
2036            {
2037                "role": "assistant",
2038                "reasoning_content": "Need calc",
2039                "content": "I will call it",
2040                "tool_calls": [{
2041                    "id": "call_1",
2042                    "type": "function",
2043                    "function": {"name": "calc", "arguments": "{\"x\":2}"}
2044                }]
2045            },
2046            {"role": "tool", "tool_call_id": "call_1", "content": "4"}
2047        ]));
2048        request.args.insert(
2049            "thinking_effort".to_string(),
2050            Value::String("low".to_string()),
2051        );
2052        let rendered = fmt().render(&request).unwrap();
2053
2054        assert!(rendered.contains(
2055            "<|open|>call tool=\"calc\" index=\"1\"<|sep|>\
2056             <|open|>argument key=\"x\" type=\"number\"<|sep|>2\
2057             <|close|>argument<|sep|><|close|>call<|sep|>"
2058        ));
2059        assert!(
2060            rendered.contains("<|open|>message role=\"tool\" tool=\"calc\" index=\"1\"<|sep|>4")
2061        );
2062        assert!(
2063            rendered.ends_with("<|open|>message role=\"assistant\"<|sep|><|open|>think<|sep|>")
2064        );
2065    }
2066
2067    #[test]
2068    fn tool_result_order_preserves_unresolved_runs() {
2069        let assistant = json!({"role": "assistant", "tool_calls": [
2070            {"id": "first", "function": {"name": "one"}},
2071            {"id": "second", "function": {"name": "two"}}
2072        ]});
2073        let second =
2074            json!({"role": "tool", "tool_call_id": "second", "content": "第二", "name": "old"});
2075        let first = json!({"role": "tool", "tool_call_id": "first", "content": "第一"});
2076        let sorted =
2077            normalize_tool_result_messages(vec![assistant.clone(), second.clone(), first]).unwrap();
2078        assert_eq!(sorted[1]["content"], "第一");
2079        assert_eq!(sorted[1]["tool"], "one");
2080        assert_eq!(sorted[2]["content"], "第二");
2081        assert_eq!(sorted[2]["tool"], "two");
2082        assert_eq!(sorted[2]["name"], "two");
2083
2084        let unresolved = vec![
2085            assistant,
2086            second,
2087            json!({"role": "tool", "tool_call_id": "unknown", "content": "keep order"}),
2088        ];
2089        assert_eq!(
2090            normalize_tool_result_messages(unresolved.clone()).unwrap(),
2091            unresolved
2092        );
2093    }
2094
2095    #[test]
2096    fn thinking_history_renders_an_empty_think_channel() {
2097        let request = Request::new(json!([
2098            {"role": "user", "content": "question"},
2099            {"role": "assistant", "content": "answer"},
2100            {"role": "user", "content": "follow-up"}
2101        ]));
2102
2103        let rendered = fmt().render(&request).unwrap();
2104
2105        assert!(rendered.contains(concat!(
2106            "<|open|>message role=\"assistant\"<|sep|>",
2107            "<|open|>think<|sep|><|close|>think<|sep|>",
2108            "<|open|>response<|sep|>answer<|close|>response<|sep|>"
2109        )));
2110    }
2111
2112    #[test]
2113    fn non_thinking_history_omits_preserved_reasoning() {
2114        let mut request = Request::new(json!([
2115            {"role": "user", "content": "question"},
2116            {
2117                "role": "assistant",
2118                "reasoning_content": "hidden reasoning",
2119                "content": "answer"
2120            },
2121            {"role": "user", "content": "follow-up"}
2122        ]));
2123        request
2124            .args
2125            .insert("thinking".to_string(), Value::Bool(false));
2126
2127        let rendered = fmt().render(&request).unwrap();
2128
2129        assert!(!rendered.contains("hidden reasoning"));
2130        assert!(!rendered.contains("<|open|>think<|sep|>"));
2131        assert!(rendered.contains(concat!(
2132            "<|open|>message role=\"assistant\"<|sep|>",
2133            "<|open|>response<|sep|>answer<|close|>response<|sep|>"
2134        )));
2135    }
2136
2137    #[test]
2138    fn tools_are_deep_sorted_before_declaration() {
2139        let mut request = Request::new(json!([{"role": "user", "content": "Weather?"}]));
2140        request
2141            .args
2142            .insert("thinking".to_string(), Value::Bool(false));
2143        request.tools = Some(json!([{
2144            "type": "function",
2145            "function": {
2146                "parameters": {"type": "object", "properties": {"city": {"type": "string"}}},
2147                "name": "weather",
2148                "description": "Get weather"
2149            }
2150        }]));
2151        let rendered = fmt().render(&request).unwrap();
2152        assert!(rendered.contains(concat!(
2153            "[{\"function\":{\"description\":\"Get weather\",",
2154            "\"name\":\"weather\",\"parameters\":{\"properties\":",
2155            "{\"city\":{\"type\":\"string\"}},\"type\":\"object\"}},",
2156            "\"type\":\"function\"}]"
2157        )));
2158    }
2159
2160    #[test]
2161    fn assistant_history_matches_python_json_spacing_and_reasoning_fallback() {
2162        let request = Request::new(json!([{
2163            "role": "assistant",
2164            "reasoning_content": "",
2165            "reasoning": "fallback",
2166            "content": null,
2167            "tool_calls": [{
2168                "type": "function",
2169                "function": {
2170                    "name": "run",
2171                    "arguments": {
2172                        "opts": {"a": 1, "b": [true, false]}
2173                    }
2174                }
2175            }]
2176        }]));
2177        let rendered = fmt().render(&request).unwrap();
2178
2179        assert!(rendered.contains("<|open|>think<|sep|>fallback<|close|>think<|sep|>"));
2180        assert!(rendered.contains(concat!(
2181            "<|open|>argument key=\"opts\" type=\"object\"<|sep|>",
2182            "{\"a\": 1, \"b\": [true, false]}",
2183            "<|close|>argument<|sep|>"
2184        )));
2185    }
2186
2187    fn pending_texts(prompt: &RenderedPrompt) -> Vec<String> {
2188        let segments = prompt.segments().unwrap();
2189        segments[segments.len() - prompt.pending_segments()..]
2190            .iter()
2191            .map(|segment| segment.text.clone())
2192            .collect()
2193    }
2194
2195    #[test]
2196    fn non_thinking_generation_stub_is_reported_as_pending_segments() {
2197        let mut request = Request::new(json!([{"role": "user", "content": "Hello"}]));
2198        request
2199            .args
2200            .insert("thinking".to_string(), Value::Bool(false));
2201        let prompt = fmt().render_prompt(&request).unwrap();
2202        assert_eq!(prompt.pending_segments(), 3);
2203        assert_eq!(pending_texts(&prompt), ["<|open|>", "response", "<|sep|>"]);
2204        // The assistant message header stays part of the prompt proper.
2205        let pending = prompt.pending_encode_segments().unwrap();
2206        assert_eq!(pending.len(), 3);
2207        assert!(pending[0].allow_special, "open marker is a control token");
2208        assert!(!pending[1].allow_special, "channel name is ordinary text");
2209        assert!(pending[2].allow_special, "sep marker is a control token");
2210        assert!(prompt.as_str().ends_with(concat!(
2211            "<|open|>message role=\"assistant\"<|sep|>",
2212            "<|open|>response<|sep|>"
2213        )));
2214    }
2215
2216    #[test]
2217    fn thinking_generation_stub_is_reported_as_pending_segments() {
2218        let request = Request::new(json!([{"role": "user", "content": "Hello"}]));
2219        let prompt = fmt().render_prompt(&request).unwrap();
2220        assert_eq!(prompt.pending_segments(), 3);
2221        assert_eq!(pending_texts(&prompt), ["<|open|>", "think", "<|sep|>"]);
2222    }
2223
2224    #[test]
2225    fn response_format_instructions_are_prompt_not_pending() {
2226        let mut request = Request::new(json!([{"role": "user", "content": "Hello"}]));
2227        request
2228            .args
2229            .insert("thinking".to_string(), Value::Bool(false));
2230        request.response_format = Some(json!({
2231            "type": "json_schema",
2232            "json_schema": {
2233                "name": "capital_answer",
2234                "schema": {
2235                    "type": "object",
2236                    "properties": {"capital": {"type": "boolean"}},
2237                    "required": ["capital"],
2238                    "additionalProperties": false
2239                }
2240            }
2241        }));
2242        let prompt = fmt().render_prompt(&request).unwrap();
2243        assert_eq!(prompt.pending_segments(), 3);
2244        assert_eq!(pending_texts(&prompt), ["<|open|>", "response", "<|sep|>"]);
2245        let stub_start = prompt.as_str().len() - "<|open|>response<|sep|>".len();
2246        let schema_at = prompt
2247            .as_str()
2248            .find("type=\"response-format\"")
2249            .expect("response_format renders a system message");
2250        assert!(
2251            schema_at < stub_start,
2252            "schema instructions precede the stub"
2253        );
2254    }
2255
2256    #[test]
2257    fn no_generation_prompt_has_no_pending_segments() {
2258        let mut request = Request::new(json!([{"role": "user", "content": "Hello"}]));
2259        request.add_generation_prompt = false;
2260        let prompt = fmt().render_prompt(&request).unwrap();
2261        assert_eq!(prompt.pending_segments(), 0);
2262        assert!(prompt.pending_encode_segments().unwrap().is_empty());
2263        assert!(prompt.as_str().ends_with("<|end_of_msg|>"));
2264    }
2265
2266    #[test]
2267    fn partial_assistant_continuation_has_no_pending_segments() {
2268        let mut request = Request::new(json!([
2269            {"role": "user", "content": "Hello"},
2270            {"role": "assistant", "content": "Hi", "partial": true}
2271        ]));
2272        request
2273            .args
2274            .insert("thinking".to_string(), Value::Bool(false));
2275        let prompt = fmt().render_prompt(&request).unwrap();
2276        assert_eq!(prompt.pending_segments(), 0);
2277    }
2278}