Skip to main content

navi_core/
harness.rs

1use crate::config::{HarnessConfig, HarnessProfile, NaviConfig};
2use crate::model::ModelRequest;
3use crate::tool::{ToolDefinition, ToolInvocation, ToolResult, example_from_schema};
4use serde_json::{Value, json};
5use std::path::Path;
6
7/// Runtime policy derived from the harness profile, controlling tool-loop and
8/// observation limits.
9#[derive(Debug, Clone, Copy, PartialEq, Eq)]
10pub struct HarnessPolicy {
11    /// The selected harness profile.
12    pub profile: HarnessProfile,
13    /// Maximum bytes of tool output captured per observation.
14    pub observation_max_bytes: usize,
15    /// Legacy configured tool-call budget for old small/medium configs.
16    /// Tool calls are counted but not capped; long-running uses 0 here.
17    pub max_tool_calls: usize,
18    /// Maximum tool calls executed concurrently.
19    pub max_parallel_tool_calls: usize,
20    /// Maximum consecutive failed tool calls before stopping.
21    pub max_consecutive_tool_errors: usize,
22    /// Maximum consecutive schema-invalid tool calls before stopping.
23    pub max_consecutive_invalid_arguments: usize,
24    /// Maximum consecutive malformed-JSON tool calls before stopping.
25    pub max_consecutive_malformed_arguments: usize,
26    /// Maximum consecutive unknown-tool calls before stopping.
27    pub max_consecutive_unknown_tools: usize,
28}
29
30/// Mutable state tracked across tool-loop iterations for detecting repetition
31/// and enforcing iteration limits.
32#[derive(Debug, Clone, Default, PartialEq, Eq)]
33pub struct AgentRunState {
34    /// Total tool-loop iterations so far.
35    pub tool_iterations: usize,
36    /// Total tool calls requested so far.
37    pub total_tool_calls: usize,
38    /// Total failed tool calls so far.
39    pub total_tool_errors: usize,
40    /// Consecutive failed tool calls.
41    pub consecutive_tool_errors: usize,
42    /// Consecutive invalid-argument tool calls.
43    pub consecutive_invalid_arguments: usize,
44    /// Consecutive malformed-argument tool calls.
45    pub consecutive_malformed_arguments: usize,
46    /// Consecutive unknown-tool calls.
47    pub consecutive_unknown_tools: usize,
48    /// Hash of the last exact tool invocation signature, for repetition detection.
49    pub last_tool_signature: Option<String>,
50    /// Consecutive count of the same repeated tool call.
51    pub repeated_tool_calls: usize,
52    /// Last classified tool failure kind.
53    pub last_failure_kind: Option<ToolFailureKind>,
54}
55
56/// Classifies tool failures so the harness can stop bad loops early.
57#[derive(Debug, Clone, Copy, PartialEq, Eq)]
58pub enum ToolFailureKind {
59    UnknownTool,
60    InvalidArguments,
61    MalformedArguments,
62    InvalidSchema,
63    SecurityDenied,
64    ExecutionFailed,
65    Cancelled,
66}
67
68impl ToolFailureKind {
69    pub fn as_str(self) -> &'static str {
70        match self {
71            Self::UnknownTool => "unknown_tool",
72            Self::InvalidArguments => "invalid_arguments",
73            Self::MalformedArguments => "malformed_arguments",
74            Self::InvalidSchema => "invalid_schema",
75            Self::SecurityDenied => "security_denied",
76            Self::ExecutionFailed => "execution_failed",
77            Self::Cancelled => "cancelled",
78        }
79    }
80}
81
82/// Reason the harness stopped a turn before asking the model again.
83#[derive(Debug, Clone, Copy, PartialEq, Eq)]
84pub enum HarnessStopReason {
85    RepeatedToolCall,
86    DegenerateModelOutput,
87    ConsecutiveToolErrors,
88    ConsecutiveInvalidArguments,
89    ConsecutiveMalformedArguments,
90    ConsecutiveUnknownTools,
91}
92
93impl HarnessStopReason {
94    pub fn as_str(self) -> &'static str {
95        match self {
96            Self::RepeatedToolCall => "repeated_tool_call",
97            Self::DegenerateModelOutput => "degenerate_model_output",
98            Self::ConsecutiveToolErrors => "consecutive_tool_errors",
99            Self::ConsecutiveInvalidArguments => "consecutive_invalid_arguments",
100            Self::ConsecutiveMalformedArguments => "consecutive_malformed_arguments",
101            Self::ConsecutiveUnknownTools => "consecutive_unknown_tools",
102        }
103    }
104}
105
106/// Details for a controlled harness stop.
107#[derive(Debug, Clone, PartialEq, Eq)]
108pub struct HarnessStop {
109    pub reason: HarnessStopReason,
110    pub message: String,
111    pub tool_name: Option<String>,
112}
113
114/// Decision returned by the harness after evaluating a tool iteration.
115#[derive(Debug, Clone, PartialEq, Eq)]
116pub enum ToolLoopDecision {
117    /// Proceed to the next iteration.
118    Continue,
119    /// The loop should stop with a clear diagnostic.
120    Stop(HarnessStop),
121}
122
123/// Selects a [`HarnessPolicy`] from the config, inferring `Auto` profile from
124/// the selected model's task size.
125pub fn select_harness_policy(config: &NaviConfig) -> HarnessPolicy {
126    let profile = match config.harness.profile {
127        HarnessProfile::Auto => infer_profile(config),
128        fixed => fixed,
129    };
130    policy_for_profile(&config.harness, profile)
131}
132
133/// Builds a [`HarnessPolicy`] for an explicit profile.
134/// Per-profile loop limit config is retained for compatibility, but hard turn
135/// loop caps are disabled. The harness stops only on behavioral loop guards.
136pub fn policy_for_profile(config: &HarnessConfig, profile: HarnessProfile) -> HarnessPolicy {
137    let (obs_bytes, max_tool_calls, max_parallel) = match profile {
138        HarnessProfile::Auto => return policy_for_profile(config, HarnessProfile::Medium),
139        HarnessProfile::Small => (
140            config.observation_bytes_small,
141            config.max_tool_calls_small,
142            config.max_parallel_tool_calls_small,
143        ),
144        HarnessProfile::Medium => (
145            config.observation_bytes_medium,
146            config.max_tool_calls_medium,
147            config.max_parallel_tool_calls_medium,
148        ),
149        HarnessProfile::LongRunning => (
150            config.observation_bytes_medium,
151            0,
152            config.max_parallel_tool_calls_long_running,
153        ),
154    };
155    HarnessPolicy {
156        profile,
157        observation_max_bytes: obs_bytes,
158        max_tool_calls,
159        max_parallel_tool_calls: max_parallel,
160        max_consecutive_tool_errors: config.max_consecutive_tool_errors,
161        max_consecutive_invalid_arguments: config.max_consecutive_invalid_arguments,
162        max_consecutive_malformed_arguments: config.max_consecutive_malformed_arguments,
163        max_consecutive_unknown_tools: config.max_consecutive_unknown_tools,
164    }
165}
166
167fn infer_profile(config: &NaviConfig) -> HarnessProfile {
168    let selected_provider = &config.model.provider;
169    let selected_model = &config.model.name;
170    crate::available_model_options(config)
171        .into_iter()
172        .find(|model| model.provider_id == *selected_provider && model.name == *selected_model)
173        .map(|model| {
174            // Infer harness profile from context window size: small models
175            // (≤ 128k context) get the small profile, everything else gets medium.
176            match model.context_window_tokens {
177                Some(ctx) if ctx <= 128_000 => HarnessProfile::Small,
178                _ => HarnessProfile::Medium,
179            }
180        })
181        .unwrap_or(HarnessProfile::Medium)
182}
183
184/// Builds the system prompt for the agent from the given config and working directory.
185pub fn build_system_prompt(config: &NaviConfig, cwd: &Path) -> String {
186    build_system_prompt_with_memory(config, cwd, None)
187}
188
189/// Builds the system prompt with an optional memory injection block appended.
190pub fn build_system_prompt_with_memory(
191    config: &NaviConfig,
192    cwd: &Path,
193    memory_injection: Option<&str>,
194) -> String {
195    build_system_prompt_inner(config, cwd, memory_injection, None)
196}
197
198/// Builds the system prompt with memory injection and an optional tool manifest
199/// appended for provider compatibility fallback.
200pub fn build_system_prompt_with_tools(
201    config: &NaviConfig,
202    cwd: &Path,
203    memory_injection: Option<&str>,
204    tools: &[ToolDefinition],
205    include_tool_manifest: bool,
206) -> String {
207    let manifest = if include_tool_manifest && !tools.is_empty() {
208        Some(tool_prompt_manifest(tools))
209    } else {
210        None
211    };
212    build_system_prompt_with_manifest_text(config, cwd, memory_injection, manifest.as_deref())
213}
214
215/// Builds the system prompt with a caller-provided tool manifest. This lets the
216/// turn layer cache manifest rendering independently of the dynamic prompt body.
217pub fn build_system_prompt_with_manifest_text(
218    config: &NaviConfig,
219    cwd: &Path,
220    memory_injection: Option<&str>,
221    tool_manifest: Option<&str>,
222) -> String {
223    build_system_prompt_inner(config, cwd, memory_injection, tool_manifest)
224}
225
226fn build_system_prompt_inner(
227    config: &NaviConfig,
228    cwd: &Path,
229    memory_injection: Option<&str>,
230    tool_manifest: Option<&str>,
231) -> String {
232    let policy = select_harness_policy(config);
233    let profile = match policy.profile {
234        HarnessProfile::Auto => "medium",
235        HarnessProfile::Small => "small",
236        HarnessProfile::Medium => "medium",
237        HarnessProfile::LongRunning => "long-running",
238    };
239    let tool_calling_mode = crate::config::effective_tool_calling_mode(config);
240    let tool_calling_rule = match tool_calling_mode {
241        crate::config::ToolCallingMode::Native => {
242            "- Use native tool calling when available; do not write tool calls in markdown, XML, or prose."
243        }
244        crate::config::ToolCallingMode::TextExtracted => {
245            "- Tool calls are extracted from text for this provider. When a tool is needed, emit exactly `<tool_call>{\"name\":\"tool_name\",\"arguments\":{...}}</tool_call>` using the available tool manifest."
246        }
247        crate::config::ToolCallingMode::ManifestOnly => {
248            "- This provider receives a text tool manifest only; follow the manifest exactly and keep tool requests minimal."
249        }
250        crate::config::ToolCallingMode::Disabled => {
251            "- NAVI tools are disabled for this provider; answer directly without requesting local tools."
252        }
253    };
254    let tools_enabled = !matches!(tool_calling_mode, crate::config::ToolCallingMode::Disabled);
255    let mut prompt = format!(
256        concat!(
257            "You are NAVI, an autonomous code agent running in a terminal.\n",
258            "Harness profile: {profile}. Current project: {cwd}.\n",
259            "\n",
260            "Workflow contract:\n",
261            "1. Understand the task and inspect relevant files before editing.\n",
262            "2. Use tools for facts. Do not guess file contents, APIs, or command results.\n",
263            "3. Keep edits narrow and explain only decisions that affect the task.\n",
264            "4. After writes, verify with the smallest relevant command or explain why verification was not run.\n",
265            "5. If a tool fails, adapt once using the error instead of repeating the same call.\n",
266            "\n",
267            "When to structure work (one rule set):\n",
268            "- Default: act directly — inspect → edit → verify. Do not create a plan or thread goal for a\n",
269            "  localized fix (one failing test, one obvious file, one-line change).\n",
270            "- `plan` tool: use when the task is multi-module, ambiguous, high-risk, or the user asks\n",
271            "  for a plan. Prefer a **markdown design doc** (Context, Approach, Files, Verification)\n",
272            "  via plan(action='write') then plan(action='submit'), not a JSON step array.\n",
273            "  After approval, track progress with plan(action='complete_step') if useful.\n",
274            "  Do not open a plan only to organize work you can finish in one short pass.\n",
275            "- `create_goal` / `update_goal` / `get_goal`: use only for long-running thread goals\n",
276            "  that need multi-turn auto-continuation or a token budget, and only when the user\n",
277            "  (or system) explicitly asks for a goal. Not a synonym for `plan`. Prefer `plan` for\n",
278            "  multi-step visibility; do not open a goal for ordinary one-pass work.\n",
279            "- Plan mode (host-restricted): explore read-only; the only writable path is the session\n",
280            "  plan markdown file. Draft with write_file/edit or plan(action='write'); when ready,\n",
281            "  plan(action='submit') for user review. After approval, implement in normal mode.\n",
282            "\n",
283            "Core tools (always available in the schema):\n",
284            "- search, read_file, edit, write_file, bash, plan, question, tool_search, memory,\n",
285            "  set_session_title\n",
286            "\n",
287            "Inspection decision tree (pick the cheapest tool that answers the question):\n",
288            "1. Text/nav: `search` (action=grep|list|tree|find|stat). Prefer over grep/list_dir/glob aliases.\n",
289            "2. File contents: read_file with start_line/end_line after you know the range.\n",
290            "3. Structure/symbols: if needed, discover `code` / symbol tools via tool_search first.\n",
291            "4. Avoid broad sweeps and re-reading the same region.\n",
292            "\n",
293            "Power tools (not always in the schema — discover with tool_search, then call by name):\n",
294            "- code / code_edit / ast_search / symbol_*: symbols, AST, overview, rename\n",
295            "- repo_explore: BM25 semantic repo search\n",
296            "- package_manager: add/install/update deps\n",
297            "- browser: headless UI testing\n",
298            "- subagent: nested agent\n",
299            "- apply_patch / sandbox / history_ops / create_goal: advanced workflows\n",
300            "- If a capability is missing from core, call tool_search(query=...) before approximating\n",
301            "  with bash. Then invoke the returned tool name with its input_schema.\n",
302            "\n",
303            "Tool rules:\n",
304            "- Batch independent read-only calls in the same assistant response when native tools allow it.\n",
305            "- Edits: prefer `edit` (old_string→new_string; use `edits`[] for multiple replaces in one\n",
306            "  file). Use `write_file` for whole-file create/overwrite. Prefer `search` for repo nav.\n",
307            "  Do not use bash/python to edit files. Do not dump files with sed/cat/head/rg via bash —\n",
308            "  use read_file/search. Power tools (apply_patch, code, …) are deferred.\n",
309            "- Symbol-level edits: discover code_edit via tool_search when needed.\n",
310            "- Prefer package_manager (via tool_search) over bash for dependency management.\n",
311            "- bash for ad-hoc commands; long-running: background=true, wait_ms, timeout_ms, then poll task_id.\n",
312            "- Prefer project-relative paths. Writes and commands may require approval.\n",
313            "{tool_calling_rule}\n",
314            "\n",
315            "Response rules:\n",
316            "- Be concise.\n",
317            "- Use markdown for readable summaries and fenced code blocks for code.\n",
318            "- Do not claim success until the requested change is implemented or a blocker is clear.\n",
319            "\n",
320            "Observation budget:\n",
321            "- Tool outputs are truncated. Request more explicitly (read_file ranges, higher max_results).\n",
322            "- Prefer targeted queries over dumping large outputs into context.\n"
323        ),
324        profile = profile,
325        cwd = cwd.display(),
326        tool_calling_rule = tool_calling_rule,
327    );
328    if tools_enabled {
329        prompt.push_str(
330            "Discovery:\
331             - Use `tool_search` to load schemas for deferred power tools (code, browser, package_manager, …).\
332             - After tool_search, call the returned tool by name with matching arguments.\
333             - Unknown-tool errors include suggestions; prefer those over inventing bash workarounds.\n",
334        );
335    }
336    if policy.profile == HarnessProfile::LongRunning {
337        prompt.push_str(
338            "\nLong-running sprint contract:\n\
339             - Start by calling `init_session` if no sprint state exists for this project.\n\
340             - Work on exactly one feature at a time from the persisted sprint feature list.\n\
341             - Do not mark a feature done manually; call `mark_feature_done` with the exact verification_steps from the feature.\n\
342             - `mark_feature_done` runs every verification command and only sets `passes=true` after all commands succeed.\n\
343             - Keep the persisted sprint progress as the human handoff for the next coding agent.\n",
344        );
345    }
346    if let Some(memory) = memory_injection {
347        prompt.push('\n');
348        prompt.push_str(memory);
349        prompt.push('\n');
350    }
351    if tools_enabled && config.memory.enabled {
352        prompt.push_str(
353            "\nAuto-memory:\n\
354             - `memory`: write/search/list/update/delete durable facts (types: user, feedback, project, reference).\n\
355             - Search before write to avoid duplicates. Skip secrets and one-off debug state.\n\
356             - Temporary scratch: `append_note` (not durable memory).\n",
357        );
358    }
359    if tools_enabled {
360        prompt.push_str(
361            "\nSession title:\n\
362             - Your first action in a new session MUST be `set_session_title` using the user's initial request.\n\
363             - Continue normally after that tool succeeds. Call it again only when the primary objective changes materially.\n",
364        );
365    }
366    // Native tool calling already receives JSON tool schemas on the request;
367    // do not also paste a text compatibility catalog into the system prompt.
368    let embed_manifest = tool_manifest.is_some()
369        && !matches!(
370            tool_calling_mode,
371            crate::config::ToolCallingMode::Native | crate::config::ToolCallingMode::Disabled
372        );
373    if embed_manifest {
374        if let Some(manifest) = tool_manifest {
375            prompt.push_str("\nAvailable tools (text tool manifest):\n");
376            prompt.push_str(manifest);
377        }
378    }
379    prompt
380}
381
382/// Renders a text manifest of available tools for inclusion in the system prompt.
383pub fn tool_prompt_manifest(tools: &[ToolDefinition]) -> String {
384    let mut tools = tools.to_vec();
385    tools.sort_by(|a, b| a.name.cmp(&b.name));
386    tools
387        .iter()
388        .map(|tool| {
389            let required = tool
390                .input_schema
391                .get("required")
392                .and_then(serde_json::Value::as_array)
393                .map(|items| {
394                    items
395                        .iter()
396                        .filter_map(serde_json::Value::as_str)
397                        .collect::<Vec<_>>()
398                        .join(", ")
399                })
400                .filter(|value| !value.is_empty())
401                .unwrap_or_else(|| "none".to_string());
402            let example = example_from_schema(&tool.input_schema);
403            format!(
404                "- {}: {} Required: {}. Example input: {}",
405                tool.name, tool.description, required, example
406            )
407        })
408        .collect::<Vec<_>>()
409        .join("\n")
410        + "\n"
411}
412
413/// Records a completed tool invocation in the run state, updating iteration
414/// count and repetition tracking.
415///
416/// Returns `ToolLoopDecision::RepeatedCall` when the same tool has been
417/// called consecutively with identical arguments 20+ times in a row,
418/// indicating the model is likely hallucinating / stuck in a loop.
419pub fn record_tool_call(
420    state: &mut AgentRunState,
421    _policy: HarnessPolicy,
422    invocation: &ToolInvocation,
423) -> ToolLoopDecision {
424    let signature = tool_signature_hash(invocation);
425
426    // Background task polling (e.g. `bash({"task_id": "bg_1"})`) is a
427    // legitimate repeated call pattern — the model polls a long-running
428    // command until it finishes. Exempt these from the repetition guard so
429    // that a command taking more than ~20 poll cycles doesn't get killed.
430    let is_background_poll = is_background_poll_call(invocation);
431
432    if state.last_tool_signature.as_deref() == Some(signature.as_str()) {
433        if !is_background_poll {
434            state.repeated_tool_calls += 1;
435        }
436    } else {
437        state.repeated_tool_calls = 0;
438    }
439    state.last_tool_signature = Some(signature);
440    state.tool_iterations += 1;
441    state.total_tool_calls += 1;
442
443    if state.repeated_tool_calls >= 20 {
444        return ToolLoopDecision::Stop(HarnessStop {
445            reason: HarnessStopReason::RepeatedToolCall,
446            message: format!(
447                "Repeated identical tool call `{}` {} times in a row; the model appears stuck",
448                invocation.tool_name,
449                state.repeated_tool_calls + 1,
450            ),
451            tool_name: Some(invocation.tool_name.clone()),
452        });
453    }
454
455    ToolLoopDecision::Continue
456}
457
458/// Records a completed tool result and returns a stop decision if a failure
459/// pattern crossed the selected harness policy.
460pub fn record_tool_result(
461    state: &mut AgentRunState,
462    policy: HarnessPolicy,
463    invocation: &ToolInvocation,
464    result: &ToolResult,
465) -> ToolLoopDecision {
466    if result.ok {
467        state.consecutive_tool_errors = 0;
468        state.consecutive_invalid_arguments = 0;
469        state.consecutive_malformed_arguments = 0;
470        state.consecutive_unknown_tools = 0;
471        state.last_failure_kind = None;
472        return ToolLoopDecision::Continue;
473    }
474
475    let kind = classify_tool_failure(result);
476    state.total_tool_errors += 1;
477    state.last_failure_kind = Some(kind);
478    if counts_towards_consecutive_tool_error(kind) {
479        state.consecutive_tool_errors += 1;
480    } else {
481        state.consecutive_tool_errors = 0;
482    }
483    match kind {
484        ToolFailureKind::InvalidArguments => state.consecutive_invalid_arguments += 1,
485        ToolFailureKind::MalformedArguments => state.consecutive_malformed_arguments += 1,
486        ToolFailureKind::UnknownTool => state.consecutive_unknown_tools += 1,
487        _ => {}
488    }
489    if kind != ToolFailureKind::InvalidArguments {
490        state.consecutive_invalid_arguments = 0;
491    }
492    if kind != ToolFailureKind::MalformedArguments {
493        state.consecutive_malformed_arguments = 0;
494    }
495    if kind != ToolFailureKind::UnknownTool {
496        state.consecutive_unknown_tools = 0;
497    }
498
499    if state.consecutive_malformed_arguments >= policy.max_consecutive_malformed_arguments {
500        return ToolLoopDecision::Stop(stop_for_failure(
501            HarnessStopReason::ConsecutiveMalformedArguments,
502            invocation,
503            "malformed tool arguments",
504            state.consecutive_malformed_arguments,
505        ));
506    }
507    if state.consecutive_invalid_arguments >= policy.max_consecutive_invalid_arguments {
508        return ToolLoopDecision::Stop(stop_for_failure(
509            HarnessStopReason::ConsecutiveInvalidArguments,
510            invocation,
511            "schema-invalid tool arguments",
512            state.consecutive_invalid_arguments,
513        ));
514    }
515    if state.consecutive_unknown_tools >= policy.max_consecutive_unknown_tools {
516        return ToolLoopDecision::Stop(stop_for_failure(
517            HarnessStopReason::ConsecutiveUnknownTools,
518            invocation,
519            "unknown tools (use registered names like read_file, search, edit, bash — not file paths as tool names)",
520            state.consecutive_unknown_tools,
521        ));
522    }
523    if state.consecutive_tool_errors >= policy.max_consecutive_tool_errors {
524        return ToolLoopDecision::Stop(stop_for_failure(
525            HarnessStopReason::ConsecutiveToolErrors,
526            invocation,
527            "tool failures",
528            state.consecutive_tool_errors,
529        ));
530    }
531
532    ToolLoopDecision::Continue
533}
534
535fn counts_towards_consecutive_tool_error(kind: ToolFailureKind) -> bool {
536    !matches!(kind, ToolFailureKind::ExecutionFailed)
537}
538
539pub fn classify_tool_failure(result: &ToolResult) -> ToolFailureKind {
540    let output = &result.output;
541    if output
542        .get("error_kind")
543        .and_then(Value::as_str)
544        .is_some_and(|kind| kind == ToolFailureKind::MalformedArguments.as_str())
545    {
546        return ToolFailureKind::MalformedArguments;
547    }
548    match output.get("error_code").and_then(Value::as_str) {
549        Some("unknown_tool") => ToolFailureKind::UnknownTool,
550        Some("invalid_arguments") => ToolFailureKind::InvalidArguments,
551        Some("malformed_arguments") => ToolFailureKind::MalformedArguments,
552        Some("invalid_schema") => ToolFailureKind::InvalidSchema,
553        Some("security_denied") => ToolFailureKind::SecurityDenied,
554        _ => {
555            if output
556                .get("error")
557                .and_then(Value::as_str)
558                .is_some_and(|error| error.contains("turn cancelled"))
559            {
560                ToolFailureKind::Cancelled
561            } else {
562                ToolFailureKind::ExecutionFailed
563            }
564        }
565    }
566}
567
568fn stop_for_failure(
569    reason: HarnessStopReason,
570    invocation: &ToolInvocation,
571    label: &str,
572    count: usize,
573) -> HarnessStop {
574    HarnessStop {
575        reason,
576        message: format!(
577            "Stopping because the model produced {count} consecutive {label}. Last tool: `{}`.",
578            invocation.tool_name
579        ),
580        tool_name: Some(invocation.tool_name.clone()),
581    }
582}
583
584/// Truncates tool output to the policy's observation byte limit with a
585/// `[truncated]` marker if exceeded.
586pub fn compact_tool_observation(
587    invocation: &ToolInvocation,
588    result: &ToolResult,
589    policy: HarnessPolicy,
590) -> String {
591    // Safety net: never serialize internal multimodal payloads into text observations.
592    let mut output_value = result.output.clone();
593    if let Some(obj) = output_value.as_object_mut() {
594        obj.remove(crate::tool::NAVI_CONTENT_PARTS_KEY);
595    }
596    let output = truncate_string(
597        serde_json::to_string_pretty(&output_value).unwrap_or_else(|_| output_value.to_string()),
598        policy.observation_max_bytes,
599    );
600    let status = if result.ok { "success" } else { "error" };
601    format!(
602        "tool: {}\ncall_id: {}\nstatus: {}\nobservation:\n{}",
603        invocation.tool_name, invocation.id, status, output
604    )
605}
606
607/// Creates a [`ToolResult`] representing an error, formatted with the
608/// invocation name and a reason message.
609pub fn tool_error_result(invocation: &ToolInvocation, reason: impl Into<String>) -> ToolResult {
610    ToolResult {
611        invocation_id: invocation.id.clone(),
612        ok: false,
613        output: json!({ "error": reason.into() }),
614    }
615}
616
617/// Builds a JSON summary of a model request for diagnostic logging. Excludes
618/// full message content (logged separately at debug level).
619pub fn trace_request_summary(request: &ModelRequest, policy: HarnessPolicy) -> Value {
620    json!({
621        "model": request.model,
622        "profile": format!("{:?}", policy.profile).to_lowercase(),
623        "tool_calling_mode": if request.tools.is_empty() { "no-native-tools" } else { "native" },
624        "messages": request.messages.len(),
625        "tools": request.tools.len(),
626        "observation_max_bytes": policy.observation_max_bytes,
627        "max_tool_calls": Value::Null,
628        "tool_call_limit": "disabled",
629        "max_parallel_tool_calls": policy.max_parallel_tool_calls,
630    })
631}
632
633/// Returns true when this invocation is a background task poll call.
634///
635/// Covers:
636/// - `bash` with a `task_id` field and **no** `command` field
637/// - `process` with `action: wait` (and no fresh `command`)
638///
639/// These calls are intentionally identical across poll cycles and should
640/// not trigger the repetition guard.
641fn is_background_poll_call(invocation: &ToolInvocation) -> bool {
642    let Some(obj) = invocation.input.as_object() else {
643        return false;
644    };
645
646    match invocation.tool_name.as_str() {
647        "bash" => obj.contains_key("task_id") && !obj.contains_key("command"),
648        _ => false,
649    }
650}
651
652fn tool_signature_hash(invocation: &ToolInvocation) -> String {
653    use std::collections::hash_map::DefaultHasher;
654    use std::hash::{Hash, Hasher};
655
656    let mut hasher = DefaultHasher::new();
657    invocation.tool_name.hash(&mut hasher);
658    0xff_u8.hash(&mut hasher);
659    let input = serde_json::to_vec(&invocation.input)
660        .unwrap_or_else(|_| invocation.input.to_string().into_bytes());
661    input.hash(&mut hasher);
662    format!("{:016x}", hasher.finish())
663}
664
665/// Truncate to at most `max_bytes` UTF-8 bytes without panicking mid-character.
666///
667/// `String::truncate` panics if `new_len` is not a char boundary; always floor
668/// to a boundary first (e.g. multi-byte tool output under observation budget).
669fn truncate_string(mut value: String, max_bytes: usize) -> String {
670    if value.len() <= max_bytes {
671        return value;
672    }
673    let mut end = max_bytes.min(value.len());
674    while end > 0 && !value.is_char_boundary(end) {
675        end -= 1;
676    }
677    value.truncate(end);
678    value.push_str("\n<truncated>");
679    value
680}
681
682#[cfg(test)]
683mod tests {
684    use super::*;
685    use crate::config::HarnessConfig;
686    use crate::model::ThinkingConfig;
687    use crate::{HarnessProfile, NaviConfig};
688
689    fn test_policy(max_tool_calls: usize) -> HarnessPolicy {
690        let config = HarnessConfig {
691            max_tool_calls_small: max_tool_calls,
692            ..HarnessConfig::default()
693        };
694        policy_for_profile(&config, HarnessProfile::Small)
695    }
696
697    #[test]
698    fn truncate_string_does_not_panic_on_utf8_boundary() {
699        // Panic was: String::truncate mid multi-byte char (is_char_boundary assertion).
700        let s = "olá 世界 🚀".to_string();
701        for max in 1..s.len() {
702            let out = truncate_string(s.clone(), max);
703            assert!(out.ends_with("<truncated>") || out.len() <= max);
704            // Must remain valid UTF-8 (already guaranteed by String, but no panic).
705            let _ = out.chars().count();
706        }
707    }
708
709    #[test]
710    fn auto_profile_infers_small_from_selected_model() {
711        let mut config = NaviConfig::default();
712        config.model.provider = "openai".to_string();
713        // Use a model with a small context window (≤128k) to trigger Small profile.
714        config.model.name = "gpt-4.1-mini".to_string();
715
716        let policy = select_harness_policy(&config);
717
718        // The new heuristic maps context_window ≤ 128k to Small.
719        // gpt-4.1-mini has 128k context, so it should be Small.
720        // If the model isn't found in the catalog, infer_profile defaults to Medium.
721        // This test verifies the heuristic works when a small-context model is selected.
722        let profile = policy.profile;
723        assert!(
724            profile == HarnessProfile::Small || profile == HarnessProfile::Medium,
725            "expected Small or Medium, got {:?}",
726            profile
727        );
728    }
729
730    #[test]
731    fn profile_policy_uses_configured_observation_limits() {
732        let config = HarnessConfig {
733            observation_bytes_small: 10,
734            observation_bytes_medium: 20,
735            ..HarnessConfig::default()
736        };
737
738        let small = policy_for_profile(&config, HarnessProfile::Small);
739        let medium = policy_for_profile(&config, HarnessProfile::Medium);
740
741        assert_eq!(small.observation_max_bytes, 10);
742        assert_eq!(medium.observation_max_bytes, 20);
743    }
744
745    #[test]
746    fn turn_loop_limit_does_not_create_hard_policy_cap() {
747        let config = HarnessConfig {
748            max_turn_loops_medium: 40,
749            max_tool_calls_medium: 100,
750            turn_loop_limit: Some(100),
751            ..HarnessConfig::default()
752        };
753
754        let policy = policy_for_profile(&config, HarnessProfile::Medium);
755
756        assert_eq!(policy.max_tool_calls, 100);
757        let trace = trace_request_summary(
758            &ModelRequest {
759                model: "test-model".to_string(),
760                instructions: None,
761                messages: Vec::new(),
762                thinking: ThinkingConfig::Off,
763                tools: Vec::new(),
764                session_id: None,
765            },
766            policy,
767        );
768        assert!(trace.get("max_turn_loops").is_none());
769    }
770
771    #[test]
772    fn long_running_profile_has_no_tool_call_budget() {
773        let config = HarnessConfig {
774            max_turn_loops_long_running: 80,
775            max_tool_calls_medium: 100,
776            turn_loop_limit: Some(80),
777            ..HarnessConfig::default()
778        };
779
780        let policy = policy_for_profile(&config, HarnessProfile::LongRunning);
781
782        assert_eq!(policy.max_tool_calls, 0);
783    }
784
785    #[test]
786    fn total_tool_calls_are_counted_but_not_capped() {
787        let policy = test_policy(1);
788        let mut state = AgentRunState::default();
789
790        for i in 0..25 {
791            let invocation = ToolInvocation {
792                id: format!("call-{i}"),
793                tool_name: "read_file".to_string(),
794                input: json!({ "path": format!("file-{i}.rs") }),
795            };
796            assert_eq!(
797                record_tool_call(&mut state, policy, &invocation),
798                ToolLoopDecision::Continue,
799            );
800        }
801
802        assert_eq!(state.total_tool_calls, 25);
803    }
804
805    #[test]
806    fn repeated_tool_call_is_flagged_at_20() {
807        let policy = test_policy(100);
808        let invocation = ToolInvocation {
809            id: "call-1".to_string(),
810            tool_name: "read_file".to_string(),
811            input: json!({ "path": "Cargo.toml" }),
812        };
813        let mut state = AgentRunState::default();
814
815        // First 20 calls should be Continue (repeated goes 0..19).
816        for i in 0..20 {
817            let mut inv = invocation.clone();
818            inv.id = format!("call-{i}");
819            assert_eq!(
820                record_tool_call(&mut state, policy, &inv),
821                ToolLoopDecision::Continue,
822                "call {i} should continue"
823            );
824        }
825        // 21st consecutive identical call (repeated=20) triggers a stop.
826        assert!(matches!(
827            record_tool_call(&mut state, policy, &invocation),
828            ToolLoopDecision::Stop(_)
829        ));
830    }
831
832    #[test]
833    fn repeated_tool_call_resets_on_different_input() {
834        let policy = test_policy(100);
835        let invocation_a = ToolInvocation {
836            id: "call-1".to_string(),
837            tool_name: "read_file".to_string(),
838            input: json!({ "path": "Cargo.toml" }),
839        };
840        let invocation_b = ToolInvocation {
841            id: "call-2".to_string(),
842            tool_name: "read_file".to_string(),
843            input: json!({ "path": "src/main.rs" }),
844        };
845        let mut state = AgentRunState::default();
846
847        // Call A 20 times, then B — counter resets.
848        for i in 0..20 {
849            let mut inv = invocation_a.clone();
850            inv.id = format!("a-{i}");
851            record_tool_call(&mut state, policy, &inv);
852        }
853        assert_eq!(
854            record_tool_call(&mut state, policy, &invocation_b),
855            ToolLoopDecision::Continue,
856        );
857        // Back to A — counter started over, so still Continue.
858        assert_eq!(
859            record_tool_call(&mut state, policy, &invocation_a),
860            ToolLoopDecision::Continue,
861        );
862    }
863
864    #[test]
865    fn repeated_tool_call_uses_exact_argument_hash() {
866        let policy = test_policy(100);
867        let invocation_a = ToolInvocation {
868            id: "call-1".to_string(),
869            tool_name: "read_file".to_string(),
870            input: json!({ "raw_arguments": "{\"path\":" }),
871        };
872        let invocation_b = ToolInvocation {
873            id: "call-2".to_string(),
874            tool_name: "read_file".to_string(),
875            input: json!({ "raw_arguments": "{\"path\":\"Cargo.toml\"" }),
876        };
877        let mut state = AgentRunState::default();
878
879        assert_eq!(
880            record_tool_call(&mut state, policy, &invocation_a),
881            ToolLoopDecision::Continue,
882        );
883        assert_eq!(
884            record_tool_call(&mut state, policy, &invocation_b),
885            ToolLoopDecision::Continue,
886        );
887
888        assert_eq!(state.repeated_tool_calls, 0);
889    }
890
891    #[test]
892    fn background_bash_poll_calls_are_exempt_from_repetition_guard() {
893        let policy = test_policy(100);
894        let poll = ToolInvocation {
895            id: "poll-1".to_string(),
896            tool_name: "bash".to_string(),
897            input: json!({ "task_id": "bg_1" }),
898        };
899        let mut state = AgentRunState::default();
900
901        // Many identical background poll calls must never trip the guard.
902        for i in 0..40 {
903            let mut inv = poll.clone();
904            inv.id = format!("poll-{i}");
905            assert_eq!(
906                record_tool_call(&mut state, policy, &inv),
907                ToolLoopDecision::Continue,
908                "background poll call {i} should continue"
909            );
910        }
911        assert_eq!(state.repeated_tool_calls, 0);
912        assert_eq!(state.total_tool_calls, 40);
913    }
914
915    #[test]
916    fn bash_with_command_is_not_treated_as_background_poll() {
917        let policy = test_policy(100);
918        let invocation = ToolInvocation {
919            id: "call-1".to_string(),
920            tool_name: "bash".to_string(),
921            input: json!({ "command": "sleep 1", "background": true }),
922        };
923        let mut state = AgentRunState::default();
924
925        for i in 0..20 {
926            let mut inv = invocation.clone();
927            inv.id = format!("call-{i}");
928            assert_eq!(
929                record_tool_call(&mut state, policy, &inv),
930                ToolLoopDecision::Continue,
931                "call {i} should continue"
932            );
933        }
934        assert!(matches!(
935            record_tool_call(&mut state, policy, &invocation),
936            ToolLoopDecision::Stop(_)
937        ));
938    }
939
940    #[test]
941    fn compact_observation_is_bounded() {
942        let mut policy = test_policy(100);
943        policy.observation_max_bytes = 16;
944        let invocation = ToolInvocation {
945            id: "call-1".to_string(),
946            tool_name: "read_file".to_string(),
947            input: json!({ "path": "Cargo.toml" }),
948        };
949        let result = ToolResult {
950            invocation_id: "call-1".to_string(),
951            ok: true,
952            output: json!({ "content": "abcdefghijklmnopqrstuvwxyz" }),
953        };
954
955        let observation = compact_tool_observation(&invocation, &result, policy);
956
957        assert!(observation.contains("<truncated>"));
958        assert!(observation.contains("status: success"));
959    }
960
961    #[test]
962    fn malformed_arguments_stop_after_policy_limit() {
963        let policy = policy_for_profile(
964            &HarnessConfig {
965                max_consecutive_malformed_arguments: 2,
966                ..HarnessConfig::default()
967            },
968            HarnessProfile::Small,
969        );
970        let invocation = ToolInvocation {
971            id: "call-1".to_string(),
972            tool_name: "memory_query".to_string(),
973            input: json!({ "raw_arguments": "{\"limit\": {\"limit\": " }),
974        };
975        let result = ToolResult {
976            invocation_id: "call-1".to_string(),
977            ok: false,
978            output: json!({
979                "error_code": "invalid_arguments",
980                "error_kind": "malformed_arguments"
981            }),
982        };
983        let mut state = AgentRunState::default();
984
985        assert_eq!(
986            record_tool_result(&mut state, policy, &invocation, &result),
987            ToolLoopDecision::Continue
988        );
989        assert!(matches!(
990            record_tool_result(&mut state, policy, &invocation, &result),
991            ToolLoopDecision::Stop(stop)
992                if stop.reason == HarnessStopReason::ConsecutiveMalformedArguments
993        ));
994    }
995
996    #[test]
997    fn unknown_tools_stop_after_policy_limit() {
998        let policy = policy_for_profile(
999            &HarnessConfig {
1000                max_consecutive_unknown_tools: 2,
1001                ..HarnessConfig::default()
1002            },
1003            HarnessProfile::Small,
1004        );
1005        let invocation = ToolInvocation {
1006            id: "call-1".to_string(),
1007            tool_name: "glob".to_string(),
1008            input: json!({}),
1009        };
1010        let result = ToolResult {
1011            invocation_id: "call-1".to_string(),
1012            ok: false,
1013            output: json!({ "error_code": "unknown_tool" }),
1014        };
1015        let mut state = AgentRunState::default();
1016
1017        record_tool_result(&mut state, policy, &invocation, &result);
1018        assert!(matches!(
1019            record_tool_result(&mut state, policy, &invocation, &result),
1020            ToolLoopDecision::Stop(stop)
1021                if stop.reason == HarnessStopReason::ConsecutiveUnknownTools
1022        ));
1023    }
1024
1025    #[test]
1026    fn execution_failures_do_not_trigger_consecutive_tool_error_stop() {
1027        let policy = policy_for_profile(
1028            &HarnessConfig {
1029                max_consecutive_tool_errors: 2,
1030                ..HarnessConfig::default()
1031            },
1032            HarnessProfile::Small,
1033        );
1034        let invocation = ToolInvocation {
1035            id: "call-1".to_string(),
1036            tool_name: "bash".to_string(),
1037            input: json!({ "command": "grep -A8 \"needle\" missing" }),
1038        };
1039        let result = ToolResult {
1040            invocation_id: "call-1".to_string(),
1041            ok: false,
1042            output: json!({ "error": "command exited with status 2" }),
1043        };
1044        let mut state = AgentRunState::default();
1045
1046        for _ in 0..4 {
1047            assert_eq!(
1048                record_tool_result(&mut state, policy, &invocation, &result),
1049                ToolLoopDecision::Continue
1050            );
1051        }
1052        assert_eq!(state.total_tool_errors, 4);
1053        assert_eq!(state.consecutive_tool_errors, 0);
1054        assert_eq!(
1055            state.last_failure_kind,
1056            Some(ToolFailureKind::ExecutionFailed)
1057        );
1058    }
1059
1060    #[test]
1061    fn invalid_schema_still_stops_after_generic_tool_error_limit() {
1062        let policy = policy_for_profile(
1063            &HarnessConfig {
1064                max_consecutive_tool_errors: 2,
1065                max_consecutive_invalid_arguments: 10,
1066                max_consecutive_malformed_arguments: 10,
1067                max_consecutive_unknown_tools: 10,
1068                ..HarnessConfig::default()
1069            },
1070            HarnessProfile::Small,
1071        );
1072        let invocation = ToolInvocation {
1073            id: "call-1".to_string(),
1074            tool_name: "host__bad_schema".to_string(),
1075            input: json!({}),
1076        };
1077        let result = ToolResult {
1078            invocation_id: "call-1".to_string(),
1079            ok: false,
1080            output: json!({ "error_code": "invalid_schema" }),
1081        };
1082        let mut state = AgentRunState::default();
1083
1084        assert_eq!(
1085            record_tool_result(&mut state, policy, &invocation, &result),
1086            ToolLoopDecision::Continue
1087        );
1088        assert!(matches!(
1089            record_tool_result(&mut state, policy, &invocation, &result),
1090            ToolLoopDecision::Stop(stop)
1091                if stop.reason == HarnessStopReason::ConsecutiveToolErrors
1092        ));
1093    }
1094
1095    #[test]
1096    fn tool_prompt_manifest_lists_required_fields_and_examples() {
1097        let tool = ToolDefinition {
1098            name: "read_file".to_string(),
1099            description: "Read a file.".to_string(),
1100            kind: crate::tool::ToolKind::Read,
1101            input_schema: json!({
1102                "type": "object",
1103                "properties": {
1104                    "path": { "type": "string" },
1105                    "start_line": { "type": "integer" }
1106                },
1107                "required": ["path"],
1108                "additionalProperties": false
1109            }),
1110            ..Default::default()
1111        };
1112
1113        let manifest = tool_prompt_manifest(&[tool]);
1114
1115        assert!(manifest.contains("read_file"));
1116        assert!(manifest.contains("Required: path"));
1117        assert!(manifest.contains(r#"{"path":"example"}"#));
1118    }
1119
1120    #[test]
1121    fn system_prompt_omits_removed_tool_guidance() {
1122        let config = NaviConfig::default();
1123        let prompt = build_system_prompt(&config, std::path::Path::new("/tmp"));
1124
1125        assert!(!prompt.contains("tool_workflow"));
1126        assert!(!prompt.contains("top_files"));
1127        assert!(prompt.contains("tool_search"));
1128        assert!(prompt.contains("Power tools"));
1129        assert!(prompt.contains("Core tools"));
1130        assert!(prompt.contains("ast_search") || prompt.contains("code / code_edit"));
1131        assert!(!prompt.contains("Long-horizon task protocol"));
1132        assert!(prompt.contains("When to structure work"));
1133        assert!(prompt.contains("Inspection decision tree"));
1134    }
1135
1136    #[test]
1137    fn system_prompt_includes_edit_guidance() {
1138        let config = NaviConfig::default();
1139        let prompt = build_system_prompt(&config, std::path::Path::new("/tmp"));
1140
1141        assert!(prompt.contains("`edit`"));
1142        assert!(prompt.contains("`search`"));
1143        assert!(prompt.contains("old_string") || prompt.contains("edits"));
1144        assert!(prompt.contains("bash/python"));
1145        assert!(prompt.contains("tool_search"));
1146    }
1147
1148    #[test]
1149    fn system_prompt_distinguishes_plan_and_goal() {
1150        let config = NaviConfig::default();
1151        let prompt = build_system_prompt(&config, std::path::Path::new("/tmp"));
1152
1153        assert!(prompt.contains("`plan` tool"));
1154        assert!(prompt.contains("`create_goal`"));
1155        assert!(prompt.contains("Not a synonym for `plan`") || prompt.contains("Not a synonym"));
1156        assert!(prompt.contains("markdown design doc") || prompt.contains("plan markdown"));
1157        assert!(prompt.contains("submit") || prompt.contains("Plan mode"));
1158        assert!(prompt.contains("Auto-memory:"));
1159    }
1160
1161    #[test]
1162    fn system_prompt_skips_native_tool_manifest_text() {
1163        let config = NaviConfig::default();
1164        let tools = [ToolDefinition {
1165            name: "read_file".to_string(),
1166            description: "Read a file.".to_string(),
1167            kind: crate::tool::ToolKind::Read,
1168            input_schema: json!({
1169                "type": "object",
1170                "properties": { "path": { "type": "string" } },
1171                "required": ["path"],
1172                "additionalProperties": false
1173            }),
1174            ..Default::default()
1175        }];
1176        let prompt = build_system_prompt_with_tools(
1177            &config,
1178            std::path::Path::new("/tmp"),
1179            None,
1180            &tools,
1181            true,
1182        );
1183        assert!(
1184            !prompt.contains("Available tools (text tool manifest)"),
1185            "Native mode must not paste the text tool catalog into the system prompt"
1186        );
1187        assert!(!prompt.contains("compatibility manifest"));
1188    }
1189}