pub(crate) mod audit;
pub(crate) mod catalog;
pub(crate) mod engine;
pub mod limits;
pub(crate) mod usage;
pub(crate) const TOOL_NAME: &str = "code_mode";
pub(crate) fn definition() -> serde_json::Value {
serde_json::json!({
"type": "function",
"name": TOOL_NAME,
"description": r#"Run sandboxed JavaScript to orchestrate Magi-Code tools without sending every intermediate result to the model.
Use for deterministic/mechanical work: repeated/batched calls; repository/file traversal; filtering, sorting, grouping, counting, deduplication; pagination; fan-out/fan-in queries (sequential calls); result-based branching; mechanical retries/validation passes; multi-step read/search/edit/check workflows; reducing large results to small summaries.
Synchronous SDK: tools.list(), tools.search(query), tools.describe(name), tools.call(name, args), tools.budget(). Calls return {success, content} or {success:false, error}. Read results use content.files[].text and content.complete, not content.text; never detect truncation by searching file contents.
Write a JavaScript function body and use return for results; console is unavailable. The code argument is JSON-decoded before JavaScript parses it: JSON \n becomes a source newline, while JSON \\n becomes a JavaScript \n escape. Literal newlines are invalid inside single- or double-quoted JavaScript strings. Use template literals for multiline text and String.raw when embedded scripts need literal backslashes. A backtick or ${ inside any template literal, including String.raw, ends the literal or starts interpolation; escape them as \` and \${, or write embedded scripts containing them to a file first. Example JavaScript (before JSON encoding):
return tools.call('bash', {command: String.raw`python3 - <<'PY'
print('hello\nworld')
PY`, intent: 'Print two lines', timeout: 30});
Keep intermediate data here; return only what further model reasoning needs. Prefer direct tools when results require semantic judgment before choosing the next step.
tools.budget() returns {inner_calls, result_bytes, runtime_ms, memory_bytes, batch_history_bytes, session_history_bytes} with {used, limit, remaining}, plus inner_result_bytes_limit and return_bytes_limit. History fields are null without session recording. Checking budget consumes no tool calls, result-byte allowance, or history records; computation and memory remain bounded. Execution budgets reset per code_mode call; only session_history_bytes is shared. Before exhaustion, stop at a safe boundary and return progress, reserving calls for checkpoint writes. History headroom is advisory, not a guarantee the next result fits. A session-history error requires compaction; other exhausted batch budgets require smaller batches. Completed effects remain applied; inspect execution evidence and re-read affected files instead of replaying the whole script.
No filesystem, process, network, or environment APIs except through exposed Magi-Code tools. Completed actions are not rolled back after failure or cancellation."#,
"parameters": {
"type": "object", "properties": {
"code": {"type": "string"},
"intent": {"type": "string", "minLength": 1, "description": "Short explanation of this batch's purpose, shown on the transcript card. Include for user-visible progress."}
},
"required": ["code"], "additionalProperties": false
}
})
}
/// Protection may replace script output, but must not erase host evidence.
pub(crate) fn preserve_execution_evidence(
result: &crate::tools::ToolResult,
provider_output: &mut String,
) {
if result.tool_name != TOOL_NAME || *provider_output == result.content {
return;
}
if let Some(summary) = result.metadata.get("code_mode_execution") {
provider_output.push_str("\nHost-owned Code Mode execution evidence and warnings:\n");
provider_output.push_str(&summary.to_string());
}
}