pub mod proxy;
use std::io::{BufRead, Write};
use std::path::{Path, PathBuf};
use std::sync::{Arc, Mutex};
use serde::{Deserialize, Serialize};
use serde_json::Value;
use sqz_engine::{SqzEngine, ToolDefinition, ToolSelector};
use sqz_engine::error::{Result, SqzError};
use sqz_engine::preset::{Preset, PresetParser};
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct ToolCallRequest {
pub tool_id: String,
pub input: Value,
pub intent: Option<String>,
}
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct ToolCallResponse {
pub tool_id: String,
pub output: String,
pub tokens_original: u32,
pub tokens_compressed: u32,
}
#[derive(Debug, Clone)]
pub enum McpTransport {
Stdio,
Sse { port: u16 },
}
#[derive(Debug, Deserialize)]
struct JsonRpcRequest {
#[allow(dead_code)]
jsonrpc: String,
id: Option<Value>,
method: String,
params: Option<Value>,
}
#[derive(Debug, Serialize)]
struct JsonRpcResponse {
jsonrpc: String,
id: Option<Value>,
#[serde(skip_serializing_if = "Option::is_none")]
result: Option<Value>,
#[serde(skip_serializing_if = "Option::is_none")]
error: Option<JsonRpcError>,
}
#[derive(Debug, Serialize)]
struct JsonRpcError {
code: i32,
message: String,
}
impl JsonRpcResponse {
fn ok(id: Option<Value>, result: Value) -> Self {
Self {
jsonrpc: "2.0".to_string(),
id,
result: Some(result),
error: None,
}
}
fn err(id: Option<Value>, code: i32, message: impl Into<String>) -> Self {
Self {
jsonrpc: "2.0".to_string(),
id,
result: None,
error: Some(JsonRpcError { code, message: message.into() }),
}
}
}
struct SharedState {
pending_preset: Mutex<Option<String>>,
tool_selector: Mutex<ToolSelector>,
registered_tools: Mutex<Vec<ToolDefinition>>,
}
pub struct McpServer {
engine: SqzEngine,
shared: Arc<SharedState>,
preset_dir: PathBuf,
}
impl McpServer {
pub fn new(preset_dir: &Path) -> Result<Self> {
let engine = SqzEngine::new()?;
Self::with_engine(preset_dir, engine)
}
#[cfg(test)]
fn new_with_store(preset_dir: &Path, store_path: &Path) -> Result<Self> {
let engine = SqzEngine::with_preset_and_store(Preset::default(), store_path)?;
Self::with_engine(preset_dir, engine)
}
fn with_engine(preset_dir: &Path, engine: SqzEngine) -> Result<Self> {
let preset = Preset::default();
let model_path = Path::new("");
let mut tool_selector = ToolSelector::new(model_path, &preset)?;
let default_tools = default_tool_definitions();
tool_selector.register_tools(&default_tools)?;
let shared = Arc::new(SharedState {
pending_preset: Mutex::new(None),
tool_selector: Mutex::new(tool_selector),
registered_tools: Mutex::new(default_tools),
});
Ok(McpServer {
engine,
shared,
preset_dir: preset_dir.to_owned(),
})
}
fn apply_pending_preset(&mut self) {
let pending = {
let mut guard = self.shared.pending_preset.lock()
.unwrap_or_else(|e| e.into_inner());
guard.take()
};
if let Some(toml_str) = pending {
match self.engine.reload_preset(&toml_str) {
Ok(()) => {
if let Ok(new_preset) = PresetParser::parse(&toml_str) {
if let Ok(mut sel) = self.shared.tool_selector.lock() {
if let Ok(mut new_sel) = ToolSelector::new(Path::new(""), &new_preset) {
if let Ok(tools) = self.shared.registered_tools.lock() {
let _ = new_sel.register_tools(&tools);
}
*sel = new_sel;
}
}
}
eprintln!("[sqz-mcp] preset applied from hot-reload");
}
Err(e) => eprintln!("[sqz-mcp] engine reload error: {e}"),
}
}
}
pub fn handle_tool_call(&mut self, request: ToolCallRequest) -> Result<ToolCallResponse> {
self.apply_pending_preset();
match request.tool_id.as_str() {
"passthrough" => self.handle_passthrough(request),
"expand" => self.handle_expand(request),
"sqz_recall" => self.handle_sqz_recall(request),
"sqz_read_file" => self.handle_sqz_read_file(request),
"sqz_grep" => self.handle_sqz_grep(request),
"sqz_list_dir" => self.handle_sqz_list_dir(request),
_ => self.handle_compress(request),
}
}
fn log_compression(&self, tool_id: &str, tokens_original: u32, tokens_compressed: u32) {
let project = std::env::current_dir().ok();
let project_str = project.as_ref().map(|p| p.to_string_lossy().to_string());
let _ = self.engine.session_store().log_compression_with_project(
tokens_original,
tokens_compressed,
&[],
tool_id,
project_str.as_deref(),
);
}
fn handle_compress(&mut self, request: ToolCallRequest) -> Result<ToolCallResponse> {
let raw_input = match request.input.get("text").and_then(|v| v.as_str()) {
Some(text) => text.to_string(),
None => serde_json::to_string(&request.input)
.map_err(|e| SqzError::Other(format!("input serialization error: {e}")))?,
};
let compressed = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {
let tokens_original = self.engine.count_tokens(&raw_input);
self.compress_cached(&raw_input, tokens_original, false, true)
.map(|(output, tokens_compressed)| (output, tokens_original, tokens_compressed))
}));
let (output, tokens_original, tokens_compressed) = match compressed {
Ok(result) => result?,
Err(_) => {
self.engine.clear_poison();
let tokens = estimate_tokens(&raw_input);
(raw_input, tokens, tokens)
}
};
self.log_compression(&request.tool_id, tokens_original, tokens_compressed);
Ok(ToolCallResponse {
tool_id: request.tool_id,
output,
tokens_original,
tokens_compressed,
})
}
fn handle_sqz_recall(&mut self, request: ToolCallRequest) -> Result<ToolCallResponse> {
let query = request
.input
.get("query")
.and_then(|v| v.as_str())
.ok_or_else(|| {
SqzError::Other("sqz_recall: input must be { \"query\": \"<terms>\" }".to_string())
})?;
let limit = request
.input
.get("limit")
.and_then(|v| v.as_u64())
.map(|n| n.clamp(1, 25) as u32)
.unwrap_or(5);
let hits = self.engine.session_store().recall_search(query, limit)?;
let mut output = if hits.is_empty() {
format!("[sqz:recall no matches for \"{query}\"]")
} else {
let mut out = format!("[sqz:recall query=\"{query}\" hits={}]\n", hits.len());
let mut any_output_kind = false;
for (i, hit) in hits.iter().enumerate() {
let snippet = hit.snippet.replace('\n', " ");
if hit.kind == "output" {
any_output_kind = true;
out.push_str(&format!(
"{}. [{} {}] ref={} — {}\n",
i + 1,
hit.kind,
hit.created_at,
&hit.ref_hash[..hit.ref_hash.len().min(16)],
snippet,
));
} else {
out.push_str(&format!(
"{}. [{} {}] id={} — {}\n",
i + 1,
hit.kind,
hit.created_at,
hit.ref_hash,
snippet,
));
}
}
if any_output_kind {
out.push_str("Pass a ref to the expand tool for the full original content.\n");
}
out
};
if output.ends_with('\n') {
output.pop();
}
let tokens = estimate_tokens(&output);
Ok(ToolCallResponse {
tool_id: request.tool_id,
output,
tokens_original: tokens,
tokens_compressed: tokens,
})
}
fn handle_passthrough(&mut self, request: ToolCallRequest) -> Result<ToolCallResponse> {
let text = match request.input.get("text").and_then(|v| v.as_str()) {
Some(s) => s.to_string(),
None => {
serde_json::to_string(&request.input)
.map_err(|e| SqzError::Other(format!("input serialization error: {e}")))?
}
};
let tokens = estimate_tokens(&text);
Ok(ToolCallResponse {
tool_id: request.tool_id,
output: text,
tokens_original: tokens,
tokens_compressed: tokens,
})
}
fn handle_expand(&mut self, request: ToolCallRequest) -> Result<ToolCallResponse> {
let raw = request
.input
.get("prefix")
.and_then(|v| v.as_str())
.ok_or_else(|| {
SqzError::Other("expand: input must be { \"prefix\": \"<hex>\" }".to_string())
})?;
let (prefix, range) = sqz_engine::parse_ref_token(raw);
let result = self.engine.cache_manager().expand_ref(raw)?;
let output = match result {
Some(sqz_engine::ExpandResult::Original { bytes, hash }) => {
let as_text = String::from_utf8_lossy(&bytes).into_owned();
match range {
Some((a, b)) => format!("[sqz:expand hash={hash} lines={a}-{b}]\n{as_text}"),
None => format!("[sqz:expand hash={hash}]\n{as_text}"),
}
}
Some(sqz_engine::ExpandResult::CompressedOnly { compressed, hash }) => {
format!(
"[sqz:expand hash={hash} note=compressed-only (predates original-capture migration)]\n{compressed}"
)
}
None => {
format!("[sqz:expand hash-not-found prefix={prefix}]")
}
};
let tokens = estimate_tokens(&output);
Ok(ToolCallResponse {
tool_id: request.tool_id,
output,
tokens_original: tokens,
tokens_compressed: tokens,
})
}
fn compress_cached(&self, text: &str, tokens: u32, lossless: bool, allow_delta: bool) -> Result<(String, u32)> {
use sqz_engine::CacheResult;
#[cfg(test)]
tests::maybe_panic();
let result = if lossless {
self.engine.compress_with_cache_lossless(text)?
} else {
self.engine.compress_with_cache(text)?
};
let tiny = tokens <= REF_TOKENS;
Ok(match result {
CacheResult::Dedup { inline_ref, token_cost } if !tiny => (inline_ref, token_cost),
CacheResult::Delta { delta_text, token_cost, .. } if allow_delta && !tiny => (delta_text, token_cost),
CacheResult::Dedup { .. } | CacheResult::Delta { .. } => return self.compress_fresh(text, lossless),
CacheResult::Fresh { output } => {
let truncated = output.stages_applied.iter().any(|s| s == "entropy_truncate");
if truncated && !self.engine.never_store(text) {
let hash = sqz_engine::CacheManager::sha256_hex(text.as_bytes());
let data = format!("{}\n[full output: call expand with prefix \"{}\"]", output.data, &hash[..16]);
let tokens = self.engine.count_tokens(&data);
(data, tokens)
} else {
(output.data, output.tokens_compressed)
}
}
})
}
fn compress_fresh(&self, text: &str, lossless: bool) -> Result<(String, u32)> {
let output = if lossless {
self.engine.compress_lossless(text)?
} else {
self.engine.compress(text)?
};
Ok((output.data, output.tokens_compressed))
}
fn handle_sqz_read_file(&mut self, request: ToolCallRequest) -> Result<ToolCallResponse> {
let path_str = request
.input
.get("path")
.and_then(|v| v.as_str())
.ok_or_else(|| {
SqzError::Other(
"sqz_read_file: input must be { \"path\": \"<file>\" }".to_string(),
)
})?;
let max_bytes = cap_from(&request.input, "max_bytes", READ_DEFAULT_MAX_BYTES);
let path = std::path::PathBuf::from(path_str);
if let Ok(meta) = std::fs::metadata(&path) {
if meta.len() > READ_HARD_LIMIT_BYTES {
return Err(SqzError::Other(format!(
"sqz_read_file: '{}' is {} MB, too large to load; use head/sed/tail through the shell for files this size",
path.display(),
meta.len() / (1024 * 1024)
)));
}
}
let bytes = match std::fs::read(&path) {
Ok(b) => b,
Err(e) => {
return Err(SqzError::Other(format!(
"sqz_read_file: could not read '{}': {e}",
path.display()
)))
}
};
let full_text = String::from_utf8_lossy(&bytes).into_owned();
let offset = request.input.get("offset").and_then(|v| v.as_u64()).map(|v| v.max(1) as usize);
let limit = request.input.get("limit").and_then(|v| v.as_u64()).map(|v| v as usize);
let total_lines = full_text.lines().count();
let (raw_text, range) = match (offset, limit) {
(None, None) => (full_text, None),
(offset, limit) => {
let start = offset.unwrap_or(1);
let end = match limit {
Some(n) => start.saturating_add(n).saturating_sub(1).min(total_lines),
None => total_lines,
};
if start > total_lines {
return Err(SqzError::Other(format!(
"sqz_read_file: offset {start} is past the end of '{}' ({total_lines} lines)",
path.display()
)));
}
(line_range(&full_text, start, end).to_string(), Some((start, end)))
}
};
let first_line = range.map(|(a, _)| a).unwrap_or(1);
let mut shown = range;
let mut cut = None;
let raw_text = match truncate_to_lines(&raw_text, max_bytes) {
Some(kept) => {
let last = first_line + kept.lines().count().max(1) - 1;
shown = Some((first_line, last));
cut = Some((last, !kept.ends_with('\n')));
kept.to_string()
}
None => raw_text,
};
let tokens_original = self.engine.count_tokens(&raw_text);
let (compressed_data, tokens_compressed) =
self.compress_cached(&raw_text, tokens_original, true, range.is_none())?;
let mut header = format!("[sqz_read_file path={} size={}", path.display(), bytes.len());
if let Some((a, b)) = shown {
header.push_str(&format!(" lines={a}-{b} of {total_lines}"));
}
if let Some((last, mid_line)) = cut {
header.push_str(&format!(" truncated_to={max_bytes}"));
if mid_line {
header.push_str(&format!(" line_cut={last}"));
}
if last < total_lines {
header.push_str(&format!(" continue_with_offset={}", last + 1));
}
}
let output = format!("{header}]\n{compressed_data}");
self.log_compression(&request.tool_id, tokens_original, tokens_compressed);
Ok(ToolCallResponse {
tool_id: request.tool_id,
output,
tokens_original,
tokens_compressed,
})
}
fn handle_sqz_list_dir(&mut self, request: ToolCallRequest) -> Result<ToolCallResponse> {
let path_str = request
.input
.get("path")
.and_then(|v| v.as_str())
.unwrap_or(".");
let max_depth = request
.input
.get("max_depth")
.and_then(|v| v.as_u64())
.unwrap_or(1)
.max(1) as usize;
let root = std::path::PathBuf::from(path_str);
let mut listing = Listing {
lines: Vec::new(),
max_depth,
max_entries: cap_from(&request.input, "max_entries", LIST_DEFAULT_MAX_ENTRIES),
capped: false,
};
list_dir_recursive(&root, &root, 1, &mut listing)?;
let raw = listing.lines.join("\n");
let tokens_original = self.engine.count_tokens(&raw);
let (compressed_data, tokens_compressed) = self.compress_cached(&raw, tokens_original, true, true)?;
self.log_compression(&request.tool_id, tokens_original, tokens_compressed);
let mut header = format!(
"[sqz_list_dir path={} entries={}",
root.display(),
listing.lines.len()
);
if listing.capped {
header.push_str(&format!(" stopped_at_max_entries={}", listing.max_entries));
}
let output = format!("{header}]\n{compressed_data}");
Ok(ToolCallResponse {
tool_id: request.tool_id,
output,
tokens_original,
tokens_compressed,
})
}
fn handle_sqz_grep(&mut self, request: ToolCallRequest) -> Result<ToolCallResponse> {
let pattern = request
.input
.get("pattern")
.and_then(|v| v.as_str())
.ok_or_else(|| {
SqzError::Other(
"sqz_grep: input must include { \"pattern\": \"<text>\" }".to_string(),
)
})?;
let path_str = request
.input
.get("path")
.and_then(|v| v.as_str())
.unwrap_or(".");
let limits = GrepLimits {
max_matches: request
.input
.get("max_matches")
.and_then(|v| v.as_u64())
.unwrap_or(200) as usize,
max_line_chars: cap_from(&request.input, "max_line_chars", GREP_DEFAULT_MAX_LINE_CHARS),
max_bytes: cap_from(&request.input, "max_bytes", GREP_DEFAULT_MAX_BYTES),
};
let use_regex = request
.input
.get("regex")
.and_then(|v| v.as_bool())
.unwrap_or(false);
let root = std::path::PathBuf::from(path_str);
let regex = if use_regex {
match regex::Regex::new(pattern) {
Ok(r) => Some(r),
Err(e) => {
return Err(SqzError::Other(format!(
"sqz_grep: invalid regex: {e}"
)))
}
}
} else {
None
};
let mut found = GrepResults::default();
grep_walk(&root, pattern, regex.as_ref(), &limits, &mut found)?;
let raw = found.lines.join("\n");
let tokens_original = self.engine.count_tokens(&raw);
let (compressed_data, tokens_compressed) = self.compress_cached(&raw, tokens_original, true, true)?;
self.log_compression(&request.tool_id, tokens_original, tokens_compressed);
let mut header = format!(
"[sqz_grep pattern={:?} root={} matches={}",
pattern,
root.display(),
found.lines.len(),
);
if found.lines.len() >= limits.max_matches {
header.push_str(&format!(" max_matches_reached={}", limits.max_matches));
}
if found.byte_capped {
header.push_str(&format!(" stopped_at_max_bytes={}", limits.max_bytes));
}
if found.clipped > 0 {
header.push_str(&format!(
" long_lines_clipped={} (max_line_chars={})",
found.clipped, limits.max_line_chars
));
}
let output = format!("{header}]\n{compressed_data}");
Ok(ToolCallResponse {
tool_id: request.tool_id,
output,
tokens_original,
tokens_compressed,
})
}
pub fn list_tools(&self, intent: Option<&str>) -> Result<Vec<ToolDefinition>> {
let tools = self.shared.registered_tools.lock()
.unwrap_or_else(|e| e.into_inner());
match intent {
Some(intent_str) if !intent_str.is_empty() => {
let selector = self.shared.tool_selector.lock()
.unwrap_or_else(|e| e.into_inner());
let selected_ids = selector.select(intent_str, 5)?;
let filtered: Vec<ToolDefinition> = tools
.iter()
.filter(|t| selected_ids.contains(&t.id))
.cloned()
.collect();
Ok(filtered)
}
_ => Ok(tools.clone()),
}
}
pub fn start(self, transport: McpTransport) -> Result<()> {
match transport {
McpTransport::Stdio => self.run_stdio(),
McpTransport::Sse { port } => self.run_sse(port),
}
}
pub fn watch_presets(&self) -> Result<notify::RecommendedWatcher> {
use notify::{Event, EventKind, RecursiveMode, Watcher};
let shared = Arc::clone(&self.shared);
let mut watcher = notify::recommended_watcher(move |res: notify::Result<Event>| {
if let Ok(event) = res {
if !matches!(event.kind, EventKind::Modify(_) | EventKind::Create(_)) {
return;
}
for path in &event.paths {
if path.extension().and_then(|e| e.to_str()) != Some("toml") {
continue;
}
match std::fs::read_to_string(path) {
Ok(toml_str) => {
match PresetParser::parse(&toml_str) {
Ok(_) => {
if let Ok(mut pending) = shared.pending_preset.lock() {
*pending = Some(toml_str);
}
eprintln!("[sqz-mcp] preset change detected: {}", path.display());
}
Err(e) => {
eprintln!("[sqz-mcp] invalid preset TOML in {}: {e}", path.display());
}
}
}
Err(e) => eprintln!("[sqz-mcp] preset file read error: {e}"),
}
}
}
})
.map_err(|e| SqzError::Other(format!("watcher init error: {e}")))?;
watcher
.watch(&self.preset_dir, RecursiveMode::NonRecursive)
.map_err(|e| SqzError::Other(format!("watcher watch error: {e}")))?;
Ok(watcher)
}
fn run_stdio(mut self) -> Result<()> {
let stdin = std::io::stdin();
let stdout = std::io::stdout();
let mut out = stdout.lock();
let mut input = stdin.lock();
let mut buf = Vec::new();
while read_line_bytes(&mut input, &mut buf)
.map_err(|e| SqzError::Other(format!("stdin read error: {e}")))?
{
let response = match std::str::from_utf8(&buf) {
Ok(line) if line.trim().is_empty() => continue,
Ok(line) => match self.handle_jsonrpc_line(line) {
Some(response) => response,
None => continue,
},
Err(e) => JsonRpcResponse::err(None, -32700, format!("parse error: {e}")),
};
let serialized = serde_json::to_string(&response)
.unwrap_or_else(|_| r#"{"jsonrpc":"2.0","error":{"code":-32700,"message":"serialize error"}}"#.to_string());
writeln!(out, "{serialized}")
.map_err(|e| SqzError::Other(format!("stdout write error: {e}")))?;
out.flush()
.map_err(|e| SqzError::Other(format!("stdout flush error: {e}")))?;
}
Ok(())
}
fn run_sse(mut self, port: u16) -> Result<()> {
use std::net::TcpListener;
use std::io::BufReader;
let listener = TcpListener::bind(format!("127.0.0.1:{port}"))
.map_err(|e| SqzError::Other(format!("SSE bind error on port {port}: {e}")))?;
eprintln!("[sqz-mcp] SSE server listening on http://127.0.0.1:{port}");
for stream in listener.incoming() {
match stream {
Ok(mut stream) => {
let mut reader = BufReader::new(stream.try_clone()
.map_err(|e| SqzError::Other(format!("stream clone error: {e}")))?);
let mut request_line = String::new();
let _ = reader.read_line(&mut request_line);
let mut content_length = 0usize;
loop {
let mut header = String::new();
let _ = reader.read_line(&mut header);
if header == "\r\n" || header.is_empty() {
break;
}
let lower = header.to_lowercase();
if lower.starts_with("content-length:") {
if let Some(v) = lower.split(':').nth(1) {
content_length = v.trim().parse().unwrap_or(0);
}
}
}
let mut body = vec![0u8; content_length];
use std::io::Read;
let _ = reader.read_exact(&mut body);
let body_str = String::from_utf8_lossy(&body);
let (status, json) = match self.handle_jsonrpc_line(body_str.trim()) {
Some(resp) => ("200 OK", serde_json::to_string(&resp).unwrap_or_default()),
None => ("204 No Content", String::new()),
};
let http_response = format!(
"HTTP/1.1 {}\r\nContent-Type: application/json\r\nContent-Length: {}\r\nAccess-Control-Allow-Origin: *\r\n\r\n{}",
status,
json.len(),
json
);
let _ = stream.write_all(http_response.as_bytes());
}
Err(e) => eprintln!("[sqz-mcp] connection error: {e}"),
}
}
Ok(())
}
fn handle_jsonrpc_line(&mut self, line: &str) -> Option<JsonRpcResponse> {
self.apply_pending_preset();
let req: JsonRpcRequest = match serde_json::from_str(line) {
Ok(r) => r,
Err(e) => return Some(JsonRpcResponse::err(None, -32700, format!("parse error: {e}"))),
};
if req.id.is_none() {
return None;
}
let id = req.id.clone();
let response = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| match req.method.as_str() {
"tools/list" => {
let intent = req.params
.as_ref()
.and_then(|p| p.get("intent"))
.and_then(|v| v.as_str())
.map(|s| s.to_string());
match self.list_tools(intent.as_deref()) {
Ok(tools) => {
let tool_list: Vec<Value> = tools.iter().map(|t| {
let mut tool_json = serde_json::json!({
"name": t.id,
"description": t.description,
"inputSchema": t.input_schema,
"sqz:transforms": t.compression_transforms,
});
if !t.output_schema.is_null() {
if let Some(obj) = tool_json.as_object_mut() {
obj.insert(
"outputSchema".to_string(),
t.output_schema.clone(),
);
}
}
tool_json
}).collect();
JsonRpcResponse::ok(req.id, serde_json::json!({ "tools": tool_list }))
}
Err(e) => JsonRpcResponse::err(req.id, -32603, e.to_string()),
}
}
"tools/call" => {
let params = match req.params {
Some(p) => p,
None => return JsonRpcResponse::err(req.id, -32602, "missing params"),
};
let tool_id = match params.get("name").and_then(|v| v.as_str()) {
Some(id) => id.to_string(),
None => return JsonRpcResponse::err(req.id, -32602, "missing params.name"),
};
let input = params.get("arguments").cloned().unwrap_or(Value::Null);
let intent = params.get("intent").and_then(|v| v.as_str()).map(|s| s.to_string());
let call_req = ToolCallRequest { tool_id, input, intent };
match self.handle_tool_call(call_req) {
Ok(resp) => JsonRpcResponse::ok(req.id, serde_json::json!({
"content": [{ "type": "text", "text": resp.output }],
"tokens_original": resp.tokens_original,
"tokens_compressed": resp.tokens_compressed,
})),
Err(e) => JsonRpcResponse::err(req.id, -32603, e.to_string()),
}
}
"initialize" => {
JsonRpcResponse::ok(req.id, serde_json::json!({
"protocolVersion": "2024-11-05",
"capabilities": {
"tools": { "listChanged": false }
},
"serverInfo": { "name": "sqz-mcp", "version": env!("CARGO_PKG_VERSION") }
}))
}
"ping" => JsonRpcResponse::ok(req.id, serde_json::json!({})),
_ => JsonRpcResponse::err(req.id, -32601, format!("method not found: {}", req.method)),
}));
Some(response.unwrap_or_else(|_| {
self.engine.clear_poison();
JsonRpcResponse::err(id, -32603, format!("internal error: {} panicked", req.method))
}))
}
}
fn read_line_bytes(reader: &mut impl BufRead, buf: &mut Vec<u8>) -> std::io::Result<bool> {
buf.clear();
if reader.read_until(b'\n', buf)? == 0 {
return Ok(false);
}
if buf.last() == Some(&b'\n') {
buf.pop();
if buf.last() == Some(&b'\r') {
buf.pop();
}
}
Ok(true)
}
pub fn default_tool_definitions() -> Vec<ToolDefinition> {
vec![
ToolDefinition {
id: "compress".to_string(),
name: "Compress Text".to_string(),
description: "Compress text or JSON you already have, such as a \
long tool result, through the sqz pipeline. It reads no files \
and runs no commands."
.to_string(),
input_schema: serde_json::json!({
"type": "object",
"properties": {
"text": {
"type": "string",
"description": "Text or JSON to compress."
}
},
"required": ["text"]
}),
compression_transforms: vec![
"sha256_cache: repeat inputs within the session return a ~13-token §ref:HASH§ token".to_string(),
"ast_extract: recognised source code collapses to signatures only".to_string(),
"ansi_strip: removes color/formatting codes".to_string(),
"condense: repeated identical lines collapsed to max 3 occurrences".to_string(),
"git_diff_fold: diff output has unchanged context lines folded".to_string(),
"log_fold: repeated log lines with timestamps folded to [xN]".to_string(),
"path_shorten: common path prefixes replaced with ~/".to_string(),
"truncate_strings: strings > 500 chars are truncated with '...'".to_string(),
"entropy_truncate: low-information segments are dropped and marked; the output then ends with the expand prefix of the full original".to_string(),
"safe_fallback: error/warning lines always preserved verbatim".to_string(),
"preservation_verifier: path-like and identifier tokens are \
checked for byte-exact survival; compression is discarded if \
coverage drops below 85%"
.to_string(),
],
..Default::default()
},
ToolDefinition {
id: "passthrough".to_string(),
name: "Passthrough (No Compression)".to_string(),
description: "Return the text unchanged, for when you need it \
raw rather than compressed."
.to_string(),
input_schema: serde_json::json!({
"type": "object",
"properties": {
"text": {
"type": "string",
"description": "Text to return as is."
}
},
"required": ["text"]
}),
compression_transforms: vec![
"none: input is returned byte-for-byte".to_string(),
],
..Default::default()
},
ToolDefinition {
id: "expand".to_string(),
name: "Expand Dedup Ref".to_string(),
description: "Return the original content behind a ref. \
`§ref:HASH§` stands for content sent earlier in the session; \
`§ref:HASH:L40-80§` for lines 40-80 of content sent in full, \
and expands to just those lines."
.to_string(),
input_schema: serde_json::json!({
"type": "object",
"properties": {
"prefix": {
"type": "string",
"description": "The ref token pasted whole, or its hex prefix."
}
},
"required": ["prefix"]
}),
compression_transforms: vec![
"none: returns cached original bytes".to_string(),
],
..Default::default()
},
ToolDefinition {
id: "sqz_recall".to_string(),
name: "Recall (Search Session Memory)".to_string(),
description: "Search everything sqz has seen, this session and \
earlier, before re-running a command or re-reading a file. \
Pass a hit's ref to `expand` for the full text."
.to_string(),
input_schema: serde_json::json!({
"type": "object",
"properties": {
"query": {
"type": "string",
"description": "Search terms; all must match."
},
"limit": {
"type": "integer",
"description": "Maximum number of hits, 1-25.",
"default": 5
}
},
"required": ["query"]
}),
compression_transforms: vec![
"bm25_rank: hits ordered by FTS5 relevance".to_string(),
"snippet: ~24 tokens of context per hit with >>match<< markers".to_string(),
],
..Default::default()
},
ToolDefinition {
id: "sqz_read_file".to_string(),
name: "Read File (Compressed)".to_string(),
description: "Read a file faithfully, stripping only ANSI \
escapes. Prefer it over the built-in read for files over \
2KB or ones you may read again: a repeat or ranged re-read \
returns a short ref (see `expand`) and a re-read after an \
edit returns only the changed lines."
.to_string(),
input_schema: serde_json::json!({
"type": "object",
"properties": {
"path": {
"type": "string",
"description": "File path, absolute or relative to the working directory."
},
"offset": {
"type": "integer",
"description": "First line to return, 1-based."
},
"limit": {
"type": "integer",
"description": "Number of lines to return."
},
"max_bytes": {
"type": "integer",
"description": "Byte cap, cut at a line boundary; 0 for no cap. When cut, the header gives continue_with_offset.",
"default": 262144
}
},
"required": ["path"]
}),
compression_transforms: vec![
"sha256_cache: repeat reads of unchanged content return a ~13-token §ref:HASH§ token".to_string(),
"slice_ref: a line range of content already read in full returns §ref:HASH:L<a>-<b>§".to_string(),
"delta: a re-read after a small edit returns only the changed lines".to_string(),
"ansi_strip: removes color/formatting codes".to_string(),
"max_bytes: output past the cap is cut at a line boundary and the header says where to continue".to_string(),
"lossless: no lines are folded or summarized".to_string(),
],
..Default::default()
},
ToolDefinition {
id: "sqz_list_dir".to_string(),
name: "List Directory (Compressed)".to_string(),
description: "List a directory, skipping hidden entries and \
node_modules, target, dist, build, vendor and __pycache__. \
Prefer it over `ls -la` for a project layout."
.to_string(),
input_schema: serde_json::json!({
"type": "object",
"properties": {
"path": {
"type": "string",
"description": "Directory to list.",
"default": "."
},
"max_depth": {
"type": "integer",
"description": "Levels to recurse; 1 lists direct children only.",
"default": 1
},
"max_entries": {
"type": "integer",
"description": "Entry cap; 0 for no cap.",
"default": 1000
}
}
}),
compression_transforms: vec![
"skip_bulk_dirs: .git, node_modules, target, dist, build, vendor, __pycache__ omitted".to_string(),
"max_entries: listing stops at the cap (default 1000); the header says so".to_string(),
"sha256_cache: repeat listings dedupe via §ref§".to_string(),
"lossless: every listed entry survives verbatim".to_string(),
],
..Default::default()
},
ToolDefinition {
id: "sqz_grep".to_string(),
name: "Grep Files (Compressed)".to_string(),
description: "Search files for a literal string or regex, \
returning `path:lineno:text` lines. Prefer it over the \
built-in grep when a search may match more than a few \
lines: output is capped and a repeat search returns a \
short ref."
.to_string(),
input_schema: serde_json::json!({
"type": "object",
"properties": {
"pattern": {
"type": "string",
"description": "Text to find, a literal substring unless `regex` is true."
},
"path": {
"type": "string",
"description": "File or directory to search.",
"default": "."
},
"regex": {
"type": "boolean",
"description": "Treat `pattern` as a regex.",
"default": false
},
"max_matches": {
"type": "integer",
"description": "Stop after this many matches.",
"default": 200
},
"max_line_chars": {
"type": "integer",
"description": "Clip longer match lines around the hit; 0 for no cap.",
"default": 400
},
"max_bytes": {
"type": "integer",
"description": "Stop at this many bytes of output; 0 for no cap.",
"default": 100000
}
},
"required": ["pattern"]
}),
compression_transforms: vec![
"max_matches: search stops after the cap (default 200); the header reports the match count".to_string(),
"max_line_chars: a match line over 400 chars is clipped around the hit and marked".to_string(),
"max_bytes: search stops at 100000 bytes of output; the header says so".to_string(),
"sha256_cache: repeat searches dedupe via §ref§".to_string(),
"lossless: match lines under the caps survive verbatim".to_string(),
],
..Default::default()
},
]
}
fn estimate_tokens(text: &str) -> u32 {
((text.len() as f64) / 4.0).ceil() as u32
}
const READ_DEFAULT_MAX_BYTES: usize = 256 * 1024;
const READ_HARD_LIMIT_BYTES: u64 = 64 * 1024 * 1024;
const GREP_DEFAULT_MAX_BYTES: usize = 100_000;
const GREP_DEFAULT_MAX_LINE_CHARS: usize = 400;
const LIST_DEFAULT_MAX_ENTRIES: usize = 1000;
const REF_TOKENS: u32 = 13;
fn cap_from(input: &serde_json::Value, key: &str, default: usize) -> usize {
match input.get(key).and_then(|v| v.as_u64()) {
Some(0) => usize::MAX,
Some(v) => usize::try_from(v).unwrap_or(usize::MAX),
None => default,
}
}
fn truncate_to_lines(text: &str, max_bytes: usize) -> Option<&str> {
if text.len() <= max_bytes {
return None;
}
let mut cut = max_bytes;
while !text.is_char_boundary(cut) {
cut -= 1;
}
match text[..cut].rfind('\n') {
Some(nl) => Some(&text[..=nl]),
None => Some(&text[..cut]),
}
}
fn clip_line(line: &str, hit: (usize, usize), max_chars: usize) -> Option<String> {
if line.len() <= max_chars {
return None;
}
let total = line.chars().count();
if total <= max_chars {
return None;
}
let (hit_start, hit_end) = hit;
let hit_start_char = line[..hit_start].chars().count();
let hit_chars = line[hit_start..hit_end].chars().count();
let lead = max_chars.saturating_sub(hit_chars) / 2;
let mut from = hit_start_char.saturating_sub(lead);
let to = (from + max_chars).min(total);
from = to.saturating_sub(max_chars);
let window: String = line.chars().skip(from).take(to - from).collect();
Some(format!(
"{}{}{} [line clipped: {total} chars]",
if from > 0 { "…" } else { "" },
window,
if to < total { "…" } else { "" },
))
}
fn line_range(text: &str, start: usize, end: usize) -> &str {
let mut line = 1;
let mut begin = None;
for (i, ch) in text.char_indices() {
if line == start && begin.is_none() {
begin = Some(i);
}
if ch == '\n' {
if line == end {
return &text[begin.unwrap_or(i)..=i];
}
line += 1;
}
}
begin.map(|b| &text[b..]).unwrap_or("")
}
fn list_dir_recursive(
root: &std::path::Path,
current: &std::path::Path,
depth: usize,
out: &mut Listing,
) -> Result<()> {
if depth > out.max_depth || out.capped {
return Ok(());
}
let entries = match std::fs::read_dir(current) {
Ok(e) => e,
Err(e) => {
return Err(SqzError::Other(format!(
"sqz_list_dir: could not read '{}': {e}",
current.display()
)))
}
};
let mut sorted: Vec<_> = entries
.filter_map(|e| e.ok())
.collect();
sorted.sort_by_key(|e| e.file_name());
for entry in sorted {
if out.capped {
break;
}
let name = entry.file_name();
let name_str = name.to_string_lossy();
if name_str.starts_with('.') {
continue;
}
if matches!(
name_str.as_ref(),
"node_modules" | "target" | "dist" | "build" | "__pycache__"
| "vendor" | ".next" | ".nuxt"
) {
continue;
}
let path = entry.path();
let rel = path.strip_prefix(root).unwrap_or(&path);
let type_char = match entry.file_type() {
Ok(ft) if ft.is_dir() => "d",
Ok(ft) if ft.is_symlink() => "l",
Ok(_) => "f",
Err(_) => "?",
};
if out.lines.len() >= out.max_entries {
out.capped = true;
break;
}
out.lines.push(format!("{type_char} {}", rel.display()));
if entry.file_type().map(|ft| ft.is_dir()).unwrap_or(false) && depth < out.max_depth {
let _ = list_dir_recursive(root, &path, depth + 1, out);
}
}
Ok(())
}
struct Listing {
lines: Vec<String>,
max_depth: usize,
max_entries: usize,
capped: bool,
}
fn grep_walk(
root: &std::path::Path,
needle: &str,
regex: Option<®ex::Regex>,
limits: &GrepLimits,
out: &mut GrepResults,
) -> Result<()> {
if out.full(limits) {
return Ok(());
}
if !root.exists() {
return Err(SqzError::Other(format!(
"sqz_grep: path '{}' does not exist",
root.display()
)));
}
if root.is_file() {
grep_one_file(root, needle, regex, limits, out)?;
return Ok(());
}
let entries = match std::fs::read_dir(root) {
Ok(e) => e,
Err(e) => {
return Err(SqzError::Other(format!(
"sqz_grep: could not read '{}': {e}",
root.display()
)))
}
};
let mut sorted: Vec<_> = entries.filter_map(|e| e.ok()).collect();
sorted.sort_by_key(|e| e.file_name());
for entry in sorted {
if out.full(limits) {
break;
}
let name = entry.file_name();
let name_str = name.to_string_lossy();
if name_str.starts_with('.') {
continue;
}
if matches!(
name_str.as_ref(),
"node_modules" | "target" | "dist" | "build" | "__pycache__"
| "vendor" | ".next" | ".nuxt"
) {
continue;
}
let path = entry.path();
match entry.file_type() {
Ok(ft) if ft.is_dir() => {
grep_walk(&path, needle, regex, limits, out)?;
}
Ok(_) => {
grep_one_file(&path, needle, regex, limits, out)?;
}
Err(_) => continue,
}
}
Ok(())
}
struct GrepLimits {
max_matches: usize,
max_line_chars: usize,
max_bytes: usize,
}
#[derive(Default)]
struct GrepResults {
lines: Vec<String>,
bytes: usize,
clipped: usize,
byte_capped: bool,
}
impl GrepResults {
fn full(&self, limits: &GrepLimits) -> bool {
self.byte_capped || self.lines.len() >= limits.max_matches
}
fn push(&mut self, line: String, clipped: bool, limits: &GrepLimits) {
let cost = line.len() + 1;
if !self.lines.is_empty() && self.bytes.saturating_add(cost) > limits.max_bytes {
self.byte_capped = true;
return;
}
self.bytes += cost;
self.lines.push(line);
if clipped {
self.clipped += 1;
}
}
}
fn grep_one_file(
file: &std::path::Path,
needle: &str,
regex: Option<®ex::Regex>,
limits: &GrepLimits,
out: &mut GrepResults,
) -> Result<()> {
if let Ok(meta) = std::fs::metadata(file) {
if meta.len() > 50 * 1024 * 1024 {
return Ok(());
}
}
let bytes = match std::fs::read(file) {
Ok(b) => b,
Err(_) => return Ok(()), };
let text = match std::str::from_utf8(&bytes) {
Ok(t) => t,
Err(_) => return Ok(()), };
for (lineno, line) in text.lines().enumerate() {
if out.full(limits) {
break;
}
let hit = match regex {
Some(r) => r.find(line).map(|m| (m.start(), m.end())),
None => line.find(needle).map(|i| (i, i + needle.len())),
};
if let Some(hit) = hit {
let clipped = clip_line(line, hit, limits.max_line_chars);
let was_clipped = clipped.is_some();
let shown = clipped.unwrap_or_else(|| line.to_string());
out.push(
format!("{}:{}:{}", file.display(), lineno + 1, shown),
was_clipped,
limits,
);
}
}
Ok(())
}
#[cfg(test)]
mod tests {
use super::*;
use std::time::{Duration, Instant};
use tempfile::TempDir;
impl McpServer {
pub(crate) fn handle_jsonrpc_line_unwrap(&mut self, line: &str) -> JsonRpcResponse {
self.handle_jsonrpc_line(line)
.expect("expected response; got None (notification). Use handle_jsonrpc_line directly if that's intended.")
}
}
fn make_server() -> (McpServer, TempDir) {
let dir = TempDir::new().expect("tempdir");
let store_path = dir.path().join("sessions.db");
let server = McpServer::new_with_store(dir.path(), &store_path)
.expect("McpServer::new_with_store");
(server, dir)
}
thread_local! {
static PANIC_NEXT_ENGINE_CALL: std::cell::Cell<bool> = const { std::cell::Cell::new(false) };
}
pub(crate) fn panic_next_engine_call() {
PANIC_NEXT_ENGINE_CALL.with(|p| p.set(true));
}
pub(crate) fn maybe_panic() {
if PANIC_NEXT_ENGINE_CALL.with(|p| p.replace(false)) {
panic!("injected engine panic");
}
}
#[test]
fn test_handle_tool_call_compresses_output() {
let (mut server, _dir) = make_server();
let input = serde_json::json!({
"status": "ok",
"data": {
"id": 1,
"name": "test",
"debug_info": null,
"trace_id": null,
"metadata": {
"internal_id": "abc123",
"created_at": "2025-01-01T00:00:00Z"
},
"items": ["a", "b", "c", "d", "e", "f", "g", "h"]
}
});
let req = ToolCallRequest {
tool_id: "compress".to_string(),
input: input.clone(),
intent: None,
};
let resp = server.handle_tool_call(req).expect("handle_tool_call");
assert_eq!(resp.tool_id, "compress");
assert!(!resp.output.is_empty(), "output should not be empty");
assert!(resp.tokens_original > 0, "tokens_original should be > 0");
}
#[test]
fn test_handle_tool_call_preserves_tool_id() {
let (mut server, _dir) = make_server();
let req = ToolCallRequest {
tool_id: "compress".to_string(),
input: serde_json::json!({ "text": "ls -la output here" }),
intent: None,
};
let resp = server.handle_tool_call(req).expect("handle_tool_call");
assert_eq!(resp.tool_id, "compress");
}
#[test]
fn test_compress_tool_compresses_the_text_field() {
let (mut server, _dir) = make_server();
let mut text: String = (0..300)
.map(|i| format!("2026-10-01T12:00:{:02}Z INFO worker {i} finished job {i} in {} ms\n", i % 60, i % 97))
.collect();
text.push_str("ERROR worker 299 failed: disk full\n");
let resp = server
.handle_tool_call(ToolCallRequest {
tool_id: "compress".to_string(),
input: serde_json::json!({ "text": text }),
intent: None,
})
.unwrap();
let head: String = resp.output.chars().take(200).collect();
assert!(!resp.output.starts_with("TOON:{text:"), "{head}");
assert!(resp.output.contains("disk full"), "{head}");
}
#[test]
fn test_compress_tool_names_the_expand_prefix_when_it_drops_segments() {
let (mut server, _dir) = make_server();
let mut text = String::from("run marker for the compress spill test\n\n");
for i in 0..12u8 {
let word: String = std::iter::repeat((b'a' + i) as char).take(10).collect();
text.push_str(&format!("{} filler-{i}\n\n", format!("{word} ").repeat(30)));
}
text.push_str("The quick brown fox jumps over the lazy dog by silver rivers today.\n");
let call = |server: &mut McpServer, tool: &str, input: serde_json::Value| {
server
.handle_tool_call(ToolCallRequest { tool_id: tool.to_string(), input, intent: None })
.unwrap()
};
let resp = call(&mut server, "compress", serde_json::json!({ "text": text }));
let hash = sqz_engine::CacheManager::sha256_hex(text.as_bytes());
assert!(resp.output.contains("low-information segments omitted"), "{}", resp.output);
assert!(
resp.output.ends_with(&format!("\n[full output: call expand with prefix \"{}\"]", &hash[..16])),
"{}",
resp.output
);
assert_eq!(resp.tokens_compressed, server.engine.count_tokens(&resp.output));
let expanded = call(&mut server, "expand", serde_json::json!({ "prefix": &hash[..16] }));
assert_eq!(expanded.output, format!("[sqz:expand hash={hash}]\n{text}"));
let compress = default_tool_definitions().into_iter().find(|t| t.id == "compress").unwrap();
assert!(compress.compression_transforms.iter().any(|t| t.starts_with("entropy_truncate:")));
}
fn high_risk_texts() -> Vec<(&'static str, String)> {
let hex = |s: &str| sqz_engine::CacheManager::sha256_hex(s.as_bytes());
let frames: String = (0..40)
.map(|i| format!(" {i:>2}: myapp::scheduler::step_{i}\n at ./src/scheduler/mod_{}.rs:{}:5\n", i % 7, 40 + 7 * i))
.collect();
let tables: String = ["customers", "orders", "refunds", "payments", "shipments", "invoices"]
.iter()
.map(|t| format!("CREATE TABLE {t} (\n id BIGSERIAL PRIMARY KEY,\n status TEXT NOT NULL DEFAULT 'pending',\n created_at TIMESTAMPTZ NOT NULL DEFAULT now()\n);\nCREATE INDEX idx_{t}_created_at ON {t} (created_at);\n"))
.collect();
let pem: String = (0..20).map(|i| hex(&format!("pem {i}")) + "\n").collect();
let creds: String = ["stripe", "sendgrid", "github", "datadog", "sentry"]
.iter()
.map(|s| format!("{s}_api_key: sk_test_{}\n{s}_endpoint: https://{s}.example.com/v1\n", &hex(s)[..32]))
.collect();
vec![
("panic", format!("thread 'main' panicked at src/scheduler/queue.rs:217:31:\ncalled `Option::unwrap()` on a `None` value\nstack backtrace:\n{frames}")),
("migration", format!("BEGIN;\n{tables}ALTER TABLE refunds ADD COLUMN reason TEXT;\nCOMMIT;\n")),
("keys", format!("-----BEGIN RSA PRIVATE KEY-----\n{pem}-----END RSA PRIVATE KEY-----\n{creds}")),
]
}
#[test]
fn test_compress_tool_returns_high_risk_text_unchanged() {
let (mut server, _dir) = make_server();
for (name, text) in high_risk_texts() {
assert!(text.len() > 500, "{name}");
for run in ["first", "second"] {
let resp = server
.handle_tool_call(ToolCallRequest {
tool_id: "compress".to_string(),
input: serde_json::json!({ "text": text }),
intent: None,
})
.unwrap();
assert_eq!(resp.output, text, "{name}, {run} call");
}
let stored = server.engine.cache_manager().check_dedup_with_meta(text.as_bytes()).unwrap();
assert!(stored.is_none(), "{name} was stored");
}
}
#[test]
fn test_list_tools_no_intent_returns_all() {
let (server, _dir) = make_server();
let tools = server.list_tools(None).expect("list_tools");
assert_eq!(tools.len(), default_tool_definitions().len());
}
#[test]
fn test_list_tools_with_intent_filters() {
let (server, _dir) = make_server();
let registered = default_tool_definitions().len();
let tools = server
.list_tools(Some("compress arbitrary text through the sqz pipeline"))
.expect("list_tools with intent should not error");
assert!(
tools.len() <= registered,
"filtered list must not exceed registered count ({registered})"
);
let tools = server.list_tools(Some("")).expect("empty intent = all tools");
assert_eq!(
tools.len(),
registered,
"empty intent is treated as `no intent` and returns every tool"
);
}
#[test]
fn test_tool_selector_latency_under_500ms() {
let (server, _dir) = make_server();
let start = Instant::now();
for _ in 0..10 {
let _ = server.list_tools(Some("search for files matching a pattern"));
}
let elapsed = start.elapsed();
assert!(
elapsed < Duration::from_millis(500),
"10 tool selections took {:?}, expected < 500ms",
elapsed
);
}
#[test]
fn test_preset_hot_reload_latency() {
let dir = TempDir::new().expect("tempdir");
let store = TempDir::new().expect("tempdir");
let server = McpServer::new_with_store(dir.path(), &store.path().join("sessions.db"))
.expect("McpServer::new_with_store");
let _watcher = server.watch_presets().expect("watch_presets");
let preset_path = dir.path().join("test.toml");
let toml_content = r#"
[preset]
name = "hot-reload-test"
version = "1.0"
[compression]
stages = []
[tool_selection]
max_tools = 5
similarity_threshold = 0.3
[budget]
warning_threshold = 0.70
ceiling_threshold = 0.85
default_window_size = 200000
[terse_mode]
enabled = false
level = "moderate"
[model]
family = "anthropic"
primary = "claude-sonnet-4-20250514"
complexity_threshold = 0.4
"#;
std::fs::write(&preset_path, toml_content).expect("write preset");
let deadline = Instant::now() + Duration::from_secs(2);
while Instant::now() < deadline {
std::thread::sleep(Duration::from_millis(50));
if let Ok(guard) = server.shared.pending_preset.lock() {
if guard.is_some() {
break;
}
}
}
let has_pending = server.shared.pending_preset.lock()
.map(|g| g.is_some())
.unwrap_or(false);
assert!(has_pending, "preset should have been hot-reloaded within 2 seconds");
}
#[test]
fn test_invalid_toml_keeps_previous_preset() {
let dir = TempDir::new().expect("tempdir");
let store = TempDir::new().expect("tempdir");
let server = McpServer::new_with_store(dir.path(), &store.path().join("sessions.db"))
.expect("McpServer::new_with_store");
let _watcher = server.watch_presets().expect("watch_presets");
let bad_path = dir.path().join("bad.toml");
std::fs::write(&bad_path, "this is not valid toml ][[[").expect("write bad preset");
std::thread::sleep(Duration::from_millis(200));
let has_pending = server.shared.pending_preset.lock()
.map(|g| g.is_some())
.unwrap_or(false);
assert!(!has_pending, "invalid TOML should not be stored as pending preset");
let tools = server.list_tools(None).expect("list_tools after bad preset");
assert!(!tools.is_empty(), "tools should still be available after invalid preset");
}
#[test]
fn test_jsonrpc_initialize() {
let (mut server, _dir) = make_server();
let line = r#"{"jsonrpc":"2.0","id":1,"method":"initialize","params":{}}"#;
let resp = server.handle_jsonrpc_line_unwrap(line);
assert!(resp.error.is_none(), "initialize should not error");
let result = resp.result.expect("initialize should have result");
assert!(result.get("protocolVersion").is_some());
}
#[test]
fn test_initialize_advertises_tools_capability() {
let (mut server, _dir) = make_server();
let line = r#"{"jsonrpc":"2.0","id":1,"method":"initialize","params":{}}"#;
let resp = server.handle_jsonrpc_line_unwrap(line);
let result = resp.result.expect("initialize should have result");
let caps = result.get("capabilities")
.expect("initialize result must include capabilities");
let tools_cap = caps.get("tools")
.expect("capabilities must include 'tools' key");
let tools_obj = tools_cap.as_object()
.expect("'tools' capability must be an object");
assert!(
!tools_obj.is_empty(),
"'tools' capability must not be empty {{}} — some MCP clients \
interpret that as no tools available. Got: {tools_cap:?}"
);
assert!(
tools_obj.contains_key("listChanged"),
"'tools' capability should include listChanged per MCP 2024-11-05 \
spec. Got: {tools_cap:?}"
);
}
#[test]
fn test_jsonrpc_tools_list() {
let (mut server, _dir) = make_server();
let line = r#"{"jsonrpc":"2.0","id":2,"method":"tools/list","params":{}}"#;
let resp = server.handle_jsonrpc_line_unwrap(line);
assert!(resp.error.is_none(), "tools/list should not error");
let result = resp.result.expect("tools/list should have result");
let tools = result.get("tools").expect("result should have tools");
assert!(tools.as_array().map(|a| !a.is_empty()).unwrap_or(false));
}
#[test]
fn test_jsonrpc_tools_call() {
let (mut server, _dir) = make_server();
let line = r#"{"jsonrpc":"2.0","id":3,"method":"tools/call","params":{"name":"compress","arguments":{"text":"lorem ipsum dolor sit amet"}}}"#;
let resp = server.handle_jsonrpc_line_unwrap(line);
assert!(resp.error.is_none(), "tools/call should not error: {:?}", resp.error);
let result = resp.result.expect("tools/call should have result");
assert!(result.get("content").is_some());
}
#[test]
fn test_jsonrpc_compress_panic_returns_the_text() {
let (mut server, _dir) = make_server();
let text = "2026-10-06T12:00:00Z INFO worker finished job in 3 ms\n".repeat(40);
let call = serde_json::json!({
"jsonrpc": "2.0", "id": 1, "method": "tools/call",
"params": { "name": "compress", "arguments": { "text": text } }
})
.to_string();
panic_next_engine_call();
let resp = server.handle_jsonrpc_line_unwrap(&call);
assert!(resp.error.is_none(), "{:?}", resp.error);
assert_eq!(resp.result.unwrap()["content"][0]["text"], text.as_str());
let resp = server.handle_jsonrpc_line_unwrap(&call);
let out = resp.result.unwrap()["content"][0]["text"].as_str().unwrap().to_string();
assert!(out.len() < text.len(), "compression works again: {out}");
}
#[test]
fn test_jsonrpc_tool_panic_returns_an_error() {
let (mut server, dir) = make_server();
let file = dir.path().join("notes.txt");
std::fs::write(&file, "first line of notes\nsecond line of notes\n").unwrap();
let call = serde_json::json!({
"jsonrpc": "2.0", "id": 2, "method": "tools/call",
"params": { "name": "sqz_read_file", "arguments": { "path": file.to_string_lossy() } }
})
.to_string();
panic_next_engine_call();
let resp = server.handle_jsonrpc_line_unwrap(&call);
assert_eq!(resp.id, Some(serde_json::json!(2)));
assert_eq!(resp.error.expect("an error").code, -32603);
let resp = server.handle_jsonrpc_line_unwrap(&call);
assert!(resp.error.is_none(), "{:?}", resp.error);
let text = resp.result.unwrap()["content"][0]["text"].as_str().unwrap().to_string();
assert!(text.contains("second line of notes"), "{text}");
}
#[test]
fn test_jsonrpc_ping() {
let (mut server, _dir) = make_server();
let resp = server.handle_jsonrpc_line_unwrap(r#"{"jsonrpc":"2.0","id":7,"method":"ping"}"#);
assert!(resp.error.is_none(), "ping should not error: {:?}", resp.error);
assert_eq!(resp.result, Some(serde_json::json!({})));
}
#[test]
fn test_jsonrpc_unknown_method() {
let (mut server, _dir) = make_server();
let line = r#"{"jsonrpc":"2.0","id":4,"method":"unknown/method","params":{}}"#;
let resp = server.handle_jsonrpc_line_unwrap(line);
assert!(resp.error.is_some());
assert_eq!(resp.error.unwrap().code, -32601);
}
#[test]
fn test_read_line_bytes_splits_like_lines() {
let text = "a\nb\r\nc\r\r\n\r\n\n\rd\ne";
let mut reader = text.as_bytes();
let mut buf = Vec::new();
let mut got = Vec::new();
while read_line_bytes(&mut reader, &mut buf).unwrap() {
got.push(String::from_utf8(buf.clone()).unwrap());
}
let want: Vec<String> = text.as_bytes().lines().map(|l| l.unwrap()).collect();
assert_eq!(got, want);
let mut reader: &[u8] = b"{\"text\":\"caf\xe9\"}\nnext\n";
assert!(read_line_bytes(&mut reader, &mut buf).unwrap());
assert_eq!(buf, b"{\"text\":\"caf\xe9\"}");
assert!(read_line_bytes(&mut reader, &mut buf).unwrap());
assert_eq!(buf, b"next");
assert!(!read_line_bytes(&mut reader, &mut buf).unwrap());
}
#[test]
fn test_jsonrpc_parse_error() {
let (mut server, _dir) = make_server();
let resp = server.handle_jsonrpc_line_unwrap("not json at all {{{");
assert!(resp.error.is_some());
assert_eq!(resp.error.unwrap().code, -32700);
}
#[test]
fn test_tools_list_outputschema_is_valid_object_or_absent() {
let (mut server, _dir) = make_server();
let line = r#"{"jsonrpc":"2.0","id":1,"method":"tools/list","params":{}}"#;
let resp = server.handle_jsonrpc_line_unwrap(line);
assert!(resp.error.is_none(), "tools/list errored: {:?}", resp.error);
let tools = resp.result
.expect("tools/list must have result")
.get("tools")
.cloned()
.expect("result must have tools array");
let tools = tools.as_array().expect("tools must be an array");
assert!(!tools.is_empty(), "no tools registered");
for tool in tools {
let name = tool.get("name").and_then(|v| v.as_str()).unwrap_or("?");
let input_type = tool
.get("inputSchema")
.and_then(|s| s.get("type"))
.and_then(|t| t.as_str());
assert_eq!(
input_type,
Some("object"),
"tool {name}: inputSchema.type must be \"object\", got {input_type:?}"
);
if let Some(out) = tool.get("outputSchema") {
if !out.is_null() {
let out_type = out.get("type").and_then(|t| t.as_str());
assert_eq!(
out_type,
Some("object"),
"tool {name}: outputSchema.type must be \"object\" \
per MCP spec; got {out_type:?}. This is the \
exact bug OpenCode reported in issue #5."
);
}
}
}
}
#[test]
fn test_default_tools_omit_outputschema() {
let (mut server, _dir) = make_server();
let line = r#"{"jsonrpc":"2.0","id":1,"method":"tools/list","params":{}}"#;
let resp = server.handle_jsonrpc_line_unwrap(line);
let tools = resp.result.unwrap().get("tools").cloned().unwrap();
for tool in tools.as_array().unwrap() {
let name = tool.get("name").and_then(|v| v.as_str()).unwrap_or("?");
assert!(
tool.get("outputSchema").is_none(),
"default tool {name} unexpectedly has outputSchema: \
{:?}. Remove it, or make tools/call also emit \
structuredContent matching the schema (MCP 2025-06-18).",
tool.get("outputSchema")
);
}
}
#[test]
fn test_tools_list_has_no_io_impostor_tools() {
let (mut server, _dir) = make_server();
let line = r#"{"jsonrpc":"2.0","id":1,"method":"tools/list","params":{}}"#;
let resp = server.handle_jsonrpc_line_unwrap(line);
let tools = resp.result.unwrap().get("tools").cloned().unwrap();
let names: Vec<String> = tools
.as_array()
.unwrap()
.iter()
.filter_map(|t| t.get("name").and_then(|v| v.as_str()).map(String::from))
.collect();
const FORBIDDEN: &[&str] = &[
"read_file",
"write_file",
"edit_file",
"execute_command",
"list_directory",
"search_files",
"create_directory",
"delete_file",
];
for forbidden in FORBIDDEN {
assert!(
!names.iter().any(|n| n == forbidden),
"sqz-mcp must not advertise {forbidden} — that name implies \
I/O we cannot perform and shadows the host's real tool. \
See the silent-write bug follow-up to issue #5. \
Tools registered: {names:?}"
);
}
}
#[test]
fn test_default_tools_advertise_compress_tool() {
let (mut server, _dir) = make_server();
let line = r#"{"jsonrpc":"2.0","id":1,"method":"tools/list","params":{}}"#;
let resp = server.handle_jsonrpc_line_unwrap(line);
let tools = resp.result.unwrap().get("tools").cloned().unwrap();
let names: Vec<String> = tools
.as_array()
.unwrap()
.iter()
.filter_map(|t| t.get("name").and_then(|v| v.as_str()).map(String::from))
.collect();
assert!(
names.iter().any(|n| n == "compress"),
"default tools must include `compress`; got {names:?}"
);
}
#[test]
fn test_default_tools_advertise_passthrough_and_expand() {
let (mut server, _dir) = make_server();
let line = r#"{"jsonrpc":"2.0","id":1,"method":"tools/list","params":{}}"#;
let resp = server.handle_jsonrpc_line_unwrap(line);
let tools = resp.result.unwrap().get("tools").cloned().unwrap();
let names: Vec<String> = tools
.as_array()
.unwrap()
.iter()
.filter_map(|t| t.get("name").and_then(|v| v.as_str()).map(String::from))
.collect();
assert!(
names.iter().any(|n| n == "passthrough"),
"passthrough tool must be advertised; got {names:?}"
);
assert!(
names.iter().any(|n| n == "expand"),
"expand tool must be advertised; got {names:?}"
);
}
#[test]
fn test_tool_descriptions_stay_short_and_explain_refs_once() {
let mut total = 0;
for tool in default_tool_definitions() {
let params: Vec<String> = tool.input_schema["properties"]
.as_object()
.map(|props| {
props
.values()
.filter_map(|p| p["description"].as_str().map(str::to_string))
.collect()
})
.unwrap_or_default();
for text in params.iter().chain(std::iter::once(&tool.description)) {
assert!(!text.contains('\u{2014}') && !text.contains('\u{2013}'), "{}: {text}", tool.id);
total += text.len();
}
assert_eq!(
tool.description.contains("§ref:HASH:L40-80§"),
tool.id == "expand",
"{}",
tool.id
);
}
assert!(total < 2000, "tool and parameter descriptions are {total} bytes");
}
#[test]
fn test_passthrough_returns_input_unchanged() {
let (mut server, _dir) = make_server();
let text = "ls -la\ntotal 42\n-rw-r--r-- 1 root root 17 Jan 1 00:00 readme.md\n";
let req = ToolCallRequest {
tool_id: "passthrough".to_string(),
input: serde_json::json!({ "text": text }),
intent: None,
};
let resp = server.handle_tool_call(req).unwrap();
assert_eq!(resp.output, text, "passthrough must return byte-exact input");
assert_eq!(
resp.tokens_original, resp.tokens_compressed,
"passthrough is 1:1 so token counts must match"
);
}
#[test]
fn test_passthrough_falls_back_to_serialising_if_no_text_field() {
let (mut server, _dir) = make_server();
let req = ToolCallRequest {
tool_id: "passthrough".to_string(),
input: serde_json::json!({ "foo": 1, "bar": "baz" }),
intent: None,
};
let resp = server.handle_tool_call(req).unwrap();
assert!(resp.output.contains("foo"));
assert!(resp.output.contains("bar"));
}
#[test]
fn test_expand_tool_returns_not_found_marker_on_miss() {
let (mut server, _dir) = make_server();
let req = ToolCallRequest {
tool_id: "expand".to_string(),
input: serde_json::json!({ "prefix": "deadbeef00000000" }),
intent: None,
};
let resp = server.handle_tool_call(req).unwrap();
assert!(resp.output.contains("hash-not-found"));
assert!(resp.output.contains("deadbeef00000000"));
}
#[test]
fn recall_tool_finds_previously_compressed_content() {
let (mut server, _dir) = make_server();
let text = format!(
"unique-recall-marker deployment rollout failed with status 503\n{}",
"surrounding output line with ordinary words\n".repeat(20)
);
let req = ToolCallRequest {
tool_id: "compress".to_string(),
input: serde_json::json!({ "text": text }),
intent: None,
};
server.handle_tool_call(req).unwrap();
let req = ToolCallRequest {
tool_id: "sqz_recall".to_string(),
input: serde_json::json!({ "query": "rollout 503" }),
intent: None,
};
let resp = server.handle_tool_call(req).unwrap();
assert!(
resp.output.contains("hits=") && resp.output.contains("ref="),
"expected a ranked hit with a ref: {}",
resp.output
);
assert!(resp.output.contains(">>rollout<<"), "snippet markers missing: {}", resp.output);
let ref_prefix = resp
.output
.split("ref=")
.nth(1)
.unwrap()
.chars()
.take_while(|c| c.is_ascii_hexdigit())
.collect::<String>();
let req = ToolCallRequest {
tool_id: "expand".to_string(),
input: serde_json::json!({ "prefix": ref_prefix }),
intent: None,
};
let resp = server.handle_tool_call(req).unwrap();
assert!(
resp.output.contains("unique-recall-marker"),
"expand must recover the indexed original: {}",
resp.output
);
}
#[test]
fn recall_tool_reports_no_matches_cleanly() {
let (mut server, _dir) = make_server();
let req = ToolCallRequest {
tool_id: "sqz_recall".to_string(),
input: serde_json::json!({ "query": "nothing-indexed-yet-zzz" }),
intent: None,
};
let resp = server.handle_tool_call(req).unwrap();
assert!(resp.output.contains("no matches"), "{}", resp.output);
}
#[test]
fn test_default_tools_advertise_recall() {
let (mut server, _dir) = make_server();
let line = r#"{"jsonrpc":"2.0","id":1,"method":"tools/list","params":{}}"#;
let resp = server.handle_jsonrpc_line_unwrap(line);
let tools = resp.result.unwrap().get("tools").cloned().unwrap();
let ids: Vec<String> = tools
.as_array()
.unwrap()
.iter()
.filter_map(|t| t.get("name").and_then(|v| v.as_str()).map(String::from))
.collect();
assert!(ids.iter().any(|n| n == "sqz_recall"), "tools: {ids:?}");
}
#[test]
fn test_expand_tool_strips_ref_token_wrapper() {
let (mut server, _dir) = make_server();
for prefix_input in [
"§ref:deadbeef00000000§",
"ref:deadbeef00000000",
"deadbeef00000000",
" deadbeef00000000 ",
] {
let req = ToolCallRequest {
tool_id: "expand".to_string(),
input: serde_json::json!({ "prefix": prefix_input }),
intent: None,
};
let resp = server.handle_tool_call(req).unwrap();
assert!(
resp.output.contains("deadbeef00000000"),
"input {prefix_input:?} did not yield expected prefix in output: {}",
resp.output
);
}
}
#[test]
fn test_notification_returns_none() {
let (mut server, _dir) = make_server();
let line = r#"{"jsonrpc":"2.0","method":"notifications/initialized","params":{}}"#;
assert!(server.handle_jsonrpc_line(line).is_none());
}
#[test]
fn test_unknown_notification_returns_none() {
let (mut server, _dir) = make_server();
let line = r#"{"jsonrpc":"2.0","method":"some/unknown/notif"}"#;
assert!(server.handle_jsonrpc_line(line).is_none());
}
#[test]
fn test_request_still_responds() {
let (mut server, _dir) = make_server();
let line = r#"{"jsonrpc":"2.0","id":1,"method":"initialize","params":{}}"#;
assert!(server.handle_jsonrpc_line(line).is_some());
}
#[test]
fn test_sqz_file_tools_are_advertised() {
let (mut server, _dir) = make_server();
let line = r#"{"jsonrpc":"2.0","id":1,"method":"tools/list","params":{}}"#;
let resp = server.handle_jsonrpc_line_unwrap(line);
let tools = resp.result.unwrap().get("tools").cloned().unwrap();
let names: Vec<String> = tools
.as_array()
.unwrap()
.iter()
.filter_map(|t| t.get("name").and_then(|v| v.as_str()).map(String::from))
.collect();
for expected in ["sqz_read_file", "sqz_list_dir", "sqz_grep"] {
assert!(
names.iter().any(|n| n == expected),
"{expected} must be advertised; got {names:?}"
);
}
}
#[test]
fn test_sqz_file_tools_have_sqz_prefix() {
let (mut server, _dir) = make_server();
let line = r#"{"jsonrpc":"2.0","id":1,"method":"tools/list","params":{}}"#;
let resp = server.handle_jsonrpc_line_unwrap(line);
let tools = resp.result.unwrap().get("tools").cloned().unwrap();
for tool in tools.as_array().unwrap() {
let name = tool.get("name").and_then(|v| v.as_str()).unwrap_or("");
if matches!(
name,
"read_file" | "grep" | "list_dir" | "list_directory"
| "search_files" | "write_file" | "delete_file"
| "edit_file" | "execute_command" | "create_directory"
) {
panic!(
"tool `{name}` shadows a host-native tool. \
I/O tools must be prefixed `sqz_` — see issue #5 \
follow-up."
);
}
}
}
#[test]
fn test_sqz_read_file_reads_real_file() {
let (mut server, dir) = make_server();
let file_path = dir.path().join("read_me.txt");
let original = "hello from sqz_read_file\nline two\nline three\n";
std::fs::write(&file_path, original).expect("write test file");
let req = ToolCallRequest {
tool_id: "sqz_read_file".to_string(),
input: serde_json::json!({ "path": file_path.to_string_lossy() }),
intent: None,
};
let resp = server.handle_tool_call(req).expect("read should succeed");
assert!(resp.output.contains("[sqz_read_file"));
assert!(resp.output.contains(&format!("size={}", original.len())));
assert!(
resp.output.contains("line two") || resp.output.contains("§ref:"),
"content must survive (possibly as a dedup ref); got: {}",
resp.output
);
}
#[test]
fn test_sqz_read_file_reports_missing_file_clearly() {
let (mut server, _dir) = make_server();
let req = ToolCallRequest {
tool_id: "sqz_read_file".to_string(),
input: serde_json::json!({ "path": "/nonexistent/path/xyz.txt" }),
intent: None,
};
let result = server.handle_tool_call(req);
assert!(result.is_err());
let err = result.unwrap_err().to_string();
assert!(err.contains("sqz_read_file"), "error should name the tool: {err}");
assert!(err.contains("xyz.txt"), "error should name the path: {err}");
}
fn numbered_lines(n: usize) -> String {
(1..=n).map(|i| format!("line {i:06} of the big file\n")).collect()
}
fn read(server: &mut McpServer, input: serde_json::Value) -> String {
server
.handle_tool_call(ToolCallRequest {
tool_id: "sqz_read_file".to_string(),
input,
intent: None,
})
.expect("read should succeed")
.output
}
#[test]
fn test_sqz_read_file_respects_max_bytes() {
let (mut server, dir) = make_server();
let file_path = dir.path().join("big.txt");
let content = numbered_lines(100);
std::fs::write(&file_path, &content).expect("write big file");
let line_len = "line 000001 of the big file\n".len();
let out = read(&mut server, serde_json::json!({ "path": file_path.to_string_lossy(), "max_bytes": line_len * 3 + 5 }));
let (header, body) = out.split_once('\n').unwrap();
assert!(
header.contains(&format!(" lines=1-3 of 100 truncated_to={} continue_with_offset=4]", line_len * 3 + 5)),
"{header}"
);
assert_eq!(body, line_range(&content, 1, 3));
let next = read(&mut server, serde_json::json!({ "path": file_path.to_string_lossy(), "offset": 4, "limit": 2 }));
assert!(next.ends_with("line 000004 of the big file\nline 000005 of the big file\n"), "{next}");
}
#[test]
fn test_sqz_read_file_default_cap_and_ranged_read_past_it() {
let (mut server, dir) = make_server();
let file_path = dir.path().join("huge.log");
let content = numbered_lines(20_000);
assert!(content.len() > READ_DEFAULT_MAX_BYTES);
std::fs::write(&file_path, &content).unwrap();
let out = read(&mut server, serde_json::json!({ "path": file_path.to_string_lossy() }));
let (header, body) = out.split_once('\n').unwrap();
assert!(header.contains(&format!("truncated_to={READ_DEFAULT_MAX_BYTES}")), "{header}");
assert!(body.len() <= READ_DEFAULT_MAX_BYTES && body.ends_with('\n'));
let shown = body.lines().count();
assert!(header.contains(&format!(" lines=1-{shown} of 20000 ")), "{header}");
assert!(header.contains(&format!("continue_with_offset={}", shown + 1)), "{header}");
let deep = read(&mut server, serde_json::json!({ "path": file_path.to_string_lossy(), "offset": 19_999 }));
assert!(deep.contains(" lines=19999-20000 of 20000]"), "{deep}");
assert!(!deep.contains("truncated_to"), "{deep}");
assert!(deep.ends_with("line 019999 of the big file\nline 020000 of the big file\n"));
let all = read(&mut server, serde_json::json!({ "path": file_path.to_string_lossy(), "max_bytes": 0 }));
assert!(!all.contains("truncated_to"));
}
#[test]
fn test_sqz_read_file_reports_a_cut_line() {
let (mut server, dir) = make_server();
let file_path = dir.path().join("min.js");
std::fs::write(&file_path, format!("{}\nsecond line\n", "é".repeat(1000))).unwrap();
let out = read(&mut server, serde_json::json!({ "path": file_path.to_string_lossy(), "max_bytes": 101 }));
let (header, body) = out.split_once('\n').unwrap();
assert!(header.contains(" lines=1-1 of 2 truncated_to=101 line_cut=1 continue_with_offset=2]"), "{header}");
assert_eq!(body, "é".repeat(50), "cut on a char boundary");
}
#[test]
fn test_sqz_read_file_refuses_files_over_the_hard_limit() {
let (mut server, dir) = make_server();
let file_path = dir.path().join("sparse.bin");
let f = std::fs::File::create(&file_path).unwrap();
f.set_len(READ_HARD_LIMIT_BYTES + 1).unwrap();
let err = server
.handle_tool_call(ToolCallRequest {
tool_id: "sqz_read_file".to_string(),
input: serde_json::json!({ "path": file_path.to_string_lossy() }),
intent: None,
})
.unwrap_err()
.to_string();
assert!(err.contains("too large") && err.contains("sparse.bin"), "{err}");
}
#[test]
fn test_sqz_read_file_dedup_fires_on_repeat_read() {
let (mut server, dir) = make_server();
let file_path = dir.path().join("dedup_me.txt");
let content = "sqz dedup regression — issue #12 follow-up\n\
this file should collapse to a 13-token ref on \
the second read, proving the cache is wired up.\n"
.repeat(10);
std::fs::write(&file_path, &content).expect("write test file");
let req = || ToolCallRequest {
tool_id: "sqz_read_file".to_string(),
input: serde_json::json!({ "path": file_path.to_string_lossy() }),
intent: None,
};
let first = server.handle_tool_call(req()).expect("first read");
assert!(
!first.output.contains("§ref:"),
"first read must emit full content, not a ref; got: {}",
first.output
);
let second = server.handle_tool_call(req()).expect("second read");
assert!(
second.output.contains("§ref:"),
"second read must emit a §ref:HASH§ token (the whole point \
of the cache). If this fails, handle_sqz_read_file is \
calling engine.compress() instead of compress_with_cache. \
Got: {}",
second.output
);
assert!(
second.tokens_compressed < 30,
"dedup ref should be ~13 tokens, got {}",
second.tokens_compressed
);
}
#[test]
fn test_sqz_read_file_never_stores_credential_mentions() {
let (mut server, dir) = make_server();
let file_path = dir.path().join("app.log");
let mut content: String = (0..40)
.map(|i| format!("2026-10-05T12:00:{i:02}Z INFO GET /healthz 200 1ms\n"))
.collect();
content.push_str("2026-10-05T12:30:00Z DEBUG db config host=db.internal user=app password=hunter2\n");
std::fs::write(&file_path, &content).expect("write test file");
for run in ["first", "second"] {
let resp = server
.handle_tool_call(ToolCallRequest {
tool_id: "sqz_read_file".to_string(),
input: serde_json::json!({ "path": file_path.to_string_lossy() }),
intent: None,
})
.unwrap();
assert!(resp.output.contains(&content), "{run} read: {}", resp.output);
}
let stored = server.engine.cache_manager().check_dedup_with_meta(content.as_bytes()).unwrap();
assert!(stored.is_none());
}
fn python_module(fns: usize) -> String {
(1..=fns)
.map(|i| format!("def handler_{i:03}(request):\n token = request.headers.get('Authorization')\n return verify(token, scope='handler_{i:03}')\n\n"))
.collect()
}
#[test]
fn test_sqz_read_file_ranged_read_returns_lines_with_header() {
let (mut server, dir) = make_server();
let file_path = dir.path().join("auth.py");
let content = python_module(50);
std::fs::write(&file_path, &content).unwrap();
let resp = server
.handle_tool_call(ToolCallRequest {
tool_id: "sqz_read_file".to_string(),
input: serde_json::json!({ "path": file_path.to_string_lossy(), "offset": 5, "limit": 4 }),
intent: None,
})
.expect("ranged read");
let (header, body) = resp.output.split_once('\n').unwrap();
assert!(header.ends_with(" lines=5-8 of 200]"), "{header}");
assert_eq!(body, line_range(&content, 5, 8));
assert_eq!(body.lines().count(), 4);
assert!(body.starts_with("def handler_002"), "{body}");
assert!(body.ends_with("\n\n"), "line 8 is the blank line after handler_002");
let tail = server
.handle_tool_call(ToolCallRequest {
tool_id: "sqz_read_file".to_string(),
input: serde_json::json!({ "path": file_path.to_string_lossy(), "offset": 197 }),
intent: None,
})
.unwrap();
assert!(tail.output.contains(" lines=197-200 of 200]"), "{}", tail.output);
let head = server
.handle_tool_call(ToolCallRequest {
tool_id: "sqz_read_file".to_string(),
input: serde_json::json!({ "path": file_path.to_string_lossy(), "limit": 3 }),
intent: None,
})
.unwrap();
assert!(head.output.contains(" lines=1-3 of 200]"), "{}", head.output);
assert!(head.output.ends_with(" return verify(token, scope='handler_001')\n"), "{}", head.output);
let past_end = server.handle_tool_call(ToolCallRequest {
tool_id: "sqz_read_file".to_string(),
input: serde_json::json!({ "path": file_path.to_string_lossy(), "offset": 500 }),
intent: None,
});
assert!(past_end.is_err());
}
#[test]
fn test_sqz_read_file_slice_after_full_read_returns_line_range_ref() {
let (mut server, dir) = make_server();
let file_path = dir.path().join("auth.py");
let content = python_module(50);
std::fs::write(&file_path, &content).unwrap();
let hash = sqz_engine::CacheManager::sha256_hex(content.as_bytes());
let full = server
.handle_tool_call(ToolCallRequest {
tool_id: "sqz_read_file".to_string(),
input: serde_json::json!({ "path": file_path.to_string_lossy() }),
intent: None,
})
.unwrap();
assert!(!full.output.contains("§ref:"));
let slice = server
.handle_tool_call(ToolCallRequest {
tool_id: "sqz_read_file".to_string(),
input: serde_json::json!({ "path": file_path.to_string_lossy(), "offset": 41, "limit": 40 }),
intent: None,
})
.unwrap();
let expected = format!("§ref:{}:L41-80§", &hash[..16]);
assert!(slice.output.ends_with(&expected), "{}", slice.output);
assert!(slice.output.contains(" lines=41-80 of 200]"));
assert!(slice.tokens_compressed < 25 && slice.tokens_original > 200, "{} / {}", slice.tokens_compressed, slice.tokens_original);
let expanded = server
.handle_tool_call(ToolCallRequest {
tool_id: "expand".to_string(),
input: serde_json::json!({ "prefix": expected }),
intent: None,
})
.unwrap();
let (header, body) = expanded.output.split_once('\n').unwrap();
assert!(header.contains("lines=41-80"), "{header}");
assert_eq!(body, line_range(&content, 41, 80));
let tiny = server
.handle_tool_call(ToolCallRequest {
tool_id: "sqz_read_file".to_string(),
input: serde_json::json!({ "path": file_path.to_string_lossy(), "offset": 1, "limit": 2 }),
intent: None,
})
.unwrap();
assert!(!tiny.output.contains("§ref:"), "{}", tiny.output);
assert!(tiny.output.contains("def handler_001"));
}
#[test]
fn test_sqz_read_file_delta_after_an_edit_numbers_lines_from_one() {
let (mut server, dir) = make_server();
let file_path = dir.path().join("rows.txt");
let rows: Vec<String> = (1..=60).map(|i| format!("row {i:02}\n")).collect();
std::fs::write(&file_path, rows.concat()).unwrap();
let path = file_path.to_string_lossy().to_string();
read(&mut server, serde_json::json!({ "path": path }));
let hash = sqz_engine::CacheManager::sha256_hex(rows.concat().as_bytes());
std::fs::write(&file_path, [&rows[..2], &rows[3..]].concat().concat()).unwrap();
let out = read(&mut server, serde_json::json!({ "path": path }));
let expected = format!(
"§delta:{}§\n @@ skip 1 unchanged lines @@\n row 02\n-[1 lines removed at L3]\n row 04\n @@ skip 56 unchanged lines @@",
&hash[..16]
);
assert!(out.ends_with(&expected), "{out}");
}
#[test]
fn test_sqz_read_file_ranged_reread_is_never_a_delta() {
let (mut server, dir) = make_server();
let file_path = dir.path().join("lines.txt");
let content: String = (1..=500)
.map(|i| format!("line {i:04} the quick brown fox needle={}\n", i % 7 == 0))
.collect();
std::fs::write(&file_path, &content).unwrap();
let path = file_path.to_string_lossy().to_string();
read(&mut server, serde_json::json!({ "path": path, "offset": 1, "limit": 50 }));
let next = read(&mut server, serde_json::json!({ "path": path, "offset": 11, "limit": 50 }));
let (header, body) = next.split_once('\n').unwrap();
assert!(header.ends_with(" lines=11-60 of 500]"), "{header}");
assert_eq!(body, line_range(&content, 11, 60));
}
#[test]
fn test_content_smaller_than_a_ref_is_never_a_ref_or_delta() {
let (mut server, dir) = make_server();
let root = project_dir(&dir);
std::fs::write(root.join("empty.txt"), "").unwrap();
std::fs::write(root.join("ok.txt"), "\x1b[32mok\x1b[0m\n").unwrap();
let compress = |server: &mut McpServer, text: &str| {
server
.handle_tool_call(ToolCallRequest {
tool_id: "compress".to_string(),
input: serde_json::json!({ "text": text }),
intent: None,
})
.unwrap()
.output
};
for run in ["first", "second"] {
assert_eq!(compress(&mut server, ""), "", "{run}");
let empty = read(&mut server, serde_json::json!({ "path": root.join("empty.txt").to_string_lossy() }));
assert!(empty.ends_with("size=0]\n"), "{run}: {empty}");
let ok = read(&mut server, serde_json::json!({ "path": root.join("ok.txt").to_string_lossy() }));
assert!(ok.ends_with("]\nok\n"), "{run}: {ok}");
let none = grep(&mut server, serde_json::json!({ "pattern": "no such text", "path": root.to_string_lossy() }));
assert!(none.ends_with("matches=0]\n"), "{run}: {none}");
}
}
#[test]
fn test_content_smaller_than_a_ref_is_still_stored() {
let (mut server, dir) = make_server();
let root = project_dir(&dir);
std::fs::write(root.join("ok.txt"), "zebra build ok\n").unwrap();
let call = |server: &mut McpServer, tool: &str, input: serde_json::Value| {
server
.handle_tool_call(ToolCallRequest { tool_id: tool.to_string(), input, intent: None })
.unwrap()
.output
};
read(&mut server, serde_json::json!({ "path": root.join("ok.txt").to_string_lossy() }));
call(&mut server, "compress", serde_json::json!({ "text": "zebra crossing seven" }));
let recalled = call(&mut server, "sqz_recall", serde_json::json!({ "query": "zebra" }));
for text in ["zebra build ok\n", "zebra crossing seven"] {
let hash = sqz_engine::CacheManager::sha256_hex(text.as_bytes());
assert!(recalled.contains(&format!("ref={}", &hash[..16])), "{recalled}");
let expanded = call(&mut server, "expand", serde_json::json!({ "prefix": &hash[..16] }));
assert_eq!(expanded, format!("[sqz:expand hash={hash}]\n{text}"));
}
}
#[test]
fn test_read_tools_count_tokens_with_one_tokenizer() {
let (mut server, dir) = make_server();
for i in 0..40 {
std::fs::write(dir.path().join(format!("f{i}.rs")), "fn needle_fn() -> Result<(), Error> { Ok(()) }\n").unwrap();
}
let resp = server
.handle_tool_call(ToolCallRequest {
tool_id: "sqz_grep".to_string(),
input: serde_json::json!({ "pattern": "needle_fn", "path": dir.path().to_string_lossy() }),
intent: None,
})
.unwrap();
assert!(
resp.tokens_compressed <= resp.tokens_original,
"verbatim result must never log as negative savings: {} -> {}",
resp.tokens_original,
resp.tokens_compressed
);
}
#[test]
fn test_sqz_list_dir_lists_directory() {
let (mut server, dir) = make_server();
std::fs::write(dir.path().join("a.txt"), "").unwrap();
std::fs::write(dir.path().join("b.rs"), "").unwrap();
std::fs::write(dir.path().join(".secret"), "").unwrap();
std::fs::create_dir_all(dir.path().join("node_modules/foo")).unwrap();
std::fs::write(dir.path().join("node_modules/foo/pkg.json"), "").unwrap();
let req = ToolCallRequest {
tool_id: "sqz_list_dir".to_string(),
input: serde_json::json!({ "path": dir.path().to_string_lossy() }),
intent: None,
};
let resp = server.handle_tool_call(req).expect("list should succeed");
assert!(resp.output.contains("a.txt"));
assert!(resp.output.contains("b.rs"));
assert!(
!resp.output.contains(".secret"),
"hidden files must be skipped; got: {}",
resp.output
);
assert!(
!resp.output.contains("node_modules"),
"node_modules must be skipped; got: {}",
resp.output
);
}
fn list(server: &mut McpServer, input: serde_json::Value) -> String {
server
.handle_tool_call(ToolCallRequest {
tool_id: "sqz_list_dir".to_string(),
input,
intent: None,
})
.expect("list should succeed")
.output
}
fn project_dir(dir: &TempDir) -> std::path::PathBuf {
let root = dir.path().join("proj");
std::fs::create_dir_all(&root).unwrap();
root
}
#[test]
fn test_sqz_list_dir_depth_one_is_immediate_children() {
let (mut server, dir) = make_server();
let root = project_dir(&dir);
std::fs::create_dir_all(root.join("src/nested")).unwrap();
std::fs::write(root.join("src/lib.rs"), "").unwrap();
std::fs::write(root.join("src/nested/deep.rs"), "").unwrap();
std::fs::write(root.join("README.md"), "").unwrap();
let path = root.to_string_lossy().to_string();
let one = list(&mut server, serde_json::json!({ "path": path }));
assert!(one.contains("entries=2]"), "{one}");
assert!(one.contains("d src") && one.contains("f README.md"), "{one}");
assert!(!one.contains("lib.rs"), "default depth lists immediate children only: {one}");
let two = list(&mut server, serde_json::json!({ "path": path, "max_depth": 2 }));
assert!(two.contains("entries=4]"), "{two}");
let nested = format!("d {}", Path::new("src").join("nested").display());
assert!(two.contains("lib.rs") && two.contains(&nested), "{two}");
assert!(!two.contains("deep.rs"), "{two}");
let zero = list(&mut server, serde_json::json!({ "path": path, "max_depth": 0 }));
assert!(zero.contains("entries=2]"), "{zero}");
}
#[test]
fn test_sqz_list_dir_max_entries() {
let (mut server, dir) = make_server();
let root = project_dir(&dir);
for i in 0..12 {
std::fs::write(root.join(format!("f{i:02}.txt")), "").unwrap();
}
let path = root.to_string_lossy().to_string();
let capped = list(&mut server, serde_json::json!({ "path": path, "max_entries": 5 }));
assert!(capped.contains("entries=5 stopped_at_max_entries=5]"), "{capped}");
assert!(capped.contains("f04.txt") && !capped.contains("f05.txt"), "{capped}");
let exact = list(&mut server, serde_json::json!({ "path": path, "max_entries": 12 }));
assert!(exact.contains("entries=12]"), "{exact}");
std::fs::create_dir(root.join("vendor")).unwrap();
let skipped = list(&mut server, serde_json::json!({ "path": path, "max_entries": 12 }));
assert!(skipped.contains("entries=12]"), "{skipped}");
}
#[test]
fn test_sqz_grep_finds_literal_matches() {
let (mut server, dir) = make_server();
std::fs::write(
dir.path().join("code.rs"),
"fn main() {\n // TODO: refactor this\n println!(\"hello\");\n}\n",
)
.unwrap();
std::fs::write(
dir.path().join("other.rs"),
"fn nothing_to_do() {\n // just a comment\n}\n",
)
.unwrap();
let req = ToolCallRequest {
tool_id: "sqz_grep".to_string(),
input: serde_json::json!({
"pattern": "TODO",
"path": dir.path().to_string_lossy(),
}),
intent: None,
};
let resp = server.handle_tool_call(req).expect("grep should succeed");
assert!(resp.output.contains("matches=1"), "header must include match count; got: {}", resp.output);
assert!(
resp.output.contains("TODO"),
"match content must be present; got: {}",
resp.output
);
assert!(
resp.output.contains("code.rs:2:"),
"output must be grep-style path:lineno:text; got: {}",
resp.output
);
}
#[test]
fn test_sqz_grep_regex_mode() {
let (mut server, dir) = make_server();
std::fs::write(
dir.path().join("t.rs"),
"fn my_test() {}\nfn your_test() {}\nfn not_matching() {}\n",
)
.unwrap();
let req = ToolCallRequest {
tool_id: "sqz_grep".to_string(),
input: serde_json::json!({
"pattern": r"fn \w+_test",
"path": dir.path().to_string_lossy(),
"regex": true,
}),
intent: None,
};
let resp = server.handle_tool_call(req).expect("grep should succeed");
assert!(resp.output.contains("matches=2"));
}
#[test]
fn test_sqz_grep_invalid_regex_errors_cleanly() {
let (mut server, dir) = make_server();
let req = ToolCallRequest {
tool_id: "sqz_grep".to_string(),
input: serde_json::json!({
"pattern": "fn (unclosed",
"path": dir.path().to_string_lossy(),
"regex": true,
}),
intent: None,
};
let result = server.handle_tool_call(req);
assert!(result.is_err());
let err = result.unwrap_err().to_string();
assert!(err.contains("sqz_grep"));
assert!(err.contains("invalid regex"));
}
#[test]
fn test_sqz_grep_caps_at_max_matches() {
let (mut server, dir) = make_server();
let content: String = (0..100).map(|i| format!("line {i}: match\n")).collect();
std::fs::write(dir.path().join("many.txt"), content).unwrap();
let req = ToolCallRequest {
tool_id: "sqz_grep".to_string(),
input: serde_json::json!({
"pattern": "match",
"path": dir.path().to_string_lossy(),
"max_matches": 10,
}),
intent: None,
};
let resp = server.handle_tool_call(req).expect("grep should succeed");
assert!(resp.output.contains("matches=10 max_matches_reached=10]"),
"should stop at max_matches; got: {}", resp.output);
}
fn grep(server: &mut McpServer, input: serde_json::Value) -> String {
server
.handle_tool_call(ToolCallRequest {
tool_id: "sqz_grep".to_string(),
input,
intent: None,
})
.expect("grep should succeed")
.output
}
#[test]
fn test_sqz_grep_clips_long_lines_around_the_hit() {
let (mut server, dir) = make_server();
let root = project_dir(&dir);
let line = format!("{}needle_here{}", "é".repeat(3000), "ü".repeat(3000));
std::fs::write(root.join("bundle.min.js"), format!("{line}\nshort needle_here line\n")).unwrap();
let input = serde_json::json!({ "pattern": "needle_here", "path": root.to_string_lossy() });
let out = grep(&mut server, input.clone());
let (header, body) = out.split_once('\n').unwrap();
assert!(header.contains("matches=2 long_lines_clipped=1 (max_line_chars=400)]"), "{header}");
let first = body.lines().next().unwrap();
let shown = first.split_once("bundle.min.js:1:").map(|(_, s)| s).unwrap();
assert_eq!(
shown,
format!("…{}needle_here{}… [line clipped: 6011 chars]", "é".repeat(194), "ü".repeat(195))
);
assert!(body.contains("bundle.min.js:2:short needle_here line"), "{body}");
let (mut fresh, _store) = make_server();
let mut uncapped = input;
uncapped["max_line_chars"] = 0.into();
uncapped["max_bytes"] = 0.into();
let whole = grep(&mut fresh, uncapped);
assert!(!whole.contains("clipped"), "{}", whole.chars().take(200).collect::<String>());
assert!(whole.contains(&line));
}
#[test]
fn test_clip_line_windows() {
let line: String = ('a'..='z').cycle().take(1000).collect();
let start = clip_line(&line, (0, 3), 100).unwrap();
assert!(start.starts_with("abc") && start.contains("… [line clipped: 1000 chars]"), "{start}");
let end = clip_line(&line, (997, 1000), 100).unwrap();
assert!(end.starts_with('…') && end.ends_with(&format!("{} [line clipped: 1000 chars]", &line[900..])), "{end}");
let long = clip_line(&line, (500, 800), 100).unwrap();
assert!(long.starts_with(&format!("…{}", &line[500..600])), "{long}");
assert!(clip_line("short", (0, 1), 100).is_none());
}
#[test]
fn test_sqz_grep_caps_total_bytes() {
let (mut server, dir) = make_server();
let root = project_dir(&dir);
let content: String = (0..2000).map(|i| format!("{i:05} match {}\n", "x".repeat(200))).collect();
std::fs::write(root.join("wide.txt"), content).unwrap();
let out = grep(&mut server, serde_json::json!({
"pattern": "match",
"path": root.to_string_lossy(),
"max_matches": 10_000
}));
let (header, body) = out.split_once('\n').unwrap();
assert!(header.contains(&format!("stopped_at_max_bytes={GREP_DEFAULT_MAX_BYTES}")), "{header}");
assert!(!header.contains("max_matches_reached"), "{header}");
assert!(body.len() <= GREP_DEFAULT_MAX_BYTES, "{}", body.len());
assert!(body.lines().count() > 100, "{}", body.lines().count());
}
}