Skip to main content

vtcode_core/tools/defuddle/
mod.rs

1//! Defuddle-backed fetch: `https://defuddle.md/{link}` -> clean markdown.
2//!
3//! Defuddle is a third-party hosted service that takes a URL and returns
4//! LLM-friendly markdown for the page. It is a polite convenience for the
5//! rare case where `web_fetch` returns a payload the agent would rather not
6//! parse itself (heavy JS, paywalled HTML, raw RSS, etc.).
7//!
8//! Because the service is rate-limited, this tool is hard-capped at one call
9//! per `DefuddleTool` instance. A new instance per session is the recommended
10//! pattern; the registry constructs the tool once per agent, so the cap
11//! effectively means "once per session".
12//!
13//! The response is returned inline (no temp file). When the cap is hit, the
14//! tool returns a structured JSON error that points the agent back to
15//! `web_fetch` so it does not loop.
16
17use super::traits::Tool;
18use crate::config::constants::tools;
19use crate::tools::web_fetch::classify_helpers::extract_http_status;
20use crate::tools::web_fetch::is_private_host;
21use anyhow::{Context, Result, anyhow};
22use async_trait::async_trait;
23use serde::Deserialize;
24use serde_json::{Value, json};
25use std::sync::Arc;
26use std::sync::atomic::{AtomicUsize, Ordering};
27use std::time::Duration;
28use url::Url;
29
30const DEFAULT_TIMEOUT_SECS: u64 = 30;
31const MAX_TIMEOUT_SECS: u64 = 60;
32const MAX_BYTES: usize = 256 * 1024;
33const DEFUDDLE_BASE_URL: &str = "https://defuddle.md/";
34
35pub(crate) const DEFUDDLE_FETCH_DESCRIPTION: &str = "Fetch a REMOTE web page (http:// or https:// URLs ONLY) through the defuddle.md markdown extraction service and return the cleaned markdown inline. DO NOT use for local files: inspect local paths with exec_command and readonly shell commands such as sed, rg, ls, or find. Use this sparingly: the hosted service is rate-limited, so this tool can be called at most ONCE per session. Accepts: { url: string (must start with http:// or https://), max_bytes?: number }. Returns { url, markdown, bytes, used_this_session, session_cap } or a structured error if the cap has been hit.";
36
37#[derive(Debug, Deserialize)]
38struct DefuddleArgs {
39    #[serde(default)]
40    url: Option<String>,
41    #[serde(default)]
42    max_bytes: Option<usize>,
43}
44
45#[derive(Clone, Default)]
46pub struct DefuddleTool {
47    /// How many calls have been issued against this instance. Hard-capped at
48    /// `SESSION_CAP` to stay well under the third-party service's quota.
49    uses: Arc<AtomicUsize>,
50}
51
52const SESSION_CAP: usize = 1;
53
54impl DefuddleTool {
55    pub fn new() -> Self {
56        Self::default()
57    }
58
59    /// Reset the per-instance counter. Used by tests; production code should
60    /// not call this — a new `DefuddleTool` is the way to start a new budget.
61    pub fn reset(&self) {
62        self.uses.store(0, Ordering::SeqCst);
63    }
64
65    fn current_uses(&self) -> usize {
66        self.uses.load(Ordering::SeqCst)
67    }
68
69    fn try_consume(&self) -> bool {
70        // Compare-and-swap loop so concurrent callers cannot both succeed
71        // once the cap is hit.
72        loop {
73            let current = self.uses.load(Ordering::SeqCst);
74            if current >= SESSION_CAP {
75                return false;
76            }
77            if self
78                .uses
79                .compare_exchange(current, current + 1, Ordering::SeqCst, Ordering::SeqCst)
80                .is_ok()
81            {
82                return true;
83            }
84        }
85    }
86
87    async fn run(&self, raw_args: Value) -> Result<Value> {
88        let args: DefuddleArgs = serde_json::from_value(raw_args)
89            .context("Invalid arguments for defuddle_fetch. Provide a 'url' string.")?;
90
91        let url = args
92            .url
93            .map(|u| u.trim().to_string())
94            .filter(|u| !u.is_empty())
95            .ok_or_else(|| anyhow!("defuddle_fetch requires a non-empty 'url'"))?;
96
97        // Reject anything that is not an HTTP/HTTPS URL before URL parsing.
98        // This prevents the LLM from accidentally using defuddle_fetch for
99        // local file reads, bare paths, or other non-web URLs.
100        let lower = url.to_ascii_lowercase();
101        if !lower.starts_with("http://") && !lower.starts_with("https://") {
102            return Err(anyhow!(
103                "defuddle_fetch only accepts http:// or https:// URLs for web content extraction. \
104                For local file reads, use exec_command with readonly shell inspection commands instead. \
105                Got: {url}"
106            ));
107        }
108
109        validate_target_url(&url)?;
110
111        if !self.try_consume() {
112            return Ok(cap_reached_response(&url));
113        }
114
115        let max_bytes = args.max_bytes.unwrap_or(MAX_BYTES).min(MAX_BYTES);
116        let timeout_secs = DEFAULT_TIMEOUT_SECS.min(MAX_TIMEOUT_SECS);
117
118        let (body, truncated) = match fetch_markdown(&url, timeout_secs, max_bytes).await {
119            Ok(t) => t,
120            Err(e) => {
121                return Ok(defuddle_fetch_error_response(&url, max_bytes, timeout_secs, &e));
122            }
123        };
124
125        let bytes = body.len();
126        Ok(json!({
127            "url": url,
128            "markdown": body,
129            "bytes": bytes,
130            "truncated": truncated,
131            "used_this_session": self.current_uses(),
132            "session_cap": SESSION_CAP,
133        }))
134    }
135}
136
137fn defuddle_fetch_error_response(url: &str, max_bytes: usize, timeout_secs: u64, err: &anyhow::Error) -> Value {
138    let message = err.to_string();
139    let lower = message.to_lowercase();
140    // Timeout / network-class failures take priority over HTTP status:
141    // a slow upstream that returns a 5xx after the body stalls will
142    // carry a "timed out" prefix, and the right hint in that case is
143    // "retry later / use web_fetch" — not "the service is down".
144    let (error_type, next_action) = if lower.contains("timeout") || lower.contains("timed out") {
145        (
146            "network_error",
147            "defuddle.md timed out. The upstream service may be slow. Use web_fetch on a known URL as a fallback.",
148        )
149    } else if lower.contains("dns") || lower.contains("connection refused") {
150        (
151            "network_error",
152            "defuddle.md is unreachable from this network. Use web_fetch on a known URL as a fallback.",
153        )
154    } else if let Some(status) = extract_http_status(&message) {
155        let action = match status {
156            403 | 429 => {
157                "defuddle.md rate-limited or rejected the request, so a retry will hit the same limit; the session cap is also exhausted. Use web_fetch on a known URL as a fallback."
158            }
159            404 => {
160                "defuddle.md returned 404. The upstream service path may have changed; use web_fetch directly instead."
161            }
162            500..=599 => "defuddle.md is having a server issue. Use web_fetch on a known URL as a fallback.",
163            _ => "defuddle.md returned an unexpected status. Use web_fetch on a known URL as a fallback.",
164        };
165        ("http_error", action)
166    } else {
167        ("unknown_error", "defuddle.md request failed. Use web_fetch on a known URL as a fallback.")
168    };
169    json!({
170        "error": format!("defuddle_fetch failed for '{}': {}", url, message),
171        "url": url,
172        "max_bytes": max_bytes,
173        "timeout_secs": timeout_secs,
174        "error_type": error_type,
175        "next_action": next_action,
176        "used_this_session": 1,
177        "session_cap": SESSION_CAP,
178    })
179}
180
181fn cap_reached_response(url: &str) -> Value {
182    json!({
183        "error": "defuddle_fetch session cap reached",
184        "url": url,
185        "session_cap": SESSION_CAP,
186        "next_action": "defuddle_fetch can be called at most once per session because the defuddle.md service is rate-limited. Use web_fetch (which goes through vtcode's safe HTTP path) for additional pages in this session."
187    })
188}
189
190/// Validate that the requested URL is something we'd want to point the
191/// public defuddle.md service at. We do not run the defuddle.md request
192/// itself until after this check, so an obviously-bad URL never leaves the
193/// process.
194fn validate_target_url(url: &str) -> Result<()> {
195    let parsed = Url::parse(url).context("defuddle_fetch: invalid url")?;
196    match parsed.scheme() {
197        "http" | "https" => {}
198        other => {
199            return Err(anyhow!("defuddle_fetch: refusing to fetch {other}:// URL (only http/https are allowed)"));
200        }
201    }
202    let host = parsed
203        .host_str()
204        .filter(|h| !h.is_empty())
205        .ok_or_else(|| anyhow!("defuddle_fetch: refusing to fetch a URL with no host"))?;
206    // Block private / loopback / link-local / broadcast / multicast
207    // hosts so a malicious URL can't turn defuddle into an SSRF relay.
208    // Mirrors the `web_fetch` check via the shared `is_private_host` helper.
209    if is_private_host(host) {
210        return Err(anyhow!("defuddle_fetch: refusing to fetch private/local host '{host}'"));
211    }
212    Ok(())
213}
214
215async fn fetch_markdown(url: &str, timeout_secs: u64, max_bytes: usize) -> Result<(String, bool)> {
216    // Per the user's note: `curl defuddle.md/{link}`. We must NOT use
217    // `Url::join` here because the Rust URL spec treats absolute URLs as
218    // absolute on join, which would mean the request goes straight to the
219    // target host and defuddle.md is bypassed entirely. Instead, encode the
220    // input URL as a single path segment and append it to the base URL.
221    //
222    // SECURITY: this is the only place the request URL is constructed. A
223    // future refactor that "cleans up" the string concatenation in favor
224    // of `Url::join` or any other URL builder will silently turn the
225    // tool into an SSRF relay. The regression test
226    // `request_url_targets_defuddle_host_not_target_host` guards this.
227    let encoded = percent_encode_path(url);
228    let target = format!("{DEFUDDLE_BASE_URL}{encoded}");
229
230    let client = reqwest::Client::builder()
231        .timeout(Duration::from_secs(timeout_secs))
232        // Do NOT follow redirects from the upstream service. Defuddle is a
233        // thin proxy; following its redirects could leak into SSRF-prone
234        // behavior, and we already validated the input URL above.
235        .redirect(reqwest::redirect::Policy::none())
236        .build()
237        .context("defuddle_fetch: failed to build HTTP client")?;
238
239    let response = client
240        .get(&target)
241        .header("Accept", "text/markdown, text/plain;q=0.9, */*;q=0.5")
242        .send()
243        .await
244        .context("defuddle_fetch: request failed")?;
245
246    let status = response.status();
247    if !status.is_success() {
248        return Err(anyhow!("defuddle.md returned HTTP {status}"));
249    }
250
251    let body = response.text().await.context("defuddle_fetch: failed to read response body")?;
252
253    Ok(truncate_markdown_body(body, max_bytes))
254}
255
256fn truncate_markdown_body(body: String, max_bytes: usize) -> (String, bool) {
257    if body.len() > max_bytes {
258        // Truncate to a hard cap so a giant page never blows up the agent.
259        let truncated = vtcode_commons::formatting::truncate_byte_budget(&body, max_bytes, "");
260        // `truncated` is `true` only when the upstream body exceeded the cap,
261        // not when it just happened to land exactly on it.
262        (truncated, true)
263    } else {
264        (body, false)
265    }
266}
267
268/// Percent-encode a string so it is safe to append as a single path segment
269/// to a base URL. We unreserved-encode everything except `:`, `/`, `?`, `#`,
270/// `[`, `]`, `@`, `!`, `$`, `&`, `'`, `(`, `)`, `*`, `+`, `,`, `;`, `=`
271/// (per RFC 3986 sub-delims) — those are left alone so an http(s) URL
272/// survives encoding. This is intentionally simpler than the full RFC 3986
273/// path-segment set because we know our input is a URL.
274fn percent_encode_path(input: &str) -> String {
275    use std::fmt::Write as _;
276
277    let mut out = String::with_capacity(input.len());
278    for byte in input.bytes() {
279        let unreserved = byte.is_ascii_alphanumeric() || matches!(byte, b'-' | b'.' | b'_' | b'~');
280        let sub_delim = matches!(
281            byte,
282            b':' | b'/'
283                | b'?'
284                | b'#'
285                | b'['
286                | b']'
287                | b'@'
288                | b'!'
289                | b'$'
290                | b'&'
291                | b'\''
292                | b'('
293                | b')'
294                | b'*'
295                | b'+'
296                | b','
297                | b';'
298                | b'='
299        );
300        if unreserved || sub_delim {
301            out.push(byte as char);
302        } else {
303            // Write directly into `out` instead of allocating a temporary
304            // String via format! for every encoded byte.
305            let _ = write!(out, "%{byte:02X}");
306        }
307    }
308    out
309}
310
311#[async_trait]
312impl Tool for DefuddleTool {
313    async fn execute(&self, args: Value) -> Result<Value> {
314        self.run(args).await
315    }
316
317    fn name(&self) -> &str {
318        tools::DEFUDDLE_FETCH
319    }
320
321    fn description(&self) -> &str {
322        DEFUDDLE_FETCH_DESCRIPTION
323    }
324}
325
326#[cfg(test)]
327mod tests {
328    use super::*;
329
330    #[test]
331    fn missing_url_is_rejected() {
332        let tool = DefuddleTool::new();
333        let result = tokio::runtime::Runtime::new().unwrap().block_on(tool.run(json!({})));
334        assert!(result.is_err());
335    }
336
337    #[test]
338    fn empty_url_is_rejected() {
339        let tool = DefuddleTool::new();
340        let result = tokio::runtime::Runtime::new()
341            .unwrap()
342            .block_on(tool.run(json!({ "url": "   " })));
343        assert!(result.is_err());
344    }
345
346    #[test]
347    fn non_http_scheme_is_rejected() {
348        let tool = DefuddleTool::new();
349        for bad in ["javascript:alert(1)", "file:///etc/passwd", "data:text/html,hi"] {
350            let result = tokio::runtime::Runtime::new()
351                .unwrap()
352                .block_on(tool.run(json!({ "url": bad })));
353            assert!(result.is_err(), "should reject {bad}");
354        }
355    }
356
357    #[test]
358    fn local_file_paths_are_rejected_before_url_parsing() {
359        let tool = DefuddleTool::new();
360        for bad in [
361            "/Users/vinhnguyenxuan/Documents/podcast/build-video.sh",
362            "./relative/path.txt",
363            "/etc/passwd",
364            "C:\\Users\\file.txt",
365        ] {
366            let result = tokio::runtime::Runtime::new()
367                .unwrap()
368                .block_on(tool.run(json!({ "url": bad })));
369            assert!(result.is_err(), "should reject local path: {bad}");
370            let err_msg = result.unwrap_err().to_string();
371            assert!(err_msg.contains("exec_command"));
372            assert!(!err_msg.contains(&format!("unified_{}", "file")));
373        }
374    }
375
376    #[test]
377    fn session_cap_rejects_second_call() {
378        let tool = DefuddleTool::new();
379        // Manually mark the cap as consumed; we don't make a real network
380        // call here. `try_consume` is the same code path the live call uses.
381        assert!(tool.try_consume());
382        let payload = tokio::runtime::Runtime::new()
383            .unwrap()
384            .block_on(tool.run(json!({ "url": "https://example.com" })))
385            .expect("cap hit must be a structured JSON, not a runtime error");
386        assert_eq!(payload["error"], "defuddle_fetch session cap reached");
387        assert_eq!(payload["session_cap"], 1);
388    }
389
390    #[test]
391    fn reset_allows_one_more_call() {
392        let tool = DefuddleTool::new();
393        assert!(tool.try_consume());
394        assert!(!tool.try_consume());
395        tool.reset();
396        assert!(tool.try_consume());
397    }
398
399    #[test]
400    fn markdown_truncation_preserves_utf8_boundaries() {
401        let (body, truncated) = truncate_markdown_body("你好".to_owned(), 4);
402
403        assert_eq!(body, "你");
404        assert!(truncated);
405    }
406
407    #[test]
408    fn validate_target_url_rejects_bad_inputs() {
409        assert!(validate_target_url("https://example.com").is_ok());
410        assert!(validate_target_url("https://example.com/path?q=1").is_ok());
411        assert!(validate_target_url("http://example.com").is_ok());
412        assert!(validate_target_url("javascript:alert(1)").is_err());
413        assert!(validate_target_url("file:///etc/passwd").is_err());
414        assert!(validate_target_url("data:text/html,hi").is_err());
415        assert!(validate_target_url("not a url at all").is_err());
416    }
417
418    /// Regression test for review H2: defuddle must NOT relay requests to
419    /// private / loopback / link-local / broadcast / multicast hosts.
420    /// Otherwise a URL like `http://192.168.0.1/admin` would tunnel an
421    /// SSRF through the public defuddle.md service.
422    #[test]
423    fn validate_target_url_blocks_private_hosts() {
424        for bad in [
425            "http://10.0.0.1/",
426            "http://192.168.1.1/admin",
427            "http://172.16.0.1/",
428            "http://127.0.0.1/",
429            "http://169.254.169.254/latest/meta-data/", // AWS IMDS
430            "http://[::1]/",
431            "http://[fc00::1]/",
432            "http://[fe80::1]/",
433            "http://255.255.255.255/",
434            "http://224.0.0.1/",
435            "http://localhost/admin",
436        ] {
437            let result = validate_target_url(bad);
438            assert!(result.is_err(), "private host should be rejected: {bad}");
439            let msg = result.unwrap_err().to_string();
440            assert!(
441                msg.contains("private") || msg.contains("local"),
442                "error should mention private/local; got: {msg} for {bad}"
443            );
444        }
445    }
446
447    #[test]
448    fn percent_encode_path_preserves_url_sub_delims() {
449        // Common http(s) URL: every byte is unreserved or sub-delim, so the
450        // encoded output should equal the input.
451        let url = "https://example.com/path?q=1&r=2#frag";
452        assert_eq!(percent_encode_path(url), url);
453    }
454
455    #[test]
456    fn percent_encode_path_escapes_spaces_and_unicode() {
457        let encoded = percent_encode_path("https://example.com/has space and \u{2603}");
458        assert!(encoded.starts_with("https://example.com/has%20space%20and%20%E2%98%83"));
459    }
460
461    #[test]
462    fn request_url_targets_defuddle_host_not_target_host() {
463        // Build the same URL the network call would build and assert it points
464        // at defuddle.md, not the user-supplied target. This guards against
465        // regressions where someone reintroduces `Url::join` (which would
466        // silently bypass defuddle.md for absolute URLs).
467        let user_url = "https://example.com/x";
468        let built = format!("{DEFUDDLE_BASE_URL}{}", percent_encode_path(user_url));
469        let parsed = Url::parse(&built).expect("defuddle URL must parse");
470        assert_eq!(parsed.host_str(), Some("defuddle.md"));
471        // The full input URL survives as a path segment on defuddle.md. We
472        // don't compare exact bytes (the URL parser may normalize `//`) —
473        // we just assert the target host is referenced somewhere in the path.
474        let path = parsed.path();
475        assert!(path.contains("example.com"), "defuddle.md URL should reference the user host in its path: {path}");
476    }
477
478    /// Regression test for turn_578: when defuddle.md rate-limits the
479    /// request, the structured error must say so and point the agent to
480    /// web_fetch as a fallback. It must not look like a generic
481    /// "PermissionDenied" (which is what the previous shape produced).
482    #[test]
483    fn error_response_classifies_403_and_suggests_web_fetch() {
484        let err = anyhow::anyhow!("defuddle.md returned HTTP 403");
485        let response = defuddle_fetch_error_response("https://example.com", MAX_BYTES, DEFAULT_TIMEOUT_SECS, &err);
486        assert_eq!(response["error_type"], "http_error");
487        assert!(
488            response["next_action"].as_str().unwrap_or("").contains("web_fetch"),
489            "next_action should fall back to web_fetch; got: {}",
490            response["next_action"]
491        );
492        assert!(response["session_cap"].is_number());
493    }
494
495    /// Network-level failures should classify distinctly from HTTP errors.
496    #[test]
497    fn error_response_classifies_timeout_as_network_error() {
498        let err = anyhow::anyhow!("request timed out after 30s");
499        let response = defuddle_fetch_error_response("https://example.com", MAX_BYTES, DEFAULT_TIMEOUT_SECS, &err);
500        assert_eq!(response["error_type"], "network_error");
501    }
502}