From 6bfc3cf82307b4a5d6737c7573d688689d3b1d05 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Sat, 3 Oct 2026 20:01:56 +0300 Subject: [PATCH 1/2] refactor(network): extract fetch logic into dedicated method Moved the HTTP GET and response rendering logic from the `WebFetchTool` implementation into a new private method `fetch_validated` on `WebFetchTool`. This separates the SSRF guard and gate disclosure from the actual fetch, making the code easier to test and reuse. Also changed the `url` parameter from `&str` to `&str` to avoid an unnecessary borrow. Auto-committed-on: dragonfly Co-authored-by: Medulla --- crates/tinytools-std/src/network/web_fetch.rs | 15 ++++++++++++++- 1 file changed, 14 insertions(+), 1 deletion(-) diff --git a/crates/tinytools-std/src/network/web_fetch.rs b/crates/tinytools-std/src/network/web_fetch.rs index 6bed91f..3434827 100644 --- a/crates/tinytools-std/src/network/web_fetch.rs +++ b/crates/tinytools-std/src/network/web_fetch.rs @@ -213,6 +213,19 @@ impl Tool for WebFetchTool { // before contacting the host. self.gate.disclose(&host_of(&url), false, false); + self.fetch_validated(&url, max_bytes, raw_requested).await + } +} + +impl WebFetchTool { + /// Issue the GET for a URL that already passed the gate and the SSRF + /// guard, and render the response for the model. + async fn fetch_validated( + &self, + url: &str, + max_bytes: usize, + raw_requested: bool, + ) -> anyhow::Result { // Disable automatic redirect following: reqwest follows up to 10 // redirects by default, and a redirect target may be on a host // outside the allowed-domains list. We surface 3xx responses to @@ -226,7 +239,7 @@ impl Tool for WebFetchTool { Err(e) => return Ok(ToolResult::error(format!("Failed to build client: {e}"))), }; - let resp = match client.get(&url).send().await { + let resp = match client.get(url).send().await { Ok(r) => r, Err(e) => return Ok(ToolResult::error(format!("Request failed: {e}"))), }; From a6b188b8b72b0e0150a9fd6e2a99fe32b91d6021 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Sat, 3 Oct 2026 20:03:11 +0300 Subject: [PATCH 2/2] feat(network): treat 4xx/5xx responses as fetch errors instead of successes HTTP error status codes (4xx and 5xx) are now returned as tool errors rather than successful page reports, so that rate-limited, blocked, or unavailable pages are properly surfaced to the agent for retry or alternative sourcing. The error message includes the status code, host, a short body excerpt, and a Retry-After hint when present, while 3xx redirects continue to be reported as successful but unfollowed responses. Auto-committed-on: dragonfly Co-authored-by: Medulla --- crates/tinytools-std/src/network/web_fetch.rs | 98 ++++++++++++ .../src/network/web_fetch_tests.rs | 147 ++++++++++++++++++ 2 files changed, 245 insertions(+) diff --git a/crates/tinytools-std/src/network/web_fetch.rs b/crates/tinytools-std/src/network/web_fetch.rs index 3434827..1775e0b 100644 --- a/crates/tinytools-std/src/network/web_fetch.rs +++ b/crates/tinytools-std/src/network/web_fetch.rs @@ -5,6 +5,11 @@ //! disk). `web_fetch` is the single-purpose "GET and read" primitive //! the agent reaches for when researching: returns the response body //! as text, capped, with a tiny preamble (status + final URL). +//! +//! A 4xx/5xx response is a failed fetch: it comes back as an error result +//! (`is_error`) naming the status and a short body excerpt, so a host that +//! budgets or retries on tool errors sees blocked and rate-limited pages for +//! what they are. 3xx responses are not followed and stay successful reports. use super::gate::{HttpLimits, NetGate, host_of}; use crate::url_guard::{normalize_allowed_domains, validate_url_with_dns_check}; @@ -255,6 +260,11 @@ impl WebFetchTool { .get(reqwest::header::CONTENT_TYPE) .and_then(|v| v.to_str().ok()) .map(str::to_string); + let retry_after = resp + .headers() + .get(reqwest::header::RETRY_AFTER) + .and_then(|v| v.to_str().ok()) + .map(str::to_string); let body = match resp.text().await { Ok(b) => b, Err(e) => return Ok(ToolResult::error(format!("Failed to read body: {e}"))), @@ -270,6 +280,30 @@ impl WebFetchTool { ))); } + // A 4xx/5xx is a failed fetch, not a page: returning it as a success + // let a blocked or rate-limited site count as research done. (3xx is + // handled above and stays a successful "not followed" report.) + if status.is_client_error() || status.is_server_error() { + let host = host_of(&final_url); + log::debug!( + "[tool.web_fetch] http error status={} host={host} retry_after_present={}", + status.as_u16(), + retry_after.is_some() + ); + let excerpt = error_body_excerpt( + self.html.as_ref(), + &body, + content_type.as_deref(), + raw_requested, + ); + return Ok(ToolResult::error(http_error_message( + status, + &host, + retry_after.as_deref(), + &excerpt, + ))); + } + let downloaded = body.len(); let (body, byte_capped) = if downloaded > max_bytes { let cut = floor_char_boundary(&body, max_bytes); @@ -314,6 +348,70 @@ impl WebFetchTool { } } +/// How much of an error response's body is quoted back to the model. +const ERROR_EXCERPT_CHARS: usize = 300; + +/// The model-facing text for a 4xx/5xx response: what happened, why it +/// matters, and what to do next, plus a short excerpt of the body. +fn http_error_message( + status: reqwest::StatusCode, + host: &str, + retry_after: Option<&str>, + excerpt: &str, +) -> String { + let code = status.as_u16(); + let reason = status.canonical_reason().unwrap_or("Unknown Status"); + let mut msg = format!("HTTP {code} {reason} from {host}; "); + match code { + 429 => { + msg.push_str("the site is rate limiting requests."); + if let Some(wait) = retry_after.map(str::trim).filter(|w| !w.is_empty()) { + msg.push_str(&format!(" Retry-After: {wait}.")); + } + msg.push_str(" Try another source, or retry later."); + } + 401 | 403 => msg.push_str("the site refused the request. Try another source."), + 404 | 410 => { + msg.push_str( + "the page does not exist at this URL. Check the URL or try another source.", + ); + } + 500..=599 => { + msg.push_str( + "the server failed to handle the request. Retry later or try another source.", + ); + } + _ => msg.push_str("the server rejected the request. Try another source."), + } + if !excerpt.is_empty() { + msg.push_str("\nResponse excerpt: "); + msg.push_str(excerpt); + } + msg +} + +/// A short, single-line, text-only excerpt of an error response body. +/// +/// HTML goes through the host extractor (unless the caller asked for `raw`) +/// so the model reads the page's words rather than its markup. +fn error_body_excerpt( + extractor: &dyn HtmlExtractor, + body: &str, + content_type: Option<&str>, + raw_requested: bool, +) -> String { + let text = if !raw_requested && is_html(extractor, body, content_type) { + extractor.to_markdown(body) + } else { + body.to_string() + }; + let collapsed = text.split_whitespace().collect::>().join(" "); + match collapsed.char_indices().nth(ERROR_EXCERPT_CHARS) { + Some((cut, _)) => format!("{}...", &collapsed[..cut]), + None => collapsed, + } +} + /// The largest index at or below `index` that is a char boundary of `s`. fn floor_char_boundary(s: &str, index: usize) -> usize { if index >= s.len() { diff --git a/crates/tinytools-std/src/network/web_fetch_tests.rs b/crates/tinytools-std/src/network/web_fetch_tests.rs index 3e550a8..239b5e9 100644 --- a/crates/tinytools-std/src/network/web_fetch_tests.rs +++ b/crates/tinytools-std/src/network/web_fetch_tests.rs @@ -231,3 +231,150 @@ async fn execute_blocks_when_rate_limited() { assert!(result.is_error); assert!(result.output().contains("Rate limit exceeded")); } + +// --- HTTP error statuses --------------------------------------------------- +// +// Loopback is refused by the SSRF guard, so these drive `fetch_validated` +// (everything after validation) against a one-shot local server. + +/// Serves one canned response and returns the base URL. +async fn serve_once(response: &str) -> String { + use tokio::io::{AsyncReadExt, AsyncWriteExt}; + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let addr = listener.local_addr().unwrap(); + let response = response.to_string(); + tokio::spawn(async move { + let (mut socket, _) = listener.accept().await.unwrap(); + let mut buf = vec![0u8; 8192]; + let _ = socket.read(&mut buf).await.unwrap(); + socket.write_all(response.as_bytes()).await.unwrap(); + let _ = socket.shutdown().await; + }); + format!("http://{addr}/page") +} + +fn http_response(status_line: &str, headers: &str, body: &str) -> String { + format!( + "HTTP/1.1 {status_line}\r\n{headers}Content-Length: {}\r\nConnection: close\r\n\r\n{body}", + body.len() + ) +} + +async fn fetch_canned(response: String) -> ToolResult { + let url = serve_once(&response).await; + fetch(test_security(), vec![], None, None) + .fetch_validated(&url, 1_000_000, false) + .await + .unwrap() +} + +#[tokio::test] +async fn a_200_is_a_successful_result_with_the_status_header() { + let result = fetch_canned(http_response( + "200 OK", + "Content-Type: text/plain\r\n", + "hello", + )) + .await; + assert!(!result.is_error, "got: {}", result.output()); + assert!(result.output().starts_with("status=200 url=")); + assert!(result.output().ends_with("hello")); +} + +#[tokio::test] +async fn a_403_is_an_error_naming_the_status_and_suggesting_another_source() { + let result = fetch_canned(http_response( + "403 Forbidden", + "Content-Type: text/plain\r\n", + "Access denied by bot protection", + )) + .await; + assert!(result.is_error, "got: {}", result.output()); + let out = result.output(); + assert!( + out.contains("HTTP 403 Forbidden from 127.0.0.1"), + "got: {out}" + ); + assert!(out.contains("refused the request"), "got: {out}"); + assert!(out.contains("another source"), "got: {out}"); + assert!( + out.contains("Access denied by bot protection"), + "got: {out}" + ); +} + +#[tokio::test] +async fn a_429_is_an_error_that_reports_rate_limiting_and_retry_after() { + let result = fetch_canned(http_response( + "429 Too Many Requests", + "Retry-After: 120\r\n", + "", + )) + .await; + assert!(result.is_error, "got: {}", result.output()); + let out = result.output(); + assert!(out.contains("HTTP 429 Too Many Requests"), "got: {out}"); + assert!(out.contains("rate limit"), "got: {out}"); + assert!(out.contains("Retry-After: 120"), "got: {out}"); +} + +#[tokio::test] +async fn a_429_without_retry_after_still_reports_rate_limiting() { + let result = fetch_canned(http_response("429 Too Many Requests", "", "")).await; + assert!(result.is_error); + assert!(result.output().contains("rate limit")); + assert!(!result.output().contains("Retry-After")); +} + +#[tokio::test] +async fn a_404_is_an_error() { + let result = fetch_canned(http_response( + "404 Not Found", + "Content-Type: text/html\r\n", + "

No such page

", + )) + .await; + assert!(result.is_error, "got: {}", result.output()); + let out = result.output(); + assert!(out.contains("HTTP 404 Not Found"), "got: {out}"); + // The excerpt is the page's text, not its markup. + assert!(out.contains("No such page"), "got: {out}"); + assert!(!out.contains("

"), "got: {out}"); +} + +#[tokio::test] +async fn a_5xx_is_an_error() { + let result = fetch_canned(http_response("503 Service Unavailable", "", "")).await; + assert!(result.is_error); + assert!(result.output().contains("HTTP 503 Service Unavailable")); +} + +#[tokio::test] +async fn an_error_body_excerpt_is_short() { + let long = "x".repeat(5_000); + let result = fetch_canned(http_response( + "500 Internal Server Error", + "Content-Type: text/plain\r\n", + &long, + )) + .await; + assert!(result.is_error); + assert!( + result.output().len() < 1_000, + "excerpt must be bounded, got {} bytes", + result.output().len() + ); +} + +#[tokio::test] +async fn a_redirect_is_still_reported_as_a_successful_result() { + let result = fetch_canned(http_response( + "301 Moved Permanently", + "Location: https://example.com/new\r\n", + "", + )) + .await; + assert!(!result.is_error, "got: {}", result.output()); + assert!(result.output().contains("status=301")); + assert!(result.output().contains("location=https://example.com/new")); +}