use anyhow::{Context, anyhow}; use forge_app::{HttpResponse, NetFetchService, ResponseContext, is_binary_content_type}; use reqwest::{Client, Url}; /// Retrieves content from URLs as markdown or raw text. Enables access to /// current online information including websites, APIs and documentation. Use /// for obtaining up-to-date information beyond training data, verifying facts, /// or retrieving specific online content. Handles HTTP/HTTPS and converts HTML /// to readable markdown by default. Cannot access private/restricted resources /// requiring authentication. Respects robots.txt and may be blocked by /// anti-scraping measures. For large pages, returns the first 40,000 characters /// and stores the complete content in a temporary file for subsequent access. #[derive(Debug)] pub struct ForgeFetch { client: Client, } impl Default for ForgeFetch { fn default() -> Self { Self::new() } } impl ForgeFetch { pub fn new() -> Self { Self { client: Client::new() } } } impl ForgeFetch { async fn check_robots_txt(&self, url: &Url) -> anyhow::Result<()> { let robots_url = format!("{}://{}/robots.txt", url.scheme(), url.authority()); let robots_response = self.client.get(&robots_url).send().await; if let Ok(robots) = robots_response && robots.status().is_success() { let robots_content = robots.text().await.unwrap_or_default(); let path = url.path(); for line in robots_content.lines() { if let Some(disallowed) = line.strip_prefix("Disallow: ") { let disallowed = disallowed.trim(); let disallowed = if !disallowed.starts_with('/') { format!("/{disallowed}") } else { disallowed.to_string() }; let path = if !path.starts_with('/') { format!("/{path}") } else { path.to_string() }; if path.starts_with(&disallowed) { return Err(anyhow!( "URL {url} cannot be fetched due to robots.txt restrictions" )); } } } } Ok(()) } async fn fetch_url(&self, url: &Url, force_raw: bool) -> anyhow::Result { self.check_robots_txt(url).await?; let response = self .client .get(url.as_str()) .send() .await .map_err(|e| anyhow!("Failed to fetch URL {url}: {e}"))?; let code = response.status().as_u16(); if !response.status().is_success() { return Err(anyhow!( "Failed to fetch {} - status code {}", url, response.status() )); } let content_type = response .headers() .get("content-type") .and_then(|v| v.to_str().ok()) .unwrap_or("") .to_string(); // Detect binary content types before attempting to read as text. // The fetch tool is designed for text/HTML content only. if is_binary_content_type(&content_type) { return Err(anyhow!( "URL {} returns binary content (Content-Type: {}). \ The fetch tool only handles text content. \ Use the shell tool with `curl -fLo ` to download binary files.", url, content_type )); } let page_raw = response .text() .await .map_err(|e| anyhow!("Failed to read response content from {url}: {e}"))?; // Use floor_char_boundary to avoid panicking on multi-byte UTF-8 chars let sniff_end = if page_raw.len() >= 100 { // Find the nearest char boundary at or before byte index 100 let mut end = 100; while end > 0 && !page_raw.is_char_boundary(end) { end -= 1; } end } else { page_raw.len() }; let is_page_html = page_raw .get(..sniff_end) .map(|s| s.contains(") -> anyhow::Result { let url = Url::parse(&url).with_context(|| format!("Failed to parse URL: {url}"))?; self.fetch_url(&url, raw.unwrap_or(false)).await } } #[cfg(test)] mod tests { use super::*; #[test] fn test_is_binary_content_type_text_types_are_not_binary() { assert!(!is_binary_content_type("text/html")); assert!(!is_binary_content_type("text/plain")); assert!(!is_binary_content_type("text/css")); assert!(!is_binary_content_type("application/json")); assert!(!is_binary_content_type("application/xml")); assert!(!is_binary_content_type("application/javascript")); assert!(!is_binary_content_type("application/yaml")); assert!(!is_binary_content_type("image/svg+xml")); assert!(!is_binary_content_type("text/csv")); assert!(!is_binary_content_type("text/markdown")); assert!(!is_binary_content_type("")); // empty = unknown, allow } #[test] fn test_is_binary_content_type_binary_types_detected() { assert!(is_binary_content_type("application/gzip")); assert!(is_binary_content_type("application/x-gzip")); assert!(is_binary_content_type("application/octet-stream")); assert!(is_binary_content_type("application/zip")); assert!(is_binary_content_type("application/x-tar")); assert!(is_binary_content_type("application/pdf")); assert!(is_binary_content_type("image/png")); assert!(is_binary_content_type("image/jpeg")); assert!(is_binary_content_type("audio/mpeg")); assert!(is_binary_content_type("video/mp4")); } #[test] fn test_is_binary_content_type_case_insensitive() { assert!(!is_binary_content_type("Application/JSON")); assert!(!is_binary_content_type("TEXT/HTML; charset=utf-8")); assert!(is_binary_content_type("Application/Gzip")); } }