| use anyhow::{Context, anyhow}; |
| use forge_app::{HttpResponse, NetFetchService, ResponseContext, is_binary_content_type}; |
| use reqwest::{Client, Url}; |
|
|
| |
| |
| |
| |
| |
| |
| |
| |
| #[derive(Debug)] |
| pub struct ForgeFetch { |
| client: Client, |
| } |
|
|
| impl Default for ForgeFetch { |
| fn default() -> Self { |
| Self::new() |
| } |
| } |
|
|
| impl ForgeFetch { |
| pub fn new() -> Self { |
| Self { client: Client::new() } |
| } |
| } |
|
|
| impl ForgeFetch { |
| async fn check_robots_txt(&self, url: &Url) -> anyhow::Result<()> { |
| let robots_url = format!("{}://{}/robots.txt", url.scheme(), url.authority()); |
| let robots_response = self.client.get(&robots_url).send().await; |
|
|
| if let Ok(robots) = robots_response |
| && robots.status().is_success() |
| { |
| let robots_content = robots.text().await.unwrap_or_default(); |
| let path = url.path(); |
| for line in robots_content.lines() { |
| if let Some(disallowed) = line.strip_prefix("Disallow: ") { |
| let disallowed = disallowed.trim(); |
| let disallowed = if !disallowed.starts_with('/') { |
| format!("/{disallowed}") |
| } else { |
| disallowed.to_string() |
| }; |
| let path = if !path.starts_with('/') { |
| format!("/{path}") |
| } else { |
| path.to_string() |
| }; |
| if path.starts_with(&disallowed) { |
| return Err(anyhow!( |
| "URL {url} cannot be fetched due to robots.txt restrictions" |
| )); |
| } |
| } |
| } |
| } |
| Ok(()) |
| } |
|
|
| async fn fetch_url(&self, url: &Url, force_raw: bool) -> anyhow::Result<HttpResponse> { |
| self.check_robots_txt(url).await?; |
|
|
| let response = self |
| .client |
| .get(url.as_str()) |
| .send() |
| .await |
| .map_err(|e| anyhow!("Failed to fetch URL {url}: {e}"))?; |
| let code = response.status().as_u16(); |
|
|
| if !response.status().is_success() { |
| return Err(anyhow!( |
| "Failed to fetch {} - status code {}", |
| url, |
| response.status() |
| )); |
| } |
|
|
| let content_type = response |
| .headers() |
| .get("content-type") |
| .and_then(|v| v.to_str().ok()) |
| .unwrap_or("") |
| .to_string(); |
|
|
| |
| |
| if is_binary_content_type(&content_type) { |
| return Err(anyhow!( |
| "URL {} returns binary content (Content-Type: {}). \ |
| The fetch tool only handles text content. \ |
| Use the shell tool with `curl -fLo <output_file> <url>` to download binary files.", |
| url, |
| content_type |
| )); |
| } |
|
|
| let page_raw = response |
| .text() |
| .await |
| .map_err(|e| anyhow!("Failed to read response content from {url}: {e}"))?; |
|
|
| |
| let sniff_end = if page_raw.len() >= 100 { |
| |
| let mut end = 100; |
| while end > 0 && !page_raw.is_char_boundary(end) { |
| end -= 1; |
| } |
| end |
| } else { |
| page_raw.len() |
| }; |
| let is_page_html = page_raw |
| .get(..sniff_end) |
| .map(|s| s.contains("<html")) |
| .unwrap_or(false) |
| || content_type.contains("text/html") |
| || content_type.is_empty(); |
|
|
| if is_page_html && !force_raw { |
| let content = html2md::parse_html(&page_raw); |
| Ok(HttpResponse { content, context: ResponseContext::Raw, code, content_type }) |
| } else { |
| Ok(HttpResponse { |
| content: page_raw, |
| context: ResponseContext::Parsed, |
| code, |
| content_type, |
| }) |
| } |
| } |
| } |
|
|
| #[async_trait::async_trait] |
| impl NetFetchService for ForgeFetch { |
| async fn fetch(&self, url: String, raw: Option<bool>) -> anyhow::Result<HttpResponse> { |
| let url = Url::parse(&url).with_context(|| format!("Failed to parse URL: {url}"))?; |
|
|
| self.fetch_url(&url, raw.unwrap_or(false)).await |
| } |
| } |
|
|
| #[cfg(test)] |
| mod tests { |
| use super::*; |
|
|
| #[test] |
| fn test_is_binary_content_type_text_types_are_not_binary() { |
| assert!(!is_binary_content_type("text/html")); |
| assert!(!is_binary_content_type("text/plain")); |
| assert!(!is_binary_content_type("text/css")); |
| assert!(!is_binary_content_type("application/json")); |
| assert!(!is_binary_content_type("application/xml")); |
| assert!(!is_binary_content_type("application/javascript")); |
| assert!(!is_binary_content_type("application/yaml")); |
| assert!(!is_binary_content_type("image/svg+xml")); |
| assert!(!is_binary_content_type("text/csv")); |
| assert!(!is_binary_content_type("text/markdown")); |
| assert!(!is_binary_content_type("")); |
| } |
|
|
| #[test] |
| fn test_is_binary_content_type_binary_types_detected() { |
| assert!(is_binary_content_type("application/gzip")); |
| assert!(is_binary_content_type("application/x-gzip")); |
| assert!(is_binary_content_type("application/octet-stream")); |
| assert!(is_binary_content_type("application/zip")); |
| assert!(is_binary_content_type("application/x-tar")); |
| assert!(is_binary_content_type("application/pdf")); |
| assert!(is_binary_content_type("image/png")); |
| assert!(is_binary_content_type("image/jpeg")); |
| assert!(is_binary_content_type("audio/mpeg")); |
| assert!(is_binary_content_type("video/mp4")); |
| } |
|
|
| #[test] |
| fn test_is_binary_content_type_case_insensitive() { |
| assert!(!is_binary_content_type("Application/JSON")); |
| assert!(!is_binary_content_type("TEXT/HTML; charset=utf-8")); |
| assert!(is_binary_content_type("Application/Gzip")); |
| } |
| } |
|
|