from __future__ import annotations import re from typing import Any from urllib.parse import urlparse import requests from bs4 import BeautifulSoup from config import Settings def read_webpage(url: str, settings: Settings, max_chars: int = 18000) -> dict[str, Any]: parsed = urlparse(url) if parsed.scheme not in {"http", "https"}: raise ValueError("Only http and https URLs are allowed.") response = requests.get( url, timeout=settings.request_timeout, headers={"User-Agent": "Mozilla/5.0 GAIAResearchAgent/1.0"}, allow_redirects=True, ) response.raise_for_status() content_type = response.headers.get("content-type", "") if "text/html" not in content_type and "text/plain" not in content_type: raise ValueError(f"Unsupported webpage content type: {content_type}") soup = BeautifulSoup(response.text, "html.parser") for tag in soup(["script", "style", "nav", "footer", "noscript"]): tag.decompose() title = soup.title.get_text(" ", strip=True) if soup.title else url text = re.sub(r"\s+", " ", soup.get_text(" ", strip=True))[:max_chars] return { "ok": True, "source": response.url, "content": f"TITLE: {title}\n{text}", "metadata": {"status": response.status_code, "content_type": content_type}, }