ContiAI / tools /scraper /no_agent /bs4_scraper.py
ziadsameh32's picture
Add login page
325b94c
Raw
History Blame Contribute Delete
1.41 kB
import requests
from bs4 import BeautifulSoup
def scrape_with_bs4(url: str) -> dict:
try:
# HTML case
response = requests.get(url, timeout=10)
soup = BeautifulSoup(response.text, "html.parser")
# text = soup.get_text(separator="\n").strip()
# imgs= soup.find_all
title = soup.title.string.strip() if soup.title else "No title"
text = " ".join([p.get_text(strip=True) for p in soup.find_all("p")])
img_urls = [img["src"] for img in soup.find_all("img", src=True)]
video_urls = [vid["src"] for vid in soup.find_all("video", src=True)]
audio_urls = [aud["src"] for aud in soup.find_all("audio", src=True)]
pdf_urls = [
a["href"]
for a in soup.find_all("a", href=True)
if a["href"].endswith(".pdf")
]
return {
"page_url": url,
"title": title,
"content": text,
"img_url": img_urls,
"video_url": video_urls,
"audio_url": audio_urls,
"pdf_url": pdf_urls,
"agent_recommendation_rank": 4.2,
"agent_recommendation_notes": "Scraped successfully using Crawlee + BeautifulSoupCrawler.",
"header": "Web Scraping Test",
"sub_header": "Crawlee Version",
}
except Exception as e:
return {"url": url, "error": str(e)}