ContiAI / tools /scraper /no_agent /crawlee_bs_scraper.py
ziadsameh32's picture
Add login page
325b94c
Raw
History Blame Contribute Delete
1.55 kB
import asyncio
from crawlee.crawlers import BeautifulSoupCrawler, BeautifulSoupCrawlingContext
async def crawl_bs_url(url: str):
"""Scrape a single URL using Crawlee + BeautifulSoupCrawler and return one record"""
crawler = BeautifulSoupCrawler(max_requests_per_crawl=1)
result_container = {"data": None}
@crawler.router.default_handler
async def handle_page(context: BeautifulSoupCrawlingContext):
soup = context.soup
title = soup.title.string.strip() if soup.title else "No title"
text = " ".join([p.get_text(strip=True) for p in soup.find_all("p")])
img_urls = [img["src"] for img in soup.find_all("img", src=True)]
video_urls = [v["src"] for v in soup.find_all("video", src=True)]
audio_urls = [a["src"] for a in soup.find_all("audio", src=True)]
pdf_urls = [
a["href"]
for a in soup.find_all("a", href=True)
if a["href"].lower().endswith(".pdf")
]
record = {
"page_url": context.request.url,
"title": title,
"content": text,
"img_url": img_urls,
"video_url": video_urls,
"audio_url": audio_urls,
"pdf_url": pdf_urls,
"agent_recommendation_rank": 4.2,
"agent_recommendation_notes": "Scraped successfully using Crawlee + BeautifulSoupCrawler.",
}
result_container["data"] = record
await context.push_data(record)
await crawler.run([url])
return result_container["data"]