""" SUPERSEDED: rag_index/ is now built by scripts/build_index_from_crawler.py from a sibling crawler project's data (respects robots.txt, resumable, categorized). Kept for reference only — not part of the current build path. See README.md. Local crawler for the top-150 digital marketing sources (scripts/sources.json). Runs entirely on the user's machine — never on the HF Space. Does NOT check robots.txt (per project decision). Rate-limits per domain and rotates a realistic User-Agent to avoid hard IP bans. Resumable: re-running skips URLs already present in a source's state.json. Usage: python crawl_sources.py python crawl_sources.py --only "Search Engine Land,Backlinko" --max-pages-per-source 10 python crawl_sources.py --sources sources.json --out ../crawled """ from __future__ import annotations import argparse import hashlib import json import random import re import time import xml.etree.ElementTree as ET from pathlib import Path from urllib.parse import urljoin, urlparse import httpx import trafilatura from bs4 import BeautifulSoup USER_AGENTS = [ "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) " "Chrome/124.0.0.0 Safari/537.36", "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 " "(KHTML, like Gecko) Version/17.4 Safari/605.1.15", "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) " "Chrome/124.0.0.0 Safari/537.36", ] SKIP_URL_PATTERNS = re.compile( r"(/tag/|/tags/|/category/|/categories/|/author/|/page/\d+|/wp-json/|/feed/?$|" r"\.(jpg|jpeg|png|gif|svg|webp|css|js|pdf|zip|mp4|mp3)$|#|mailto:|javascript:)", re.IGNORECASE, ) MIN_TEXT_LENGTH = 200 REQUEST_TIMEOUT = 20 MAX_RETRIES = 3 def slugify(name: str) -> str: return re.sub(r"[^a-z0-9]+", "-", name.lower()).strip("-") def url_hash(url: str) -> str: return hashlib.sha1(url.encode("utf-8")).hexdigest()[:16] class RateLimiter: def __init__(self): self._last_request = {} def wait(self, domain: str, min_interval: float): last = self._last_request.get(domain, 0) elapsed = time.monotonic() - last if elapsed < min_interval: time.sleep(min_interval - elapsed) self._last_request[domain] = time.monotonic() def fetch(client: httpx.Client, url: str) -> httpx.Response | None: for attempt in range(1, MAX_RETRIES + 1): try: resp = client.get( url, headers={ "User-Agent": random.choice(USER_AGENTS), "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8", "Accept-Language": "en-US,en;q=0.9", }, timeout=REQUEST_TIMEOUT, follow_redirects=True, ) if resp.status_code == 429 or resp.status_code >= 500: time.sleep(2**attempt) continue return resp except httpx.RequestError: time.sleep(2**attempt) return None def discover_via_feed(client: httpx.Client, feed_url: str) -> list[str]: resp = fetch(client, feed_url) if resp is None or resp.status_code != 200: return [] text = resp.text urls: list[str] = [] try: root = ET.fromstring(text) except ET.ParseError: return [] tag = root.tag.lower() if "urlset" in tag or "sitemapindex" in tag: # XML sitemap (or sitemap index) — collect entries. for loc in root.iter(): if loc.tag.lower().endswith("loc") and loc.text: urls.append(loc.text.strip()) return urls # RSS / Atom feed for item in root.iter(): local = item.tag.lower() if local.endswith("item") or local.endswith("entry"): link = None for child in item: ctag = child.tag.lower() if ctag.endswith("link"): link = child.get("href") or (child.text.strip() if child.text else None) if link: break if link: urls.append(link) return urls def discover_via_crawl( client: httpx.Client, base_url: str, crawl_depth: int, max_pages: int ) -> list[str]: domain = urlparse(base_url).netloc seen = {base_url} frontier = [(base_url, 0)] collected = [] while frontier and len(collected) < max_pages: url, depth = frontier.pop(0) resp = fetch(client, url) if resp is None or resp.status_code != 200: continue collected.append(url) if depth >= crawl_depth: continue soup = BeautifulSoup(resp.text, "html.parser") for a in soup.find_all("a", href=True): link = urljoin(url, a["href"]) parsed = urlparse(link) if parsed.netloc != domain: continue if SKIP_URL_PATTERNS.search(link): continue if link not in seen: seen.add(link) frontier.append((link, depth + 1)) return collected def load_state(state_path: Path) -> dict: if state_path.exists(): return json.loads(state_path.read_text(encoding="utf-8")) return {} def save_state(state_path: Path, state: dict): state_path.write_text(json.dumps(state, indent=2), encoding="utf-8") def crawl_source( client: httpx.Client, source: dict, out_dir: Path, rate_limiter: RateLimiter, max_pages: int, log_fh, ): name = source["name"] slug = slugify(name) domain = urlparse(source["base_url"]).netloc rate_limit = source.get("rate_limit_seconds", 3) source_dir = out_dir / slug pages_dir = source_dir / "pages" pages_dir.mkdir(parents=True, exist_ok=True) state_path = source_dir / "state.json" state = load_state(state_path) print(f"[{name}] discovering URLs...") urls: list[str] = [] if source.get("sitemap_or_feed"): rate_limiter.wait(domain, rate_limit) urls = discover_via_feed(client, source["sitemap_or_feed"]) if not urls: urls = discover_via_crawl( client, source["base_url"], source.get("crawl_depth", 2), max_pages ) urls = urls[:max_pages] print(f"[{name}] {len(urls)} candidate URLs") fetched, skipped, errors = 0, 0, 0 for url in urls: h = url_hash(url) if h in state and state[h]["status"] in ("ok", "skipped_short"): continue # resumable: already processed rate_limiter.wait(domain, rate_limit) resp = fetch(client, url) if resp is None or resp.status_code != 200: state[h] = {"url": url, "status": "error", "code": getattr(resp, "status_code", None)} errors += 1 log_fh.write(json.dumps({"source": name, "url": url, "status": "error"}) + "\n") continue html = resp.text extracted = trafilatura.extract(html, include_comments=False, include_tables=False) if not extracted or len(extracted) < MIN_TEXT_LENGTH: state[h] = {"url": url, "status": "skipped_short"} skipped += 1 log_fh.write( json.dumps({"source": name, "url": url, "status": "skipped_short_or_paywalled"}) + "\n" ) continue (pages_dir / f"{h}.html").write_text(html, encoding="utf-8", errors="ignore") (pages_dir / f"{h}.txt").write_text(extracted, encoding="utf-8") state[h] = {"url": url, "status": "ok", "fetched_at": time.time()} fetched += 1 log_fh.write(json.dumps({"source": name, "url": url, "status": "ok"}) + "\n") save_state(state_path, state) print(f"[{name}] done: {fetched} fetched, {skipped} skipped, {errors} errors") def main(): parser = argparse.ArgumentParser(description="Local crawler for digital marketing sources") parser.add_argument("--sources", default="sources.json") parser.add_argument("--out", default="../crawled") parser.add_argument("--max-pages-per-source", type=int, default=40) parser.add_argument("--only", default=None, help="comma-separated source names to restrict to") args = parser.parse_args() sources = json.loads(Path(args.sources).read_text(encoding="utf-8")) if args.only: wanted = {n.strip() for n in args.only.split(",")} sources = [s for s in sources if s["name"] in wanted] out_dir = Path(args.out) out_dir.mkdir(parents=True, exist_ok=True) log_path = out_dir / "crawl_log.jsonl" rate_limiter = RateLimiter() with httpx.Client(http2=True) as client, open(log_path, "a", encoding="utf-8") as log_fh: for source in sources: try: crawl_source( client, source, out_dir, rate_limiter, args.max_pages_per_source, log_fh ) except Exception as exc: # keep going across 150 sources print(f"[{source['name']}] FAILED: {exc}") log_fh.write( json.dumps({"source": source["name"], "status": "source_failed", "error": str(exc)}) + "\n" ) if __name__ == "__main__": main()