amogaddy commited on
Commit
ed3c6cc
Β·
verified Β·
1 Parent(s): 33d4a41

Upload scraper.py with huggingface_hub

Browse files
Files changed (1) hide show
  1. scraper.py +74 -0
scraper.py ADDED
@@ -0,0 +1,74 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import trafilatura
2
+ from duckduckgo_search import DDGS
3
+ from errors import get_logger, GenerAIError, ErrorCode, fmt_exc
4
+
5
+ log = get_logger("scraper")
6
+
7
+
8
+ def search_and_extract(query: str, max_results: int = 3) -> list[dict]:
9
+ """Search DuckDuckGo and extract clean text. Returns [] on total failure."""
10
+ log.info("Ricerca web per: %r", query)
11
+
12
+ # ── 1. DuckDuckGo search ───────────────────────────────────────────────────
13
+ try:
14
+ with DDGS() as ddgs:
15
+ hits = list(ddgs.text(query, max_results=max_results * 2))
16
+ log.debug("DuckDuckGo ha restituito %d risultati grezzi", len(hits))
17
+ except Exception as e:
18
+ err = GenerAIError(ErrorCode.WEB_SEARCH_FAILED, f"DuckDuckGo non raggiungibile: {fmt_exc(e)}", cause=e)
19
+ err.log(log)
20
+ return []
21
+
22
+ if not hits:
23
+ log.warning("[%s] Nessun risultato DuckDuckGo per: %r", ErrorCode.WEB_NO_RESULTS.value, query)
24
+ return []
25
+
26
+ # ── 2. Fetch + extract each URL ────────────────────────────────────────────
27
+ results = []
28
+ for hit in hits:
29
+ if len(results) >= max_results:
30
+ break
31
+
32
+ url = hit.get("href", "")
33
+ if not url:
34
+ log.debug("Hit senza URL, saltato: %s", hit)
35
+ continue
36
+
37
+ log.debug("Fetching: %s", url)
38
+ try:
39
+ downloaded = trafilatura.fetch_url(url, timeout=10)
40
+ except Exception as e:
41
+ log.warning("[%s] fetch_url fallito per %s β€” %s", ErrorCode.WEB_FETCH_FAILED.value, url, fmt_exc(e))
42
+ continue
43
+
44
+ if not downloaded:
45
+ log.debug("fetch_url ha restituito None per: %s", url)
46
+ continue
47
+
48
+ try:
49
+ text = trafilatura.extract(
50
+ downloaded,
51
+ include_links=False,
52
+ include_images=False,
53
+ include_tables=False,
54
+ no_fallback=False,
55
+ )
56
+ except Exception as e:
57
+ log.warning("[%s] estrazione testo fallita per %s β€” %s", ErrorCode.WEB_EXTRACT_FAILED.value, url, fmt_exc(e))
58
+ continue
59
+
60
+ if not text or len(text) < 150:
61
+ log.debug("Testo troppo corto (%d chars) per: %s", len(text) if text else 0, url)
62
+ continue
63
+
64
+ log.info("Estratti %d chars da: %s", len(text), url)
65
+ results.append({
66
+ "title": hit.get("title", ""),
67
+ "url": url,
68
+ "text": text[:2000],
69
+ })
70
+
71
+ if not results:
72
+ log.warning("[%s] Nessun testo utile estratto per: %r", ErrorCode.WEB_NO_RESULTS.value, query)
73
+
74
+ return results