File size: 5,589 Bytes
28a08e7
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
"""
semantic_nav.py — Semantic Anchor Scoring (SAS)

Scoring a 3 fattori:
  1. Keyword Match   (peso 0.50) — intent semantico agente
  2. Visual Position (peso 0.30) — struttura HTML semantica
  3. Path Depth      (peso 0.20) — profondita URL

auto_keywords_from_url(): fallback automatico se nessun intent esplicito.
Backward-compatible: fallback regex se beautifulsoup4 non disponibile.
"""
import re
import logging
from urllib.parse import urljoin, urlparse

_logger = logging.getLogger("tools.semantic_nav")

KEYWORD_MATCH_WEIGHT   = 0.50
VISUAL_POSITION_WEIGHT = 0.30
PATH_DEPTH_WEIGHT      = 0.20

VISUAL_POSITION_SCORES = {
    "nav":     1.0,
    "header":  0.8,
    "main":    0.7,
    "article": 0.6,
    "section": 0.5,
    "form":    0.4,
    "aside":   0.3,
    "footer":  0.1,
    "default": 0.4,
}

_SKIP = re.compile(
    r"(twitter\.com|x\.com|facebook\.com|instagram\.com|linkedin\.com|"
    r"youtube\.com|cdn\.|analytics|pixel|tracking|googletagmanager|doubleclick|"
    r"javascript:|mailto:|tel:|#$)",
    re.I,
)


def auto_keywords_from_url(url):
    """Estrae keywords dall URL come fallback intent."""
    parsed = urlparse(url)
    parts = []
    host_parts = parsed.netloc.replace("www.", "").split(".")
    parts.extend(host_parts[:-1])
    path_parts = [s for s in parsed.path.split("/") if s and len(s) > 2]
    parts.extend(path_parts)
    for kv in (parsed.query or "").split("&"):
        if "=" in kv:
            k, v = kv.split("=", 1)
            parts.extend([k, v])
    _stop = {"the", "a", "an", "and", "or", "in", "on", "at", "to", "for", "of", "with"}
    return [
        p.lower().strip("-_")
        for p in parts
        if p and not p.isdigit() and p.lower() not in _stop and len(p) > 1
    ][:12]


def _get_parent_tag(tag):
    current = tag
    while current is not None:
        name = getattr(current, "name", None)
        if name in VISUAL_POSITION_SCORES:
            return name
        current = getattr(current, "parent", None)
    return "default"


def _score_link(text, url, tag, keywords):
    score = 0.0
    text_lower = text.lower()
    for kw in keywords:
        if kw.lower() in text_lower:
            score += KEYWORD_MATCH_WEIGHT * 0.6
            break
    parent = getattr(tag, "parent", None)
    ctx = parent.get_text(separator=" ", strip=True)[:300] if parent else ""
    for kw in keywords:
        if kw.lower() in ctx.lower():
            score += KEYWORD_MATCH_WEIGHT * 0.4
            break
    parent_tag = _get_parent_tag(tag)
    score += VISUAL_POSITION_SCORES.get(parent_tag, VISUAL_POSITION_SCORES["default"]) * VISUAL_POSITION_WEIGHT
    path_segs = [s for s in urlparse(url).path.split("/") if s]
    score += min(len(path_segs) / 5.0, 1.0) * PATH_DEPTH_WEIGHT
    return max(0.0, min(score, 1.0))


def _fallback_extract(html, base_url, keywords, max_links):
    """Regex fallback quando bs4 non disponibile."""
    base_domain = urlparse(base_url).netloc
    links = []
    kw_lower = [k.lower() for k in keywords]
    for m in re.finditer(r'<a[^>]+href=["\']([ ^"\'>#][^"\']*)["\'\'][^>]*>([^<]{2,80})</a>', html, re.I):
        raw_href, text = m.group(1).strip(), m.group(2).strip()
        if not raw_href or _SKIP.search(raw_href):
            continue
        full_url = urljoin(base_url, raw_href)
        if not full_url.startswith(("http://", "https://")):
            continue
        url_lower = full_url.lower()
        kw_hit = any(kw in text.lower() or kw in url_lower for kw in kw_lower)
        same_domain = urlparse(full_url).netloc == base_domain
        score = (0.5 if kw_hit else 0.2) + (0.2 if same_domain else 0.0)
        links.append({"text": text[:60], "url": full_url, "score": round(score, 2), "context": ""})
    links.sort(key=lambda x: x["score"], reverse=True)
    return links[:max_links]


def extract_and_score_links(html_content, base_url, navigation_intent_keywords=None, max_links=10):
    """
    Estrae e prioritizza link con Semantic Anchor Scoring.
    Returns: list[dict] con keys text, url, score, context. Ordinati per score.
    """
    keywords = navigation_intent_keywords or auto_keywords_from_url(base_url)
    if not keywords:
        keywords = ["main", "content", "article"]

    try:
        from bs4 import BeautifulSoup
    except ImportError:
        _logger.debug("[SAS] beautifulsoup4 non disponibile — uso fallback regex")
        return _fallback_extract(html_content, base_url, keywords, max_links)

    soup = BeautifulSoup(html_content, "html.parser")
    results = []

    for a_tag in soup.find_all("a", href=True):
        text = a_tag.get_text(strip=True)
        raw_href = a_tag["href"]
        if not raw_href or _SKIP.search(raw_href):
            continue
        full_url = urljoin(base_url, raw_href)
        if not full_url.startswith(("http://", "https://")):
            continue
        score = _score_link(text, full_url, a_tag, keywords)
        ctx = ""
        parent = a_tag.parent
        if parent:
            parent_text = parent.get_text(separator=" ", strip=True)
            idx = parent_text.find(text)
            if idx != -1:
                ctx = parent_text[max(0, idx - 80):idx + len(text) + 80].strip()
            else:
                ctx = parent_text[:160].strip()
        results.append({"text": text[:80], "url": full_url, "score": round(score, 3), "context": ctx[:200]})

    results.sort(key=lambda x: x["score"], reverse=True)
    top = results[:max_links]
    _logger.debug("[SAS] %s: %d scored, top %d | kw=%s", base_url, len(results), len(top), keywords[:3])
    return top