AI_Chatbot / services /crawler_utils.py
Hamza-Naimat's picture
production: harden catalog tenants analytics and widget install
59bd3b5
Raw
History Blame Contribute Delete
101 kB
"""
services/crawler_utils.py — Page classification, chunking, and product metadata helpers.
No shared mutable app state. Safe to import from anywhere.
"""
from __future__ import annotations
import hashlib
import html as _html_mod
import json
import logging
import re
import urllib.parse
logger = logging.getLogger(__name__)
from datetime import datetime, timezone
from typing import Any
from services.safety import (
_clean_text,
_looks_structural_page,
_looks_like_product_page,
_looks_like_catalog_page,
_url_never_product,
_strip_storefront_boilerplate,
_dedupe_repeated_lines,
_canonical_product_title,
_trusted_content_metrics,
_BOILERPLATE_SIGNAL_RE,
_GENERIC_SECTION_SPLIT_RE,
_CONTAMINATION_HINTS_RE,
_POLICY_URL_RE,
_POLICY_TEXT_RE,
_CATEGORY_URL_RE,
_ARTICLE_URL_RE,
_PRODUCT_PRICE_LINE_RE,
_PRODUCT_PRICE_CAPTURE_RE,
_PRODUCT_AVAIL_RE,
)
_product_db_cache: dict[str, bool] = {} # db_name → is_product_db (stable per collection)
# Stable per-page URL identity used for dedupe/replace across crawls.
def _canonical_source_url(url: str) -> str:
try:
u = (url or "").strip()
if not u:
return ""
p = urllib.parse.urlparse(u)
scheme = (p.scheme or "https").lower()
netloc = (p.netloc or "").lower()
if netloc.startswith("www."):
netloc = netloc[4:]
path = re.sub(r"/+", "/", p.path or "/")
if path != "/":
path = path.rstrip("/")
return f"{scheme}://{netloc}{path}"
except Exception:
return (url or "").strip()
# ── Category propagation from crawl graph ───────────────────────────────────
# Product pages rarely contain their own category word ("ROG Strix SCAR" never
# says "laptop"). The crawl graph knows it structurally: products are discovered
# FROM category-listing pages. These helpers turn listing URLs into category
# names so chunks can carry a "categories" metadata string retrieval anchors on.
# Path segments that introduce a category section (high-precision, marker-based).
_CAT_MARKER_SEGS = {
"collections", "collection", "category", "categories",
"product-category", "product-categories", "shop-category", "c",
}
# Segments that end a category run inside a URL (/collections/toys/products/x).
_CAT_STOP_SEGS = {"products", "product", "items", "item", "p", "page", "pages"}
# Structural slugs that are never a real category name.
_CAT_GENERIC_WORDS = {
"all", "index", "home", "default", "frontpage", "main", "page", "pages",
"shop", "store", "catalog", "catalogue", "products", "product", "items",
"item", "collections", "collection", "category", "categories", "new",
"sale", "search", "cart", "checkout", "account", "login", "wishlist",
}
def _humanize_category_slug(seg: str):
"""'travel_2' → 'travel', 'rc-toys' → 'rc toys'; None for structural/junk."""
s = (seg or "").strip().lower()
if not s or "." in s or len(s) > 60:
return None
s = re.sub(r"[_-]\d+$", "", s) # pagination/id counters: travel_2, page-3
s = re.sub(r"[-_]+", " ", s).strip()
if len(s) < 2 or len(s) > 40 or not re.search(r"[a-z]", s):
return None
if s in _CAT_GENERIC_WORDS:
return None
return s
def category_slugs_from_url(url: str) -> list:
"""Marker-based category names from a URL path (/collections/rc-toys/...,
/catalogue/category/books/travel_2/...). Empty list when no marker."""
try:
path = urllib.parse.urlparse(str(url or "")).path.lower()
except Exception:
return []
segs = [s for s in path.split("/") if s]
out, i = [], 0
while i < len(segs):
if segs[i] in _CAT_MARKER_SEGS:
j = i + 1
while j < len(segs) and segs[j] not in _CAT_MARKER_SEGS and segs[j] not in _CAT_STOP_SEGS:
h = _humanize_category_slug(segs[j])
if h:
out.append(h)
j += 1
i = j
else:
i += 1
seen, res = set(), []
for c in out:
if c not in seen:
seen.add(c)
res.append(c)
return res[:6]
def listing_slug_from_url(url: str):
"""Category name for a page KNOWN to be a listing (classified catalog/category).
Marker-based when possible; else last meaningful path segment, walking back
over pagination/index segments (…/travel_2/page-2.html → 'travel')."""
marker = category_slugs_from_url(url)
if marker:
return marker[-1]
try:
path = urllib.parse.urlparse(str(url or "")).path.lower()
except Exception:
return None
for seg in reversed([s for s in path.split("/") if s]):
if seg in _CAT_STOP_SEGS:
return None # …/product/545 is a detail page, not a listing
h = _humanize_category_slug(seg)
if h:
return h
return None
def categories_for_page(url: str, parent_urls=()) -> list:
"""Save-time categories: marker-derived from the page URL itself plus any
crawl-graph parent URLs (collection-scoped hrefs, category sections)."""
cats = list(category_slugs_from_url(url))
for p in parent_urls or ():
cats.extend(category_slugs_from_url(p))
seen, out = set(), []
for c in cats:
if c not in seen:
seen.add(c)
out.append(c)
return out[:6]
# Per-DB docs-path hints (e.g. ["/roman/", "/arabic/"]) loaded from
# databases/<name>/config.json "docs_path_hints" by app.py at crawl start.
# Module-level is safe: only one crawl runs at a time (manual guard + auto-crawl sem=1).
_DOCS_PATH_HINTS: list = []
def set_docs_path_hints(hints) -> None:
global _DOCS_PATH_HINTS
_DOCS_PATH_HINTS = [str(h).lower() for h in (hints or []) if str(h).strip()]
# A docs corpus has no products. Per-page product/catalog detection still misfires
# on non-/docs content pages (e.g. /authors, /about): a title + descriptive prose
# but no price gets shaped into a "Product:/Full specs:" card with triple-title +
# nav boilerplate, which then outranks/garbles the real answer. When this flag is
# set at crawl start for a docs-only DB, every page is treated as docs content.
_DOCS_ONLY_DB: bool = False
def set_docs_only_db(flag) -> None:
global _DOCS_ONLY_DB
_DOCS_ONLY_DB = bool(flag)
def _looks_like_docs_page(url: str, body: str = "") -> bool:
"""Heuristic for docs/tutorial/chapter pages that should not be forced into product shaping."""
u = (url or "").lower()
if any(seg in u for seg in ("/docs/", "/guide/", "/tutorial", "/chapter-", "/lesson-", "/lessons/",
"/api/", "/reference/", "/concepts/", "/overview", "/getting-started",
"/quickstart", "/changelog/", "/releases/")):
return True
if _DOCS_PATH_HINTS and any(seg in u for seg in _DOCS_PATH_HINTS):
return True
b = (body or "").lower()
_has_labels = bool(re.search(r"(?m)^(?:parameters|returns|example|usage|response|request):", b))
_has_func_sig = bool(re.search(r"def \w+\(|\w+\(\) ->", b))
_has_learning = bool(re.search(r"(?i)\b(?:learning outcomes|learning goals|by completing|by the end of|chapter|lesson|exercise|quiz)\b", b))
if _has_labels or _has_func_sig or _has_learning:
return True
# Code fences alone: only when no strong e-commerce signal is present
if "```" in b:
return not bool(re.search(r"\$[\d,.]+|\b(?:add to cart|buy now|in stock|out of stock)\b", b))
return False
def _looks_generic_title(title: str) -> bool:
cand = _canonical_product_title(title or "").strip(" -:|")
if not cand:
return True
if len(cand) < 4:
return True
if "|" in cand:
return True
if re.search(r"(?i)\b(web scraper test sites|all rights reserved|privacy policy|terms of service|home\s*\|\s*[^|]+)$", cand):
return True
low = cand.lower()
if re.search(
r"\b(?:web scraper|cloud scraper)\b.*\b(?:extension|pricing|marketplace|learn|documentation|video tutorials|test sites|forum|install|login|company|about us|contact|privacy policy|media kit|resources|blog|screenshots|status)\b",
low,
):
return True
if re.fullmatch(
r"(?:web scraper|cloud scraper|test sites|forum|documentation|video tutorials|pricing|marketplace|learn|install|login|about us|contact us)(?:\s*[-|]\s*.*)?",
low,
):
return True
if re.fullmatch(r"[\W_]+", cand):
return True
words = re.findall(r"[A-Za-z0-9]+", cand)
if len(words) <= 1 and re.match(r"(?i)^(?:black|white|blue|grey|gray|silver|gold|red|green|pink|purple|yellow|orange|brown|beige|navy|teal|lavender|maroon|violet|golden|transparent|clear|unknown|default|variant|color|colour)$", cand):
return True
return False
def _derive_page_title(title_hint: str, cleaned: str, product: dict | None = None, *, prefer_title_hint: bool = False) -> str:
candidates: list[str] = []
body = cleaned or ""
ordered_candidates = [title_hint or "", (product or {}).get("title") or "", (product or {}).get("canonical_title") or ""]
if not prefer_title_hint:
for pat in (
r"(?m)^\s*(?:##|###)\s+(.+?)\s*$",
r"(?i)\b(?:name|title|product)\s*:\s*([^\n\.]{4,180})",
r"(?m)^(?!https?://)([A-Z][^\n]{4,120})$",
):
for mm in re.finditer(pat, body):
cand = _canonical_product_title((mm.group(1) or "").strip())
if cand and cand not in candidates:
candidates.append(cand)
break
for cand in ordered_candidates:
cand = _canonical_product_title(str(cand or "")).strip(" -:|")
if cand and cand not in candidates:
candidates.append(cand)
if prefer_title_hint:
for pat in (
r"(?m)^\s*(?:##|###)\s+(.+?)\s*$",
r"(?i)\b(?:name|title|product)\s*:\s*([^\n\.]{4,180})",
r"(?m)^(?!https?://)([A-Z][^\n]{4,120})$",
):
for mm in re.finditer(pat, body):
cand = _canonical_product_title((mm.group(1) or "").strip())
if cand and cand not in candidates:
candidates.append(cand)
break
non_generic = [c for c in candidates if not _looks_generic_title(c)]
if non_generic:
candidates = non_generic + [c for c in candidates if c not in non_generic]
return candidates[0] if candidates else ""
_DOCS_DB_TYPES = {"docs", "documentation", "text", "rag", "knowledge", "kb", "corpus"}
def _check_is_product_db(db, db_name: str = "", cfg: dict | None = None) -> bool:
"""Return True if the DB collection looks like a product/catalog DB.
Result cached per db_name so the sample query runs at most once per DB.
Config authority (mirrors catalog_query.is_catalog_db) runs BEFORE the chunk
sample: an explicit flag / docs db_type / docs_path_hints marks a docs corpus,
so a stray product-tagged enrichment chunk can't flip a docs DB to product
(which wrongly fires the product-retry/live-rescue that clobbers docs context)."""
cfg = cfg or {}
_flag = cfg.get("is_product_db")
if _flag is None:
_flag = cfg.get("product_db")
if _flag is True:
return True
if _flag is False:
return False
_db_type = str(cfg.get("db_type") or cfg.get("mode") or cfg.get("catalog_mode") or "").lower().strip()
if _db_type in _DOCS_DB_TYPES:
return False
if cfg.get("docs_path_hints"):
return False
if db_name and db_name in _product_db_cache:
return _product_db_cache[db_name]
result = False
if db:
try:
# Authoritative + insertion-order independent: does ANY chunk carry
# product metadata? The 24-chunk head sample missed product pages on
# large DBs (tsc_pk's first chunks are home/policy/nav), so is_product_db
# flipped False per-container and silently disabled the deterministic
# product answerers.
for _w in ({"chunk_kind": "product"}, {"content_type": "product"}):
try:
_hit = db._collection.get(where=_w, limit=1)
if (_hit.get("ids") or _hit.get("documents")):
result = True
break
except Exception:
pass
if result:
if db_name:
_product_db_cache[db_name] = True
return True
sample = db._collection.get(limit=24, include=["documents", "metadatas"])
docs = sample.get("documents") or []
metas = sample.get("metadatas") or []
for i, m in enumerate(metas):
m = m or {}
text = str(docs[i] if i < len(docs) else "" or "")
source = str(m.get("source") or "").lower()
if m.get("price") is not None or m.get("ram_gb") is not None or m.get("gpu_vram_gb") is not None:
result = True
break
if m.get("content_type") == "product":
result = True
break
if re.search(r"/(?:products?|items?)(?:/|$|#)", source) or "/collections/" in source:
result = True
break
tl = text.lower()
if ("rs." in tl or "pkr" in tl or "$" in tl or "£" in tl or "€" in tl) and ("add to cart" in tl or "shopping cart" in tl):
result = True
break
except Exception:
pass
# Only cache True: a False result may be stale (DB was empty or pre-crawl when sampled).
if db_name and result:
_product_db_cache[db_name] = result
return result
_PRODUCT_QUERY_STOP = {
"what", "is", "the", "price", "pricing", "cost", "of", "for", "a", "an", "item", "product",
"products", "much", "how", "does", "do", "you", "have", "tell", "me", "about", "details",
}
def _product_query_rerank_score(question: str, doc) -> float:
source = str(((getattr(doc, "metadata", None) or {}).get("source")) or "")
text = str(getattr(doc, "page_content", "") or "")
if not source and not text:
return 0.0
q_tokens = {
t for t in re.findall(r"[a-z0-9]+", (question or "").lower())
if len(t) >= 2 and t not in _PRODUCT_QUERY_STOP
}
source_tokens = set(re.findall(r"[a-z0-9]+", source.lower().replace("-", " ").replace("_", " ")))
head_tokens = set(re.findall(r"[a-z0-9]+", text[:500].lower()))
combined_tokens = source_tokens | head_tokens
overlap = len(q_tokens & combined_tokens)
slug_overlap = len(q_tokens & source_tokens)
head_overlap = len(q_tokens & head_tokens)
q_numbers = set(re.findall(r"\b\d+\b", question or ""))
doc_numbers = set(re.findall(r"\b\d+\b", f"{source} {text[:500]}"))
numbers_hit = len(q_numbers & doc_numbers)
normalized_q = re.sub(r"\s+", " ", re.sub(r"[^a-z0-9]+", " ", (question or "").lower())).strip()
normalized_doc = re.sub(r"\s+", " ", re.sub(r"[^a-z0-9]+", " ", f"{source} {text[:500]}".lower())).strip()
score = overlap + (slug_overlap * 3.0) + (head_overlap * 1.25) + (numbers_hit * 2.0)
if normalized_q and normalized_q in normalized_doc:
score += 8.0
if re.search(r"\b(price|pricing|cost)\b", question or "", re.I) and re.search(r"(?i)\b(?:\brs\.?|\bpkr|\$|£|€)\s*[\d,]+", text[:500]):
score += 1.0
return score
def _extract_product_summary(text: str, url: str, title_hint: str = "", authority_title: str = "", authority_price: float = 0.0, authority_currency: str = "Rs.") -> dict:
import html as _html_mod
title_hint = _html_mod.unescape(title_hint or "") # "Toy &ndash; Babyfy" → "Toy – Babyfy"
body = _strip_storefront_boilerplate(text or "")
used_structured_fields: list[str] = []
body_fallback_used = False
def _canonicalize_title(candidate: str) -> str:
import html as _html
# Shopify <title> tags arrive entity-encoded ("Toy &ndash; Babyfy") — unescape
# first so the "– Site" suffix splitter and dedup actually see the dash.
return _canonical_product_title(_html.unescape(candidate or "")).strip(" -:|")
def _is_generic_title(candidate: str) -> bool:
cand = _canonicalize_title(candidate)
if not cand:
return True
if len(cand) < 4:
return True
# A price is not a product name ("Rs.1", "$24", "£51.77")
if re.fullmatch(r'(?i)(?:rs\.?|pkr|\$|£|€)\s*[\d.,]*\s*', cand):
return True
if "|" in cand or "web scraper test sites" in cand.lower():
return True
# Spec-table rows are never product names ("Price (incl. tax) £51.77 Tax £0.00 ...")
if re.search(r'(?i)\bprice\s*\(|[\$£€]\s*\d|\b(?:incl|excl)\.\s*tax\b|\bavailability\b|\bin\s+stock\b|\bnumber\s+of\s+reviews?\b', cand):
return True
# Variant-swatch / price-fragment shapes are never product names:
# "PKR Pink - Rs.1", "PKR Red - Sold Out", "Diecast Model Ducati Diavel Rs.4".
# A standalone currency CODE as the first word is swatch text (a real title
# like "The $100 Startup" has the symbol glued to digits, not "PKR <word>").
if re.match(r'(?i)^(?:rs\.?|pkr|usd|eur|gbp|aed)\s', cand):
return True
if re.search(r'(?i)\bsold\s*out\b', cand):
return True
# Trailing currency fragment ("… Rs.4", "… - Rs.1,2") = truncated price tail.
if re.search(r'(?i)[\s\-–](?:rs\.?|pkr|\$|£|€)\s*[\d.,]*$', cand):
return True
# Truncated card text from the source site itself ("Teach children...").
if cand.endswith("...") or re.search(r'(?i)\bloading\.{0,3}$|\btranslation\s+missing\b', cand):
return True
if re.fullmatch(r"[\W_]+", cand):
return True
words = re.findall(r"[A-Za-z0-9]+", cand)
if len(words) <= 1 and re.match(r"(?i)^(?:black|white|blue|grey|gray|silver|gold|red|green|pink|purple|yellow|orange|brown|beige|navy|teal|lavender|maroon|violet|golden|transparent|clear|unknown|default|variant|color|colour)$", cand):
return True
return False
def _dedupe_repeated_phrase(text: str) -> str:
toks = [t for t in re.split(r"\s+", (text or "").strip()) if t]
if len(toks) >= 4 and len(toks) % 2 == 0:
half = len(toks) // 2
if toks[:half] == toks[half:]:
return " ".join(toks[:half]).strip(" -:|")
return (text or "").strip(" -:|")
def _score_title(candidate: str) -> tuple[int, int, int]:
cand = _canonicalize_title(candidate)
if not cand:
return (0, 0, 0, -10)
words = re.findall(r"[A-Za-z0-9]+", cand)
generic = 1 if _is_generic_title(cand) else 0
has_modelish_token = 1 if re.search(r"(?:\d|[A-Z]{2,}\d|\d+[A-Za-z][A-Za-z0-9\-]*)", cand) else 0
penalty = 0
if generic:
penalty += 3
if len(words) < 2:
penalty += 1
# Prefer non-generic product names first; among those, prefer titles
# with model-like tokens and then the richer titles.
return (1 - generic, has_modelish_token, len(words), len(cand), -penalty)
title_candidates: list[str] = []
def _push_title(candidate: str, label: str) -> None:
cand = _canonicalize_title(candidate)
if not cand:
return
if cand not in title_candidates:
title_candidates.append(cand)
if label:
used_structured_fields.append(label)
def _title_from_body_line(line: str) -> str:
raw = re.sub(r"\s+", " ", (line or "")).strip(" -:|")
if not raw:
return ""
raw = re.sub(r'(?i)^(?:product|name|title|model)\s*:\s*', "", raw).strip(" -:|")
# Prefer the model-like phrase immediately after a price marker.
for pm in re.finditer(r'(?i)(?:\$|£|€|\brs\.?|\bpkr)\s*[\d,]+(?:\.\d{1,2})?\s+((?-i:[A-Z0-9]).{3,119})', raw):
tail = pm.group(1).strip(" -:|")
tail = re.split(
r'(?i)\s+(?:hdd:|ssd:|ram:|processor:|display:|os:|availability:|reviews?|'
r'add to cart|wishlist|toggle navigation|cloud scraper|pricing|marketplace|'
r'learn documentation|video tutorials|test sites|forum|contact us|copyright|description:)\b',
tail,
)[0].strip(" -:|")
# Spec-table tail, not a product name ("£51.77 In stock (22 available)...")
if re.match(r'(?i)^(?:in\s+stock|out\s+of\s+stock|tax\b|availability|number\s+of|qty|quantity|customer|reviews?)\b', tail):
continue
# Sentence boundary inside the tail → prose fragment from a description
# ("$500 million were stolen from the Museum. It remains..."), not a title.
if ". " in tail[:90]:
continue
tail = tail.split(",")[0].strip(" -:|")
tail = _dedupe_repeated_phrase(tail)
tail_candidates = []
if tail:
tail_candidates.append(_canonicalize_title(tail))
tail_spans = re.findall(
r'([A-Z][A-Za-z0-9&\'"()\-]+(?:\s+[A-Z0-9][A-Za-z0-9&\'"()\-]+){1,8})',
tail,
)
tail_candidates.extend(s.strip(" -:|") for s in tail_spans if s and len(s.strip()) <= 90)
tail_candidates = [s for s in tail_candidates if s and len(s) <= 120]
tail_candidates = [s for s in tail_candidates if not _is_generic_title(s)]
cand = max(tail_candidates, key=_score_title) if tail_candidates else _canonicalize_title(tail)
if cand and not _is_generic_title(cand):
return cand
# Strip obvious spec / boilerplate suffixes from generic lines.
raw = re.split(
r'(?i)\s+(?:hdd:|ssd:|ram:|processor:|display:|os:|availability:|reviews?|'
r'add to cart|wishlist|toggle navigation|cloud scraper|pricing|marketplace|'
r'learn documentation|video tutorials|test sites|forum|contact us|copyright|description:)\b',
raw,
)[0].strip(" -:|")
raw = raw.split(",")[0].strip(" -:|")
# On long mixed lines, prefer the longest capitalized product-like span.
spans = re.findall(
r'([A-Z][A-Za-z0-9&\'"()\-]+(?:\s+[A-Z0-9][A-Za-z0-9&\'"()\-]+){1,8})',
raw,
)
spans = [s.strip(" -:|") for s in spans if s and len(s.strip()) <= 90]
spans = [s for s in spans if not _is_generic_title(s)]
if spans:
return max(spans, key=_score_title)
return _canonicalize_title(raw).strip(" -:|")
def _price_tail_title(text: str) -> str:
raw = re.sub(r"\s+", " ", (text or "")).strip(" -:|")
if not raw:
return ""
raw = re.sub(r'(?i)^(?:product|name|title|model)\s*:\s*', "", raw).strip(" -:|")
for pm in re.finditer(r'(?i)(?:\$|£|€|\brs\.?|\bpkr)\s*[\d,]+(?:\.\d{1,2})?\s+((?-i:[A-Z0-9]).{3,119})', raw):
tail = pm.group(1).strip(" -:|")
tail = re.split(
r'(?i)\s+(?:hdd:|ssd:|ram:|processor:|display:|os:|availability:|reviews?|'
r'add to cart|wishlist|toggle navigation|cloud scraper|pricing|marketplace|'
r'learn documentation|video tutorials|test sites|forum|contact us|copyright|description:)\b',
tail,
)[0].strip(" -:|")
# Spec-table tail, not a product name ("£51.77 In stock (22 available)...")
if re.match(r'(?i)^(?:in\s+stock|out\s+of\s+stock|tax\b|availability|number\s+of|qty|quantity|customer|reviews?)\b', tail):
continue
# Sentence boundary inside the tail → prose fragment from a description
# ("$500 million were stolen from the Museum. It remains..."), not a title.
if ". " in tail[:90]:
continue
tail = tail.split(",")[0].strip(" -:|")
tail = _dedupe_repeated_phrase(tail)
tail_candidates = []
if tail:
tail_candidates.append(_canonicalize_title(tail))
tail_spans = re.findall(
r'([A-Z][A-Za-z0-9&\'"()\-]+(?:\s+[A-Z0-9][A-Za-z0-9&\'"()\-]+){1,8})',
tail,
)
tail_candidates.extend(s.strip(" -:|") for s in tail_spans if s and len(s.strip()) <= 90)
tail_candidates = [s for s in tail_candidates if s and len(s) <= 120]
tail_candidates = [s for s in tail_candidates if not _is_generic_title(s)]
if tail_candidates:
cand = max(tail_candidates, key=_score_title)
if cand and not _is_generic_title(cand):
return cand
return ""
# Prefer a body-derived title when the page title is generic or boilerplate.
price_title = _price_tail_title(body)
if price_title:
_push_title(price_title, "price_title")
body_lines = [re.sub(r"\s+", " ", ln).strip(" -:|") for ln in re.split(r"[\r\n]+", body) if len(ln.strip()) >= 4]
for ln in body_lines[:180]:
cand = _title_from_body_line(ln)
if not cand or len(cand) > 120:
continue
if _is_generic_title(cand):
continue
if re.search(r'(?i)\b(?:add to cart|wishlist|reviews?|toggle navigation|cloud scraper|pricing|marketplace|learn documentation|video tutorials|test sites|forum|privacy policy|terms of service|all rights reserved)\b', cand):
continue
if re.search(r'(?i)^\s*(?:\brs\.?|\bpkr|\$|£|€)\s*[\d,]+', cand):
continue
_push_title(cand, "body_title")
break
title_match = re.search(r'(?i)\bname:\s*([^\n\.]{4,180})', body)
if title_match:
_push_title(title_match.group(1), "name")
if title_hint:
_push_title(title_hint, "title_hint")
# "<Product Name> | <Site Name>" is the dominant title-tag convention —
# the bare hint dies on the generic-title "|" rule, so also push the
# first segment ("A Light in the Attic | Books to Scrape" → product name).
_hint_head = re.split(r'\s*[|–—]\s*| - ', title_hint, maxsplit=1)[0].strip()
if _hint_head and _hint_head.lower() != title_hint.strip().lower():
_push_title(_hint_head, "title_hint_head")
slug = (url or "").rstrip("/").split("/")[-1]
_push_title(re.sub(r'[-_]+', ' ', slug).strip().title(), "slug")
# If the best title is still generic, try to salvage a better one from the
# early body lines before falling back to the weak variant.
if title_candidates:
title = max(title_candidates, key=_score_title)
# The page's own <title> head is authoritative: if the scored winner is just
# the hint head with extra LEADING junk ("Sandbox Sharp Objects" vs
# "Sharp Objects"), prefer the clean hint head.
if title_hint:
_hh = _canonicalize_title(re.split(r'\s*[|–—]\s*| - ', title_hint, maxsplit=1)[0].strip())
if (_hh and not _is_generic_title(_hh)
and title.lower() != _hh.lower()
and (title.lower().endswith(" " + _hh.lower())
or re.match(re.escape(_hh.lower()) + r'\s*[–—|-]', title.lower()))):
title = _hh
# URL-slug agreement: when the hint head matches the page's own slug
# ("into-the-wild" ≈ "Into the Wild") it IS the product name — prefer it
# over breadcrumb-mangled winners ("Sandbox Home Books Nonfiction Into").
if _hh and not _is_generic_title(_hh) and title.lower() != _hh.lower():
_slug_seg = re.sub(r'\.html?$', '', re.sub(r'(?i)/index\.html?$', '', urllib.parse.urlparse(str(url)).path.rstrip('/')).rsplit('/', 1)[-1])
_slug_norm = re.sub(r'[\s\-_]*\d+$', '', re.sub(r'[^a-z0-9]+', ' ', _slug_seg.lower()).strip()).strip()
_hh_norm = re.sub(r'[^a-z0-9]+', ' ', _hh.lower()).strip()
if len(_slug_norm) >= 6 and _hh_norm and (_hh_norm.startswith(_slug_norm) or _slug_norm.startswith(_hh_norm)):
title = _hh
if _is_generic_title(title):
body_candidates: list[str] = []
for pat in (
r'(?i)\bfull specs?\s*:\s*([^\n]+)',
r'(?i)\bdescription\s*:\s*([^\n]+)',
):
mm = re.search(pat, body)
if not mm:
continue
cand = _canonicalize_title((mm.group(1) or "").split(",")[0])
if not cand or len(cand) > 120:
continue
if _is_generic_title(cand):
continue
if re.search(r'(?i)\b(?:price|availability|add to cart|wishlist|reviews?)\b', cand):
continue
if re.search(r'(?i)\b(?:\brs\.?|\bpkr|\$|£|€)\s*[\d,]+', cand):
continue
body_candidates.append(cand)
for ln in body_lines[:220]:
cand = _title_from_body_line(ln)
if not cand or len(cand) > 120:
continue
if _is_generic_title(cand):
continue
body_candidates.append(cand)
for cand in body_candidates:
_push_title(cand, "body_title")
non_generic_titles = [cand for cand in title_candidates if not _is_generic_title(cand)]
if non_generic_titles:
title = max(non_generic_titles, key=_score_title)
else:
title = max(title_candidates, key=_score_title)
else:
title = ""
# On storefront pages, a valid price-tail title is the strongest signal and
# should win over generic site chrome or breadcrumb noise.
if price_title and not _is_generic_title(price_title):
title = price_title
price_label = ""
price_num = None
if authority_price and authority_price > 0:
# og:price:amount / JSON-LD offer price — the storefront's own declared
# price, carried as a NUMBER through every text transform. Body regexes
# mis-fire on model names ("DJI RS 2" → Rs.2) and piece counts
# ("Colors 24" → 24); the authority is immune and never wiped.
price_num = float(authority_price)
_ap_disp = f"{price_num:,.2f}".rstrip("0").rstrip(".")
price_label = f"{authority_currency or 'Rs.'}{_ap_disp}"
used_structured_fields.append("price_authority")
elif authority_price and authority_price < 0:
# Storefront declared price 0 (OOS placeholder) — no price exists on
# this page; regex extraction would only find junk.
pass
else:
pm = _PRODUCT_PRICE_LINE_RE.search(body) or _PRODUCT_PRICE_CAPTURE_RE.search(body)
if pm:
if pm.re is _PRODUCT_PRICE_LINE_RE:
currency = (pm.group(1) or "").strip() or "Rs."
digits = pm.group(2)
else:
whole = pm.group(0)
digits = pm.group(1)
currency = whole.replace(digits, "").strip() or "Rs."
price_label = f"{currency}{digits}"
used_structured_fields.append("price")
try:
price_num = float(digits.replace(",", ""))
except Exception:
price_num = None
if price_num is not None and price_num <= 0:
# "Rs.0.00" = OOS placeholder / cart remnant, never a price.
price_label = ""
price_num = None
used_structured_fields.remove("price")
avail = ""
# Authoritative: the product's OWN schema.org availability emitted by structured
# extraction ("Avail: .../InStock", "Availability: .../OutOfStock"). camelCase
# InStock has no space, so the loose phrase scan below can't see it — and would
# otherwise grab the FIRST stray "Out of Stock" on the page (a variant badge or a
# JS-rendered related-product card), mis-flagging an in-stock product as OOS.
sm = re.search(r'(?i)\bavail(?:ability)?:\s*(\S+)', body)
if sm:
tok = sm.group(1).lower()
if re.search(r'outofstock|out[_\s]?of[_\s]?stock|soldout|sold[_\s]?out|discontinued|backorder', tok):
avail = "out of stock"
elif re.search(r'instock|in[_\s]?stock|preorder|pre[_\s-]?order|available', tok):
avail = "available"
if avail:
used_structured_fields.append("availability")
if not avail:
am = _PRODUCT_AVAIL_RE.search(body)
if am:
avail = am.group(1).strip()
used_structured_fields.append("availability")
desc = ""
dm = re.search(r'(?i)\bdescription:\s*([^\n]{20,600})', body)
if not dm:
dm = re.search(r'(?m)(?i)^description\s*$\n([^\n]{20,600})', body)
if dm:
desc = dm.group(1).strip()
used_structured_fields.append("description")
if not desc:
body_fallback_used = True
sentences = re.split(r'(?<=[.!?])\s+', body)
useful = []
for sent in sentences:
s = sent.strip()
if len(s) < 25:
continue
if re.search(r'(?i)\b(add to cart|wishlist|recently viewed|you may also like|customers also bought|checkout|subtotal)\b', s):
continue
# A price inside a description sentence is related-product carousel
# leakage ("Number Book: Teach children... Rs.1,050.00 PKR"), not prose —
# the product's own price already lives in price_label. Same for
# truncated card text ("...") and loading/i18n widget artifacts.
if re.search(r'(?i)(?:\brs\.?|\bpkr\b|\$|£|€)\s*[\d,]+|\d[\d,]*\s*(?:pkr|rs\.?|usd|eur|gbp)\b', s):
continue
if "..." in s or re.search(r'(?i)\bloading\b|\btranslation\s+missing\b', s):
continue
# Shipping/delivery chrome repeats on every product page — never prose.
if re.search(r'(?i)\b(?:free (?:delivery|shipping)|delivery summary|order placed|dispatched|estimated delivery|cash on delivery)\b', s):
continue
useful.append(s)
if len(" ".join(useful)) >= 420:
break
desc = " ".join(useful[:3]).strip()
contaminated = bool(_CONTAMINATION_HINTS_RE.search(body))
# Final authority: a title-tag head matching the URL slug IS the product name.
# Descriptions naming OTHER priced products ("The $100 Startup") can otherwise
# hijack both the title and its paired price_label.
try:
_hh_fin = _canonicalize_title(re.split(r'\s*[|–—]\s*| - ', title_hint or '', maxsplit=1)[0].strip())
_path_fin = re.sub(r'(?i)/index\.html?$', '', urllib.parse.urlparse(str(url)).path.rstrip('/'))
_slug_fin = re.sub(r'[\s\-_]*\d+$', '', re.sub(r'[^a-z0-9]+', ' ', re.sub(r'\.html?$', '', _path_fin.rsplit('/', 1)[-1]).lower()).strip()).strip()
_hh_fin_n = re.sub(r'[^a-z0-9]+', ' ', _hh_fin.lower()).strip()
if (len(_slug_fin) >= 6 and _hh_fin_n and title
and title.strip().lower() != _hh_fin.lower()
and (_hh_fin_n.startswith(_slug_fin) or _slug_fin.startswith(_hh_fin_n))):
title = _hh_fin
if "price_authority" not in used_structured_fields:
price_label = "" # stale pair from hijacked title — chunker fallback re-derives
price_num = None
except Exception:
pass
# ABSOLUTE title authority: og:title / JSON-LD Product.name is the page's
# own declaration of its product name. Variant-picker text ("Single Piece
# Black", "15ml / BLACK") is structurally indistinguishable from a name —
# shape heuristics cannot flag it; the authoritative source always wins.
if authority_title:
try:
_auth = _canonicalize_title(re.split(r'\s*[|–—]\s*| - ', authority_title, maxsplit=1)[0].strip())
if _auth and len(_auth) <= 120 and not _is_generic_title(_auth) and title.strip().lower() != _auth.lower():
title = _auth
except Exception:
pass
canonical_title = _canonicalize_title(title)
return {
"title": title.strip(),
"canonical_title": canonical_title,
"price_label": price_label.strip(),
"price_num": price_num,
"availability": avail,
"description": desc[:900],
"contaminated": contaminated,
"used_structured_fields": list(dict.fromkeys(used_structured_fields)),
"body_fallback_used": body_fallback_used,
}
def _classify_page_type(url: str, cleaned: str, product_like: bool, structural: bool, docs_like: bool = False) -> str:
source = (url or "").lower()
body = cleaned or ""
metrics = _trusted_content_metrics(body)
if docs_like:
if structural:
return "structural"
if re.search(r"(?i)\b(by completing|by the end of this (?:chapter|lesson)|you will be able to|learning outcomes|objectives|goals)\b", body):
return "article"
if metrics["prose_chars"] >= 160 or metrics["sentence_count"] >= 2:
return "article"
return "structural" if metrics["nav_hits"] >= 6 else "article"
if _looks_like_catalog_page(source, body):
return "catalog"
if product_like:
return "product"
if structural:
return "structural"
# Outcomes/learning-goals marker override (universal):
# Many docs index pages look "category/structural" but still contain the
# canonical "By completing..., you will:" goals list. Never classify those
# as category-only content.
if re.search(r"(?i)\b(by completing|by the end of this (?:chapter|lesson)|you will be able to|learning outcomes|objectives|goals)\b", body):
return "article"
if len(_GENERIC_SECTION_SPLIT_RE.split(body)) >= 3:
return "faq"
if _CATEGORY_URL_RE.search(source) or (metrics["category_hits"] >= 2 and metrics["sentence_count"] <= 3):
return "category"
if _ARTICLE_URL_RE.search(source) or (
metrics["prose_chars"] > 300 and metrics["sentence_count"] >= 2 and metrics["nav_hits"] <= max(2, metrics["sentence_count"])
):
return "article"
if _POLICY_URL_RE.search(source):
return "policy"
if _POLICY_TEXT_RE.search(body) and metrics["policy_hits"] >= 2 and metrics["prose_chars"] < 1400 and metrics["sentence_count"] <= 8:
return "policy"
return "unknown"
def _quality_score(
cleaned: str,
*,
page_type: str,
used_structured_fields: list[str] | None = None,
body_fallback_used: bool = False,
structural: bool = False,
contaminated: bool = False,
had_boilerplate: bool = False,
docs_like: bool = False,
) -> float:
score = 0.0
used_structured_fields = used_structured_fields or []
metrics = _trusted_content_metrics(cleaned)
if used_structured_fields:
score += 0.4
if any(f in used_structured_fields for f in ("price", "availability", "description")):
score += 0.2
if metrics["prose_chars"] > 200:
score += 0.2
if not had_boilerplate:
score += 0.2
if body_fallback_used:
score -= 0.1
if structural or contaminated:
score = min(score, 0.2)
if docs_like and page_type in {"article", "structural"}:
score = max(score, 0.35)
if metrics["prose_chars"] > 120:
score += 0.15
if metrics["sentence_count"] >= 2:
score += 0.1
if not had_boilerplate:
score += 0.05
if page_type == "unknown":
score = min(score, 0.49)
return round(max(0.0, min(1.0, score)), 2)
def _page_classifier_confidence(
cleaned: str,
*,
page_type: str,
product_like: bool = False,
structural: bool = False,
used_structured_fields: list[str] | None = None,
body_fallback_used: bool = False,
docs_like: bool = False,
) -> float:
used_structured_fields = used_structured_fields or []
metrics = _trusted_content_metrics(cleaned)
if page_type == "product":
conf = 0.55
if product_like:
conf += 0.15
conf += min(0.2, 0.05 * len(used_structured_fields))
if body_fallback_used:
conf -= 0.1
return round(max(0.2, min(0.98, conf)), 2)
if page_type == "structural":
conf = 0.6 if structural else 0.45
if metrics["nav_hits"] >= 8:
conf += 0.15
return round(max(0.2, min(0.98, conf)), 2)
if page_type == "faq":
conf = 0.7 if len(_GENERIC_SECTION_SPLIT_RE.split(cleaned)) >= 3 else 0.5
return round(max(0.2, min(0.98, conf)), 2)
if page_type == "policy":
conf = 0.75 if metrics["policy_hits"] else 0.55
return round(max(0.2, min(0.98, conf)), 2)
if page_type == "category":
conf = 0.65 if metrics["category_hits"] else 0.5
return round(max(0.2, min(0.98, conf)), 2)
if page_type == "catalog":
conf = 0.68 if metrics["category_hits"] or metrics["prose_chars"] > 300 else 0.52
if metrics["sentence_count"] >= 3:
conf += 0.08
return round(max(0.2, min(0.98, conf)), 2)
if page_type == "article":
conf = 0.55
if metrics["prose_chars"] > 300:
conf += 0.15
if metrics["sentence_count"] >= 3:
conf += 0.1
return round(max(0.2, min(0.98, conf)), 2)
if docs_like:
conf = 0.72
if metrics["prose_chars"] > 200:
conf += 0.08
if metrics["sentence_count"] >= 2:
conf += 0.05
return round(max(0.2, min(0.98, conf)), 2)
return 0.35
def _prepare_crawl_page(text: str, url: str, title_hint: str = "", authority_title: str = "", authority_price: float = 0.0, authority_currency: str = "Rs.") -> tuple[str, dict]:
import html as _html_mod
# Hints arrive entity-encoded from raw <title> regex extraction; they feed
# page_title and the chunker's "## " heading lines downstream of every
# body-text unescape, so decode here or "&ndash;" persists into chunks.
title_hint = _html_mod.unescape(title_hint or "")
authority_title = _html_mod.unescape(authority_title or "")
raw = _clean_text(text or "")
docs_like = True if _DOCS_ONLY_DB else _looks_like_docs_page(url, raw)
catalog_like = False if docs_like else _looks_like_catalog_page(url, raw)
product_like = False if docs_like or catalog_like else _looks_like_product_page(url, raw)
cleaned = _strip_storefront_boilerplate(raw) if product_like else _dedupe_repeated_lines(raw)
cleaned = _clean_text(cleaned)
had_boilerplate = cleaned != raw and bool(_BOILERPLATE_SIGNAL_RE.search(raw))
structural = _looks_structural_page(url, cleaned)
page_type = _classify_page_type(url, cleaned, product_like, structural, docs_like=docs_like)
now_iso = datetime.now(timezone.utc).isoformat()
content_hash = hashlib.sha256(cleaned.encode("utf-8", errors="ignore")).hexdigest() if cleaned else ""
meta = {
"structural": structural,
"page_type": page_type,
"docs_like": docs_like,
"contaminated": False,
"used_structured_fields": [],
"body_fallback_used": False,
"had_boilerplate": had_boilerplate,
"extraction_mode": "body_text",
"crawled_at": now_iso,
"last_verified_at": now_iso,
"source_status": "live",
"content_hash": content_hash,
}
# Persist the hints so the flush-time boilerplate scrub can re-run this
# function on scrubbed text with the same inputs (HTML is gone by then).
meta["title_hint"] = (title_hint or "")[:300]
meta["authority_title"] = (authority_title or "")[:300]
meta["authority_price"] = float(authority_price or 0.0)
meta["authority_currency"] = (authority_currency or "Rs.")[:8]
_dt_hint = authority_title or title_hint
# Docs pages should derive their title from the actual page text first.
# Generic site chrome titles are a common failure mode on docs-heavy sites,
# so only treat the title hint as a fallback signal here.
meta["page_title"] = _derive_page_title(_dt_hint, cleaned, prefer_title_hint=docs_like)
meta["catalog_listing"] = False
if catalog_like:
meta["catalog_listing"] = True
meta["page_type"] = "catalog"
meta["page_title"] = _derive_page_title(_dt_hint, cleaned, prefer_title_hint=docs_like)
if product_like:
product = _extract_product_summary(cleaned, url, title_hint=title_hint, authority_title=authority_title, authority_price=authority_price, authority_currency=authority_currency)
meta["product"] = product
meta["contaminated"] = bool(product.get("contaminated"))
meta["used_structured_fields"] = list(product.get("used_structured_fields") or [])
meta["body_fallback_used"] = bool(product.get("body_fallback_used"))
meta["extraction_mode"] = "structured_product" if meta["used_structured_fields"] else "body_text"
if product.get("title") and (product.get("price_label") or product.get("description") or meta["used_structured_fields"]):
meta["page_type"] = "product"
meta["page_title"] = _derive_page_title(_dt_hint, cleaned, product, prefer_title_hint=docs_like)
elif not docs_like and meta["page_type"] in {"policy", "unknown", "category", "article"} and not _url_never_product(url):
# Product/catalog pages often carry policy/footer boilerplate that can
# overwhelm the classifier. If structured product signals are present,
# promote them even when the URL path itself is generic.
# GUARD: never promote the site root or info/account/policy pages — on
# WooCommerce/WordPress the global cart+price chrome makes _extract_product_summary
# find a "title+price" on About/Contact/Policy pages, mis-tagging them product.
product = _extract_product_summary(cleaned, url, title_hint=title_hint, authority_title=authority_title, authority_price=authority_price, authority_currency=authority_currency)
multiple_price_hits = len(_PRODUCT_PRICE_CAPTURE_RE.findall(cleaned)) + len(_PRODUCT_PRICE_LINE_RE.findall(cleaned))
if product.get("title") and multiple_price_hits >= 2:
meta["catalog_listing"] = True
meta["catalog_item_count"] = multiple_price_hits
meta["page_type"] = "catalog"
meta["page_title"] = _derive_page_title(_dt_hint, cleaned, product)
elif product.get("title") and (product.get("price_label") or product.get("description") or product.get("used_structured_fields")):
meta["product"] = product
meta["page_type"] = "product"
meta["contaminated"] = bool(product.get("contaminated"))
meta["used_structured_fields"] = list(product.get("used_structured_fields") or [])
meta["body_fallback_used"] = bool(product.get("body_fallback_used"))
meta["extraction_mode"] = "structured_product" if meta["used_structured_fields"] else "body_text"
meta["page_title"] = _derive_page_title(_dt_hint, cleaned, product, prefer_title_hint=docs_like)
else:
meta["page_title"] = _derive_page_title(_dt_hint, cleaned, prefer_title_hint=docs_like)
meta["quality_score"] = _quality_score(
cleaned,
page_type=meta["page_type"],
used_structured_fields=meta.get("used_structured_fields") or [],
body_fallback_used=bool(meta.get("body_fallback_used")),
structural=bool(meta.get("structural")),
contaminated=bool(meta.get("contaminated")),
had_boilerplate=bool(meta.get("had_boilerplate")),
docs_like=docs_like,
)
meta["page_classifier_confidence"] = _page_classifier_confidence(
cleaned,
page_type=meta["page_type"],
product_like=product_like,
structural=bool(meta.get("structural")),
used_structured_fields=meta.get("used_structured_fields") or [],
body_fallback_used=bool(meta.get("body_fallback_used")),
docs_like=docs_like,
)
quarantine_reason = ""
retrieve_eligible = True
if re.search(r'/(?:cart|checkout|account|login|register|logout|orders?|addresses|wishlist|search)(?:/|$|\?|#)', url or "", re.I):
# Storefront utility pages (cart/checkout/account/login/search) are pure
# chrome, never content. A populated cart page ("Your cart is currently
# empty … Top Selling") clears the category word-count gate below, so
# guard by URL FIRST — and however the page entered the frontier
# (sitemap/seed URLs bypass the discovery junk-scorer that lists 'cart').
retrieve_eligible = False
quarantine_reason = "utility_page"
elif meta["page_type"] in {"structural", "category"}:
# Large structural/category pages are often the canonical overview pages
# for doc sets. Keep docs-like overviews searchable even when they are
# compact, because many chapter/outcomes pages are short but meaningful.
_wc = len(cleaned.split())
_heading_hits = len(re.findall(r"(?m)^\s*(?:##|###)\s+", cleaned))
_min_docs_words = 40 if docs_like else 150
if docs_like and (_wc >= 25 or _heading_hits >= 1 or len(_GENERIC_SECTION_SPLIT_RE.split(cleaned)) >= 2):
retrieve_eligible = True
quarantine_reason = ""
elif _wc > _min_docs_words:
retrieve_eligible = True
quarantine_reason = ""
else:
retrieve_eligible = False
quarantine_reason = meta["page_type"]
elif meta.get("contaminated"):
retrieve_eligible = False
quarantine_reason = "contaminated"
elif meta["page_type"] == "product" and meta.get("body_fallback_used") and len(meta.get("used_structured_fields") or []) < 2:
retrieve_eligible = False
meta["quality_score"] = min(float(meta.get("quality_score") or 0.0), 0.35)
quarantine_reason = "weak_product_fallback"
elif meta["page_type"] in {"product", "catalog", "unknown"} and float(meta.get("quality_score") or 0.0) < 0.5:
# Article/faq/policy pages have no structured fields so their score
# is capped at 0.4 by design — don't quarantine them on score alone.
retrieve_eligible = False
quarantine_reason = "low_quality"
meta["retrieve_eligible"] = retrieve_eligible
meta["quarantine_reason"] = quarantine_reason
return cleaned.strip(), meta
def _classify_and_chunk(
text: str, url: str, *, title_hint: str = "", authority_title: str = "", discovery_layer: str = ""
) -> tuple[str, dict, list]:
"""Classify, prep, and chunk a fetched page. Single call site for both crawl paths."""
cleaned, page_meta = _prepare_crawl_page(text, url, title_hint=title_hint, authority_title=authority_title)
page_meta = dict(page_meta or {})
page_meta["source_canonical"] = _canonical_source_url(url)
if discovery_layer:
page_meta["discovery_layer"] = discovery_layer
docs = _smart_chunk_page(cleaned, url, page_meta=page_meta)
return cleaned, page_meta, docs
def _chroma_safe_metadata(metadata: dict | None) -> dict:
"""Chroma metadata values must be scalar; encode richer crawl fields compactly."""
safe = {}
for key, value in (metadata or {}).items():
if value is None:
continue
if isinstance(value, (str, int, float, bool)):
safe[key] = value
elif isinstance(value, (list, tuple, set)):
values = [str(v) for v in value if v is not None and str(v) != ""]
if values:
safe[key] = ", ".join(values)
elif isinstance(value, dict):
try:
safe[key] = json.dumps(value, ensure_ascii=True, sort_keys=True)
except Exception:
safe[key] = str(value)
else:
safe[key] = str(value)
return safe
def _sanitize_docs_for_chroma(docs: list) -> list:
for doc in docs or []:
doc.metadata = _chroma_safe_metadata(getattr(doc, "metadata", None) or {})
return docs
def _merge_variant_docs(docs: list) -> list:
if not docs:
return docs
grouped = {}
for doc in docs:
meta = getattr(doc, "metadata", None) or {}
title = _canonical_product_title(meta.get("product_title") or "")
if not title:
grouped[id(doc)] = [doc]
continue
price_key = str(meta.get("price") or "").strip()
key = (title.lower(), price_key)
grouped.setdefault(key, []).append(doc)
merged = []
for group in grouped.values():
if len(group) == 1:
merged.extend(group)
continue
base = group[0]
variants = []
for doc in group:
text = getattr(doc, "page_content", "") or ""
mm = re.findall(r'(?i)\b(?:color|colour|size|variant|pack(?: of)?|piece(?:s)?)\b[^\n,;]*', text)
variants.extend(v.strip() for v in mm if v.strip())
uniq_variants = list(dict.fromkeys(variants))
if uniq_variants:
base.page_content = f"{base.page_content}\nVariants: {'; '.join(uniq_variants[:8])}".strip()
base.metadata["variant_count"] = len(uniq_variants)
base.metadata["dedup_applied"] = len(group) > 1
merged.append(base)
return merged
# ── Product spec extraction regexes (used by smart chunker) ──────────────────
_PROD_PRICE_RE = re.compile(r'(?i)(?:\$|£|€|\brs\.?\s*|\bpkr\s*)(\d[\d,]*\.?\d*)')
_PROD_SPEC_RE = re.compile(r'\b(?:processor|cpu|ram|memory|storage|ssd|hdd|gpu|graphics|display|battery|os|android|windows|linux|screen)\b', re.I)
_PROD_SPLIT_RE = re.compile(r'(?i)(\$|£|€|\brs\.?\s*|\bpkr\s*)(\d[\d,]*\.?\d*)\s+([A-Z][A-Za-z0-9 \(\)\-\.]+?(?:,[^\$£€]{10,400}?))(?=\s*(?:\$|£|€|\brs\.?\s*|\bpkr\s*)|\s*\Z)', re.S)
_FAQ_SPLIT_RE = re.compile(r'(?m)^(?=(?:Q:|Question:|How |What |Why |When |Where |Who |Can |Do |Is |Are |Does |Should ))', re.I)
def _smart_chunk_page(text: str, url: str, chunk_size: int = 400, chunk_step: int = 320, page_meta: dict | None = None) -> list:
from langchain_core.documents import Document
"""
Smart page chunker. Three modes, tried in order:
1. Product page — $PRICE + spec keywords → one Document per product, price metadata
2. FAQ page — Q&A or heading sections → one Document per section
3. Generic — existing word-based sliding window (unchanged fallback)
Returns list[Document]. Never raises.
"""
# Important: preserve newline structure for list-ish pages (outcomes, policies, release notes, etc.).
# _clean_text() is intentionally aggressive and tends to flatten whitespace, which destroys bullet boundaries.
# Markdown "## {title}" headings arrive entity-encoded ("Set Of 8 Pcs &ndash;
# thestationerycompany.pk") from raw <title>/H-tag extraction — they become the
# chunker's heading lines (Mode 2b) and leak "&ndash;/&amp;/&#39;" into chunk
# bodies. Decode at the chunker entry so every downstream heading/body is clean.
raw_text = _html_mod.unescape(text or "")
raw_text = raw_text.replace("\r\n", "\n").replace("\r", "\n")
raw_text = re.sub(r"[ \t]+", " ", raw_text)
raw_text = re.sub(r"\n{3,}", "\n\n", raw_text)
clean = _clean_text(raw_text)
# Cart-drawer widgets survive tag-stripping on many Shopify themes and inject
# "Subtotal: Rs.0.00" next to real products — scrub before chunking. Drawer
# tails vary by theme ("Check Out" spaced, "GO TO OUR COLLECTION", "Continue
# shopping" — danytech matched none of the old tails), so after the tailed
# strip also drop the bare sentence: it is pure chrome on every theme.
clean = re.sub(r'(?is)(?:(?:your\s+)?cart\s*[×x]?\s*)?your cart is currently empty\.?'
r'.{0,160}?(?:check\s*out|view cart|continue shopping|go to (?:our )?collections?)', ' ', clean)
clean = re.sub(r'(?i)(?:your\s+cart\s*[×x]?\s*)?your cart is currently empty\.?', ' ', clean)
clean = re.sub(r'(?i)subtotal:?\s*(?:rs\.?|pkr|\$|£|€)\s*0(?:\.00)?\b', ' ', clean)
docs = []
page_meta = page_meta or {}
# page_meta["page_title"] is the raw <title> ("Terms of service &ndash;
# thestationerycompany.pk") — it seeds Mode-2c "## {heading}" lines and
# _base_meta, NEITHER of which flow through the raw_text unescape above. Decode
# a shallow copy here so seed headings never leak &ndash;/&amp;/&#39;.
if page_meta.get("page_title"):
try:
page_meta = {**page_meta, "page_title": _html_mod.unescape(str(page_meta["page_title"]))}
except Exception:
pass
logger.info(f"[CHUNK-IN] {str(url)[:70]} text={len(text or '')} clean={len(clean)} pt={page_meta.get('page_type')}")
_canon_source = _canonical_source_url(str((page_meta or {}).get("source_canonical") or url or ""))
_base_meta = {"source": (_canon_source or url), "source_canonical": (_canon_source or str(url or ""))}
for _k in ("crawled_at", "last_verified_at", "source_status", "content_hash"):
if page_meta.get(_k) is not None:
_base_meta[_k] = page_meta.get(_k)
if page_meta.get("structural"):
_base_meta["structural"] = True
if page_meta.get("contaminated"):
_base_meta["contaminated"] = True
if page_meta.get("page_type"):
_base_meta["content_type"] = page_meta["page_type"]
if page_meta.get("page_title"):
# Helps disambiguate chapter/part/policy pages with similar boilerplate phrases.
_base_meta["page_title"] = str(page_meta.get("page_title") or "")[:200]
if "quality_score" in page_meta:
_base_meta["quality_score"] = page_meta.get("quality_score")
if "page_classifier_confidence" in page_meta:
_base_meta["page_classifier_confidence"] = page_meta.get("page_classifier_confidence")
if page_meta.get("used_structured_fields"):
_base_meta["used_structured_fields"] = list(page_meta.get("used_structured_fields") or [])
if "body_fallback_used" in page_meta:
_base_meta["body_fallback_used"] = bool(page_meta.get("body_fallback_used"))
if "retrieve_eligible" in page_meta:
_base_meta["retrieve_eligible"] = bool(page_meta.get("retrieve_eligible"))
if page_meta.get("quarantine_reason"):
_base_meta["quarantine_reason"] = page_meta.get("quarantine_reason")
if page_meta.get("extraction_mode"):
_base_meta["extraction_mode"] = page_meta.get("extraction_mode")
if page_meta.get("docs_like"):
_base_meta["docs_like"] = True
if page_meta.get("categories"):
# Crawl-graph category names — retrieval anchors match these so products
# answer category queries their body text never mentions ("laptop").
_base_meta["categories"] = str(page_meta.get("categories"))[:300]
_base_meta["dedup_applied"] = False
def _finalize_docs(out_docs: list) -> list:
logger.info(f"[CHUNK] {str(url)[:70]} -> {len(out_docs or [])} docs (clean={len(clean)} pt={page_meta.get('page_type')} cl={page_meta.get('catalog_listing')})")
for _idx, _doc in enumerate(out_docs or []):
_m = dict(getattr(_doc, "metadata", None) or {})
_m["chunk_index"] = int(_idx)
_m["section_id"] = str(_m.get("section_id") or f"s{_idx}")
if not _m.get("chunk_kind"):
sid = str(_m.get("section_id") or "").lower()
if sid.startswith("product_"):
_m["chunk_kind"] = "product"
elif sid.startswith("catalog_"):
_m["chunk_kind"] = "catalog"
elif sid.startswith("category_"):
_m["chunk_kind"] = "category"
elif sid.startswith("faq_"):
_m["chunk_kind"] = "faq"
elif sid.startswith("head_"):
_m["chunk_kind"] = "heading"
elif sid.startswith("para_"):
_m["chunk_kind"] = "paragraph"
elif sid.startswith("table_"):
_m["chunk_kind"] = "tabular"
elif sid.startswith("list_"):
_m["chunk_kind"] = "list"
else:
_m["chunk_kind"] = "generic"
# Universal non-positive-price scrub: storefronts declare 0.00 for
# unsellable placeholder listings (e.g. Jmary MT-33 Vlogging Kit). A
# price=0 key poisons cheapest/bounds ranking, so drop it regardless of
# which path set it. Last chokepoint before docs are returned.
if "price" in _m:
try:
if float(_m["price"]) <= 0:
_m.pop("price", None)
except (TypeError, ValueError):
_m.pop("price", None)
# Universal short-title scrub: a <4-char product_title ("Fun", "Set")
# is brand/fragment leakage the gate rejects and that pollutes
# retrieval. Promote the canonical title if it's longer, else drop both
# title keys. Last chokepoint — catches every assignment path.
_pt = str(_m.get("product_title") or "").strip()
if _pt and len(_pt) < 4:
_ct = str(_m.get("canonical_product_title") or "").strip()
if len(_ct) >= 4:
_m["product_title"] = _ct
else:
_m.pop("product_title", None)
_m.pop("canonical_product_title", None)
# Universal nav-chrome scrub: storefront button/badge text ("Shop Now",
# "Click to enlarge", "Sold out", "Pre order") leaks into titles + the
# chunk head when boilerplate misses a varying-prefix banner. These tokens
# are never product content — strip from titles (gate invariant) and body.
_CHROME_RE = re.compile(r"(?i)\b(?:shop now|buy online|click to (?:enlarge|zoom)|add to (?:cart|wishlist|bag)|quick view|view (?:cart|details)|sold out|pre[\s-]?order|location click|location)\b")
for _tk in ("product_title", "canonical_product_title"):
_tv = str(_m.get(_tk) or "")
if _tv and _CHROME_RE.search(_tv):
_tv2 = re.sub(r"\s{2,}", " ", _CHROME_RE.sub(" ", _tv)).strip(" -:|")
if len(_tv2) >= 4:
_m[_tk] = _tv2
else:
_m.pop(_tk, None)
_body0 = str(getattr(_doc, "page_content", "") or "")
if _CHROME_RE.search(_body0):
_body1 = re.sub(r"[ \t]{2,}", " ", _CHROME_RE.sub(" ", _body0))
if _body1.strip():
_doc.page_content = _body1
# Universal HTML-entity scrub (last chokepoint): "## {title}" heading lines
# and titles are assembled from fields that unescape only ONCE, so
# double-encoded source ("&amp;ndash;" → "&ndash;" after one pass) leaves a
# literal entity in the body that crawl_gate flags as junk → quarantine.
# Loop-unescape body + title fields until stable so none survives.
def _unescape_stable(_s: str) -> str:
for _ in range(4):
_u = _html_mod.unescape(_s)
if _u == _s:
return _s
_s = _u
return _s
_eb = str(getattr(_doc, "page_content", "") or "")
_eu = _unescape_stable(_eb)
if _eu != _eb:
_doc.page_content = _eu
for _tk in ("product_title", "canonical_product_title", "page_title"):
_tv = _m.get(_tk)
if _tv:
_ts = _unescape_stable(str(_tv))
if _ts != str(_tv):
_m[_tk] = _ts
_sample = str(getattr(_doc, "page_content", "") or "")
_m["chunk_hash"] = hashlib.sha256(_sample.encode("utf-8", errors="ignore")).hexdigest()
_doc.metadata = _m
return out_docs
_docs_guard = bool(page_meta.get("docs_like"))
product_meta = page_meta.get("product") or {}
# Catalog-misclassification guard: a category/listing page can slip through as
# page_type=product (its title heuristic grabs the first product card). A real
# product page repeats ONE name across variants and ONE price in its spec table;
# a listing has MANY distinct names AND MANY distinct price values. Require both
# so variant tables ("£51.77 ... Tax £0.00") and spec rows don't false-positive.
_CARD_NAME_STOP = re.compile(r'(?i)^(?:in\s+stock|out\s+of\s+stock|tax|availability|number|qty|quantity|shipping|delivery|add\s+to|reviews?|incl|excl)\b')
_card_prices, _card_names = set(), set()
for m in re.finditer(r'(?:\$|£|€|\brs\.?\s*|\bpkr\s*)(\d[\d,]*\.?\d*)\s+([A-Z][A-Za-z0-9][^$£€\n]{2,40})', clean):
_nm = (m.group(2) or "")[:20].strip()
if not _nm or _CARD_NAME_STOP.match(_nm):
continue
_card_names.add(_nm.lower())
_card_prices.add(m.group(1))
# Detail-page exemption: product pages commonly carry a related-items carousel
# (≥3 other names+prices) — when the URL slug names THIS product, it's a detail
# page, not a listing, regardless of how many recommendations it shows.
_is_detailish = False
_pm_title = str((product_meta or {}).get("title") or "")
if _pm_title:
_slug_seg = re.sub(r'\.html?$', '', re.sub(r'(?i)/index\.html?$', '', urllib.parse.urlparse(str(url)).path.rstrip('/')).rsplit('/', 1)[-1])
_slug_n = re.sub(r'[\s\-_]*\d+$', '', re.sub(r'[^a-z0-9]+', ' ', _slug_seg.lower()).strip()).strip()
_pm_t_n = re.sub(r'[^a-z0-9]+', ' ', _pm_title.lower()).strip()
if len(_slug_n) >= 6 and _pm_t_n and (_pm_t_n.startswith(_slug_n) or _slug_n.startswith(_pm_t_n) or _slug_n in _pm_t_n):
_is_detailish = True
if len(_card_names) >= 3 and len(_card_prices) >= 3 and page_meta.get("page_type") == "product" and not _is_detailish:
page_meta = dict(page_meta)
page_meta["page_type"] = "catalog"
page_meta["catalog_listing"] = True
product_meta = {}
if (not _docs_guard and page_meta.get("page_type") == "product" and product_meta.get("title")
and not product_meta.get("price_label")
and float(page_meta.get("authority_price") or 0.0) >= 0): # <0: storefront declared price 0 — no price exists
# Universal fallback: many product pages print the price adjacent to the
# title in body text ("$24.99 Nokia 123" / "Nokia 123 ... Rs. 2,499").
# Without this the chunk gets no "Price:" line and no price metadata,
# which disables price-ranking retrieval for the whole DB.
# The title can recur inside the description next to prose amounts
# ("...an $80,000 vase..."), so score EVERY title occurrence by distance
# to the nearest price and keep the tightest pairing — the page header's
# title+price sit a few chars apart; prose mentions are looser.
# Titles under 4 chars ("A") match inside ordinary words ("Availability"
# right after "Tax £0.00") — too ambiguous to anchor a price search.
_t_full = str(product_meta["title"]).strip()
_t_esc = re.escape(_t_full[:30]) if len(_t_full) >= 4 else None
def _fb_val_ok(_v: str) -> bool:
try:
return float(_v.replace(",", "")) > 0 # £0.00 tax rows are not prices
except Exception:
return False
_fb_best = None # (distance, symbol, number)
for _tm in (re.finditer(_t_esc, clean, re.I) if _t_esc else ()):
_pre = clean[max(0, _tm.start() - 14):_tm.start()]
_pm_pre = re.search(r'(?i)(\$|£|€|\brs\.?\s*|\bpkr\s*)([\d,]+\.?\d*)\s{0,3}$', _pre)
if _pm_pre and _fb_val_ok(_pm_pre.group(2)):
_cand = (0, _pm_pre.group(1).strip(), _pm_pre.group(2))
if _fb_best is None or _cand[0] < _fb_best[0]:
_fb_best = _cand
continue
_tail = clean[_tm.end():_tm.end() + 90]
_pm_post = re.search(r'(?i)(\$|£|€|\brs\.?\s*|\bpkr\s*)([\d,]+\.?\d*)', _tail)
if _pm_post and _fb_val_ok(_pm_post.group(2)) and (_fb_best is None or _pm_post.start() < _fb_best[0]):
_fb_best = (_pm_post.start(), _pm_post.group(1).strip(), _pm_post.group(2))
if _fb_best:
product_meta = dict(product_meta)
product_meta["price_label"] = f"{_fb_best[1]}{_fb_best[2]}"
try:
product_meta["price_num"] = float(_fb_best[2].replace(",", ""))
except Exception:
pass
if not _docs_guard and page_meta.get("page_type") == "product" and product_meta.get("title") and (product_meta.get("price_label") or product_meta.get("description")):
lines = [f"Product: {product_meta['title']}"]
if product_meta.get("price_label"):
lines.append(f"Price: {product_meta['price_label']}")
if product_meta.get("availability"):
lines.append(f"Availability: {product_meta['availability']}")
if product_meta.get("description"):
lines.append(f"Description: {product_meta['description']}")
lines.append(f"Full specs: {product_meta['title']}, {product_meta['description']}")
_prod_text = "\n".join(lines).strip()
_prod_meta = dict(_base_meta)
_prod_meta["product_title"] = product_meta["title"]
_prod_meta["canonical_product_title"] = product_meta.get("canonical_title") or _canonical_product_title(product_meta["title"])
_prod_meta["page_title"] = product_meta.get("title") or _base_meta.get("page_title") or ""
try:
_pn = float(product_meta.get("price_num") or 0.0)
except Exception:
_pn = 0.0
# Storefronts declare 0.00 for unsellable placeholder listings — a 0
# price key poisons cheapest/bounds ranking, so omit the key entirely.
if _pn > 0:
_prod_meta["price"] = _pn
if product_meta.get("availability"):
_prod_meta["availability"] = product_meta["availability"]
_prod_meta["used_structured_fields"] = list(product_meta.get("used_structured_fields") or _base_meta.get("used_structured_fields") or [])
_prod_meta["body_fallback_used"] = bool(product_meta.get("body_fallback_used"))
_prod_meta.update(_extract_product_metadata(_prod_text))
_prod_meta["section_id"] = "product_0"
_prod_meta["chunk_kind"] = "product"
docs.append(Document(page_content=_prod_text, metadata=_prod_meta))
return _finalize_docs(_merge_variant_docs(docs))
# Mode 1b: catalog/listing page – split repeated product cards into per-item chunks.
# Gate on REAL card signals (≥3 distinct names+prices, computed above), not raw
# price-hit count: a product detail page has 2+ price hits from its spec table
# (price + £0.00 tax) plus prose amounts ("gave $25,000 to charity, ...") which
# the card splitter then mints into garbage products.
if not _docs_guard and (page_meta.get("catalog_listing") or (len(_card_names) >= 3 and len(_card_prices) >= 3)):
products = []
for m in _PROD_SPLIT_RE.finditer(clean):
currency_raw = (m.group(1) or "").strip()
price_str = m.group(2).replace(',', '')
try:
price_num = float(price_str)
except Exception:
continue
raw = m.group(3).strip()
comma_idx = raw.find(',')
name = raw[:comma_idx].strip() if comma_idx > 0 else raw
specs = raw[comma_idx + 1:].strip() if comma_idx > 0 else ''
name = re.sub(r'^[^\s]+(?:\s+[^\s]+){0,3}\.{2,}\s+', '', name).strip()
name = re.sub(r'\s+reviews?\s*$', '', name, flags=re.I).strip()
# Compare-at/sale price pairs ("Rs.3,299.00 Rs.2,999.00") make the
# splitter capture a price fragment as the card name — never a title.
if re.match(r'(?i)^(?:rs\.?|pkr|[\$£€])\s*[\d.,]*$', name):
continue
# A bare ≤3-char single word ("Fun" from "WinFun…", "Set") is brand/
# fragment leakage, never a full card title — and the gate rejects any
# product_title < 4 chars. Align the splitter with that invariant.
if not name or len(name.strip()) < 4:
continue
# Sentence fragments masquerading as cards ("Fun, mini fish-shaped
# erasers for kids…"): a single-word name whose specs continue in
# lowercase is marketing copy split at a comma, not a product card —
# and the paired price belongs to the PREVIOUS card.
if len(re.findall(r'[A-Za-z0-9]+', name)) == 1 and re.match(r'^[a-z]', specs):
continue
currency_label = "Rs." if re.match(r"(?i)^(?:rs\.?|pkr)$", currency_raw.replace(" ", "")) else (currency_raw or "Rs.")
products.append((price_num, f"{currency_label}{price_str}", name, specs))
if products:
for price_num, price_label, name, specs in products:
lines = [f"Product: {name}", f"Price: {price_label}"]
_attr_tags = []
if re.search(r'geforce|gtx|rtx|radeon\s+r[579x]|radeon\s+rx', specs, re.I):
_attr_tags.append("gaming laptop dedicated GPU")
if re.search(r'\btouch\b', specs, re.I) or re.search(r'\btouch\b', name, re.I):
_attr_tags.append("touchscreen display")
if re.search(r'2\s*in\s*1|360|yoga|spin\b', name, re.I):
_attr_tags.append("convertible 2-in-1 laptop")
if re.search(r'\bssd\b', specs, re.I):
_attr_tags.append("fast SSD storage")
if re.search(r'windows', specs, re.I):
_attr_tags.append("Windows laptop")
if re.search(r'android', specs, re.I):
_attr_tags.append("Android device")
if _attr_tags:
lines.append("Features: " + ", ".join(_attr_tags))
for part in [s.strip() for s in specs.split(',')]:
pl = part.lower()
if re.search(r'geforce|nvidia|radeon|amd\s+r|gtx|rtx|mx\d', pl):
lines.append(f"GPU: {part}")
elif re.search(r'\d+\s*gb(?:\s+ram)?$|\bddr\b', pl):
lines.append(f"RAM: {part}")
elif re.search(r'\d+\s*(?:gb|tb)\s+(?:ssd|hdd|emmc)|\d+\s*tb\b', pl):
lines.append(f"Storage: {part}")
elif re.search(r'core\s+i\d|celeron|pentium|ryzen|athlon|snapdragon', pl):
lines.append(f"Processor: {part}")
elif re.search(r'windows|linux|dos|macos|android|chrome\s*os', pl):
lines.append(f"OS: {part}")
elif re.search(r'\d+\.?\d*\"\s*(?:hd|fhd|uhd|ips|touch)?|(?:hd|fhd|uhd|ips)\s+display', pl):
lines.append(f"Display: {part}")
if specs:
lines.append(f"Full specs: {name}, {specs}")
_card_text = "\n".join(lines).strip()
_card_meta = dict(_base_meta)
_card_meta["product_title"] = name
_card_meta["canonical_product_title"] = _canonical_product_title(name)
_card_meta["page_title"] = _base_meta.get("page_title") or name
if price_num > 0:
_card_meta["price"] = price_num
_card_meta["section_id"] = f"catalog_{len(docs)}"
_card_meta["chunk_kind"] = "catalog"
_card_meta["catalog_listing"] = True
_card_meta.update(_extract_product_metadata(_card_text))
docs.append(Document(page_content=_card_text, metadata=_card_meta))
if docs:
return _finalize_docs(_merge_variant_docs(docs))
if len(clean.split()) > 80:
_cat_meta = dict(_base_meta)
_cat_meta["section_id"] = "category_0"
_cat_meta["chunk_kind"] = "category"
_cat_meta["catalog_listing"] = True
_cat_meta["page_title"] = _base_meta.get("page_title") or _derive_page_title(title_hint, clean)
docs.append(Document(page_content=clean[:5000], metadata=_cat_meta))
return _finalize_docs(docs)
# ── Mode 1: product page ──────────────────────────────────────────────────
if len(_PROD_PRICE_RE.findall(clean)) >= 1 and _PROD_SPEC_RE.search(clean):
products = []
for m in _PROD_SPLIT_RE.finditer(clean):
currency_raw = (m.group(1) or "").strip()
price_str = m.group(2).replace(',', '')
try:
price_num = float(price_str)
except:
continue
raw = m.group(3).strip()
comma_idx = raw.find(',')
name = raw[:comma_idx].strip() if comma_idx > 0 else raw
specs = raw[comma_idx + 1:].strip() if comma_idx > 0 else ''
# Strip breadcrumb prefix "Dell Inspiron... Dell Inspiron 15"
name = re.sub(r'^[^\s]+(?:\s+[^\s]+){0,3}\.{2,}\s+', '', name).strip()
name = re.sub(r'\s+reviews?\s*$', '', name, flags=re.I).strip()
# Compare-at/sale price pairs ("Rs.3,299.00 Rs.2,999.00") make the
# splitter capture a price fragment as the card name — never a title.
if re.match(r'(?i)^(?:rs\.?|pkr|[\$£€])\s*[\d.,]*$', name):
continue
if not name or len(name) < 3:
continue
# Sentence fragments masquerading as cards — same guard as Mode 1b.
if len(re.findall(r'[A-Za-z0-9]+', name)) == 1 and re.match(r'^[a-z]', specs):
continue
currency_label = "Rs." if re.match(r"(?i)^(?:rs\.?|pkr)$", currency_raw.replace(" ", "")) else (currency_raw or "Rs.")
products.append((price_num, f"{currency_label}{price_str}", name, specs))
if len(products) >= 1:
for price_num, price_label, name, specs in products:
lines = [f"Product: {name}", f"Price: {price_label}"]
# ── Attribute normalization: inject user-vocabulary tags ──────
_attr_tags = []
if re.search(r'geforce|gtx|rtx|radeon\s+r[579x]|radeon\s+rx', specs, re.I):
_attr_tags.append("gaming laptop dedicated GPU")
if re.search(r'\btouch\b', specs, re.I) or re.search(r'\btouch\b', name, re.I):
_attr_tags.append("touchscreen display")
if re.search(r'2\s*in\s*1|360|yoga|spin\b', name, re.I):
_attr_tags.append("convertible 2-in-1 laptop")
if re.search(r'\bssd\b', specs, re.I):
_attr_tags.append("fast SSD storage")
if re.search(r'windows', specs, re.I):
_attr_tags.append("Windows laptop")
if re.search(r'android', specs, re.I):
_attr_tags.append("Android device")
if _attr_tags:
lines.append("Features: " + ", ".join(_attr_tags))
for part in [s.strip() for s in specs.split(',')]:
pl = part.lower()
if re.search(r'geforce|nvidia|radeon|amd\s+r|gtx|rtx|mx\d', pl):
lines.append(f"GPU: {part}")
elif re.search(r'\d+\s*gb(?:\s+ram)?$|\bddr\b', pl):
lines.append(f"RAM: {part}")
elif re.search(r'\d+\s*(?:gb|tb)\s+(?:ssd|hdd|emmc)|\d+\s*tb\b', pl):
lines.append(f"Storage: {part}")
elif re.search(r'core\s+i\d|celeron|pentium|ryzen|athlon|snapdragon', pl):
lines.append(f"Processor: {part}")
elif re.search(r'windows|linux|dos|macos|android|chrome\s*os', pl):
lines.append(f"OS: {part}")
elif re.search(r'\d+\.?\d*\"\s*(?:hd|fhd|uhd|ips|touch)?|(?:hd|fhd|uhd|ips)\s+display', pl):
lines.append(f"Display: {part}")
if specs:
lines.append(f"Full specs: {name}, {specs}")
_prod_text = "\n".join(lines)
_prod_meta = dict(_base_meta)
_prod_meta["product_title"] = name
if price_num > 0:
_prod_meta["price"] = price_num
_prod_meta["canonical_product_title"] = _canonical_product_title(name)
_prod_meta.update(_extract_product_metadata(_prod_text))
_prod_meta["section_id"] = f"product_{len(docs)}"
docs.append(Document(page_content=_prod_text, metadata=_prod_meta))
if docs:
return _finalize_docs(_merge_variant_docs(docs))
# Mode 2: FAQ / section page
# The split regex fires on any line starting with How/What/Why/etc., so it
# over-triggers on long-form prose. The old code then truncated each section
# with sec[:2000], silently discarding everything past 2000 chars — on long
# pages this dropped >90% of the content and made it unretrievable. Chunk long
# sections (sentence-bounded) instead of truncating, so no content is lost.
sections = _FAQ_SPLIT_RE.split(clean)
if len(sections) >= 3:
for sec in sections:
sec = sec.strip()
if len(sec) <= 40:
continue
if len(sec) <= 1800:
_m2 = dict(_base_meta)
_m2["section_id"] = f"faq_{len(docs)}"
docs.append(Document(page_content=sec, metadata=_m2))
continue
# Long section: pack whole sentences into ~1400-char chunks with a
# one-sentence tail overlap rather than dropping the overflow.
_fs = [s.strip() for s in re.split(r"(?<=[.!?])\s+(?=[A-Z0-9“\"'(])", sec) if s and s.strip()] or [sec]
_fb: list[str] = []
_fc = 0
for _s in _fs:
if _fc and (_fc + len(_s) + 1 > 1400) and _fc >= 300:
_m2 = dict(_base_meta)
_m2["section_id"] = f"faq_{len(docs)}"
docs.append(Document(page_content=" ".join(_fb).strip(), metadata=_m2))
_tail = _fb[-1] if (_fb and len(_fb[-1]) < 400) else ""
_fb, _fc = ([_tail], len(_tail)) if _tail else ([], 0)
if len(_s) > 1800:
for _w in _s.split():
_fb.append(_w)
_fc += len(_w) + 1
if _fc > 1400:
_m2 = dict(_base_meta)
_m2["section_id"] = f"faq_{len(docs)}"
docs.append(Document(page_content=" ".join(_fb).strip(), metadata=_m2))
_fb, _fc = [], 0
continue
_fb.append(_s)
_fc += len(_s) + 1
if _fb and " ".join(_fb).strip():
_m2 = dict(_base_meta)
_m2["section_id"] = f"faq_{len(docs)}"
docs.append(Document(page_content=" ".join(_fb).strip(), metadata=_m2))
if docs:
return _finalize_docs(docs)
# Mode 2b: heading-aware page
try:
_heading_hits = list(re.finditer(r"(?m)^(?:##|###)\s+", raw_text))
if _heading_hits:
_starts = [m.start() for m in _heading_hits] + [len(raw_text)]
_pending_small_segments: list[tuple[str, str | None]] = []
def _emit_docs_from_text(seg_text: str, heading: str | None):
seg_text = (seg_text or "").strip()
if not seg_text:
return
if heading and not re.match(r"(?m)^\s*(?:##|###)\s+", seg_text):
seg_text = f"## {heading}\n\n{seg_text}"
if len(seg_text) <= 1600:
_meta = dict(_base_meta)
_meta["section_id"] = f"head_{len(docs)}"
_meta["chunk_kind"] = "heading"
if heading:
_meta["heading"] = heading
docs.append(Document(page_content=seg_text, metadata=_meta))
return
# Large sections: split on paragraphs / sentence boundaries and
# keep a small overlap so answers spanning a boundary are not lost.
blocks = [p.strip() for p in re.split(r"\n{2,}|(?<=[.?!])\s{2,}", seg_text) if p.strip()]
# A single block can still be huge (long code block, prose with no
# paragraph breaks). bge-small embeds only ~512 tokens (~2000 chars),
# so any oversized block must be hard-split into windows or its tail
# is never embedded (the silent recall killer). Enforce per-block.
_MAX_BLK = 1400
_bounded = []
for _blk in blocks:
if len(_blk) <= _MAX_BLK:
_bounded.append(_blk)
else:
_bounded.extend(_blk[i:i + 1200] for i in range(0, len(_blk), 1200))
blocks = _bounded or [seg_text[i:i + 1200] for i in range(0, len(seg_text), 1200)]
packed = []
buf = []
buf_chars = 0
for blk in blocks:
if buf and (buf_chars + len(blk) + 2 > 1400):
packed.append("\n\n".join(buf).strip())
tail = packed[-1][-180:].strip()
buf = [tail] if tail else []
buf_chars = len(tail) + (2 if tail else 0)
buf.append(blk)
buf_chars += len(blk) + 2
if buf:
packed.append("\n\n".join(buf).strip())
for idx, piece in enumerate(packed):
if not piece:
continue
if idx and packed[idx - 1]:
prev_tail = packed[idx - 1][-120:].strip()
if prev_tail and not piece.startswith(prev_tail):
piece = f"{prev_tail}\n\n{piece}"
_meta = dict(_base_meta)
_meta["section_id"] = f"head_{len(docs)}"
_meta["chunk_kind"] = "heading"
if heading:
_meta["heading"] = heading
docs.append(Document(page_content=piece, metadata=_meta))
def _flush_small_segments():
nonlocal _pending_small_segments
if not _pending_small_segments:
return
merged_parts = []
for seg_text, seg_heading in _pending_small_segments:
seg_text = (seg_text or "").strip()
if not seg_text:
continue
if seg_heading and not re.match(r"(?m)^\s*(?:##|###)\s+", seg_text):
seg_text = f"## {seg_heading}\n\n{seg_text}"
merged_parts.append(seg_text)
merged = "\n\n".join(merged_parts).strip()
_pending_small_segments = []
if merged:
_emit_docs_from_text(merged, None)
for idx, start in enumerate(_starts[:-1]):
seg = raw_text[start:_starts[idx + 1]].strip()
hm = re.match(r"^(?:##|###)\s+(.+?)\s*(?=\Z|(?:##|###)\s+)", seg, re.S)
heading = hm.group(1).strip() if hm else None
if len(seg) <= 280:
_pending_small_segments.append((seg, heading))
continue
_flush_small_segments()
_emit_docs_from_text(seg, heading)
_flush_small_segments()
if docs:
return _finalize_docs(docs)
except Exception:
pass
# Mode 2c: paragraph-aware grouping
try:
_paras = [p.strip() for p in re.split(r"\n{2,}", raw_text) if p and p.strip()]
if len(_paras) >= 2:
_seed_heading = str((page_meta or {}).get("page_title") or "").strip()
if _seed_heading and re.search(r"(?i)(web scraper test sites|all rights reserved|privacy policy|terms of service|home\s*\|\s*[^|]+)$", _seed_heading):
_seed_heading = ""
_last_heading = _seed_heading
_buf: list[str] = []
_chars = 0
_last_emitted: str | None = None
def _emit_para_chunk(chunk_text: str):
nonlocal _last_emitted
chunk_text = (chunk_text or "").strip()
if not chunk_text:
return
if _last_emitted:
tail = _last_emitted[-120:].strip()
if tail and not chunk_text.startswith(tail):
chunk_text = f"{tail}\n\n{chunk_text}"
_last_emitted = chunk_text
_meta = dict(_base_meta)
_meta["section_id"] = f"para_{len(docs)}"
_meta["chunk_kind"] = "paragraph"
if _last_heading:
_meta["heading"] = _last_heading
docs.append(Document(page_content=chunk_text, metadata=_meta))
def _flush_para_buf():
nonlocal _buf, _chars
if not _buf:
return
_chunk = "\n\n".join(_buf).strip()
if not _chunk:
_buf, _chars = [], 0
return
if _last_heading and not re.match(r"(?m)^\s*(?:##|###)\s+", _chunk):
_chunk = f"## {_last_heading}\n\n{_chunk}"
_emit_para_chunk(_chunk)
_buf, _chars = [], 0
for para in _paras:
_hm = re.match(r"^\s*(?:##|###)\s+(.+?)\s*$", para)
if _hm:
_flush_para_buf()
_last_heading = _hm.group(1).strip()
continue
# A single paragraph with no internal breaks can exceed the
# embedding window; window-split it so its tail still gets embedded.
_para_pieces = ([para] if len(para) <= 1600
else [para[i:i + 1200] for i in range(0, len(para), 1200)])
for _para_txt in _para_pieces:
if _buf and (_chars + len(_para_txt) + 2 > 1600):
_flush_para_buf()
_buf.append(_para_txt)
_chars += len(_para_txt) + 2
_flush_para_buf()
if docs:
return _finalize_docs(docs)
except Exception:
pass
# Mode 2d: tabular / key-value / code-ish page
try:
_lines = [ln.rstrip() for ln in raw_text.splitlines()]
_nonempty_lines = [ln.strip() for ln in _lines if ln.strip()]
def _is_rowish(ln: str) -> bool:
s = ln.strip()
if not s:
return False
if re.match(r"^\s*(?:##|###)\s+", s):
return False
if s.startswith(("```", "~~~")):
return True
if "\t" in s or " | " in s:
return True
if re.match(r"^[A-Za-z][A-Za-z0-9 _/\-]{1,40}\s*:\s+\S", s):
return True
if re.match(r"^[A-Za-z][A-Za-z0-9 _/\-]{1,40}\s*=\s*\S", s):
return True
if re.match(r"^(?:[-*•]|\d{1,2}[.)])\s+\S", s):
return True
if re.match(r"^\s{2,}\S", ln):
return True
return False
_row_lines = [ln for ln in _nonempty_lines if _is_rowish(ln)]
if len(_row_lines) >= 4 and len(_row_lines) >= max(4, int(len(_nonempty_lines) * 0.45)):
_last_heading = ""
_buf: list[str] = []
_chars = 0
def _flush_row_buf():
nonlocal _buf, _chars
if not _buf:
return
_chunk = "\n".join(_buf).strip()
if _last_heading and not re.match(r"(?m)^\s*(?:##|###)\s+", _chunk):
_chunk = f"## {_last_heading}\n\n{_chunk}"
_meta = dict(_base_meta)
_meta["section_id"] = f"table_{len(docs)}"
_meta["chunk_kind"] = "tabular"
if _last_heading:
_meta["heading"] = _last_heading
docs.append(Document(page_content=_chunk[:2400], metadata=_meta))
_buf, _chars = [], 0
for ln in _lines:
s = ln.strip()
if not s:
continue
hm = re.match(r"^\s*(?:##|###)\s+(.+?)\s*$", s)
if hm:
_flush_row_buf()
_last_heading = hm.group(1).strip()
continue
if not _is_rowish(ln):
continue
if _buf and (_chars + len(s) + 1 > 1600):
_flush_row_buf()
_buf.append(s)
_chars += len(s) + 1
_flush_row_buf()
if docs:
return _finalize_docs(docs)
except Exception:
pass
# Mode 2.5: bullet/list page
try:
_raw_lines = [ln.strip() for ln in raw_text.split("\n") if ln and ln.strip()]
_bullet_re = re.compile(r"^(?:[-*•]|\d{1,2}[.)])\s+")
_bullet_lines = [ln for ln in _raw_lines if _bullet_re.match(ln)]
if len(_bullet_lines) >= 3:
buf = []
buf_chars = 0
for ln in _raw_lines:
if len(ln) > 500:
continue
if not _bullet_re.match(ln):
continue
if buf and (buf_chars + len(ln) + 1 > 2000):
_m3 = dict(_base_meta)
_m3["section_id"] = f"list_{len(docs)}"
_m3["chunk_kind"] = "list"
docs.append(Document(page_content="\n".join(buf).strip(), metadata=_m3))
buf, buf_chars = [], 0
buf.append(ln)
buf_chars += len(ln) + 1
if buf:
_m4 = dict(_base_meta)
_m4["section_id"] = f"list_{len(docs)}"
_m4["chunk_kind"] = "list"
docs.append(Document(page_content="\n".join(buf).strip(), metadata=_m4))
if docs:
return _finalize_docs(docs)
except Exception:
pass
# Mode 3 (legacy word-split) — kept verbatim for any non-prose page that slips
# through to the fallback (product/catalog/category). A raw word-count window is
# fine for price-card text and must not change, so product DBs are untouched.
def _legacy_word_split():
words = clean.split()
for j in range(0, max(1, len(words)), chunk_step):
chunk = " ".join(words[j:j + chunk_size])
if len(chunk) > 20:
_m5 = dict(_base_meta)
_m5["section_id"] = f"generic_{len(docs)}"
docs.append(Document(page_content=chunk, metadata=_m5))
if j + chunk_size >= len(words):
break
return _finalize_docs(docs)
if str(page_meta.get("page_type") or "") in {"product", "catalog", "category"}:
return _legacy_word_split()
# Mode 3 (prose): structure-aware, sentence-bounded, heading-prefixed.
# The old raw word-count window cut sentences mid-phrase and stripped section
# context, diluting chunk embeddings and hurting retrieval recall (the dominant
# failure mode on docs DBs). We now pack WHOLE sentences to a focused size and
# prepend the page title as heading context (mirrors Mode 2c). Smaller, coherent,
# context-anchored chunks embed far more precisely.
_heading_ctx = str(_base_meta.get("page_title") or "").strip()
if _heading_ctx and re.search(r"(?i)(all rights reserved|privacy policy|terms of service|home\s*\|)$", _heading_ctx):
_heading_ctx = ""
_prose_src = re.sub(r"\s+", " ", clean).strip()
_sents = [s.strip() for s in re.split(r"(?<=[.!?])\s+(?=[A-Z0-9“\"'(])", _prose_src) if s and s.strip()]
if not _sents and _prose_src:
_sents = [_prose_src]
_TARGET = 1100 # ~180 words: focused enough for precise embeddings
_HARDMAX = 1600
_pbuf: list[str] = []
_plen = 0
def _emit_prose():
nonlocal _pbuf, _plen
_body = " ".join(_pbuf).strip()
if not _body:
_pbuf, _plen = [], 0
return
if _heading_ctx:
_body = f"## {_heading_ctx}\n\n{_body}"
_m = dict(_base_meta)
_m["section_id"] = f"prose_{len(docs)}"
_m["chunk_kind"] = "prose"
if _heading_ctx:
_m["heading"] = _heading_ctx
docs.append(Document(page_content=_body, metadata=_m))
# carry the last sentence as overlap for cross-chunk continuity
_carry = _pbuf[-1] if _pbuf else ""
if _carry and len(_carry) < 400:
_pbuf, _plen = [_carry], len(_carry)
else:
_pbuf, _plen = [], 0
for _s in _sents:
if _plen and (_plen + len(_s) + 1 > _TARGET) and _plen >= 300:
_emit_prose()
if len(_s) > _HARDMAX: # pathological single sentence: word-pack it
for _w in _s.split():
_pbuf.append(_w)
_plen += len(_w) + 1
if _plen > _TARGET:
_emit_prose()
continue
_pbuf.append(_s)
_plen += len(_s) + 1
if _pbuf:
_emit_prose()
if not docs: # safety: sentence packing yielded nothing
return _legacy_word_split()
return _finalize_docs(docs)
def _extract_product_metadata(text: str) -> dict:
"""Parse product-catalog chunk text into structured ChromaDB metadata fields."""
meta = {}
pm = re.search(r'Price:\s*(?:\$|£|€|\brs\.?\s*|\bpkr\s*)?([\d,]+\.?\d*)', text, re.I)
if pm:
try:
_pv = float(pm.group(1).replace(',', ''))
if _pv > 0: # "Price: 0.00" = OOS placeholder, never a price
meta['price'] = _pv
except:
pass
rm = re.search(r'RAM:\s*(\d+)\s*GB', text, re.I)
if rm:
try:
meta['ram_gb'] = int(rm.group(1))
except:
pass
gm = re.search(r'GPU:[^\n]*?(\d+)\s*GB', text, re.I)
if gm:
try:
meta['gpu_vram_gb'] = int(gm.group(1)); meta['has_gpu'] = 1
except:
pass
else:
meta['has_gpu'] = 0
meta['has_touch'] = 1 if re.search(r'Display:[^\n]*\bTouch\b', text) else 0
meta['is_convertible'] = 1 if re.search(r'\bconvertible\b', text, re.I) else 0
meta['has_ssd'] = 1 if re.search(r'\bSSD\b', text) else 0
rvm = re.search(r'(\d+)\s+(?:customer\s+)?reviews?\b', text, re.I)
if rvm:
try:
meta['review_count'] = int(rvm.group(1))
except Exception:
pass
return meta
def _enrich_docs_metadata(docs: list) -> list:
"""Auto-detect product catalog chunks and enrich with structured metadata."""
for doc in docs:
if re.search(r'^Product:\s+\S', doc.page_content, re.M):
extracted = _extract_product_metadata(doc.page_content)
if doc.metadata:
doc.metadata.update(extracted)
else:
doc.metadata = extracted
return docs