Spaces:
Sleeping
Sleeping
File size: 101,214 Bytes
f5c68e4 8b4670b 58b01ce 99e7daa 8b4670b 8a45089 8b4670b 5e8014c 8a45089 58b01ce 8b4670b 7f8f089 104a091 8b4670b f5c68e4 8b4670b 5e8014c 365db39 aa50665 30d7c8e ca05721 9de344b ca05721 aa50665 ca05721 54c22a8 9de344b 54c22a8 ca05721 0a2afc3 0329348 0a2afc3 cdb7537 0a2afc3 cdb7537 0a2afc3 cdb7537 e442d44 cdb7537 0a2afc3 762e903 b926f3e 762e903 8b4670b 183b1dc b926f3e d24a834 8ba7755 b926f3e 96e99ca b926f3e 8b4670b d24a834 8b4670b 6f14423 8b4670b c8ccdd3 8226429 8b4670b f7f0554 8b4670b f7f0554 8226429 f7f0554 8226429 f7f0554 96e99ca 09675c1 f7f0554 c884cb3 f7f0554 c884cb3 f7f0554 c884cb3 f7f0554 c884cb3 f7f0554 c884cb3 f7f0554 0c27e35 e523e8a 2091b96 e523e8a 96e99ca 2091b96 e523e8a c884cb3 166d6ef c884cb3 e523e8a 0c27e35 c884cb3 2091b96 c884cb3 96e99ca 2091b96 c884cb3 e442d44 c884cb3 e442d44 0c27e35 e442d44 6f14423 e442d44 8b4670b f7f0554 96e99ca f7f0554 96e99ca 368331c 96e99ca 2091b96 f7f0554 6f14423 f7f0554 0c27e35 f7f0554 3ca4b24 c884cb3 f7f0554 8b4670b c8ccdd3 8b4670b 1e8a065 8b4670b 079e10b 8b4670b 09675c1 8b4670b abf1a60 c8ccdd3 abf1a60 dd27b77 f7f0554 8b4670b f7f0554 8b4670b cdb7537 8b4670b 4cadc87 cdb7537 7f8f089 8b4670b 997466c 4cadc87 997466c 4cadc87 8b4670b 4cadc87 8b4670b cdb7537 8b4670b cdb7537 8b4670b cdb7537 8b4670b 0a2afc3 8b4670b cdb7537 8b4670b c8ccdd3 af037c3 8b4670b 30d7c8e ca05721 8b4670b cdb7537 58b01ce 8b4670b ca05721 8b4670b 58b01ce 8b4670b dd27b77 c8ccdd3 dd27b77 4bebdf1 dd27b77 0a2afc3 7f8f089 dd27b77 8b4670b c8ccdd3 8b4670b 4cadc87 dd27b77 104a091 4cadc87 104a091 c8ccdd3 b58e55e 0a2afc3 dd27b77 0a2afc3 4cadc87 dd27b77 0a2afc3 dd27b77 8b4670b cdb7537 8b4670b cdb7537 8b4670b 6500ef8 f5c68e4 4bebdf1 f5c68e4 2f36ac4 8b4670b 0a2afc3 8b4670b f5c68e4 8b4670b 9e0bb04 dd27b77 9e0bb04 dd27b77 9e0bb04 8b4670b f5c68e4 6f14423 8b4670b 6513922 8b4670b f5c68e4 8b4670b f097462 99e7daa f097462 9e5dbc9 59bd3b5 9e5dbc9 8b4670b 067b095 8a45089 5e8014c 58b01ce 8b4670b e419e83 8b4670b 9de344b 365db39 8b4670b 58b01ce 8a45089 58b01ce f7f0554 0a2afc3 f7f0554 99e7daa 067b095 183b1dc 104a091 58b01ce 9de344b 8b4670b 7b7217d 96e99ca 6f14423 96e99ca 2091b96 7b7217d c8ccdd3 792287f 09a3400 c72226d 09a3400 c72226d 09a3400 c72226d 09a3400 c72226d 09a3400 792287f 09a3400 792287f 09a3400 792287f 9de344b 8b4670b f7f0554 0a2afc3 af037c3 8b4670b 0a2afc3 8b4670b 58b01ce 8b4670b 9de344b 2091b96 0a2afc3 6513922 0a2afc3 6513922 0a2afc3 892a6d5 99e7daa 0a2afc3 af037c3 6513922 0a2afc3 af037c3 0a2afc3 f5c68e4 8b4670b 6513922 8b4670b 6513922 8b4670b 892a6d5 8b4670b af037c3 6513922 8b4670b f5c68e4 8b4670b af037c3 f7f0554 8b4670b 58b01ce 8b4670b 58b01ce 8b4670b f5c68e4 cb5e492 8b4670b cb5e492 58b01ce cb5e492 8b4670b 58b01ce 8b4670b f5c68e4 5052a4a f5c68e4 5f66c38 dbbb57b 5f66c38 f5c68e4 5f66c38 f5c68e4 5f66c38 30d7c8e 5f66c38 f5c68e4 5f66c38 f5c68e4 5f66c38 dbbb57b 5f66c38 f5c68e4 5052a4a f5c68e4 5f66c38 dbbb57b 5f66c38 f5c68e4 fbb5702 f5c68e4 5f66c38 f5c68e4 5052a4a 5f66c38 f5c68e4 2f36ac4 f5c68e4 30d7c8e f5c68e4 4cadc87 f5c68e4 2f36ac4 f5c68e4 1e369b6 f097462 1e369b6 bf14afb 1e369b6 58b01ce 1e369b6 cb5e492 58b01ce 8b4670b 6f14423 8b4670b af037c3 8b4670b 9e5dbc9 8b4670b | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448 449 450 451 452 453 454 455 456 457 458 459 460 461 462 463 464 465 466 467 468 469 470 471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488 489 490 491 492 493 494 495 496 497 498 499 500 501 502 503 504 505 506 507 508 509 510 511 512 513 514 515 516 517 518 519 520 521 522 523 524 525 526 527 528 529 530 531 532 533 534 535 536 537 538 539 540 541 542 543 544 545 546 547 548 549 550 551 552 553 554 555 556 557 558 559 560 561 562 563 564 565 566 567 568 569 570 571 572 573 574 575 576 577 578 579 580 581 582 583 584 585 586 587 588 589 590 591 592 593 594 595 596 597 598 599 600 601 602 603 604 605 606 607 608 609 610 611 612 613 614 615 616 617 618 619 620 621 622 623 624 625 626 627 628 629 630 631 632 633 634 635 636 637 638 639 640 641 642 643 644 645 646 647 648 649 650 651 652 653 654 655 656 657 658 659 660 661 662 663 664 665 666 667 668 669 670 671 672 673 674 675 676 677 678 679 680 681 682 683 684 685 686 687 688 689 690 691 692 693 694 695 696 697 698 699 700 701 702 703 704 705 706 707 708 709 710 711 712 713 714 715 716 717 718 719 720 721 722 723 724 725 726 727 728 729 730 731 732 733 734 735 736 737 738 739 740 741 742 743 744 745 746 747 748 749 750 751 752 753 754 755 756 757 758 759 760 761 762 763 764 765 766 767 768 769 770 771 772 773 774 775 776 777 778 779 780 781 782 783 784 785 786 787 788 789 790 791 792 793 794 795 796 797 798 799 800 801 802 803 804 805 806 807 808 809 810 811 812 813 814 815 816 817 818 819 820 821 822 823 824 825 826 827 828 829 830 831 832 833 834 835 836 837 838 839 840 841 842 843 844 845 846 847 848 849 850 851 852 853 854 855 856 857 858 859 860 861 862 863 864 865 866 867 868 869 870 871 872 873 874 875 876 877 878 879 880 881 882 883 884 885 886 887 888 889 890 891 892 893 894 895 896 897 898 899 900 901 902 903 904 905 906 907 908 909 910 911 912 913 914 915 916 917 918 919 920 921 922 923 924 925 926 927 928 929 930 931 932 933 934 935 936 937 938 939 940 941 942 943 944 945 946 947 948 949 950 951 952 953 954 955 956 957 958 959 960 961 962 963 964 965 966 967 968 969 970 971 972 973 974 975 976 977 978 979 980 981 982 983 984 985 986 987 988 989 990 991 992 993 994 995 996 997 998 999 1000 1001 1002 1003 1004 1005 1006 1007 1008 1009 1010 1011 1012 1013 1014 1015 1016 1017 1018 1019 1020 1021 1022 1023 1024 1025 1026 1027 1028 1029 1030 1031 1032 1033 1034 1035 1036 1037 1038 1039 1040 1041 1042 1043 1044 1045 1046 1047 1048 1049 1050 1051 1052 1053 1054 1055 1056 1057 1058 1059 1060 1061 1062 1063 1064 1065 1066 1067 1068 1069 1070 1071 1072 1073 1074 1075 1076 1077 1078 1079 1080 1081 1082 1083 1084 1085 1086 1087 1088 1089 1090 1091 1092 1093 1094 1095 1096 1097 1098 1099 1100 1101 1102 1103 1104 1105 1106 1107 1108 1109 1110 1111 1112 1113 1114 1115 1116 1117 1118 1119 1120 1121 1122 1123 1124 1125 1126 1127 1128 1129 1130 1131 1132 1133 1134 1135 1136 1137 1138 1139 1140 1141 1142 1143 1144 1145 1146 1147 1148 1149 1150 1151 1152 1153 1154 1155 1156 1157 1158 1159 1160 1161 1162 1163 1164 1165 1166 1167 1168 1169 1170 1171 1172 1173 1174 1175 1176 1177 1178 1179 1180 1181 1182 1183 1184 1185 1186 1187 1188 1189 1190 1191 1192 1193 1194 1195 1196 1197 1198 1199 1200 1201 1202 1203 1204 1205 1206 1207 1208 1209 1210 1211 1212 1213 1214 1215 1216 1217 1218 1219 1220 1221 1222 1223 1224 1225 1226 1227 1228 1229 1230 1231 1232 1233 1234 1235 1236 1237 1238 1239 1240 1241 1242 1243 1244 1245 1246 1247 1248 1249 1250 1251 1252 1253 1254 1255 1256 1257 1258 1259 1260 1261 1262 1263 1264 1265 1266 1267 1268 1269 1270 1271 1272 1273 1274 1275 1276 1277 1278 1279 1280 1281 1282 1283 1284 1285 1286 1287 1288 1289 1290 1291 1292 1293 1294 1295 1296 1297 1298 1299 1300 1301 1302 1303 1304 1305 1306 1307 1308 1309 1310 1311 1312 1313 1314 1315 1316 1317 1318 1319 1320 1321 1322 1323 1324 1325 1326 1327 1328 1329 1330 1331 1332 1333 1334 1335 1336 1337 1338 1339 1340 1341 1342 1343 1344 1345 1346 1347 1348 1349 1350 1351 1352 1353 1354 1355 1356 1357 1358 1359 1360 1361 1362 1363 1364 1365 1366 1367 1368 1369 1370 1371 1372 1373 1374 1375 1376 1377 1378 1379 1380 1381 1382 1383 1384 1385 1386 1387 1388 1389 1390 1391 1392 1393 1394 1395 1396 1397 1398 1399 1400 1401 1402 1403 1404 1405 1406 1407 1408 1409 1410 1411 1412 1413 1414 1415 1416 1417 1418 1419 1420 1421 1422 1423 1424 1425 1426 1427 1428 1429 1430 1431 1432 1433 1434 1435 1436 1437 1438 1439 1440 1441 1442 1443 1444 1445 1446 1447 1448 1449 1450 1451 1452 1453 1454 1455 1456 1457 1458 1459 1460 1461 1462 1463 1464 1465 1466 1467 1468 1469 1470 1471 1472 1473 1474 1475 1476 1477 1478 1479 1480 1481 1482 1483 1484 1485 1486 1487 1488 1489 1490 1491 1492 1493 1494 1495 1496 1497 1498 1499 1500 1501 1502 1503 1504 1505 1506 1507 1508 1509 1510 1511 1512 1513 1514 1515 1516 1517 1518 1519 1520 1521 1522 1523 1524 1525 1526 1527 1528 1529 1530 1531 1532 1533 1534 1535 1536 1537 1538 1539 1540 1541 1542 1543 1544 1545 1546 1547 1548 1549 1550 1551 1552 1553 1554 1555 1556 1557 1558 1559 1560 1561 1562 1563 1564 1565 1566 1567 1568 1569 1570 1571 1572 1573 1574 1575 1576 1577 1578 1579 1580 1581 1582 1583 1584 1585 1586 1587 1588 1589 1590 1591 1592 1593 1594 1595 1596 1597 1598 1599 1600 1601 1602 1603 1604 1605 1606 1607 1608 1609 1610 1611 1612 1613 1614 1615 1616 1617 1618 1619 1620 1621 1622 1623 1624 1625 1626 1627 1628 1629 1630 1631 1632 1633 1634 1635 1636 1637 1638 1639 1640 1641 1642 1643 1644 1645 1646 1647 1648 1649 1650 1651 1652 1653 1654 1655 1656 1657 1658 1659 1660 1661 1662 1663 1664 1665 1666 1667 1668 1669 1670 1671 1672 1673 1674 1675 1676 1677 1678 1679 1680 1681 1682 1683 1684 1685 1686 1687 1688 1689 1690 1691 1692 1693 1694 1695 1696 1697 1698 1699 1700 1701 1702 1703 1704 1705 1706 1707 1708 1709 1710 1711 1712 1713 1714 1715 1716 1717 1718 1719 1720 1721 1722 1723 1724 1725 1726 1727 1728 1729 1730 1731 1732 1733 1734 1735 1736 1737 1738 1739 1740 1741 1742 1743 1744 1745 1746 1747 1748 1749 1750 1751 1752 1753 1754 1755 1756 1757 1758 1759 1760 1761 1762 1763 1764 1765 1766 1767 1768 1769 1770 1771 1772 1773 1774 1775 1776 1777 1778 1779 1780 1781 1782 1783 1784 1785 1786 1787 1788 1789 1790 1791 1792 1793 1794 1795 1796 1797 1798 1799 1800 1801 1802 1803 1804 1805 1806 1807 1808 1809 1810 1811 1812 1813 1814 1815 1816 1817 1818 1819 1820 1821 1822 1823 1824 1825 1826 1827 1828 1829 1830 1831 1832 1833 1834 1835 1836 1837 1838 1839 1840 1841 1842 1843 1844 1845 1846 1847 1848 1849 1850 1851 1852 1853 1854 1855 1856 1857 1858 1859 1860 1861 1862 1863 1864 1865 1866 1867 1868 1869 1870 1871 1872 1873 1874 1875 1876 1877 1878 1879 1880 1881 1882 1883 1884 1885 1886 1887 1888 1889 1890 1891 1892 1893 1894 1895 1896 1897 1898 1899 1900 1901 1902 1903 1904 1905 1906 1907 1908 1909 1910 1911 1912 1913 1914 1915 1916 1917 1918 1919 1920 1921 1922 1923 1924 1925 1926 1927 1928 1929 1930 1931 1932 1933 1934 1935 1936 1937 1938 1939 1940 1941 1942 1943 1944 1945 1946 1947 1948 1949 1950 1951 1952 1953 1954 1955 1956 1957 1958 1959 1960 1961 1962 1963 1964 1965 1966 1967 1968 1969 1970 1971 1972 1973 1974 1975 1976 1977 1978 1979 1980 1981 1982 1983 1984 1985 1986 1987 1988 1989 1990 1991 1992 1993 1994 1995 1996 1997 1998 1999 2000 2001 2002 2003 2004 2005 2006 2007 2008 2009 2010 2011 2012 2013 2014 2015 2016 2017 2018 2019 | """
services/crawler_utils.py — Page classification, chunking, and product metadata helpers.
No shared mutable app state. Safe to import from anywhere.
"""
from __future__ import annotations
import hashlib
import html as _html_mod
import json
import logging
import re
import urllib.parse
logger = logging.getLogger(__name__)
from datetime import datetime, timezone
from typing import Any
from services.safety import (
_clean_text,
_looks_structural_page,
_looks_like_product_page,
_looks_like_catalog_page,
_url_never_product,
_strip_storefront_boilerplate,
_dedupe_repeated_lines,
_canonical_product_title,
_trusted_content_metrics,
_BOILERPLATE_SIGNAL_RE,
_GENERIC_SECTION_SPLIT_RE,
_CONTAMINATION_HINTS_RE,
_POLICY_URL_RE,
_POLICY_TEXT_RE,
_CATEGORY_URL_RE,
_ARTICLE_URL_RE,
_PRODUCT_PRICE_LINE_RE,
_PRODUCT_PRICE_CAPTURE_RE,
_PRODUCT_AVAIL_RE,
)
_product_db_cache: dict[str, bool] = {} # db_name → is_product_db (stable per collection)
# Stable per-page URL identity used for dedupe/replace across crawls.
def _canonical_source_url(url: str) -> str:
try:
u = (url or "").strip()
if not u:
return ""
p = urllib.parse.urlparse(u)
scheme = (p.scheme or "https").lower()
netloc = (p.netloc or "").lower()
if netloc.startswith("www."):
netloc = netloc[4:]
path = re.sub(r"/+", "/", p.path or "/")
if path != "/":
path = path.rstrip("/")
return f"{scheme}://{netloc}{path}"
except Exception:
return (url or "").strip()
# ── Category propagation from crawl graph ───────────────────────────────────
# Product pages rarely contain their own category word ("ROG Strix SCAR" never
# says "laptop"). The crawl graph knows it structurally: products are discovered
# FROM category-listing pages. These helpers turn listing URLs into category
# names so chunks can carry a "categories" metadata string retrieval anchors on.
# Path segments that introduce a category section (high-precision, marker-based).
_CAT_MARKER_SEGS = {
"collections", "collection", "category", "categories",
"product-category", "product-categories", "shop-category", "c",
}
# Segments that end a category run inside a URL (/collections/toys/products/x).
_CAT_STOP_SEGS = {"products", "product", "items", "item", "p", "page", "pages"}
# Structural slugs that are never a real category name.
_CAT_GENERIC_WORDS = {
"all", "index", "home", "default", "frontpage", "main", "page", "pages",
"shop", "store", "catalog", "catalogue", "products", "product", "items",
"item", "collections", "collection", "category", "categories", "new",
"sale", "search", "cart", "checkout", "account", "login", "wishlist",
}
def _humanize_category_slug(seg: str):
"""'travel_2' → 'travel', 'rc-toys' → 'rc toys'; None for structural/junk."""
s = (seg or "").strip().lower()
if not s or "." in s or len(s) > 60:
return None
s = re.sub(r"[_-]\d+$", "", s) # pagination/id counters: travel_2, page-3
s = re.sub(r"[-_]+", " ", s).strip()
if len(s) < 2 or len(s) > 40 or not re.search(r"[a-z]", s):
return None
if s in _CAT_GENERIC_WORDS:
return None
return s
def category_slugs_from_url(url: str) -> list:
"""Marker-based category names from a URL path (/collections/rc-toys/...,
/catalogue/category/books/travel_2/...). Empty list when no marker."""
try:
path = urllib.parse.urlparse(str(url or "")).path.lower()
except Exception:
return []
segs = [s for s in path.split("/") if s]
out, i = [], 0
while i < len(segs):
if segs[i] in _CAT_MARKER_SEGS:
j = i + 1
while j < len(segs) and segs[j] not in _CAT_MARKER_SEGS and segs[j] not in _CAT_STOP_SEGS:
h = _humanize_category_slug(segs[j])
if h:
out.append(h)
j += 1
i = j
else:
i += 1
seen, res = set(), []
for c in out:
if c not in seen:
seen.add(c)
res.append(c)
return res[:6]
def listing_slug_from_url(url: str):
"""Category name for a page KNOWN to be a listing (classified catalog/category).
Marker-based when possible; else last meaningful path segment, walking back
over pagination/index segments (…/travel_2/page-2.html → 'travel')."""
marker = category_slugs_from_url(url)
if marker:
return marker[-1]
try:
path = urllib.parse.urlparse(str(url or "")).path.lower()
except Exception:
return None
for seg in reversed([s for s in path.split("/") if s]):
if seg in _CAT_STOP_SEGS:
return None # …/product/545 is a detail page, not a listing
h = _humanize_category_slug(seg)
if h:
return h
return None
def categories_for_page(url: str, parent_urls=()) -> list:
"""Save-time categories: marker-derived from the page URL itself plus any
crawl-graph parent URLs (collection-scoped hrefs, category sections)."""
cats = list(category_slugs_from_url(url))
for p in parent_urls or ():
cats.extend(category_slugs_from_url(p))
seen, out = set(), []
for c in cats:
if c not in seen:
seen.add(c)
out.append(c)
return out[:6]
# Per-DB docs-path hints (e.g. ["/roman/", "/arabic/"]) loaded from
# databases/<name>/config.json "docs_path_hints" by app.py at crawl start.
# Module-level is safe: only one crawl runs at a time (manual guard + auto-crawl sem=1).
_DOCS_PATH_HINTS: list = []
def set_docs_path_hints(hints) -> None:
global _DOCS_PATH_HINTS
_DOCS_PATH_HINTS = [str(h).lower() for h in (hints or []) if str(h).strip()]
# A docs corpus has no products. Per-page product/catalog detection still misfires
# on non-/docs content pages (e.g. /authors, /about): a title + descriptive prose
# but no price gets shaped into a "Product:/Full specs:" card with triple-title +
# nav boilerplate, which then outranks/garbles the real answer. When this flag is
# set at crawl start for a docs-only DB, every page is treated as docs content.
_DOCS_ONLY_DB: bool = False
def set_docs_only_db(flag) -> None:
global _DOCS_ONLY_DB
_DOCS_ONLY_DB = bool(flag)
def _looks_like_docs_page(url: str, body: str = "") -> bool:
"""Heuristic for docs/tutorial/chapter pages that should not be forced into product shaping."""
u = (url or "").lower()
if any(seg in u for seg in ("/docs/", "/guide/", "/tutorial", "/chapter-", "/lesson-", "/lessons/",
"/api/", "/reference/", "/concepts/", "/overview", "/getting-started",
"/quickstart", "/changelog/", "/releases/")):
return True
if _DOCS_PATH_HINTS and any(seg in u for seg in _DOCS_PATH_HINTS):
return True
b = (body or "").lower()
_has_labels = bool(re.search(r"(?m)^(?:parameters|returns|example|usage|response|request):", b))
_has_func_sig = bool(re.search(r"def \w+\(|\w+\(\) ->", b))
_has_learning = bool(re.search(r"(?i)\b(?:learning outcomes|learning goals|by completing|by the end of|chapter|lesson|exercise|quiz)\b", b))
if _has_labels or _has_func_sig or _has_learning:
return True
# Code fences alone: only when no strong e-commerce signal is present
if "```" in b:
return not bool(re.search(r"\$[\d,.]+|\b(?:add to cart|buy now|in stock|out of stock)\b", b))
return False
def _looks_generic_title(title: str) -> bool:
cand = _canonical_product_title(title or "").strip(" -:|")
if not cand:
return True
if len(cand) < 4:
return True
if "|" in cand:
return True
if re.search(r"(?i)\b(web scraper test sites|all rights reserved|privacy policy|terms of service|home\s*\|\s*[^|]+)$", cand):
return True
low = cand.lower()
if re.search(
r"\b(?:web scraper|cloud scraper)\b.*\b(?:extension|pricing|marketplace|learn|documentation|video tutorials|test sites|forum|install|login|company|about us|contact|privacy policy|media kit|resources|blog|screenshots|status)\b",
low,
):
return True
if re.fullmatch(
r"(?:web scraper|cloud scraper|test sites|forum|documentation|video tutorials|pricing|marketplace|learn|install|login|about us|contact us)(?:\s*[-|]\s*.*)?",
low,
):
return True
if re.fullmatch(r"[\W_]+", cand):
return True
words = re.findall(r"[A-Za-z0-9]+", cand)
if len(words) <= 1 and re.match(r"(?i)^(?:black|white|blue|grey|gray|silver|gold|red|green|pink|purple|yellow|orange|brown|beige|navy|teal|lavender|maroon|violet|golden|transparent|clear|unknown|default|variant|color|colour)$", cand):
return True
return False
def _derive_page_title(title_hint: str, cleaned: str, product: dict | None = None, *, prefer_title_hint: bool = False) -> str:
candidates: list[str] = []
body = cleaned or ""
ordered_candidates = [title_hint or "", (product or {}).get("title") or "", (product or {}).get("canonical_title") or ""]
if not prefer_title_hint:
for pat in (
r"(?m)^\s*(?:##|###)\s+(.+?)\s*$",
r"(?i)\b(?:name|title|product)\s*:\s*([^\n\.]{4,180})",
r"(?m)^(?!https?://)([A-Z][^\n]{4,120})$",
):
for mm in re.finditer(pat, body):
cand = _canonical_product_title((mm.group(1) or "").strip())
if cand and cand not in candidates:
candidates.append(cand)
break
for cand in ordered_candidates:
cand = _canonical_product_title(str(cand or "")).strip(" -:|")
if cand and cand not in candidates:
candidates.append(cand)
if prefer_title_hint:
for pat in (
r"(?m)^\s*(?:##|###)\s+(.+?)\s*$",
r"(?i)\b(?:name|title|product)\s*:\s*([^\n\.]{4,180})",
r"(?m)^(?!https?://)([A-Z][^\n]{4,120})$",
):
for mm in re.finditer(pat, body):
cand = _canonical_product_title((mm.group(1) or "").strip())
if cand and cand not in candidates:
candidates.append(cand)
break
non_generic = [c for c in candidates if not _looks_generic_title(c)]
if non_generic:
candidates = non_generic + [c for c in candidates if c not in non_generic]
return candidates[0] if candidates else ""
_DOCS_DB_TYPES = {"docs", "documentation", "text", "rag", "knowledge", "kb", "corpus"}
def _check_is_product_db(db, db_name: str = "", cfg: dict | None = None) -> bool:
"""Return True if the DB collection looks like a product/catalog DB.
Result cached per db_name so the sample query runs at most once per DB.
Config authority (mirrors catalog_query.is_catalog_db) runs BEFORE the chunk
sample: an explicit flag / docs db_type / docs_path_hints marks a docs corpus,
so a stray product-tagged enrichment chunk can't flip a docs DB to product
(which wrongly fires the product-retry/live-rescue that clobbers docs context)."""
cfg = cfg or {}
_flag = cfg.get("is_product_db")
if _flag is None:
_flag = cfg.get("product_db")
if _flag is True:
return True
if _flag is False:
return False
_db_type = str(cfg.get("db_type") or cfg.get("mode") or cfg.get("catalog_mode") or "").lower().strip()
if _db_type in _DOCS_DB_TYPES:
return False
if cfg.get("docs_path_hints"):
return False
if db_name and db_name in _product_db_cache:
return _product_db_cache[db_name]
result = False
if db:
try:
# Authoritative + insertion-order independent: does ANY chunk carry
# product metadata? The 24-chunk head sample missed product pages on
# large DBs (tsc_pk's first chunks are home/policy/nav), so is_product_db
# flipped False per-container and silently disabled the deterministic
# product answerers.
for _w in ({"chunk_kind": "product"}, {"content_type": "product"}):
try:
_hit = db._collection.get(where=_w, limit=1)
if (_hit.get("ids") or _hit.get("documents")):
result = True
break
except Exception:
pass
if result:
if db_name:
_product_db_cache[db_name] = True
return True
sample = db._collection.get(limit=24, include=["documents", "metadatas"])
docs = sample.get("documents") or []
metas = sample.get("metadatas") or []
for i, m in enumerate(metas):
m = m or {}
text = str(docs[i] if i < len(docs) else "" or "")
source = str(m.get("source") or "").lower()
if m.get("price") is not None or m.get("ram_gb") is not None or m.get("gpu_vram_gb") is not None:
result = True
break
if m.get("content_type") == "product":
result = True
break
if re.search(r"/(?:products?|items?)(?:/|$|#)", source) or "/collections/" in source:
result = True
break
tl = text.lower()
if ("rs." in tl or "pkr" in tl or "$" in tl or "£" in tl or "€" in tl) and ("add to cart" in tl or "shopping cart" in tl):
result = True
break
except Exception:
pass
# Only cache True: a False result may be stale (DB was empty or pre-crawl when sampled).
if db_name and result:
_product_db_cache[db_name] = result
return result
_PRODUCT_QUERY_STOP = {
"what", "is", "the", "price", "pricing", "cost", "of", "for", "a", "an", "item", "product",
"products", "much", "how", "does", "do", "you", "have", "tell", "me", "about", "details",
}
def _product_query_rerank_score(question: str, doc) -> float:
source = str(((getattr(doc, "metadata", None) or {}).get("source")) or "")
text = str(getattr(doc, "page_content", "") or "")
if not source and not text:
return 0.0
q_tokens = {
t for t in re.findall(r"[a-z0-9]+", (question or "").lower())
if len(t) >= 2 and t not in _PRODUCT_QUERY_STOP
}
source_tokens = set(re.findall(r"[a-z0-9]+", source.lower().replace("-", " ").replace("_", " ")))
head_tokens = set(re.findall(r"[a-z0-9]+", text[:500].lower()))
combined_tokens = source_tokens | head_tokens
overlap = len(q_tokens & combined_tokens)
slug_overlap = len(q_tokens & source_tokens)
head_overlap = len(q_tokens & head_tokens)
q_numbers = set(re.findall(r"\b\d+\b", question or ""))
doc_numbers = set(re.findall(r"\b\d+\b", f"{source} {text[:500]}"))
numbers_hit = len(q_numbers & doc_numbers)
normalized_q = re.sub(r"\s+", " ", re.sub(r"[^a-z0-9]+", " ", (question or "").lower())).strip()
normalized_doc = re.sub(r"\s+", " ", re.sub(r"[^a-z0-9]+", " ", f"{source} {text[:500]}".lower())).strip()
score = overlap + (slug_overlap * 3.0) + (head_overlap * 1.25) + (numbers_hit * 2.0)
if normalized_q and normalized_q in normalized_doc:
score += 8.0
if re.search(r"\b(price|pricing|cost)\b", question or "", re.I) and re.search(r"(?i)\b(?:\brs\.?|\bpkr|\$|£|€)\s*[\d,]+", text[:500]):
score += 1.0
return score
def _extract_product_summary(text: str, url: str, title_hint: str = "", authority_title: str = "", authority_price: float = 0.0, authority_currency: str = "Rs.") -> dict:
import html as _html_mod
title_hint = _html_mod.unescape(title_hint or "") # "Toy – Babyfy" → "Toy – Babyfy"
body = _strip_storefront_boilerplate(text or "")
used_structured_fields: list[str] = []
body_fallback_used = False
def _canonicalize_title(candidate: str) -> str:
import html as _html
# Shopify <title> tags arrive entity-encoded ("Toy – Babyfy") — unescape
# first so the "– Site" suffix splitter and dedup actually see the dash.
return _canonical_product_title(_html.unescape(candidate or "")).strip(" -:|")
def _is_generic_title(candidate: str) -> bool:
cand = _canonicalize_title(candidate)
if not cand:
return True
if len(cand) < 4:
return True
# A price is not a product name ("Rs.1", "$24", "£51.77")
if re.fullmatch(r'(?i)(?:rs\.?|pkr|\$|£|€)\s*[\d.,]*\s*', cand):
return True
if "|" in cand or "web scraper test sites" in cand.lower():
return True
# Spec-table rows are never product names ("Price (incl. tax) £51.77 Tax £0.00 ...")
if re.search(r'(?i)\bprice\s*\(|[\$£€]\s*\d|\b(?:incl|excl)\.\s*tax\b|\bavailability\b|\bin\s+stock\b|\bnumber\s+of\s+reviews?\b', cand):
return True
# Variant-swatch / price-fragment shapes are never product names:
# "PKR Pink - Rs.1", "PKR Red - Sold Out", "Diecast Model Ducati Diavel Rs.4".
# A standalone currency CODE as the first word is swatch text (a real title
# like "The $100 Startup" has the symbol glued to digits, not "PKR <word>").
if re.match(r'(?i)^(?:rs\.?|pkr|usd|eur|gbp|aed)\s', cand):
return True
if re.search(r'(?i)\bsold\s*out\b', cand):
return True
# Trailing currency fragment ("… Rs.4", "… - Rs.1,2") = truncated price tail.
if re.search(r'(?i)[\s\-–](?:rs\.?|pkr|\$|£|€)\s*[\d.,]*$', cand):
return True
# Truncated card text from the source site itself ("Teach children...").
if cand.endswith("...") or re.search(r'(?i)\bloading\.{0,3}$|\btranslation\s+missing\b', cand):
return True
if re.fullmatch(r"[\W_]+", cand):
return True
words = re.findall(r"[A-Za-z0-9]+", cand)
if len(words) <= 1 and re.match(r"(?i)^(?:black|white|blue|grey|gray|silver|gold|red|green|pink|purple|yellow|orange|brown|beige|navy|teal|lavender|maroon|violet|golden|transparent|clear|unknown|default|variant|color|colour)$", cand):
return True
return False
def _dedupe_repeated_phrase(text: str) -> str:
toks = [t for t in re.split(r"\s+", (text or "").strip()) if t]
if len(toks) >= 4 and len(toks) % 2 == 0:
half = len(toks) // 2
if toks[:half] == toks[half:]:
return " ".join(toks[:half]).strip(" -:|")
return (text or "").strip(" -:|")
def _score_title(candidate: str) -> tuple[int, int, int]:
cand = _canonicalize_title(candidate)
if not cand:
return (0, 0, 0, -10)
words = re.findall(r"[A-Za-z0-9]+", cand)
generic = 1 if _is_generic_title(cand) else 0
has_modelish_token = 1 if re.search(r"(?:\d|[A-Z]{2,}\d|\d+[A-Za-z][A-Za-z0-9\-]*)", cand) else 0
penalty = 0
if generic:
penalty += 3
if len(words) < 2:
penalty += 1
# Prefer non-generic product names first; among those, prefer titles
# with model-like tokens and then the richer titles.
return (1 - generic, has_modelish_token, len(words), len(cand), -penalty)
title_candidates: list[str] = []
def _push_title(candidate: str, label: str) -> None:
cand = _canonicalize_title(candidate)
if not cand:
return
if cand not in title_candidates:
title_candidates.append(cand)
if label:
used_structured_fields.append(label)
def _title_from_body_line(line: str) -> str:
raw = re.sub(r"\s+", " ", (line or "")).strip(" -:|")
if not raw:
return ""
raw = re.sub(r'(?i)^(?:product|name|title|model)\s*:\s*', "", raw).strip(" -:|")
# Prefer the model-like phrase immediately after a price marker.
for pm in re.finditer(r'(?i)(?:\$|£|€|\brs\.?|\bpkr)\s*[\d,]+(?:\.\d{1,2})?\s+((?-i:[A-Z0-9]).{3,119})', raw):
tail = pm.group(1).strip(" -:|")
tail = re.split(
r'(?i)\s+(?:hdd:|ssd:|ram:|processor:|display:|os:|availability:|reviews?|'
r'add to cart|wishlist|toggle navigation|cloud scraper|pricing|marketplace|'
r'learn documentation|video tutorials|test sites|forum|contact us|copyright|description:)\b',
tail,
)[0].strip(" -:|")
# Spec-table tail, not a product name ("£51.77 In stock (22 available)...")
if re.match(r'(?i)^(?:in\s+stock|out\s+of\s+stock|tax\b|availability|number\s+of|qty|quantity|customer|reviews?)\b', tail):
continue
# Sentence boundary inside the tail → prose fragment from a description
# ("$500 million were stolen from the Museum. It remains..."), not a title.
if ". " in tail[:90]:
continue
tail = tail.split(",")[0].strip(" -:|")
tail = _dedupe_repeated_phrase(tail)
tail_candidates = []
if tail:
tail_candidates.append(_canonicalize_title(tail))
tail_spans = re.findall(
r'([A-Z][A-Za-z0-9&\'"()\-]+(?:\s+[A-Z0-9][A-Za-z0-9&\'"()\-]+){1,8})',
tail,
)
tail_candidates.extend(s.strip(" -:|") for s in tail_spans if s and len(s.strip()) <= 90)
tail_candidates = [s for s in tail_candidates if s and len(s) <= 120]
tail_candidates = [s for s in tail_candidates if not _is_generic_title(s)]
cand = max(tail_candidates, key=_score_title) if tail_candidates else _canonicalize_title(tail)
if cand and not _is_generic_title(cand):
return cand
# Strip obvious spec / boilerplate suffixes from generic lines.
raw = re.split(
r'(?i)\s+(?:hdd:|ssd:|ram:|processor:|display:|os:|availability:|reviews?|'
r'add to cart|wishlist|toggle navigation|cloud scraper|pricing|marketplace|'
r'learn documentation|video tutorials|test sites|forum|contact us|copyright|description:)\b',
raw,
)[0].strip(" -:|")
raw = raw.split(",")[0].strip(" -:|")
# On long mixed lines, prefer the longest capitalized product-like span.
spans = re.findall(
r'([A-Z][A-Za-z0-9&\'"()\-]+(?:\s+[A-Z0-9][A-Za-z0-9&\'"()\-]+){1,8})',
raw,
)
spans = [s.strip(" -:|") for s in spans if s and len(s.strip()) <= 90]
spans = [s for s in spans if not _is_generic_title(s)]
if spans:
return max(spans, key=_score_title)
return _canonicalize_title(raw).strip(" -:|")
def _price_tail_title(text: str) -> str:
raw = re.sub(r"\s+", " ", (text or "")).strip(" -:|")
if not raw:
return ""
raw = re.sub(r'(?i)^(?:product|name|title|model)\s*:\s*', "", raw).strip(" -:|")
for pm in re.finditer(r'(?i)(?:\$|£|€|\brs\.?|\bpkr)\s*[\d,]+(?:\.\d{1,2})?\s+((?-i:[A-Z0-9]).{3,119})', raw):
tail = pm.group(1).strip(" -:|")
tail = re.split(
r'(?i)\s+(?:hdd:|ssd:|ram:|processor:|display:|os:|availability:|reviews?|'
r'add to cart|wishlist|toggle navigation|cloud scraper|pricing|marketplace|'
r'learn documentation|video tutorials|test sites|forum|contact us|copyright|description:)\b',
tail,
)[0].strip(" -:|")
# Spec-table tail, not a product name ("£51.77 In stock (22 available)...")
if re.match(r'(?i)^(?:in\s+stock|out\s+of\s+stock|tax\b|availability|number\s+of|qty|quantity|customer|reviews?)\b', tail):
continue
# Sentence boundary inside the tail → prose fragment from a description
# ("$500 million were stolen from the Museum. It remains..."), not a title.
if ". " in tail[:90]:
continue
tail = tail.split(",")[0].strip(" -:|")
tail = _dedupe_repeated_phrase(tail)
tail_candidates = []
if tail:
tail_candidates.append(_canonicalize_title(tail))
tail_spans = re.findall(
r'([A-Z][A-Za-z0-9&\'"()\-]+(?:\s+[A-Z0-9][A-Za-z0-9&\'"()\-]+){1,8})',
tail,
)
tail_candidates.extend(s.strip(" -:|") for s in tail_spans if s and len(s.strip()) <= 90)
tail_candidates = [s for s in tail_candidates if s and len(s) <= 120]
tail_candidates = [s for s in tail_candidates if not _is_generic_title(s)]
if tail_candidates:
cand = max(tail_candidates, key=_score_title)
if cand and not _is_generic_title(cand):
return cand
return ""
# Prefer a body-derived title when the page title is generic or boilerplate.
price_title = _price_tail_title(body)
if price_title:
_push_title(price_title, "price_title")
body_lines = [re.sub(r"\s+", " ", ln).strip(" -:|") for ln in re.split(r"[\r\n]+", body) if len(ln.strip()) >= 4]
for ln in body_lines[:180]:
cand = _title_from_body_line(ln)
if not cand or len(cand) > 120:
continue
if _is_generic_title(cand):
continue
if re.search(r'(?i)\b(?:add to cart|wishlist|reviews?|toggle navigation|cloud scraper|pricing|marketplace|learn documentation|video tutorials|test sites|forum|privacy policy|terms of service|all rights reserved)\b', cand):
continue
if re.search(r'(?i)^\s*(?:\brs\.?|\bpkr|\$|£|€)\s*[\d,]+', cand):
continue
_push_title(cand, "body_title")
break
title_match = re.search(r'(?i)\bname:\s*([^\n\.]{4,180})', body)
if title_match:
_push_title(title_match.group(1), "name")
if title_hint:
_push_title(title_hint, "title_hint")
# "<Product Name> | <Site Name>" is the dominant title-tag convention —
# the bare hint dies on the generic-title "|" rule, so also push the
# first segment ("A Light in the Attic | Books to Scrape" → product name).
_hint_head = re.split(r'\s*[|–—]\s*| - ', title_hint, maxsplit=1)[0].strip()
if _hint_head and _hint_head.lower() != title_hint.strip().lower():
_push_title(_hint_head, "title_hint_head")
slug = (url or "").rstrip("/").split("/")[-1]
_push_title(re.sub(r'[-_]+', ' ', slug).strip().title(), "slug")
# If the best title is still generic, try to salvage a better one from the
# early body lines before falling back to the weak variant.
if title_candidates:
title = max(title_candidates, key=_score_title)
# The page's own <title> head is authoritative: if the scored winner is just
# the hint head with extra LEADING junk ("Sandbox Sharp Objects" vs
# "Sharp Objects"), prefer the clean hint head.
if title_hint:
_hh = _canonicalize_title(re.split(r'\s*[|–—]\s*| - ', title_hint, maxsplit=1)[0].strip())
if (_hh and not _is_generic_title(_hh)
and title.lower() != _hh.lower()
and (title.lower().endswith(" " + _hh.lower())
or re.match(re.escape(_hh.lower()) + r'\s*[–—|-]', title.lower()))):
title = _hh
# URL-slug agreement: when the hint head matches the page's own slug
# ("into-the-wild" ≈ "Into the Wild") it IS the product name — prefer it
# over breadcrumb-mangled winners ("Sandbox Home Books Nonfiction Into").
if _hh and not _is_generic_title(_hh) and title.lower() != _hh.lower():
_slug_seg = re.sub(r'\.html?$', '', re.sub(r'(?i)/index\.html?$', '', urllib.parse.urlparse(str(url)).path.rstrip('/')).rsplit('/', 1)[-1])
_slug_norm = re.sub(r'[\s\-_]*\d+$', '', re.sub(r'[^a-z0-9]+', ' ', _slug_seg.lower()).strip()).strip()
_hh_norm = re.sub(r'[^a-z0-9]+', ' ', _hh.lower()).strip()
if len(_slug_norm) >= 6 and _hh_norm and (_hh_norm.startswith(_slug_norm) or _slug_norm.startswith(_hh_norm)):
title = _hh
if _is_generic_title(title):
body_candidates: list[str] = []
for pat in (
r'(?i)\bfull specs?\s*:\s*([^\n]+)',
r'(?i)\bdescription\s*:\s*([^\n]+)',
):
mm = re.search(pat, body)
if not mm:
continue
cand = _canonicalize_title((mm.group(1) or "").split(",")[0])
if not cand or len(cand) > 120:
continue
if _is_generic_title(cand):
continue
if re.search(r'(?i)\b(?:price|availability|add to cart|wishlist|reviews?)\b', cand):
continue
if re.search(r'(?i)\b(?:\brs\.?|\bpkr|\$|£|€)\s*[\d,]+', cand):
continue
body_candidates.append(cand)
for ln in body_lines[:220]:
cand = _title_from_body_line(ln)
if not cand or len(cand) > 120:
continue
if _is_generic_title(cand):
continue
body_candidates.append(cand)
for cand in body_candidates:
_push_title(cand, "body_title")
non_generic_titles = [cand for cand in title_candidates if not _is_generic_title(cand)]
if non_generic_titles:
title = max(non_generic_titles, key=_score_title)
else:
title = max(title_candidates, key=_score_title)
else:
title = ""
# On storefront pages, a valid price-tail title is the strongest signal and
# should win over generic site chrome or breadcrumb noise.
if price_title and not _is_generic_title(price_title):
title = price_title
price_label = ""
price_num = None
if authority_price and authority_price > 0:
# og:price:amount / JSON-LD offer price — the storefront's own declared
# price, carried as a NUMBER through every text transform. Body regexes
# mis-fire on model names ("DJI RS 2" → Rs.2) and piece counts
# ("Colors 24" → 24); the authority is immune and never wiped.
price_num = float(authority_price)
_ap_disp = f"{price_num:,.2f}".rstrip("0").rstrip(".")
price_label = f"{authority_currency or 'Rs.'}{_ap_disp}"
used_structured_fields.append("price_authority")
elif authority_price and authority_price < 0:
# Storefront declared price 0 (OOS placeholder) — no price exists on
# this page; regex extraction would only find junk.
pass
else:
pm = _PRODUCT_PRICE_LINE_RE.search(body) or _PRODUCT_PRICE_CAPTURE_RE.search(body)
if pm:
if pm.re is _PRODUCT_PRICE_LINE_RE:
currency = (pm.group(1) or "").strip() or "Rs."
digits = pm.group(2)
else:
whole = pm.group(0)
digits = pm.group(1)
currency = whole.replace(digits, "").strip() or "Rs."
price_label = f"{currency}{digits}"
used_structured_fields.append("price")
try:
price_num = float(digits.replace(",", ""))
except Exception:
price_num = None
if price_num is not None and price_num <= 0:
# "Rs.0.00" = OOS placeholder / cart remnant, never a price.
price_label = ""
price_num = None
used_structured_fields.remove("price")
avail = ""
# Authoritative: the product's OWN schema.org availability emitted by structured
# extraction ("Avail: .../InStock", "Availability: .../OutOfStock"). camelCase
# InStock has no space, so the loose phrase scan below can't see it — and would
# otherwise grab the FIRST stray "Out of Stock" on the page (a variant badge or a
# JS-rendered related-product card), mis-flagging an in-stock product as OOS.
sm = re.search(r'(?i)\bavail(?:ability)?:\s*(\S+)', body)
if sm:
tok = sm.group(1).lower()
if re.search(r'outofstock|out[_\s]?of[_\s]?stock|soldout|sold[_\s]?out|discontinued|backorder', tok):
avail = "out of stock"
elif re.search(r'instock|in[_\s]?stock|preorder|pre[_\s-]?order|available', tok):
avail = "available"
if avail:
used_structured_fields.append("availability")
if not avail:
am = _PRODUCT_AVAIL_RE.search(body)
if am:
avail = am.group(1).strip()
used_structured_fields.append("availability")
desc = ""
dm = re.search(r'(?i)\bdescription:\s*([^\n]{20,600})', body)
if not dm:
dm = re.search(r'(?m)(?i)^description\s*$\n([^\n]{20,600})', body)
if dm:
desc = dm.group(1).strip()
used_structured_fields.append("description")
if not desc:
body_fallback_used = True
sentences = re.split(r'(?<=[.!?])\s+', body)
useful = []
for sent in sentences:
s = sent.strip()
if len(s) < 25:
continue
if re.search(r'(?i)\b(add to cart|wishlist|recently viewed|you may also like|customers also bought|checkout|subtotal)\b', s):
continue
# A price inside a description sentence is related-product carousel
# leakage ("Number Book: Teach children... Rs.1,050.00 PKR"), not prose —
# the product's own price already lives in price_label. Same for
# truncated card text ("...") and loading/i18n widget artifacts.
if re.search(r'(?i)(?:\brs\.?|\bpkr\b|\$|£|€)\s*[\d,]+|\d[\d,]*\s*(?:pkr|rs\.?|usd|eur|gbp)\b', s):
continue
if "..." in s or re.search(r'(?i)\bloading\b|\btranslation\s+missing\b', s):
continue
# Shipping/delivery chrome repeats on every product page — never prose.
if re.search(r'(?i)\b(?:free (?:delivery|shipping)|delivery summary|order placed|dispatched|estimated delivery|cash on delivery)\b', s):
continue
useful.append(s)
if len(" ".join(useful)) >= 420:
break
desc = " ".join(useful[:3]).strip()
contaminated = bool(_CONTAMINATION_HINTS_RE.search(body))
# Final authority: a title-tag head matching the URL slug IS the product name.
# Descriptions naming OTHER priced products ("The $100 Startup") can otherwise
# hijack both the title and its paired price_label.
try:
_hh_fin = _canonicalize_title(re.split(r'\s*[|–—]\s*| - ', title_hint or '', maxsplit=1)[0].strip())
_path_fin = re.sub(r'(?i)/index\.html?$', '', urllib.parse.urlparse(str(url)).path.rstrip('/'))
_slug_fin = re.sub(r'[\s\-_]*\d+$', '', re.sub(r'[^a-z0-9]+', ' ', re.sub(r'\.html?$', '', _path_fin.rsplit('/', 1)[-1]).lower()).strip()).strip()
_hh_fin_n = re.sub(r'[^a-z0-9]+', ' ', _hh_fin.lower()).strip()
if (len(_slug_fin) >= 6 and _hh_fin_n and title
and title.strip().lower() != _hh_fin.lower()
and (_hh_fin_n.startswith(_slug_fin) or _slug_fin.startswith(_hh_fin_n))):
title = _hh_fin
if "price_authority" not in used_structured_fields:
price_label = "" # stale pair from hijacked title — chunker fallback re-derives
price_num = None
except Exception:
pass
# ABSOLUTE title authority: og:title / JSON-LD Product.name is the page's
# own declaration of its product name. Variant-picker text ("Single Piece
# Black", "15ml / BLACK") is structurally indistinguishable from a name —
# shape heuristics cannot flag it; the authoritative source always wins.
if authority_title:
try:
_auth = _canonicalize_title(re.split(r'\s*[|–—]\s*| - ', authority_title, maxsplit=1)[0].strip())
if _auth and len(_auth) <= 120 and not _is_generic_title(_auth) and title.strip().lower() != _auth.lower():
title = _auth
except Exception:
pass
canonical_title = _canonicalize_title(title)
return {
"title": title.strip(),
"canonical_title": canonical_title,
"price_label": price_label.strip(),
"price_num": price_num,
"availability": avail,
"description": desc[:900],
"contaminated": contaminated,
"used_structured_fields": list(dict.fromkeys(used_structured_fields)),
"body_fallback_used": body_fallback_used,
}
def _classify_page_type(url: str, cleaned: str, product_like: bool, structural: bool, docs_like: bool = False) -> str:
source = (url or "").lower()
body = cleaned or ""
metrics = _trusted_content_metrics(body)
if docs_like:
if structural:
return "structural"
if re.search(r"(?i)\b(by completing|by the end of this (?:chapter|lesson)|you will be able to|learning outcomes|objectives|goals)\b", body):
return "article"
if metrics["prose_chars"] >= 160 or metrics["sentence_count"] >= 2:
return "article"
return "structural" if metrics["nav_hits"] >= 6 else "article"
if _looks_like_catalog_page(source, body):
return "catalog"
if product_like:
return "product"
if structural:
return "structural"
# Outcomes/learning-goals marker override (universal):
# Many docs index pages look "category/structural" but still contain the
# canonical "By completing..., you will:" goals list. Never classify those
# as category-only content.
if re.search(r"(?i)\b(by completing|by the end of this (?:chapter|lesson)|you will be able to|learning outcomes|objectives|goals)\b", body):
return "article"
if len(_GENERIC_SECTION_SPLIT_RE.split(body)) >= 3:
return "faq"
if _CATEGORY_URL_RE.search(source) or (metrics["category_hits"] >= 2 and metrics["sentence_count"] <= 3):
return "category"
if _ARTICLE_URL_RE.search(source) or (
metrics["prose_chars"] > 300 and metrics["sentence_count"] >= 2 and metrics["nav_hits"] <= max(2, metrics["sentence_count"])
):
return "article"
if _POLICY_URL_RE.search(source):
return "policy"
if _POLICY_TEXT_RE.search(body) and metrics["policy_hits"] >= 2 and metrics["prose_chars"] < 1400 and metrics["sentence_count"] <= 8:
return "policy"
return "unknown"
def _quality_score(
cleaned: str,
*,
page_type: str,
used_structured_fields: list[str] | None = None,
body_fallback_used: bool = False,
structural: bool = False,
contaminated: bool = False,
had_boilerplate: bool = False,
docs_like: bool = False,
) -> float:
score = 0.0
used_structured_fields = used_structured_fields or []
metrics = _trusted_content_metrics(cleaned)
if used_structured_fields:
score += 0.4
if any(f in used_structured_fields for f in ("price", "availability", "description")):
score += 0.2
if metrics["prose_chars"] > 200:
score += 0.2
if not had_boilerplate:
score += 0.2
if body_fallback_used:
score -= 0.1
if structural or contaminated:
score = min(score, 0.2)
if docs_like and page_type in {"article", "structural"}:
score = max(score, 0.35)
if metrics["prose_chars"] > 120:
score += 0.15
if metrics["sentence_count"] >= 2:
score += 0.1
if not had_boilerplate:
score += 0.05
if page_type == "unknown":
score = min(score, 0.49)
return round(max(0.0, min(1.0, score)), 2)
def _page_classifier_confidence(
cleaned: str,
*,
page_type: str,
product_like: bool = False,
structural: bool = False,
used_structured_fields: list[str] | None = None,
body_fallback_used: bool = False,
docs_like: bool = False,
) -> float:
used_structured_fields = used_structured_fields or []
metrics = _trusted_content_metrics(cleaned)
if page_type == "product":
conf = 0.55
if product_like:
conf += 0.15
conf += min(0.2, 0.05 * len(used_structured_fields))
if body_fallback_used:
conf -= 0.1
return round(max(0.2, min(0.98, conf)), 2)
if page_type == "structural":
conf = 0.6 if structural else 0.45
if metrics["nav_hits"] >= 8:
conf += 0.15
return round(max(0.2, min(0.98, conf)), 2)
if page_type == "faq":
conf = 0.7 if len(_GENERIC_SECTION_SPLIT_RE.split(cleaned)) >= 3 else 0.5
return round(max(0.2, min(0.98, conf)), 2)
if page_type == "policy":
conf = 0.75 if metrics["policy_hits"] else 0.55
return round(max(0.2, min(0.98, conf)), 2)
if page_type == "category":
conf = 0.65 if metrics["category_hits"] else 0.5
return round(max(0.2, min(0.98, conf)), 2)
if page_type == "catalog":
conf = 0.68 if metrics["category_hits"] or metrics["prose_chars"] > 300 else 0.52
if metrics["sentence_count"] >= 3:
conf += 0.08
return round(max(0.2, min(0.98, conf)), 2)
if page_type == "article":
conf = 0.55
if metrics["prose_chars"] > 300:
conf += 0.15
if metrics["sentence_count"] >= 3:
conf += 0.1
return round(max(0.2, min(0.98, conf)), 2)
if docs_like:
conf = 0.72
if metrics["prose_chars"] > 200:
conf += 0.08
if metrics["sentence_count"] >= 2:
conf += 0.05
return round(max(0.2, min(0.98, conf)), 2)
return 0.35
def _prepare_crawl_page(text: str, url: str, title_hint: str = "", authority_title: str = "", authority_price: float = 0.0, authority_currency: str = "Rs.") -> tuple[str, dict]:
import html as _html_mod
# Hints arrive entity-encoded from raw <title> regex extraction; they feed
# page_title and the chunker's "## " heading lines downstream of every
# body-text unescape, so decode here or "–" persists into chunks.
title_hint = _html_mod.unescape(title_hint or "")
authority_title = _html_mod.unescape(authority_title or "")
raw = _clean_text(text or "")
docs_like = True if _DOCS_ONLY_DB else _looks_like_docs_page(url, raw)
catalog_like = False if docs_like else _looks_like_catalog_page(url, raw)
product_like = False if docs_like or catalog_like else _looks_like_product_page(url, raw)
cleaned = _strip_storefront_boilerplate(raw) if product_like else _dedupe_repeated_lines(raw)
cleaned = _clean_text(cleaned)
had_boilerplate = cleaned != raw and bool(_BOILERPLATE_SIGNAL_RE.search(raw))
structural = _looks_structural_page(url, cleaned)
page_type = _classify_page_type(url, cleaned, product_like, structural, docs_like=docs_like)
now_iso = datetime.now(timezone.utc).isoformat()
content_hash = hashlib.sha256(cleaned.encode("utf-8", errors="ignore")).hexdigest() if cleaned else ""
meta = {
"structural": structural,
"page_type": page_type,
"docs_like": docs_like,
"contaminated": False,
"used_structured_fields": [],
"body_fallback_used": False,
"had_boilerplate": had_boilerplate,
"extraction_mode": "body_text",
"crawled_at": now_iso,
"last_verified_at": now_iso,
"source_status": "live",
"content_hash": content_hash,
}
# Persist the hints so the flush-time boilerplate scrub can re-run this
# function on scrubbed text with the same inputs (HTML is gone by then).
meta["title_hint"] = (title_hint or "")[:300]
meta["authority_title"] = (authority_title or "")[:300]
meta["authority_price"] = float(authority_price or 0.0)
meta["authority_currency"] = (authority_currency or "Rs.")[:8]
_dt_hint = authority_title or title_hint
# Docs pages should derive their title from the actual page text first.
# Generic site chrome titles are a common failure mode on docs-heavy sites,
# so only treat the title hint as a fallback signal here.
meta["page_title"] = _derive_page_title(_dt_hint, cleaned, prefer_title_hint=docs_like)
meta["catalog_listing"] = False
if catalog_like:
meta["catalog_listing"] = True
meta["page_type"] = "catalog"
meta["page_title"] = _derive_page_title(_dt_hint, cleaned, prefer_title_hint=docs_like)
if product_like:
product = _extract_product_summary(cleaned, url, title_hint=title_hint, authority_title=authority_title, authority_price=authority_price, authority_currency=authority_currency)
meta["product"] = product
meta["contaminated"] = bool(product.get("contaminated"))
meta["used_structured_fields"] = list(product.get("used_structured_fields") or [])
meta["body_fallback_used"] = bool(product.get("body_fallback_used"))
meta["extraction_mode"] = "structured_product" if meta["used_structured_fields"] else "body_text"
if product.get("title") and (product.get("price_label") or product.get("description") or meta["used_structured_fields"]):
meta["page_type"] = "product"
meta["page_title"] = _derive_page_title(_dt_hint, cleaned, product, prefer_title_hint=docs_like)
elif not docs_like and meta["page_type"] in {"policy", "unknown", "category", "article"} and not _url_never_product(url):
# Product/catalog pages often carry policy/footer boilerplate that can
# overwhelm the classifier. If structured product signals are present,
# promote them even when the URL path itself is generic.
# GUARD: never promote the site root or info/account/policy pages — on
# WooCommerce/WordPress the global cart+price chrome makes _extract_product_summary
# find a "title+price" on About/Contact/Policy pages, mis-tagging them product.
product = _extract_product_summary(cleaned, url, title_hint=title_hint, authority_title=authority_title, authority_price=authority_price, authority_currency=authority_currency)
multiple_price_hits = len(_PRODUCT_PRICE_CAPTURE_RE.findall(cleaned)) + len(_PRODUCT_PRICE_LINE_RE.findall(cleaned))
if product.get("title") and multiple_price_hits >= 2:
meta["catalog_listing"] = True
meta["catalog_item_count"] = multiple_price_hits
meta["page_type"] = "catalog"
meta["page_title"] = _derive_page_title(_dt_hint, cleaned, product)
elif product.get("title") and (product.get("price_label") or product.get("description") or product.get("used_structured_fields")):
meta["product"] = product
meta["page_type"] = "product"
meta["contaminated"] = bool(product.get("contaminated"))
meta["used_structured_fields"] = list(product.get("used_structured_fields") or [])
meta["body_fallback_used"] = bool(product.get("body_fallback_used"))
meta["extraction_mode"] = "structured_product" if meta["used_structured_fields"] else "body_text"
meta["page_title"] = _derive_page_title(_dt_hint, cleaned, product, prefer_title_hint=docs_like)
else:
meta["page_title"] = _derive_page_title(_dt_hint, cleaned, prefer_title_hint=docs_like)
meta["quality_score"] = _quality_score(
cleaned,
page_type=meta["page_type"],
used_structured_fields=meta.get("used_structured_fields") or [],
body_fallback_used=bool(meta.get("body_fallback_used")),
structural=bool(meta.get("structural")),
contaminated=bool(meta.get("contaminated")),
had_boilerplate=bool(meta.get("had_boilerplate")),
docs_like=docs_like,
)
meta["page_classifier_confidence"] = _page_classifier_confidence(
cleaned,
page_type=meta["page_type"],
product_like=product_like,
structural=bool(meta.get("structural")),
used_structured_fields=meta.get("used_structured_fields") or [],
body_fallback_used=bool(meta.get("body_fallback_used")),
docs_like=docs_like,
)
quarantine_reason = ""
retrieve_eligible = True
if re.search(r'/(?:cart|checkout|account|login|register|logout|orders?|addresses|wishlist|search)(?:/|$|\?|#)', url or "", re.I):
# Storefront utility pages (cart/checkout/account/login/search) are pure
# chrome, never content. A populated cart page ("Your cart is currently
# empty … Top Selling") clears the category word-count gate below, so
# guard by URL FIRST — and however the page entered the frontier
# (sitemap/seed URLs bypass the discovery junk-scorer that lists 'cart').
retrieve_eligible = False
quarantine_reason = "utility_page"
elif meta["page_type"] in {"structural", "category"}:
# Large structural/category pages are often the canonical overview pages
# for doc sets. Keep docs-like overviews searchable even when they are
# compact, because many chapter/outcomes pages are short but meaningful.
_wc = len(cleaned.split())
_heading_hits = len(re.findall(r"(?m)^\s*(?:##|###)\s+", cleaned))
_min_docs_words = 40 if docs_like else 150
if docs_like and (_wc >= 25 or _heading_hits >= 1 or len(_GENERIC_SECTION_SPLIT_RE.split(cleaned)) >= 2):
retrieve_eligible = True
quarantine_reason = ""
elif _wc > _min_docs_words:
retrieve_eligible = True
quarantine_reason = ""
else:
retrieve_eligible = False
quarantine_reason = meta["page_type"]
elif meta.get("contaminated"):
retrieve_eligible = False
quarantine_reason = "contaminated"
elif meta["page_type"] == "product" and meta.get("body_fallback_used") and len(meta.get("used_structured_fields") or []) < 2:
retrieve_eligible = False
meta["quality_score"] = min(float(meta.get("quality_score") or 0.0), 0.35)
quarantine_reason = "weak_product_fallback"
elif meta["page_type"] in {"product", "catalog", "unknown"} and float(meta.get("quality_score") or 0.0) < 0.5:
# Article/faq/policy pages have no structured fields so their score
# is capped at 0.4 by design — don't quarantine them on score alone.
retrieve_eligible = False
quarantine_reason = "low_quality"
meta["retrieve_eligible"] = retrieve_eligible
meta["quarantine_reason"] = quarantine_reason
return cleaned.strip(), meta
def _classify_and_chunk(
text: str, url: str, *, title_hint: str = "", authority_title: str = "", discovery_layer: str = ""
) -> tuple[str, dict, list]:
"""Classify, prep, and chunk a fetched page. Single call site for both crawl paths."""
cleaned, page_meta = _prepare_crawl_page(text, url, title_hint=title_hint, authority_title=authority_title)
page_meta = dict(page_meta or {})
page_meta["source_canonical"] = _canonical_source_url(url)
if discovery_layer:
page_meta["discovery_layer"] = discovery_layer
docs = _smart_chunk_page(cleaned, url, page_meta=page_meta)
return cleaned, page_meta, docs
def _chroma_safe_metadata(metadata: dict | None) -> dict:
"""Chroma metadata values must be scalar; encode richer crawl fields compactly."""
safe = {}
for key, value in (metadata or {}).items():
if value is None:
continue
if isinstance(value, (str, int, float, bool)):
safe[key] = value
elif isinstance(value, (list, tuple, set)):
values = [str(v) for v in value if v is not None and str(v) != ""]
if values:
safe[key] = ", ".join(values)
elif isinstance(value, dict):
try:
safe[key] = json.dumps(value, ensure_ascii=True, sort_keys=True)
except Exception:
safe[key] = str(value)
else:
safe[key] = str(value)
return safe
def _sanitize_docs_for_chroma(docs: list) -> list:
for doc in docs or []:
doc.metadata = _chroma_safe_metadata(getattr(doc, "metadata", None) or {})
return docs
def _merge_variant_docs(docs: list) -> list:
if not docs:
return docs
grouped = {}
for doc in docs:
meta = getattr(doc, "metadata", None) or {}
title = _canonical_product_title(meta.get("product_title") or "")
if not title:
grouped[id(doc)] = [doc]
continue
price_key = str(meta.get("price") or "").strip()
key = (title.lower(), price_key)
grouped.setdefault(key, []).append(doc)
merged = []
for group in grouped.values():
if len(group) == 1:
merged.extend(group)
continue
base = group[0]
variants = []
for doc in group:
text = getattr(doc, "page_content", "") or ""
mm = re.findall(r'(?i)\b(?:color|colour|size|variant|pack(?: of)?|piece(?:s)?)\b[^\n,;]*', text)
variants.extend(v.strip() for v in mm if v.strip())
uniq_variants = list(dict.fromkeys(variants))
if uniq_variants:
base.page_content = f"{base.page_content}\nVariants: {'; '.join(uniq_variants[:8])}".strip()
base.metadata["variant_count"] = len(uniq_variants)
base.metadata["dedup_applied"] = len(group) > 1
merged.append(base)
return merged
# ── Product spec extraction regexes (used by smart chunker) ──────────────────
_PROD_PRICE_RE = re.compile(r'(?i)(?:\$|£|€|\brs\.?\s*|\bpkr\s*)(\d[\d,]*\.?\d*)')
_PROD_SPEC_RE = re.compile(r'\b(?:processor|cpu|ram|memory|storage|ssd|hdd|gpu|graphics|display|battery|os|android|windows|linux|screen)\b', re.I)
_PROD_SPLIT_RE = re.compile(r'(?i)(\$|£|€|\brs\.?\s*|\bpkr\s*)(\d[\d,]*\.?\d*)\s+([A-Z][A-Za-z0-9 \(\)\-\.]+?(?:,[^\$£€]{10,400}?))(?=\s*(?:\$|£|€|\brs\.?\s*|\bpkr\s*)|\s*\Z)', re.S)
_FAQ_SPLIT_RE = re.compile(r'(?m)^(?=(?:Q:|Question:|How |What |Why |When |Where |Who |Can |Do |Is |Are |Does |Should ))', re.I)
def _smart_chunk_page(text: str, url: str, chunk_size: int = 400, chunk_step: int = 320, page_meta: dict | None = None) -> list:
from langchain_core.documents import Document
"""
Smart page chunker. Three modes, tried in order:
1. Product page — $PRICE + spec keywords → one Document per product, price metadata
2. FAQ page — Q&A or heading sections → one Document per section
3. Generic — existing word-based sliding window (unchanged fallback)
Returns list[Document]. Never raises.
"""
# Important: preserve newline structure for list-ish pages (outcomes, policies, release notes, etc.).
# _clean_text() is intentionally aggressive and tends to flatten whitespace, which destroys bullet boundaries.
# Markdown "## {title}" headings arrive entity-encoded ("Set Of 8 Pcs –
# thestationerycompany.pk") from raw <title>/H-tag extraction — they become the
# chunker's heading lines (Mode 2b) and leak "–/&/'" into chunk
# bodies. Decode at the chunker entry so every downstream heading/body is clean.
raw_text = _html_mod.unescape(text or "")
raw_text = raw_text.replace("\r\n", "\n").replace("\r", "\n")
raw_text = re.sub(r"[ \t]+", " ", raw_text)
raw_text = re.sub(r"\n{3,}", "\n\n", raw_text)
clean = _clean_text(raw_text)
# Cart-drawer widgets survive tag-stripping on many Shopify themes and inject
# "Subtotal: Rs.0.00" next to real products — scrub before chunking. Drawer
# tails vary by theme ("Check Out" spaced, "GO TO OUR COLLECTION", "Continue
# shopping" — danytech matched none of the old tails), so after the tailed
# strip also drop the bare sentence: it is pure chrome on every theme.
clean = re.sub(r'(?is)(?:(?:your\s+)?cart\s*[×x]?\s*)?your cart is currently empty\.?'
r'.{0,160}?(?:check\s*out|view cart|continue shopping|go to (?:our )?collections?)', ' ', clean)
clean = re.sub(r'(?i)(?:your\s+cart\s*[×x]?\s*)?your cart is currently empty\.?', ' ', clean)
clean = re.sub(r'(?i)subtotal:?\s*(?:rs\.?|pkr|\$|£|€)\s*0(?:\.00)?\b', ' ', clean)
docs = []
page_meta = page_meta or {}
# page_meta["page_title"] is the raw <title> ("Terms of service –
# thestationerycompany.pk") — it seeds Mode-2c "## {heading}" lines and
# _base_meta, NEITHER of which flow through the raw_text unescape above. Decode
# a shallow copy here so seed headings never leak –/&/'.
if page_meta.get("page_title"):
try:
page_meta = {**page_meta, "page_title": _html_mod.unescape(str(page_meta["page_title"]))}
except Exception:
pass
logger.info(f"[CHUNK-IN] {str(url)[:70]} text={len(text or '')} clean={len(clean)} pt={page_meta.get('page_type')}")
_canon_source = _canonical_source_url(str((page_meta or {}).get("source_canonical") or url or ""))
_base_meta = {"source": (_canon_source or url), "source_canonical": (_canon_source or str(url or ""))}
for _k in ("crawled_at", "last_verified_at", "source_status", "content_hash"):
if page_meta.get(_k) is not None:
_base_meta[_k] = page_meta.get(_k)
if page_meta.get("structural"):
_base_meta["structural"] = True
if page_meta.get("contaminated"):
_base_meta["contaminated"] = True
if page_meta.get("page_type"):
_base_meta["content_type"] = page_meta["page_type"]
if page_meta.get("page_title"):
# Helps disambiguate chapter/part/policy pages with similar boilerplate phrases.
_base_meta["page_title"] = str(page_meta.get("page_title") or "")[:200]
if "quality_score" in page_meta:
_base_meta["quality_score"] = page_meta.get("quality_score")
if "page_classifier_confidence" in page_meta:
_base_meta["page_classifier_confidence"] = page_meta.get("page_classifier_confidence")
if page_meta.get("used_structured_fields"):
_base_meta["used_structured_fields"] = list(page_meta.get("used_structured_fields") or [])
if "body_fallback_used" in page_meta:
_base_meta["body_fallback_used"] = bool(page_meta.get("body_fallback_used"))
if "retrieve_eligible" in page_meta:
_base_meta["retrieve_eligible"] = bool(page_meta.get("retrieve_eligible"))
if page_meta.get("quarantine_reason"):
_base_meta["quarantine_reason"] = page_meta.get("quarantine_reason")
if page_meta.get("extraction_mode"):
_base_meta["extraction_mode"] = page_meta.get("extraction_mode")
if page_meta.get("docs_like"):
_base_meta["docs_like"] = True
if page_meta.get("categories"):
# Crawl-graph category names — retrieval anchors match these so products
# answer category queries their body text never mentions ("laptop").
_base_meta["categories"] = str(page_meta.get("categories"))[:300]
_base_meta["dedup_applied"] = False
def _finalize_docs(out_docs: list) -> list:
logger.info(f"[CHUNK] {str(url)[:70]} -> {len(out_docs or [])} docs (clean={len(clean)} pt={page_meta.get('page_type')} cl={page_meta.get('catalog_listing')})")
for _idx, _doc in enumerate(out_docs or []):
_m = dict(getattr(_doc, "metadata", None) or {})
_m["chunk_index"] = int(_idx)
_m["section_id"] = str(_m.get("section_id") or f"s{_idx}")
if not _m.get("chunk_kind"):
sid = str(_m.get("section_id") or "").lower()
if sid.startswith("product_"):
_m["chunk_kind"] = "product"
elif sid.startswith("catalog_"):
_m["chunk_kind"] = "catalog"
elif sid.startswith("category_"):
_m["chunk_kind"] = "category"
elif sid.startswith("faq_"):
_m["chunk_kind"] = "faq"
elif sid.startswith("head_"):
_m["chunk_kind"] = "heading"
elif sid.startswith("para_"):
_m["chunk_kind"] = "paragraph"
elif sid.startswith("table_"):
_m["chunk_kind"] = "tabular"
elif sid.startswith("list_"):
_m["chunk_kind"] = "list"
else:
_m["chunk_kind"] = "generic"
# Universal non-positive-price scrub: storefronts declare 0.00 for
# unsellable placeholder listings (e.g. Jmary MT-33 Vlogging Kit). A
# price=0 key poisons cheapest/bounds ranking, so drop it regardless of
# which path set it. Last chokepoint before docs are returned.
if "price" in _m:
try:
if float(_m["price"]) <= 0:
_m.pop("price", None)
except (TypeError, ValueError):
_m.pop("price", None)
# Universal short-title scrub: a <4-char product_title ("Fun", "Set")
# is brand/fragment leakage the gate rejects and that pollutes
# retrieval. Promote the canonical title if it's longer, else drop both
# title keys. Last chokepoint — catches every assignment path.
_pt = str(_m.get("product_title") or "").strip()
if _pt and len(_pt) < 4:
_ct = str(_m.get("canonical_product_title") or "").strip()
if len(_ct) >= 4:
_m["product_title"] = _ct
else:
_m.pop("product_title", None)
_m.pop("canonical_product_title", None)
# Universal nav-chrome scrub: storefront button/badge text ("Shop Now",
# "Click to enlarge", "Sold out", "Pre order") leaks into titles + the
# chunk head when boilerplate misses a varying-prefix banner. These tokens
# are never product content — strip from titles (gate invariant) and body.
_CHROME_RE = re.compile(r"(?i)\b(?:shop now|buy online|click to (?:enlarge|zoom)|add to (?:cart|wishlist|bag)|quick view|view (?:cart|details)|sold out|pre[\s-]?order|location click|location)\b")
for _tk in ("product_title", "canonical_product_title"):
_tv = str(_m.get(_tk) or "")
if _tv and _CHROME_RE.search(_tv):
_tv2 = re.sub(r"\s{2,}", " ", _CHROME_RE.sub(" ", _tv)).strip(" -:|")
if len(_tv2) >= 4:
_m[_tk] = _tv2
else:
_m.pop(_tk, None)
_body0 = str(getattr(_doc, "page_content", "") or "")
if _CHROME_RE.search(_body0):
_body1 = re.sub(r"[ \t]{2,}", " ", _CHROME_RE.sub(" ", _body0))
if _body1.strip():
_doc.page_content = _body1
# Universal HTML-entity scrub (last chokepoint): "## {title}" heading lines
# and titles are assembled from fields that unescape only ONCE, so
# double-encoded source ("&ndash;" → "–" after one pass) leaves a
# literal entity in the body that crawl_gate flags as junk → quarantine.
# Loop-unescape body + title fields until stable so none survives.
def _unescape_stable(_s: str) -> str:
for _ in range(4):
_u = _html_mod.unescape(_s)
if _u == _s:
return _s
_s = _u
return _s
_eb = str(getattr(_doc, "page_content", "") or "")
_eu = _unescape_stable(_eb)
if _eu != _eb:
_doc.page_content = _eu
for _tk in ("product_title", "canonical_product_title", "page_title"):
_tv = _m.get(_tk)
if _tv:
_ts = _unescape_stable(str(_tv))
if _ts != str(_tv):
_m[_tk] = _ts
_sample = str(getattr(_doc, "page_content", "") or "")
_m["chunk_hash"] = hashlib.sha256(_sample.encode("utf-8", errors="ignore")).hexdigest()
_doc.metadata = _m
return out_docs
_docs_guard = bool(page_meta.get("docs_like"))
product_meta = page_meta.get("product") or {}
# Catalog-misclassification guard: a category/listing page can slip through as
# page_type=product (its title heuristic grabs the first product card). A real
# product page repeats ONE name across variants and ONE price in its spec table;
# a listing has MANY distinct names AND MANY distinct price values. Require both
# so variant tables ("£51.77 ... Tax £0.00") and spec rows don't false-positive.
_CARD_NAME_STOP = re.compile(r'(?i)^(?:in\s+stock|out\s+of\s+stock|tax|availability|number|qty|quantity|shipping|delivery|add\s+to|reviews?|incl|excl)\b')
_card_prices, _card_names = set(), set()
for m in re.finditer(r'(?:\$|£|€|\brs\.?\s*|\bpkr\s*)(\d[\d,]*\.?\d*)\s+([A-Z][A-Za-z0-9][^$£€\n]{2,40})', clean):
_nm = (m.group(2) or "")[:20].strip()
if not _nm or _CARD_NAME_STOP.match(_nm):
continue
_card_names.add(_nm.lower())
_card_prices.add(m.group(1))
# Detail-page exemption: product pages commonly carry a related-items carousel
# (≥3 other names+prices) — when the URL slug names THIS product, it's a detail
# page, not a listing, regardless of how many recommendations it shows.
_is_detailish = False
_pm_title = str((product_meta or {}).get("title") or "")
if _pm_title:
_slug_seg = re.sub(r'\.html?$', '', re.sub(r'(?i)/index\.html?$', '', urllib.parse.urlparse(str(url)).path.rstrip('/')).rsplit('/', 1)[-1])
_slug_n = re.sub(r'[\s\-_]*\d+$', '', re.sub(r'[^a-z0-9]+', ' ', _slug_seg.lower()).strip()).strip()
_pm_t_n = re.sub(r'[^a-z0-9]+', ' ', _pm_title.lower()).strip()
if len(_slug_n) >= 6 and _pm_t_n and (_pm_t_n.startswith(_slug_n) or _slug_n.startswith(_pm_t_n) or _slug_n in _pm_t_n):
_is_detailish = True
if len(_card_names) >= 3 and len(_card_prices) >= 3 and page_meta.get("page_type") == "product" and not _is_detailish:
page_meta = dict(page_meta)
page_meta["page_type"] = "catalog"
page_meta["catalog_listing"] = True
product_meta = {}
if (not _docs_guard and page_meta.get("page_type") == "product" and product_meta.get("title")
and not product_meta.get("price_label")
and float(page_meta.get("authority_price") or 0.0) >= 0): # <0: storefront declared price 0 — no price exists
# Universal fallback: many product pages print the price adjacent to the
# title in body text ("$24.99 Nokia 123" / "Nokia 123 ... Rs. 2,499").
# Without this the chunk gets no "Price:" line and no price metadata,
# which disables price-ranking retrieval for the whole DB.
# The title can recur inside the description next to prose amounts
# ("...an $80,000 vase..."), so score EVERY title occurrence by distance
# to the nearest price and keep the tightest pairing — the page header's
# title+price sit a few chars apart; prose mentions are looser.
# Titles under 4 chars ("A") match inside ordinary words ("Availability"
# right after "Tax £0.00") — too ambiguous to anchor a price search.
_t_full = str(product_meta["title"]).strip()
_t_esc = re.escape(_t_full[:30]) if len(_t_full) >= 4 else None
def _fb_val_ok(_v: str) -> bool:
try:
return float(_v.replace(",", "")) > 0 # £0.00 tax rows are not prices
except Exception:
return False
_fb_best = None # (distance, symbol, number)
for _tm in (re.finditer(_t_esc, clean, re.I) if _t_esc else ()):
_pre = clean[max(0, _tm.start() - 14):_tm.start()]
_pm_pre = re.search(r'(?i)(\$|£|€|\brs\.?\s*|\bpkr\s*)([\d,]+\.?\d*)\s{0,3}$', _pre)
if _pm_pre and _fb_val_ok(_pm_pre.group(2)):
_cand = (0, _pm_pre.group(1).strip(), _pm_pre.group(2))
if _fb_best is None or _cand[0] < _fb_best[0]:
_fb_best = _cand
continue
_tail = clean[_tm.end():_tm.end() + 90]
_pm_post = re.search(r'(?i)(\$|£|€|\brs\.?\s*|\bpkr\s*)([\d,]+\.?\d*)', _tail)
if _pm_post and _fb_val_ok(_pm_post.group(2)) and (_fb_best is None or _pm_post.start() < _fb_best[0]):
_fb_best = (_pm_post.start(), _pm_post.group(1).strip(), _pm_post.group(2))
if _fb_best:
product_meta = dict(product_meta)
product_meta["price_label"] = f"{_fb_best[1]}{_fb_best[2]}"
try:
product_meta["price_num"] = float(_fb_best[2].replace(",", ""))
except Exception:
pass
if not _docs_guard and page_meta.get("page_type") == "product" and product_meta.get("title") and (product_meta.get("price_label") or product_meta.get("description")):
lines = [f"Product: {product_meta['title']}"]
if product_meta.get("price_label"):
lines.append(f"Price: {product_meta['price_label']}")
if product_meta.get("availability"):
lines.append(f"Availability: {product_meta['availability']}")
if product_meta.get("description"):
lines.append(f"Description: {product_meta['description']}")
lines.append(f"Full specs: {product_meta['title']}, {product_meta['description']}")
_prod_text = "\n".join(lines).strip()
_prod_meta = dict(_base_meta)
_prod_meta["product_title"] = product_meta["title"]
_prod_meta["canonical_product_title"] = product_meta.get("canonical_title") or _canonical_product_title(product_meta["title"])
_prod_meta["page_title"] = product_meta.get("title") or _base_meta.get("page_title") or ""
try:
_pn = float(product_meta.get("price_num") or 0.0)
except Exception:
_pn = 0.0
# Storefronts declare 0.00 for unsellable placeholder listings — a 0
# price key poisons cheapest/bounds ranking, so omit the key entirely.
if _pn > 0:
_prod_meta["price"] = _pn
if product_meta.get("availability"):
_prod_meta["availability"] = product_meta["availability"]
_prod_meta["used_structured_fields"] = list(product_meta.get("used_structured_fields") or _base_meta.get("used_structured_fields") or [])
_prod_meta["body_fallback_used"] = bool(product_meta.get("body_fallback_used"))
_prod_meta.update(_extract_product_metadata(_prod_text))
_prod_meta["section_id"] = "product_0"
_prod_meta["chunk_kind"] = "product"
docs.append(Document(page_content=_prod_text, metadata=_prod_meta))
return _finalize_docs(_merge_variant_docs(docs))
# Mode 1b: catalog/listing page – split repeated product cards into per-item chunks.
# Gate on REAL card signals (≥3 distinct names+prices, computed above), not raw
# price-hit count: a product detail page has 2+ price hits from its spec table
# (price + £0.00 tax) plus prose amounts ("gave $25,000 to charity, ...") which
# the card splitter then mints into garbage products.
if not _docs_guard and (page_meta.get("catalog_listing") or (len(_card_names) >= 3 and len(_card_prices) >= 3)):
products = []
for m in _PROD_SPLIT_RE.finditer(clean):
currency_raw = (m.group(1) or "").strip()
price_str = m.group(2).replace(',', '')
try:
price_num = float(price_str)
except Exception:
continue
raw = m.group(3).strip()
comma_idx = raw.find(',')
name = raw[:comma_idx].strip() if comma_idx > 0 else raw
specs = raw[comma_idx + 1:].strip() if comma_idx > 0 else ''
name = re.sub(r'^[^\s]+(?:\s+[^\s]+){0,3}\.{2,}\s+', '', name).strip()
name = re.sub(r'\s+reviews?\s*$', '', name, flags=re.I).strip()
# Compare-at/sale price pairs ("Rs.3,299.00 Rs.2,999.00") make the
# splitter capture a price fragment as the card name — never a title.
if re.match(r'(?i)^(?:rs\.?|pkr|[\$£€])\s*[\d.,]*$', name):
continue
# A bare ≤3-char single word ("Fun" from "WinFun…", "Set") is brand/
# fragment leakage, never a full card title — and the gate rejects any
# product_title < 4 chars. Align the splitter with that invariant.
if not name or len(name.strip()) < 4:
continue
# Sentence fragments masquerading as cards ("Fun, mini fish-shaped
# erasers for kids…"): a single-word name whose specs continue in
# lowercase is marketing copy split at a comma, not a product card —
# and the paired price belongs to the PREVIOUS card.
if len(re.findall(r'[A-Za-z0-9]+', name)) == 1 and re.match(r'^[a-z]', specs):
continue
currency_label = "Rs." if re.match(r"(?i)^(?:rs\.?|pkr)$", currency_raw.replace(" ", "")) else (currency_raw or "Rs.")
products.append((price_num, f"{currency_label}{price_str}", name, specs))
if products:
for price_num, price_label, name, specs in products:
lines = [f"Product: {name}", f"Price: {price_label}"]
_attr_tags = []
if re.search(r'geforce|gtx|rtx|radeon\s+r[579x]|radeon\s+rx', specs, re.I):
_attr_tags.append("gaming laptop dedicated GPU")
if re.search(r'\btouch\b', specs, re.I) or re.search(r'\btouch\b', name, re.I):
_attr_tags.append("touchscreen display")
if re.search(r'2\s*in\s*1|360|yoga|spin\b', name, re.I):
_attr_tags.append("convertible 2-in-1 laptop")
if re.search(r'\bssd\b', specs, re.I):
_attr_tags.append("fast SSD storage")
if re.search(r'windows', specs, re.I):
_attr_tags.append("Windows laptop")
if re.search(r'android', specs, re.I):
_attr_tags.append("Android device")
if _attr_tags:
lines.append("Features: " + ", ".join(_attr_tags))
for part in [s.strip() for s in specs.split(',')]:
pl = part.lower()
if re.search(r'geforce|nvidia|radeon|amd\s+r|gtx|rtx|mx\d', pl):
lines.append(f"GPU: {part}")
elif re.search(r'\d+\s*gb(?:\s+ram)?$|\bddr\b', pl):
lines.append(f"RAM: {part}")
elif re.search(r'\d+\s*(?:gb|tb)\s+(?:ssd|hdd|emmc)|\d+\s*tb\b', pl):
lines.append(f"Storage: {part}")
elif re.search(r'core\s+i\d|celeron|pentium|ryzen|athlon|snapdragon', pl):
lines.append(f"Processor: {part}")
elif re.search(r'windows|linux|dos|macos|android|chrome\s*os', pl):
lines.append(f"OS: {part}")
elif re.search(r'\d+\.?\d*\"\s*(?:hd|fhd|uhd|ips|touch)?|(?:hd|fhd|uhd|ips)\s+display', pl):
lines.append(f"Display: {part}")
if specs:
lines.append(f"Full specs: {name}, {specs}")
_card_text = "\n".join(lines).strip()
_card_meta = dict(_base_meta)
_card_meta["product_title"] = name
_card_meta["canonical_product_title"] = _canonical_product_title(name)
_card_meta["page_title"] = _base_meta.get("page_title") or name
if price_num > 0:
_card_meta["price"] = price_num
_card_meta["section_id"] = f"catalog_{len(docs)}"
_card_meta["chunk_kind"] = "catalog"
_card_meta["catalog_listing"] = True
_card_meta.update(_extract_product_metadata(_card_text))
docs.append(Document(page_content=_card_text, metadata=_card_meta))
if docs:
return _finalize_docs(_merge_variant_docs(docs))
if len(clean.split()) > 80:
_cat_meta = dict(_base_meta)
_cat_meta["section_id"] = "category_0"
_cat_meta["chunk_kind"] = "category"
_cat_meta["catalog_listing"] = True
_cat_meta["page_title"] = _base_meta.get("page_title") or _derive_page_title(title_hint, clean)
docs.append(Document(page_content=clean[:5000], metadata=_cat_meta))
return _finalize_docs(docs)
# ── Mode 1: product page ──────────────────────────────────────────────────
if len(_PROD_PRICE_RE.findall(clean)) >= 1 and _PROD_SPEC_RE.search(clean):
products = []
for m in _PROD_SPLIT_RE.finditer(clean):
currency_raw = (m.group(1) or "").strip()
price_str = m.group(2).replace(',', '')
try:
price_num = float(price_str)
except:
continue
raw = m.group(3).strip()
comma_idx = raw.find(',')
name = raw[:comma_idx].strip() if comma_idx > 0 else raw
specs = raw[comma_idx + 1:].strip() if comma_idx > 0 else ''
# Strip breadcrumb prefix "Dell Inspiron... Dell Inspiron 15"
name = re.sub(r'^[^\s]+(?:\s+[^\s]+){0,3}\.{2,}\s+', '', name).strip()
name = re.sub(r'\s+reviews?\s*$', '', name, flags=re.I).strip()
# Compare-at/sale price pairs ("Rs.3,299.00 Rs.2,999.00") make the
# splitter capture a price fragment as the card name — never a title.
if re.match(r'(?i)^(?:rs\.?|pkr|[\$£€])\s*[\d.,]*$', name):
continue
if not name or len(name) < 3:
continue
# Sentence fragments masquerading as cards — same guard as Mode 1b.
if len(re.findall(r'[A-Za-z0-9]+', name)) == 1 and re.match(r'^[a-z]', specs):
continue
currency_label = "Rs." if re.match(r"(?i)^(?:rs\.?|pkr)$", currency_raw.replace(" ", "")) else (currency_raw or "Rs.")
products.append((price_num, f"{currency_label}{price_str}", name, specs))
if len(products) >= 1:
for price_num, price_label, name, specs in products:
lines = [f"Product: {name}", f"Price: {price_label}"]
# ── Attribute normalization: inject user-vocabulary tags ──────
_attr_tags = []
if re.search(r'geforce|gtx|rtx|radeon\s+r[579x]|radeon\s+rx', specs, re.I):
_attr_tags.append("gaming laptop dedicated GPU")
if re.search(r'\btouch\b', specs, re.I) or re.search(r'\btouch\b', name, re.I):
_attr_tags.append("touchscreen display")
if re.search(r'2\s*in\s*1|360|yoga|spin\b', name, re.I):
_attr_tags.append("convertible 2-in-1 laptop")
if re.search(r'\bssd\b', specs, re.I):
_attr_tags.append("fast SSD storage")
if re.search(r'windows', specs, re.I):
_attr_tags.append("Windows laptop")
if re.search(r'android', specs, re.I):
_attr_tags.append("Android device")
if _attr_tags:
lines.append("Features: " + ", ".join(_attr_tags))
for part in [s.strip() for s in specs.split(',')]:
pl = part.lower()
if re.search(r'geforce|nvidia|radeon|amd\s+r|gtx|rtx|mx\d', pl):
lines.append(f"GPU: {part}")
elif re.search(r'\d+\s*gb(?:\s+ram)?$|\bddr\b', pl):
lines.append(f"RAM: {part}")
elif re.search(r'\d+\s*(?:gb|tb)\s+(?:ssd|hdd|emmc)|\d+\s*tb\b', pl):
lines.append(f"Storage: {part}")
elif re.search(r'core\s+i\d|celeron|pentium|ryzen|athlon|snapdragon', pl):
lines.append(f"Processor: {part}")
elif re.search(r'windows|linux|dos|macos|android|chrome\s*os', pl):
lines.append(f"OS: {part}")
elif re.search(r'\d+\.?\d*\"\s*(?:hd|fhd|uhd|ips|touch)?|(?:hd|fhd|uhd|ips)\s+display', pl):
lines.append(f"Display: {part}")
if specs:
lines.append(f"Full specs: {name}, {specs}")
_prod_text = "\n".join(lines)
_prod_meta = dict(_base_meta)
_prod_meta["product_title"] = name
if price_num > 0:
_prod_meta["price"] = price_num
_prod_meta["canonical_product_title"] = _canonical_product_title(name)
_prod_meta.update(_extract_product_metadata(_prod_text))
_prod_meta["section_id"] = f"product_{len(docs)}"
docs.append(Document(page_content=_prod_text, metadata=_prod_meta))
if docs:
return _finalize_docs(_merge_variant_docs(docs))
# Mode 2: FAQ / section page
# The split regex fires on any line starting with How/What/Why/etc., so it
# over-triggers on long-form prose. The old code then truncated each section
# with sec[:2000], silently discarding everything past 2000 chars — on long
# pages this dropped >90% of the content and made it unretrievable. Chunk long
# sections (sentence-bounded) instead of truncating, so no content is lost.
sections = _FAQ_SPLIT_RE.split(clean)
if len(sections) >= 3:
for sec in sections:
sec = sec.strip()
if len(sec) <= 40:
continue
if len(sec) <= 1800:
_m2 = dict(_base_meta)
_m2["section_id"] = f"faq_{len(docs)}"
docs.append(Document(page_content=sec, metadata=_m2))
continue
# Long section: pack whole sentences into ~1400-char chunks with a
# one-sentence tail overlap rather than dropping the overflow.
_fs = [s.strip() for s in re.split(r"(?<=[.!?])\s+(?=[A-Z0-9“\"'(])", sec) if s and s.strip()] or [sec]
_fb: list[str] = []
_fc = 0
for _s in _fs:
if _fc and (_fc + len(_s) + 1 > 1400) and _fc >= 300:
_m2 = dict(_base_meta)
_m2["section_id"] = f"faq_{len(docs)}"
docs.append(Document(page_content=" ".join(_fb).strip(), metadata=_m2))
_tail = _fb[-1] if (_fb and len(_fb[-1]) < 400) else ""
_fb, _fc = ([_tail], len(_tail)) if _tail else ([], 0)
if len(_s) > 1800:
for _w in _s.split():
_fb.append(_w)
_fc += len(_w) + 1
if _fc > 1400:
_m2 = dict(_base_meta)
_m2["section_id"] = f"faq_{len(docs)}"
docs.append(Document(page_content=" ".join(_fb).strip(), metadata=_m2))
_fb, _fc = [], 0
continue
_fb.append(_s)
_fc += len(_s) + 1
if _fb and " ".join(_fb).strip():
_m2 = dict(_base_meta)
_m2["section_id"] = f"faq_{len(docs)}"
docs.append(Document(page_content=" ".join(_fb).strip(), metadata=_m2))
if docs:
return _finalize_docs(docs)
# Mode 2b: heading-aware page
try:
_heading_hits = list(re.finditer(r"(?m)^(?:##|###)\s+", raw_text))
if _heading_hits:
_starts = [m.start() for m in _heading_hits] + [len(raw_text)]
_pending_small_segments: list[tuple[str, str | None]] = []
def _emit_docs_from_text(seg_text: str, heading: str | None):
seg_text = (seg_text or "").strip()
if not seg_text:
return
if heading and not re.match(r"(?m)^\s*(?:##|###)\s+", seg_text):
seg_text = f"## {heading}\n\n{seg_text}"
if len(seg_text) <= 1600:
_meta = dict(_base_meta)
_meta["section_id"] = f"head_{len(docs)}"
_meta["chunk_kind"] = "heading"
if heading:
_meta["heading"] = heading
docs.append(Document(page_content=seg_text, metadata=_meta))
return
# Large sections: split on paragraphs / sentence boundaries and
# keep a small overlap so answers spanning a boundary are not lost.
blocks = [p.strip() for p in re.split(r"\n{2,}|(?<=[.?!])\s{2,}", seg_text) if p.strip()]
# A single block can still be huge (long code block, prose with no
# paragraph breaks). bge-small embeds only ~512 tokens (~2000 chars),
# so any oversized block must be hard-split into windows or its tail
# is never embedded (the silent recall killer). Enforce per-block.
_MAX_BLK = 1400
_bounded = []
for _blk in blocks:
if len(_blk) <= _MAX_BLK:
_bounded.append(_blk)
else:
_bounded.extend(_blk[i:i + 1200] for i in range(0, len(_blk), 1200))
blocks = _bounded or [seg_text[i:i + 1200] for i in range(0, len(seg_text), 1200)]
packed = []
buf = []
buf_chars = 0
for blk in blocks:
if buf and (buf_chars + len(blk) + 2 > 1400):
packed.append("\n\n".join(buf).strip())
tail = packed[-1][-180:].strip()
buf = [tail] if tail else []
buf_chars = len(tail) + (2 if tail else 0)
buf.append(blk)
buf_chars += len(blk) + 2
if buf:
packed.append("\n\n".join(buf).strip())
for idx, piece in enumerate(packed):
if not piece:
continue
if idx and packed[idx - 1]:
prev_tail = packed[idx - 1][-120:].strip()
if prev_tail and not piece.startswith(prev_tail):
piece = f"{prev_tail}\n\n{piece}"
_meta = dict(_base_meta)
_meta["section_id"] = f"head_{len(docs)}"
_meta["chunk_kind"] = "heading"
if heading:
_meta["heading"] = heading
docs.append(Document(page_content=piece, metadata=_meta))
def _flush_small_segments():
nonlocal _pending_small_segments
if not _pending_small_segments:
return
merged_parts = []
for seg_text, seg_heading in _pending_small_segments:
seg_text = (seg_text or "").strip()
if not seg_text:
continue
if seg_heading and not re.match(r"(?m)^\s*(?:##|###)\s+", seg_text):
seg_text = f"## {seg_heading}\n\n{seg_text}"
merged_parts.append(seg_text)
merged = "\n\n".join(merged_parts).strip()
_pending_small_segments = []
if merged:
_emit_docs_from_text(merged, None)
for idx, start in enumerate(_starts[:-1]):
seg = raw_text[start:_starts[idx + 1]].strip()
hm = re.match(r"^(?:##|###)\s+(.+?)\s*(?=\Z|(?:##|###)\s+)", seg, re.S)
heading = hm.group(1).strip() if hm else None
if len(seg) <= 280:
_pending_small_segments.append((seg, heading))
continue
_flush_small_segments()
_emit_docs_from_text(seg, heading)
_flush_small_segments()
if docs:
return _finalize_docs(docs)
except Exception:
pass
# Mode 2c: paragraph-aware grouping
try:
_paras = [p.strip() for p in re.split(r"\n{2,}", raw_text) if p and p.strip()]
if len(_paras) >= 2:
_seed_heading = str((page_meta or {}).get("page_title") or "").strip()
if _seed_heading and re.search(r"(?i)(web scraper test sites|all rights reserved|privacy policy|terms of service|home\s*\|\s*[^|]+)$", _seed_heading):
_seed_heading = ""
_last_heading = _seed_heading
_buf: list[str] = []
_chars = 0
_last_emitted: str | None = None
def _emit_para_chunk(chunk_text: str):
nonlocal _last_emitted
chunk_text = (chunk_text or "").strip()
if not chunk_text:
return
if _last_emitted:
tail = _last_emitted[-120:].strip()
if tail and not chunk_text.startswith(tail):
chunk_text = f"{tail}\n\n{chunk_text}"
_last_emitted = chunk_text
_meta = dict(_base_meta)
_meta["section_id"] = f"para_{len(docs)}"
_meta["chunk_kind"] = "paragraph"
if _last_heading:
_meta["heading"] = _last_heading
docs.append(Document(page_content=chunk_text, metadata=_meta))
def _flush_para_buf():
nonlocal _buf, _chars
if not _buf:
return
_chunk = "\n\n".join(_buf).strip()
if not _chunk:
_buf, _chars = [], 0
return
if _last_heading and not re.match(r"(?m)^\s*(?:##|###)\s+", _chunk):
_chunk = f"## {_last_heading}\n\n{_chunk}"
_emit_para_chunk(_chunk)
_buf, _chars = [], 0
for para in _paras:
_hm = re.match(r"^\s*(?:##|###)\s+(.+?)\s*$", para)
if _hm:
_flush_para_buf()
_last_heading = _hm.group(1).strip()
continue
# A single paragraph with no internal breaks can exceed the
# embedding window; window-split it so its tail still gets embedded.
_para_pieces = ([para] if len(para) <= 1600
else [para[i:i + 1200] for i in range(0, len(para), 1200)])
for _para_txt in _para_pieces:
if _buf and (_chars + len(_para_txt) + 2 > 1600):
_flush_para_buf()
_buf.append(_para_txt)
_chars += len(_para_txt) + 2
_flush_para_buf()
if docs:
return _finalize_docs(docs)
except Exception:
pass
# Mode 2d: tabular / key-value / code-ish page
try:
_lines = [ln.rstrip() for ln in raw_text.splitlines()]
_nonempty_lines = [ln.strip() for ln in _lines if ln.strip()]
def _is_rowish(ln: str) -> bool:
s = ln.strip()
if not s:
return False
if re.match(r"^\s*(?:##|###)\s+", s):
return False
if s.startswith(("```", "~~~")):
return True
if "\t" in s or " | " in s:
return True
if re.match(r"^[A-Za-z][A-Za-z0-9 _/\-]{1,40}\s*:\s+\S", s):
return True
if re.match(r"^[A-Za-z][A-Za-z0-9 _/\-]{1,40}\s*=\s*\S", s):
return True
if re.match(r"^(?:[-*•]|\d{1,2}[.)])\s+\S", s):
return True
if re.match(r"^\s{2,}\S", ln):
return True
return False
_row_lines = [ln for ln in _nonempty_lines if _is_rowish(ln)]
if len(_row_lines) >= 4 and len(_row_lines) >= max(4, int(len(_nonempty_lines) * 0.45)):
_last_heading = ""
_buf: list[str] = []
_chars = 0
def _flush_row_buf():
nonlocal _buf, _chars
if not _buf:
return
_chunk = "\n".join(_buf).strip()
if _last_heading and not re.match(r"(?m)^\s*(?:##|###)\s+", _chunk):
_chunk = f"## {_last_heading}\n\n{_chunk}"
_meta = dict(_base_meta)
_meta["section_id"] = f"table_{len(docs)}"
_meta["chunk_kind"] = "tabular"
if _last_heading:
_meta["heading"] = _last_heading
docs.append(Document(page_content=_chunk[:2400], metadata=_meta))
_buf, _chars = [], 0
for ln in _lines:
s = ln.strip()
if not s:
continue
hm = re.match(r"^\s*(?:##|###)\s+(.+?)\s*$", s)
if hm:
_flush_row_buf()
_last_heading = hm.group(1).strip()
continue
if not _is_rowish(ln):
continue
if _buf and (_chars + len(s) + 1 > 1600):
_flush_row_buf()
_buf.append(s)
_chars += len(s) + 1
_flush_row_buf()
if docs:
return _finalize_docs(docs)
except Exception:
pass
# Mode 2.5: bullet/list page
try:
_raw_lines = [ln.strip() for ln in raw_text.split("\n") if ln and ln.strip()]
_bullet_re = re.compile(r"^(?:[-*•]|\d{1,2}[.)])\s+")
_bullet_lines = [ln for ln in _raw_lines if _bullet_re.match(ln)]
if len(_bullet_lines) >= 3:
buf = []
buf_chars = 0
for ln in _raw_lines:
if len(ln) > 500:
continue
if not _bullet_re.match(ln):
continue
if buf and (buf_chars + len(ln) + 1 > 2000):
_m3 = dict(_base_meta)
_m3["section_id"] = f"list_{len(docs)}"
_m3["chunk_kind"] = "list"
docs.append(Document(page_content="\n".join(buf).strip(), metadata=_m3))
buf, buf_chars = [], 0
buf.append(ln)
buf_chars += len(ln) + 1
if buf:
_m4 = dict(_base_meta)
_m4["section_id"] = f"list_{len(docs)}"
_m4["chunk_kind"] = "list"
docs.append(Document(page_content="\n".join(buf).strip(), metadata=_m4))
if docs:
return _finalize_docs(docs)
except Exception:
pass
# Mode 3 (legacy word-split) — kept verbatim for any non-prose page that slips
# through to the fallback (product/catalog/category). A raw word-count window is
# fine for price-card text and must not change, so product DBs are untouched.
def _legacy_word_split():
words = clean.split()
for j in range(0, max(1, len(words)), chunk_step):
chunk = " ".join(words[j:j + chunk_size])
if len(chunk) > 20:
_m5 = dict(_base_meta)
_m5["section_id"] = f"generic_{len(docs)}"
docs.append(Document(page_content=chunk, metadata=_m5))
if j + chunk_size >= len(words):
break
return _finalize_docs(docs)
if str(page_meta.get("page_type") or "") in {"product", "catalog", "category"}:
return _legacy_word_split()
# Mode 3 (prose): structure-aware, sentence-bounded, heading-prefixed.
# The old raw word-count window cut sentences mid-phrase and stripped section
# context, diluting chunk embeddings and hurting retrieval recall (the dominant
# failure mode on docs DBs). We now pack WHOLE sentences to a focused size and
# prepend the page title as heading context (mirrors Mode 2c). Smaller, coherent,
# context-anchored chunks embed far more precisely.
_heading_ctx = str(_base_meta.get("page_title") or "").strip()
if _heading_ctx and re.search(r"(?i)(all rights reserved|privacy policy|terms of service|home\s*\|)$", _heading_ctx):
_heading_ctx = ""
_prose_src = re.sub(r"\s+", " ", clean).strip()
_sents = [s.strip() for s in re.split(r"(?<=[.!?])\s+(?=[A-Z0-9“\"'(])", _prose_src) if s and s.strip()]
if not _sents and _prose_src:
_sents = [_prose_src]
_TARGET = 1100 # ~180 words: focused enough for precise embeddings
_HARDMAX = 1600
_pbuf: list[str] = []
_plen = 0
def _emit_prose():
nonlocal _pbuf, _plen
_body = " ".join(_pbuf).strip()
if not _body:
_pbuf, _plen = [], 0
return
if _heading_ctx:
_body = f"## {_heading_ctx}\n\n{_body}"
_m = dict(_base_meta)
_m["section_id"] = f"prose_{len(docs)}"
_m["chunk_kind"] = "prose"
if _heading_ctx:
_m["heading"] = _heading_ctx
docs.append(Document(page_content=_body, metadata=_m))
# carry the last sentence as overlap for cross-chunk continuity
_carry = _pbuf[-1] if _pbuf else ""
if _carry and len(_carry) < 400:
_pbuf, _plen = [_carry], len(_carry)
else:
_pbuf, _plen = [], 0
for _s in _sents:
if _plen and (_plen + len(_s) + 1 > _TARGET) and _plen >= 300:
_emit_prose()
if len(_s) > _HARDMAX: # pathological single sentence: word-pack it
for _w in _s.split():
_pbuf.append(_w)
_plen += len(_w) + 1
if _plen > _TARGET:
_emit_prose()
continue
_pbuf.append(_s)
_plen += len(_s) + 1
if _pbuf:
_emit_prose()
if not docs: # safety: sentence packing yielded nothing
return _legacy_word_split()
return _finalize_docs(docs)
def _extract_product_metadata(text: str) -> dict:
"""Parse product-catalog chunk text into structured ChromaDB metadata fields."""
meta = {}
pm = re.search(r'Price:\s*(?:\$|£|€|\brs\.?\s*|\bpkr\s*)?([\d,]+\.?\d*)', text, re.I)
if pm:
try:
_pv = float(pm.group(1).replace(',', ''))
if _pv > 0: # "Price: 0.00" = OOS placeholder, never a price
meta['price'] = _pv
except:
pass
rm = re.search(r'RAM:\s*(\d+)\s*GB', text, re.I)
if rm:
try:
meta['ram_gb'] = int(rm.group(1))
except:
pass
gm = re.search(r'GPU:[^\n]*?(\d+)\s*GB', text, re.I)
if gm:
try:
meta['gpu_vram_gb'] = int(gm.group(1)); meta['has_gpu'] = 1
except:
pass
else:
meta['has_gpu'] = 0
meta['has_touch'] = 1 if re.search(r'Display:[^\n]*\bTouch\b', text) else 0
meta['is_convertible'] = 1 if re.search(r'\bconvertible\b', text, re.I) else 0
meta['has_ssd'] = 1 if re.search(r'\bSSD\b', text) else 0
rvm = re.search(r'(\d+)\s+(?:customer\s+)?reviews?\b', text, re.I)
if rvm:
try:
meta['review_count'] = int(rvm.group(1))
except Exception:
pass
return meta
def _enrich_docs_metadata(docs: list) -> list:
"""Auto-detect product catalog chunks and enrich with structured metadata."""
for doc in docs:
if re.search(r'^Product:\s+\S', doc.page_content, re.M):
extracted = _extract_product_metadata(doc.page_content)
if doc.metadata:
doc.metadata.update(extracted)
else:
doc.metadata = extracted
return docs
|