Spaces:
Sleeping
Sleeping
File size: 9,217 Bytes
ce8f04a | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 | """Document intel: Trafilatura HTML, PyMuPDF/pypdf, PL/EU legal citations."""
from __future__ import annotations
from pathlib import Path
import pytest
SAMPLE_HTML = """
<!DOCTYPE html>
<html><head><title>Regulamin naboru SMART</title></head>
<body>
<nav>Menu · Kontakt · Logowanie · Cookie banner</nav>
<main>
<article>
<h1>Regulamin naboru Ścieżka SMART</h1>
<p>Niniejszy regulamin określa zasady ubiegania się o dofinansowanie
zgodnie z rozporządzeniem (UE) 2021/1058 oraz CELEX 32021R1058.</p>
<p>Podstawa prawna: Dz.U. z 2021 r. poz. 1234. Termin składania wniosków
do 30.09.2027 r.</p>
<p>Art. 14 ust. 1 określa obowiązki beneficjenta w zakresie kwalifikowalności.</p>
<table><tr><th>Kryterium</th><th>Punkty</th></tr>
<tr><td>Innowacyjność</td><td>30</td></tr></table>
</article>
</main>
<footer>Copyright PARP · Polityka prywatności · Mapa strony</footer>
</body></html>
"""
def test_html_main_content_strips_chrome():
from core.document_intel.html_extract import html_to_clean_text, extract_main_content
result = html_to_clean_text(SAMPLE_HTML, url="https://www.parp.gov.pl/smart")
text = result["text"]
assert result["chars"] >= 80
assert "Regulamin naboru" in text or "SMART" in text
assert "Cookie banner" not in text or result["extractor"] in ("trafilatura", "bs4")
# main content must keep legal bits
assert "2021" in text
assert extract_main_content(SAMPLE_HTML)
def test_legal_citations_celex_du_ue():
from core.document_intel.legal_citations import extract_legal_citations
text = (
"Zgodnie z CELEX:32021R1058 oraz rozporządzeniem (UE) 2021/1058 "
"oraz Dz.U. z 2021 r. poz. 1234 Art. 14 ust. 1."
)
out = extract_legal_citations(text)
assert out["count"] >= 2
assert any("32021R1058" in c for c in out["celex_ids"])
assert any("DU/2021/1234" == c for c in out["du_refs"])
assert out["ue_regs"]
# no invent
empty = extract_legal_citations("Brak podstawy prawnej w tym akapicie.")
assert empty["count"] == 0 or not empty["celex_ids"]
def test_extract_from_html_pipeline_includes_legal():
from core.document_intel.pipeline import extract_from_html
out = extract_from_html(SAMPLE_HTML, url="https://example.gov.pl/reg")
assert out["ok"] is True
assert out["legal"]["count"] >= 1
assert out["extractor"] in ("trafilatura", "bs4")
def test_pdf_extract_from_simple_pdf(tmp_path: Path):
"""Generate minimal PDF via reportlab or skip if unavailable; prefer pymupdf write."""
pdf_path = tmp_path / "reg.pdf"
try:
import fitz
doc = fitz.open()
page = doc.new_page()
page.insert_text(
(72, 72),
"Regulamin testowy. CELEX 32021R0695. Dz.U. z 2022 r. poz. 55.",
)
doc.save(str(pdf_path))
doc.close()
except Exception:
pytest.skip("pymupdf cannot create PDF in this env")
from core.document_intel.pdf_extract import extract_pdf_text
from core.document_intel.legal_citations import extract_legal_citations
result = extract_pdf_text(pdf_path)
assert result["parser"] in ("pymupdf", "pypdf")
assert result["chars"] >= 20
assert "Regulamin" in result["text"] or "CELEX" in result["text"]
cites = extract_legal_citations(result["text"])
assert cites["count"] >= 1
def test_pdf_parser_local_cascade_uses_document_intel(tmp_path: Path):
try:
import fitz
except ImportError:
pytest.skip("pymupdf missing")
pdf_path = tmp_path / "local.pdf"
doc = fitz.open()
page = doc.new_page()
page.insert_text((72, 72), "Dokument grantowy PARP SMART § 1. Podstawa CELEX 32021R1058.")
doc.save(str(pdf_path))
doc.close()
from rag_pipeline.pdf_parser import _parse_local_pdf_sync
out = _parse_local_pdf_sync(str(pdf_path))
assert out["text"]
assert out["parser"] in ("pymupdf", "pypdf")
assert "SMART" in out["text"] or "CELEX" in out["text"] or "PARP" in out["text"]
def test_eurlex_batch_still_rejects_free_text_titles():
"""Regression: document intel must not weaken EUR-Lex legal-ID gate."""
from integrations.eurlex_client import _sanitize_search_query, is_valid_eurlex_query
assert _sanitize_search_query("PARP Harmonogram naborów SMART") == ""
assert is_valid_eurlex_query("32021R1058")
def test_stealth_fetch_module_and_domain_gate():
from core.document_intel.stealth_fetch import (
is_stealth_fetch_enabled,
should_use_stealth,
stealth_get,
)
assert is_stealth_fetch_enabled() is True
assert should_use_stealth("https://www.parp.gov.pl/component/grants")
assert should_use_stealth("https://www.bgk.pl/oferta/")
assert not should_use_stealth("https://example.com/page")
# Offline: invalid URL
bad = stealth_get("not-a-url")
assert bad["ok"] is False
# Mocked success path — no network
from unittest.mock import MagicMock, patch
class FakeResp:
status_code = 200
text = "<html><body><main>" + ("Regulamin SMART " * 40) + "</main></body></html>"
content = text.encode()
headers = {"content-type": "text/html"}
url = "https://www.parp.gov.pl/x"
fake_mod = MagicMock()
fake_mod.get = MagicMock(return_value=FakeResp())
with patch.dict("sys.modules", {"curl_cffi": MagicMock(), "curl_cffi.requests": fake_mod}):
# re-import path uses from curl_cffi import requests
with patch("core.document_intel.stealth_fetch.stealth_get") as direct:
# unit the real function with patched import inside
pass
# Call real implementation with patched curl_cffi.requests
import core.document_intel.stealth_fetch as sf
with patch.object(sf, "is_stealth_fetch_enabled", return_value=True):
with patch(
"curl_cffi.requests.get",
return_value=FakeResp(),
):
# stealth_get does `from curl_cffi import requests as curl_requests`
import curl_cffi.requests as cr
with patch.object(cr, "get", return_value=FakeResp()):
out = stealth_get("https://www.parp.gov.pl/demo", timeout=5)
assert out["ok"] is True
assert out["status_code"] == 200
assert len(out["text"]) >= 200
assert out["impersonate"]
def test_crawl4ai_hard_blocked_uses_stealth_not_empty(monkeypatch):
"""PARP URL must try stealth instead of hard-skipping to empty."""
import asyncio
from unittest.mock import AsyncMock, patch
from core.crawl4ai_client import scrape_url_to_markdown
async def fake_stealth(url, **kwargs):
return {
"ok": True,
"status_code": 200,
"text": (
"<html><body><main><h1>PARP SMART</h1>"
+ ("<p>Treść regulaminu naboru. " * 30)
+ "</main></body></html>"
),
"content": b"",
"headers": {},
"final_url": url,
"impersonate": "chrome",
"error": None,
}
async def _run():
with patch(
"core.document_intel.stealth_fetch.stealth_get_async",
new=AsyncMock(side_effect=fake_stealth),
):
md = await scrape_url_to_markdown("https://www.parp.gov.pl/component/grants/grants")
assert md
assert "SMART" in md or "regulaminu" in md.lower() or "naboru" in md.lower()
assert len(md) >= 80
asyncio.run(_run())
@pytest.mark.integration
def test_live_stealth_parp_optional():
"""Optional live check — skip if network/WAF blocks CI."""
import os
if os.environ.get("RUN_LIVE_STEALTH", "").lower() not in ("1", "true", "yes"):
pytest.skip("Set RUN_LIVE_STEALTH=1 to hit PARP live")
from core.document_intel.stealth_fetch import stealth_get
out = stealth_get("https://www.parp.gov.pl/", timeout=25)
assert out["ok"] is True
assert out["status_code"] == 200
assert len(out.get("text") or "") > 1000
def test_page_fetcher_hash_uses_main_content_when_available():
"""HttpPageFetcher should prefer trafilatura main-text hash when possible."""
import asyncio
from unittest.mock import MagicMock, patch
from core.grants.page_fetcher import HttpPageFetcher, content_hash
from core.document_intel.html_extract import extract_main_content
main = extract_main_content(SAMPLE_HTML)
assert main
fake_resp = MagicMock()
fake_resp.text = SAMPLE_HTML
fake_resp.status_code = 200
async def _run():
with patch("requests.get", return_value=fake_resp):
fetcher = HttpPageFetcher(timeout=5)
page = await fetcher.fetch("https://www.parp.gov.pl/demo")
assert page.status_code == 200
assert page.body == SAMPLE_HTML
# hash should match main content when trafilatura/bs4 works
assert page.content_hash == content_hash(main) or page.content_hash == content_hash(
SAMPLE_HTML
)
assert page.source in ("http+trafilatura", "http")
asyncio.run(_run())
|