ContiAI / core /books /fetch_url.py
ziadsameh32's picture
Add login page
325b94c
Raw
History Blame Contribute Delete
2.52 kB
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from requests.exceptions import RequestException
PDF_HEADERS = {
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/121.0.0.0 Safari/537.36"
),
"Accept": "application/pdf,application/octet-stream;q=0.9,*/*;q=0.8",
"Accept-Language": "en-US,en;q=0.9,ar;q=0.8",
"Referer": "https://www.google.com/",
}
def fetch_url(url: str):
try:
r = requests.get(
url,
timeout=20,
allow_redirects=True,
headers=PDF_HEADERS,
)
r.raise_for_status()
ct = r.headers.get("Content-Type", "").lower()
return ct, r.content, r.url
except RequestException:
return None, None, None
def looks_like_pdf(content: bytes) -> bool:
return bool(content and content.startswith(b"%PDF"))
def get_pdf_bytes(source_url: str) -> bytes | None:
ct, content, final_url = fetch_url(source_url)
if not ct or not content:
return None
# -------------------------
# 1️⃣ PDF مباشر (صريح أو كذاب)
# -------------------------
if (
"application/pdf" in ct
or final_url.lower().endswith(".pdf")
or looks_like_pdf(content)
):
return content
# -------------------------
# 2️⃣ HTML wrapper
# -------------------------
if "text/html" in ct:
try:
soup = BeautifulSoup(
content.decode("utf-8", errors="ignore"),
"html.parser",
)
# a[href]
for a in soup.find_all("a", href=True):
href = a["href"].strip()
if ".pdf" in href.lower():
pdf_url = urljoin(final_url, href)
ct2, content2, _ = fetch_url(pdf_url)
if ct2 and ("application/pdf" in ct2 or looks_like_pdf(content2)):
return content2
# iframe / embed
for tag in soup.find_all(["iframe", "embed"], src=True):
src = tag["src"]
if ".pdf" in src.lower():
pdf_url = urljoin(final_url, src)
ct2, content2, _ = fetch_url(pdf_url)
if ct2 and ("application/pdf" in ct2 or looks_like_pdf(content2)):
return content2
except Exception:
return None
return None