File size: 18,289 Bytes
cdc7861
94c35fd
 
 
 
 
 
 
 
 
e9beec1
 
cdc7861
 
 
 
e9beec1
cdc7861
e9beec1
 
cdc7861
 
 
 
e9beec1
 
 
 
 
 
94c35fd
2217911
 
 
 
 
 
 
e9beec1
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
94c35fd
e9beec1
 
94c35fd
 
e9beec1
 
94c35fd
 
 
 
 
 
 
 
 
 
 
 
 
 
2217911
 
 
 
94c35fd
 
 
 
 
 
 
e9beec1
 
 
 
 
 
 
 
 
 
7d15083
 
 
 
2217911
7d15083
e9beec1
 
 
ad29f85
7d15083
 
 
 
e9beec1
 
 
 
 
 
94c35fd
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e9beec1
 
 
 
 
 
 
2217911
 
 
 
 
 
 
 
 
 
 
 
 
 
94c35fd
cdc7861
ad29f85
 
94c35fd
2217911
94c35fd
e9beec1
 
94c35fd
e9beec1
 
 
cdc7861
e9beec1
cdc7861
e9beec1
 
ad29f85
2217911
 
 
e9beec1
2217911
ad29f85
 
 
e9beec1
94c35fd
e9beec1
 
 
 
 
 
94c35fd
 
 
 
 
 
 
 
 
 
e9beec1
 
ad29f85
e9beec1
94c35fd
ad29f85
 
 
 
 
e9beec1
94c35fd
 
 
 
 
 
 
 
e9beec1
 
 
 
ad29f85
94c35fd
 
 
 
2217911
 
 
 
 
 
ad29f85
e9beec1
 
 
 
 
 
 
2217911
 
 
 
 
e9beec1
 
 
2217911
 
 
 
 
 
e9beec1
 
 
 
 
 
 
 
 
 
2217911
ad29f85
e9beec1
ad29f85
 
 
2217911
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
ad29f85
2217911
 
e9beec1
 
 
 
 
 
 
 
 
 
 
cdc7861
e9beec1
 
 
 
 
 
cdc7861
e9beec1
 
 
 
 
 
 
 
 
 
 
cdc7861
e9beec1
 
 
 
cdc7861
e9beec1
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
cdc7861
ad29f85
 
94c35fd
ad29f85
94c35fd
cdc7861
e9beec1
94c35fd
cdc7861
 
ad29f85
 
 
 
 
 
 
 
e9beec1
 
 
 
 
 
 
 
 
 
 
 
 
 
cdc7861
 
e9beec1
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
"""
HBL PDF crawler + extractor (v2)
----------------------------------
Changes from v1:
- Seeds every {function}/Downloads.aspx?library={lib}&site=/{function} combo
  directly, since MyWorkspace.aspx does not link out to other departments.
- Handles SharePoint's "load more" postback (only ~67 items load per initial
  GET; the rest need repeated postback clicks).
- Everything else (PDF download via cookie-cloned session, pdfplumber
  extraction, OCR fallback, JSON output) is unchanged from v1.
Run this on an internet-connected machine, NOT the air-gapped server.
Copy the output folder over afterward, same as your other offline transfers.
"""

import os
import re
import json
import time
import hashlib
from pathlib import Path
from urllib.parse import urljoin, urlparse

import requests
from bs4 import BeautifulSoup
from tqdm import tqdm

import pdfplumber

from selenium import webdriver
from selenium.webdriver.edge.options import Options as EdgeOptions
from selenium.webdriver.common.by import By
from selenium.common.exceptions import WebDriverException

try:
    from requests_negotiate_sspi import HttpNegotiateAuth
    HAS_SSPI = True
except ImportError:
    HAS_SSPI = False

# Optional OCR fallback β€” only imported if a scanned PDF is actually encountered
def _ocr_pdf(pdf_path):
    import pytesseract
    from pdf2image import convert_from_path
    images = convert_from_path(pdf_path)
    text = ""
    for i, image in enumerate(images):
        text += f"\n--- OCR page {i+1} ---\n"
        text += pytesseract.image_to_string(image)
    return text

# ---------------------------------------------------------------------------
# Config
# ---------------------------------------------------------------------------
BASE_DOMAIN = "iamhbl.com"
OUTPUT_DIR = Path("./hbl_pdf_crawl")
PDF_DIR = OUTPUT_DIR / "pdfs"
MAX_PAGES_TO_CRAWL = 3000
REQUEST_DELAY_SECONDS = 0.5
PAGE_LOAD_WAIT_SECONDS = 1.5   # let JS-rendered content finish loading before reading the page
POSTBACK_WAIT_SECONDS = 1.2    # wait after each load-more click for postback to settle
MAX_LOAD_MORE_CLICKS = 60      # safety cap per page
REQUEST_TIMEOUT = 20

# ---------------------------------------------------------------------------
# Seed matrix β€” every department function x library combo.
# CONFIRM library spelling per department before running at scale (Circular vs
# Circulars, Forms vs Form, etc. β€” SharePoint libraries are not guaranteed to
# be named consistently). Open one Downloads.aspx per function in a browser
# and read the tab hrefs to get the exact library= value.
# ---------------------------------------------------------------------------
FUNCTIONS = [
    "HOD", "CASB", "GCG", "ISE", "RB", "RM", "GT",
    "BOAS", "CIB", "CS", "DF", "EPT", "FCB", "FIGRB",
    "IA", "IB", "IBAN", "IFI", "ITSE", "LL", "MB", "OS",
    "RSC", "SQ", "SS", "TTPO",
]  # from the department dropdown seen in Downloads.aspx; trim/extend as confirmed

LIBRARIES = ["Circulars", "Forms", "FAQ", "Policies", "Reports and Reviews", "Procedures"]  # confirmed spelling from run output

MAX_NETWORK_RETRIES = 4
NETWORK_RETRY_BASE_DELAY = 5  # seconds; doubles each retry

SEED_URLS = [
    f"https://iamhbl.com/{fn}/Pages/Downloads.aspx?library={lib}&site=%2f{fn}"
    for fn in FUNCTIONS
    for lib in LIBRARIES
]

# ---------------------------------------------------------------------------
# AUTH β€” attaches to an Edge window YOU already logged into manually, instead
# of copying cookies by hand. Before running this script:
#   1. Close every Edge window (check Task Manager for msedge.exe).
#   2. Run:  msedge.exe --remote-debugging-port=9222 --user-data-dir="C:\edge-debug-profile"
#   3. Log into iamhbl.com in that window.
#   4. Then run this script β€” it attaches to that same window and session.
# ---------------------------------------------------------------------------
EDGE_DEBUGGER_ADDRESS = "127.0.0.1:9222"

# If Selenium's automatic driver download hangs or fails on this network-restricted
# laptop, download msedgedriver.exe manually (matching your Edge version) from
# https://developer.microsoft.com/en-us/microsoft-edge/tools/webdriver/ and set
# its path here. Leave as None to let Selenium try to auto-manage it.
MSEDGEDRIVER_PATH = r"C:\Users\syed.zaki1\Downloads\edgedriver_win64\msedgedriver.exe"  # e.g. r"C:\edgedriver\msedgedriver.exe"

def get_driver():
    options = EdgeOptions()
    options.debugger_address = EDGE_DEBUGGER_ADDRESS
    options.page_load_strategy = "eager"  # return once DOM is ready, don't wait for every image/tracker to finish
    if MSEDGEDRIVER_PATH:
        from selenium.webdriver.edge.service import Service
        service = Service(executable_path=MSEDGEDRIVER_PATH)
        return webdriver.Edge(service=service, options=options)
    return webdriver.Edge(options=options)

def looks_like_login_page(html_lower):
    return any(marker in html_lower for marker in
               ('id="username"', 'id="password"', 'sign in', 'log in to continue'))

def looks_like_error_or_empty(html_lower):
    # SharePoint returns a rendered page even for a bad library= value, but the
    # repeater is empty. Catches the common "no items" / webpart error markers.
    return any(marker in html_lower for marker in
               ('an unexpected error has occurred', 'this list is empty', 'webpartzone1'))

# ---------------------------------------------------------------------------
# Load-more postback handling
# ---------------------------------------------------------------------------
LOAD_MORE_CANDIDATE_STRATEGIES = [
    # (By, selector) pairs tried in order; first visible match wins.
    (By.XPATH, "//a[contains(translate(text(),'LOADMORE','loadmore'),'load more')]"),
    (By.XPATH, "//button[contains(translate(text(),'LOADMORE','loadmore'),'load more')]"),
    (By.XPATH, "//*[contains(@id,'loadmore') or contains(@id,'LoadMore')]"),
    (By.CSS_SELECTOR, "[id*='loadmore' i]"),
]

def find_load_more_button(driver):
    for by, selector in LOAD_MORE_CANDIDATE_STRATEGIES:
        try:
            elements = driver.find_elements(by, selector)
            for el in elements:
                if el.is_displayed() and el.is_enabled():
                    return el
        except Exception:
            continue
    return None

def click_load_more_until_done(driver, max_clicks=MAX_LOAD_MORE_CLICKS):
    """Repeatedly clicks the load-more control until it disappears or the page
    stops growing. Returns the number of clicks performed."""
    clicks = 0
    prev_html_len = len(driver.page_source)
    for _ in range(max_clicks):
        btn = find_load_more_button(driver)
        if btn is None:
            break
        try:
            driver.execute_script("arguments[0].scrollIntoView(true);", btn)
            btn.click()
        except Exception as e:
            print(f"    [load-more click failed] {e}")
            break
        time.sleep(POSTBACK_WAIT_SECONDS)
        clicks += 1
        new_html_len = len(driver.page_source)
        if new_html_len <= prev_html_len:
            # postback happened but content didn't grow β€” stop to avoid a loop
            break
        prev_html_len = new_html_len
    return clicks

# ---------------------------------------------------------------------------
# Crawl: find every PDF URL, driving the real logged-in browser
# ---------------------------------------------------------------------------
def is_same_domain(url):
    netloc = urlparse(url).netloc.lower()
    return netloc == BASE_DOMAIN or netloc.endswith("." + BASE_DOMAIN)

def safe_driver_get(driver, url, max_retries=MAX_NETWORK_RETRIES):
    """driver.get() with retry+backoff. Returns True on success, False if the
    page never loaded after all retries (caller should skip and move on)."""
    for attempt in range(max_retries):
        try:
            driver.get(url)
            return True
        except WebDriverException as e:
            delay = NETWORK_RETRY_BASE_DELAY * (2 ** attempt)
            print(f"  [network error, retry {attempt+1}/{max_retries} in {delay}s] {url}")
            print(f"    {str(e).splitlines()[0]}")
            time.sleep(delay)
    return False

def crawl_and_download(driver, sess, seed_urls):
    visited_pages = set()
    pdf_links_found = set()
    downloaded_count = 0
    empty_seeds = []
    failed_seeds = []
    queue = list(seed_urls)
    login_warned = False

    pbar = tqdm(total=min(MAX_PAGES_TO_CRAWL, len(queue)), desc="Crawling seeded pages")
    while queue and len(visited_pages) < MAX_PAGES_TO_CRAWL:
        url = queue.pop(0)
        if url in visited_pages or not is_same_domain(url):
            continue

        visited_pages.add(url)
        pbar.update(1)

        page_t0 = time.time()
        if not safe_driver_get(driver, url):
            print(f"  [gave up after {MAX_NETWORK_RETRIES} retries] {url}")
            failed_seeds.append(url)
            continue
        time.sleep(PAGE_LOAD_WAIT_SECONDS)
        page_elapsed = time.time() - page_t0
        if page_elapsed > 8:
            print(f"  [slow page: {page_elapsed:.1f}s] {url}")

        html_lower = driver.page_source.lower()

        if looks_like_login_page(html_lower) and not login_warned:
            print(f"\n  WARNING: {url} looks like a login page β€” your Edge session may not be logged in.")
            print("  Log in manually in the attached Edge window, then rerun.\n")
            login_warned = True

        if looks_like_error_or_empty(html_lower):
            empty_seeds.append(url)

        # Exhaust pagination before parsing the final DOM
        clicks = click_load_more_until_done(driver)
        if clicks:
            pbar.set_postfix(last_clicks=clicks)

        soup = BeautifulSoup(driver.page_source, "html.parser")
        page_pdf_count = 0
        for a in soup.find_all("a", href=True):
            href = urljoin(url, a["href"]).split("#")[0]
            href_path = urlparse(href).path.lower()
            if href_path.endswith(".pdf"):
                page_pdf_count += 1
                if href not in pdf_links_found:
                    pdf_links_found.add(href)
                    if download_one_pdf(sess, href):
                        downloaded_count += 1
                        pbar.set_postfix(pdfs=downloaded_count)
            elif is_same_domain(href) and href not in visited_pages:
                # Only follow same-domain links that are themselves other
                # Downloads.aspx pages (e.g. sub-folder drilldowns) β€” avoids
                # wandering off into unrelated intranet pages.
                if "downloads.aspx" in href.lower():
                    queue.append(href)

        if page_pdf_count == 0:
            print(f"  [0 pdfs] {url}")

        time.sleep(REQUEST_DELAY_SECONDS)

    pbar.close()
    print(f"\nCrawled {len(visited_pages)} pages. Found {len(pdf_links_found)} PDF links, downloaded {downloaded_count}.")
    if empty_seeds:
        print(f"\n{len(empty_seeds)} seed URLs looked empty/errored β€” check library= spelling for these:")
        for u in empty_seeds:
            print(f"  {u}")
    if failed_seeds:
        print(f"\n{len(failed_seeds)} seed URLs failed after {MAX_NETWORK_RETRIES} retries (likely WiFi/DNS drop) β€” rerun these:")
        for u in failed_seeds:
            print(f"  {u}")
        (OUTPUT_DIR / "failed_seeds.txt").write_text("\n".join(failed_seeds), encoding="utf-8")
        print(f"  Written to {OUTPUT_DIR / 'failed_seeds.txt'}")
    return pdf_links_found

# ---------------------------------------------------------------------------
# Download PDFs β€” reuse the browser's cookies in a requests.Session so
# binary downloads are fast and don't go through the browser's download UI
# ---------------------------------------------------------------------------
def session_from_driver(driver):
    sess = requests.Session()
    # Cookies alone won't authenticate against Windows Integrated Auth (NTLM/
    # Kerberos) sites β€” Edge handles that handshake transparently because
    # it's domain-joined, but a plain requests.Session never does the
    # handshake and gets 401s on every resource. Cloning cookies is kept as
    # a harmless supplement in case any part of the site is cookie-based.
    for cookie in driver.get_cookies():
        sess.cookies.set(cookie["name"], cookie["value"], domain=cookie.get("domain"))
    sess.headers.update({"User-Agent": driver.execute_script("return navigator.userAgent;")})
    if HAS_SSPI:
        sess.auth = HttpNegotiateAuth()
    else:
        print("WARNING: requests_negotiate_sspi not installed.")
        print("  Run: pip install requests-negotiate-sspi")
        print("  Without it, PDF downloads will likely 401 on this NTLM/Kerberos intranet.")
    return sess

def safe_filename(url):
    name = os.path.basename(urlparse(url).path) or "document.pdf"
    name = re.sub(r"[^\w\-.]", "_", name)
    if not name.lower().endswith(".pdf"):
        name += ".pdf"
    url_hash = hashlib.md5(url.encode()).hexdigest()[:8]
    return f"{url_hash}_{name}"

def download_one_pdf(sess, url, max_retries=MAX_NETWORK_RETRIES):
    """Downloads a single PDF immediately. Returns True if a new file was saved."""
    PDF_DIR.mkdir(parents=True, exist_ok=True)
    dest = PDF_DIR / safe_filename(url)
    if dest.exists():
        return False

    for attempt in range(max_retries):
        try:
            resp = sess.get(url, timeout=REQUEST_TIMEOUT)
            resp.raise_for_status()
            if b"%PDF" not in resp.content[:1024]:
                print(f"  [not a PDF, likely an auth redirect] {url}")
                return False
            dest.write_bytes(resp.content)
            return True
        except (requests.exceptions.ConnectionError, requests.exceptions.Timeout) as e:
            delay = NETWORK_RETRY_BASE_DELAY * (2 ** attempt)
            print(f"  [network error, retry {attempt+1}/{max_retries} in {delay}s] {url}")
            time.sleep(delay)
        except requests.RequestException as e:
            print(f"  [failed] {url} ({e})")
            return False
    print(f"  [gave up after {max_retries} retries] {url}")
    return False

# ---------------------------------------------------------------------------
# Extract text + tables from each PDF, with OCR fallback
# ---------------------------------------------------------------------------
def extract_pdf(pdf_path):
    """Returns (text, tables, used_ocr). Tries pdfplumber first; if a page's
    text comes back essentially empty (common for scanned/image-only PDFs),
    falls back to OCR for the whole document."""
    text_parts = []
    tables = []
    total_chars = 0

    try:
        with pdfplumber.open(pdf_path) as pdf:
            for page_num, page in enumerate(pdf.pages, 1):
                page_text = page.extract_text() or ""
                total_chars += len(page_text.strip())
                text_parts.append(f"\n--- Page {page_num} ---\n{page_text}")

                for table in page.extract_tables():
                    if table:
                        tables.append({"page": page_num, "rows": table})
    except Exception as e:
        print(f"  [pdfplumber error] {pdf_path.name}: {e}")

    # Heuristic: if we got almost no text relative to file size, it's likely
    # a scanned/image PDF β€” fall back to OCR.
    avg_chars_per_page = total_chars / max(len(text_parts), 1)
    used_ocr = False
    if avg_chars_per_page < 20:
        try:
            ocr_text = _ocr_pdf(pdf_path)
            if len(ocr_text.strip()) > total_chars:
                text_parts = [ocr_text]
                used_ocr = True
        except Exception as e:
            print(f"  [OCR unavailable/failed] {pdf_path.name}: {e}")

    return "\n".join(text_parts).strip(), tables, used_ocr

def extract_all(downloaded):
    records = []
    for url, pdf_path in tqdm(downloaded, desc="Extracting text"):
        text, tables, used_ocr = extract_pdf(pdf_path)
        records.append({
            "source_url": url,
            "local_file": str(pdf_path),
            "text": text,
            "tables": tables,
            "used_ocr": used_ocr,
            "char_count": len(text),
        })
    return records

# ---------------------------------------------------------------------------
# Main
# ---------------------------------------------------------------------------
def main():
    OUTPUT_DIR.mkdir(parents=True, exist_ok=True)

    print(f"Attaching to Edge on {EDGE_DEBUGGER_ADDRESS} ...")
    try:
        driver = get_driver()
    except Exception as e:
        print(f"Could not attach to Edge: {e}")
        print("Make sure Edge is running with --remote-debugging-port=9222 and you're logged into iamhbl.com.")
        return

    sess = session_from_driver(driver)

    print(f"Seeding {len(SEED_URLS)} function x library pages ({len(FUNCTIONS)} functions x {len(LIBRARIES)} libraries).")
    print("PDFs will be saved to the pdfs folder as they're found β€” check there anytime.\n")
    pdf_links = crawl_and_download(driver, sess, SEED_URLS)

    if not pdf_links:
        print("No PDFs found β€” check the login warning above, the empty-seed list, and confirm library= spelling.")
        return

    # Build the list of successfully downloaded files from disk (covers files
    # downloaded on this run and any left over from a previous interrupted run)
    downloaded = []
    for url in pdf_links:
        dest = PDF_DIR / safe_filename(url)
        if dest.exists():
            downloaded.append((url, dest))
    print(f"\n{len(downloaded)}/{len(pdf_links)} PDFs available on disk for extraction.\n")

    records = extract_all(downloaded)

    out_path = OUTPUT_DIR / "hbl_pdf_extracted.json"
    with open(out_path, "w", encoding="utf-8") as f:
        json.dump(records, f, ensure_ascii=False, indent=2)

    ocr_count = sum(1 for r in records if r["used_ocr"])
    empty_count = sum(1 for r in records if r["char_count"] < 20)

    print(f"\nDone. Extracted {len(records)} PDFs.")
    print(f"  - {ocr_count} needed OCR fallback (scanned/image PDFs)")
    print(f"  - {empty_count} still came back nearly empty β€” worth manually checking these")
    print(f"  - Metadata written to {out_path}")

if __name__ == "__main__":
    main()