File size: 3,397 Bytes
5796881
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
30d9456
5796881
30d9456
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5796881
 
 
 
 
 
 
 
 
 
 
 
 
 
232eb80
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
import os
import sys
from pathlib import Path

import pytest_asyncio

# The app modules (serp.py, scrap.py, ops.py, utils.py) live at the repo
# root, not inside a package - make sure they're importable regardless of
# how pytest resolves its rootdir.
REPO_ROOT = Path(__file__).resolve().parent.parent
sys.path.insert(0, str(REPO_ROOT))


def _find_chromium_executable() -> str | None:
    """Fallback lookup for environments where the browser Playwright's pip
    package expects doesn't match what's actually on disk (e.g. a sandbox
    image that pre-installs a specific Chromium revision separately from
    however `pip install playwright` resolved). Mirrors the workaround
    documented for the JS Playwright test runner in this environment.
    """
    browsers_path = os.environ.get("PLAYWRIGHT_BROWSERS_PATH")
    if not browsers_path:
        return None
    root = Path(browsers_path)
    if not root.is_dir():
        return None
    candidates = sorted(root.glob("chromium-*/chrome-linux/chrome"))
    return str(candidates[-1]) if candidates else None


@pytest_asyncio.fixture
async def browser():
    """A real headless Chromium instance, exactly like the one `app.py`'s
    lifespan starts in production.

    Function-scoped (one launch per test) rather than session-scoped: each
    pytest-asyncio test runs on its own event loop by default, and a
    session-scoped instance would be created on one loop's setup and then
    awaited from a different, later loop in the next test. Playwright's
    connection keeps a background reader task bound to the loop it was
    created on, so calls from another loop don't error - they just hang
    forever waiting for a reply the old loop's task will never deliver.
    Confirmed by reproducing this exact hang with a session-scoped version
    of this fixture, then ruling out the scraping logic itself by running
    the same call in a plain script outside pytest (single event loop),
    where it completed instantly. A fresh browser per test costs roughly a
    second of extra startup but sidesteps the whole cross-loop class of bug.
    """
    from playwright.async_api import async_playwright

    pw = await async_playwright().start()
    try:
        b = await pw.chromium.launch(headless=True)
    except Exception:
        executable_path = _find_chromium_executable()
        if not executable_path:
            raise
        b = await pw.chromium.launch(headless=True, executable_path=executable_path)
    yield b
    await b.close()
    await pw.stop()


@pytest_asyncio.fixture
async def page_factory(browser):
    """Returns a callable producing a real Page with fixture HTML loaded.

    This is the whole seam the Playwright-driven tests need now that every
    scraper is split into a URL builder, a navigation step and an
    `_extract_*` function: the extractors take an already-loaded page, so
    `set_content` reaches them directly. It replaces the goto-patching
    Browser wrapper that used to be required, which had to accept the
    navigation URL and throw it away - and so hid every URL bug.
    """
    contexts = []

    async def make(html: str):
        context = await browser.new_context()
        contexts.append(context)
        page = await context.new_page()
        await page.set_content(html)
        return page

    yield make

    for context in contexts:
        await context.close()