SERPent / tests /conftest.py
Claude
Claude Opus 5
Test the search URLs, and delete the harness that hid them
232eb80 unverified
Raw
History Blame Contribute Delete
3.4 kB
import os
import sys
from pathlib import Path
import pytest_asyncio
# The app modules (serp.py, scrap.py, ops.py, utils.py) live at the repo
# root, not inside a package - make sure they're importable regardless of
# how pytest resolves its rootdir.
REPO_ROOT = Path(__file__).resolve().parent.parent
sys.path.insert(0, str(REPO_ROOT))
def _find_chromium_executable() -> str | None:
"""Fallback lookup for environments where the browser Playwright's pip
package expects doesn't match what's actually on disk (e.g. a sandbox
image that pre-installs a specific Chromium revision separately from
however `pip install playwright` resolved). Mirrors the workaround
documented for the JS Playwright test runner in this environment.
"""
browsers_path = os.environ.get("PLAYWRIGHT_BROWSERS_PATH")
if not browsers_path:
return None
root = Path(browsers_path)
if not root.is_dir():
return None
candidates = sorted(root.glob("chromium-*/chrome-linux/chrome"))
return str(candidates[-1]) if candidates else None
@pytest_asyncio.fixture
async def browser():
"""A real headless Chromium instance, exactly like the one `app.py`'s
lifespan starts in production.
Function-scoped (one launch per test) rather than session-scoped: each
pytest-asyncio test runs on its own event loop by default, and a
session-scoped instance would be created on one loop's setup and then
awaited from a different, later loop in the next test. Playwright's
connection keeps a background reader task bound to the loop it was
created on, so calls from another loop don't error - they just hang
forever waiting for a reply the old loop's task will never deliver.
Confirmed by reproducing this exact hang with a session-scoped version
of this fixture, then ruling out the scraping logic itself by running
the same call in a plain script outside pytest (single event loop),
where it completed instantly. A fresh browser per test costs roughly a
second of extra startup but sidesteps the whole cross-loop class of bug.
"""
from playwright.async_api import async_playwright
pw = await async_playwright().start()
try:
b = await pw.chromium.launch(headless=True)
except Exception:
executable_path = _find_chromium_executable()
if not executable_path:
raise
b = await pw.chromium.launch(headless=True, executable_path=executable_path)
yield b
await b.close()
await pw.stop()
@pytest_asyncio.fixture
async def page_factory(browser):
"""Returns a callable producing a real Page with fixture HTML loaded.
This is the whole seam the Playwright-driven tests need now that every
scraper is split into a URL builder, a navigation step and an
`_extract_*` function: the extractors take an already-loaded page, so
`set_content` reaches them directly. It replaces the goto-patching
Browser wrapper that used to be required, which had to accept the
navigation URL and throw it away - and so hid every URL bug.
"""
contexts = []
async def make(html: str):
context = await browser.new_context()
contexts.append(context)
page = await context.new_page()
await page.set_content(html)
return page
yield make
for context in contexts:
await context.close()