Spaces:
Sleeping
Sleeping
| import os | |
| import sys | |
| from pathlib import Path | |
| import pytest_asyncio | |
| # The app modules (serp.py, scrap.py, ops.py, utils.py) live at the repo | |
| # root, not inside a package - make sure they're importable regardless of | |
| # how pytest resolves its rootdir. | |
| REPO_ROOT = Path(__file__).resolve().parent.parent | |
| sys.path.insert(0, str(REPO_ROOT)) | |
| def _find_chromium_executable() -> str | None: | |
| """Fallback lookup for environments where the browser Playwright's pip | |
| package expects doesn't match what's actually on disk (e.g. a sandbox | |
| image that pre-installs a specific Chromium revision separately from | |
| however `pip install playwright` resolved). Mirrors the workaround | |
| documented for the JS Playwright test runner in this environment. | |
| """ | |
| browsers_path = os.environ.get("PLAYWRIGHT_BROWSERS_PATH") | |
| if not browsers_path: | |
| return None | |
| root = Path(browsers_path) | |
| if not root.is_dir(): | |
| return None | |
| candidates = sorted(root.glob("chromium-*/chrome-linux/chrome")) | |
| return str(candidates[-1]) if candidates else None | |
| async def browser(): | |
| """A real headless Chromium instance, exactly like the one `app.py`'s | |
| lifespan starts in production. | |
| Function-scoped (one launch per test) rather than session-scoped: each | |
| pytest-asyncio test runs on its own event loop by default, and a | |
| session-scoped instance would be created on one loop's setup and then | |
| awaited from a different, later loop in the next test. Playwright's | |
| connection keeps a background reader task bound to the loop it was | |
| created on, so calls from another loop don't error - they just hang | |
| forever waiting for a reply the old loop's task will never deliver. | |
| Confirmed by reproducing this exact hang with a session-scoped version | |
| of this fixture, then ruling out the scraping logic itself by running | |
| the same call in a plain script outside pytest (single event loop), | |
| where it completed instantly. A fresh browser per test costs roughly a | |
| second of extra startup but sidesteps the whole cross-loop class of bug. | |
| """ | |
| from playwright.async_api import async_playwright | |
| pw = await async_playwright().start() | |
| try: | |
| b = await pw.chromium.launch(headless=True) | |
| except Exception: | |
| executable_path = _find_chromium_executable() | |
| if not executable_path: | |
| raise | |
| b = await pw.chromium.launch(headless=True, executable_path=executable_path) | |
| yield b | |
| await b.close() | |
| await pw.stop() | |
| async def page_factory(browser): | |
| """Returns a callable producing a real Page with fixture HTML loaded. | |
| This is the whole seam the Playwright-driven tests need now that every | |
| scraper is split into a URL builder, a navigation step and an | |
| `_extract_*` function: the extractors take an already-loaded page, so | |
| `set_content` reaches them directly. It replaces the goto-patching | |
| Browser wrapper that used to be required, which had to accept the | |
| navigation URL and throw it away - and so hid every URL bug. | |
| """ | |
| contexts = [] | |
| async def make(html: str): | |
| context = await browser.new_context() | |
| contexts.append(context) | |
| page = await context.new_page() | |
| await page.set_content(html) | |
| return page | |
| yield make | |
| for context in contexts: | |
| await context.close() | |