import os import sys from pathlib import Path import pytest_asyncio # The app modules (serp.py, scrap.py, ops.py, utils.py) live at the repo # root, not inside a package - make sure they're importable regardless of # how pytest resolves its rootdir. REPO_ROOT = Path(__file__).resolve().parent.parent sys.path.insert(0, str(REPO_ROOT)) def _find_chromium_executable() -> str | None: """Fallback lookup for environments where the browser Playwright's pip package expects doesn't match what's actually on disk (e.g. a sandbox image that pre-installs a specific Chromium revision separately from however `pip install playwright` resolved). Mirrors the workaround documented for the JS Playwright test runner in this environment. """ browsers_path = os.environ.get("PLAYWRIGHT_BROWSERS_PATH") if not browsers_path: return None root = Path(browsers_path) if not root.is_dir(): return None candidates = sorted(root.glob("chromium-*/chrome-linux/chrome")) return str(candidates[-1]) if candidates else None @pytest_asyncio.fixture async def browser(): """A real headless Chromium instance, exactly like the one `app.py`'s lifespan starts in production. Function-scoped (one launch per test) rather than session-scoped: each pytest-asyncio test runs on its own event loop by default, and a session-scoped instance would be created on one loop's setup and then awaited from a different, later loop in the next test. Playwright's connection keeps a background reader task bound to the loop it was created on, so calls from another loop don't error - they just hang forever waiting for a reply the old loop's task will never deliver. Confirmed by reproducing this exact hang with a session-scoped version of this fixture, then ruling out the scraping logic itself by running the same call in a plain script outside pytest (single event loop), where it completed instantly. A fresh browser per test costs roughly a second of extra startup but sidesteps the whole cross-loop class of bug. """ from playwright.async_api import async_playwright pw = await async_playwright().start() try: b = await pw.chromium.launch(headless=True) except Exception: executable_path = _find_chromium_executable() if not executable_path: raise b = await pw.chromium.launch(headless=True, executable_path=executable_path) yield b await b.close() await pw.stop() @pytest_asyncio.fixture async def page_factory(browser): """Returns a callable producing a real Page with fixture HTML loaded. This is the whole seam the Playwright-driven tests need now that every scraper is split into a URL builder, a navigation step and an `_extract_*` function: the extractors take an already-loaded page, so `set_content` reaches them directly. It replaces the goto-patching Browser wrapper that used to be required, which had to accept the navigation URL and throw it away - and so hid every URL bug. """ contexts = [] async def make(html: str): context = await browser.new_context() contexts.append(context) page = await context.new_page() await page.set_content(html) return page yield make for context in contexts: await context.close()