"""Tests for the DOM-extraction half of the Playwright-driven scrapers. Each scraper in serp.py is three separable pieces: a URL builder (pure, tested in test_serp_urls.py), a navigation step, and an `_extract_*` function that reads an already-loaded page. These test the third piece against a real headless Chromium - the same engine production uses, so selector behaviour is genuinely exercised - with the fixture HTML loaded via `set_content`. This used to require a wrapper that replaced `page.goto` on a purpose-built fake Browser, because fetch and parse weren't separated. That wrapper accepted the navigation URL and discarded it, which is precisely why URL bugs were invisible to the suite; splitting the scrapers made it unnecessary and it has been deleted. """ import pytest from helpers import load_fixture from serp import ( BraveSearchBlockedException, BrowserUnavailableError, GoogleScholarBlockedException, GoogleScholarUnavailableError, _block_stylesheet_and_image_resources, _extract_bing_results, _extract_brave_results, _extract_google_patents_results, _extract_google_scholar_results, playwright_open_page, ) async def test_bing_results_are_extracted(page_factory): page = await page_factory(load_fixture("bing_results.html")) results = await _extract_bing_results(page, 10) assert results == [ {"title": "Result A", "href": "https://example.org/a", "body": "Snippet A"}, {"title": "Result B", "href": "https://example.org/b", "body": "Snippet B"}, ] async def test_bing_extraction_honours_the_result_limit(page_factory): page = await page_factory(load_fixture("bing_results.html")) results = await _extract_bing_results(page, 1) assert [r["title"] for r in results] == ["Result A"] async def test_brave_results_are_extracted(page_factory): page = await page_factory(load_fixture("brave_results.html")) results = await _extract_brave_results(page, 10) assert results == [ {"title": "Brave Result A", "body": "Brave snippet A", "href": "https://example.org/brave-a"}, {"title": "Brave Result B", "body": "Brave snippet B", "href": "https://example.org/brave-b"}, ] async def test_brave_extraction_honours_the_result_limit(page_factory): page = await page_factory(load_fixture("brave_results.html")) results = await _extract_brave_results(page, 1) assert [r["title"] for r in results] == ["Brave Result A"] async def test_brave_block_page_raises(page_factory): """Brave's anti-bot interstitial has no result cards, so it would otherwise look like a legitimate zero-result search - and the `/serp/search` fallback chain needs an exception to move to the next backend. """ page = await page_factory(load_fixture("brave_blocked.html")) with pytest.raises(BraveSearchBlockedException): await _extract_brave_results(page, 10) async def test_google_scholar_results_are_extracted(page_factory): page = await page_factory(load_fixture("scholar_results.html")) results = await _extract_google_scholar_results(page, 10) assert results == [ {"title": "Scholar Paper A", "body": "Scholar snippet A", "href": "https://example.org/scholar-a"}, {"title": "Scholar Paper B", "body": "Scholar snippet B", "href": "https://example.org/scholar-b"}, ] async def test_google_scholar_extraction_honours_the_result_limit(page_factory): page = await page_factory(load_fixture("scholar_results.html")) results = await _extract_google_scholar_results(page, 1) assert [r["title"] for r in results] == ["Scholar Paper A"] async def test_google_patents_results_are_extracted(page_factory): page = await page_factory(load_fixture("patents", "search_results.html")) results = await _extract_google_patents_results(page, 10) assert results == [ { "id": "US11930446B2", "href": "https://patents.google.com/patent/US11930446B2/en", "title": "Widget apparatus", "body": "An apparatus comprising a widget.", }, { "id": "EP4760514A1", "href": "https://patents.google.com/patent/EP4760514A1/en", "title": "Gadget system", "body": "A system comprising a gadget.", }, ] async def test_google_patents_extraction_honours_the_result_limit(page_factory): page = await page_factory(load_fixture("patents", "search_results.html")) results = await _extract_google_patents_results(page, 1) assert [r["id"] for r in results] == ["US11930446B2"] # --------------------------------- shared plumbing --------------------------------- async def test_playwright_open_page_raises_when_browser_unavailable(): with pytest.raises(BrowserUnavailableError): async with playwright_open_page(None): pass class _FakeRoute: def __init__(self): self.aborted = False self.continued = False async def abort(self): self.aborted = True async def continue_(self): self.continued = True class _FakeRequest: def __init__(self, resource_type): self.resource_type = resource_type @pytest.mark.parametrize("resource_type", ["stylesheet", "image"]) async def test_decorative_resources_are_blocked(resource_type): """Only page text matters to these scrapers, and skipping stylesheets and images meaningfully speeds up navigation.""" route = _FakeRoute() await _block_stylesheet_and_image_resources(route, _FakeRequest(resource_type)) assert route.aborted and not route.continued @pytest.mark.parametrize("resource_type", ["document", "script", "xhr", "fetch"]) async def test_everything_else_is_allowed_through(resource_type): """Blocking the main document (or the XHR that renders results) would break the scrape entirely.""" route = _FakeRoute() await _block_stylesheet_and_image_resources(route, _FakeRequest(resource_type)) assert route.continued and not route.aborted # ----------------------------- Google Scholar blocking ----------------------------- # Scholar serves an anti-bot interstitial to datacenter IPs instead of # results. Because that page simply has no `div.gs_ri`, the scraper used to # sit in `wait_for_selector` for the full 30s timeout and then report a bare # "Timeout 30000ms exceeded" - slow, and it reads like a broken selector # rather than a blocked deployment. Observed live on the deployed Space. async def test_google_scholar_block_page_raises(page_factory): page = await page_factory(load_fixture("scholar_blocked.html")) with pytest.raises(GoogleScholarBlockedException): await _extract_google_scholar_results(page, 10) async def test_google_scholar_block_is_detected_without_waiting_for_the_timeout(page_factory): """The whole point: a challenge page is recognised immediately rather than costing every caller a full selector timeout.""" import time page = await page_factory(load_fixture("scholar_blocked.html")) started = time.perf_counter() with pytest.raises(GoogleScholarBlockedException): await _extract_google_scholar_results(page, 10) elapsed = time.perf_counter() - started assert elapsed < 5, f"took {elapsed:.1f}s to notice a block page" async def test_google_scholar_block_is_detected_from_page_text_alone(page_factory): """Google varies its interstitial. When none of the known challenge elements are present, fall back to the wording before giving up. This path is deliberately reached only *after* the selector wait times out, so the test passes a short timeout rather than sitting through the real one. Sniffing the text up front would be faster but wrong: a legitimate Scholar search for "unusual traffic" returns papers whose snippets contain that phrase, and would be misreported as a block. The fuzzy check is only safe once the precise one has already failed. """ page = await page_factory(load_fixture("scholar_blocked_no_markers.html")) with pytest.raises(GoogleScholarBlockedException): await _extract_google_scholar_results(page, 10, timeout_ms=1000) async def test_a_page_with_neither_results_nor_a_block_is_not_called_a_block(page_factory): """An unrecognised page must not be reported as a block - that would hide a genuine selector regression behind a plausible excuse.""" page = await page_factory("

something else entirely

") with pytest.raises(GoogleScholarUnavailableError) as exc_info: await _extract_google_scholar_results(page, 10, timeout_ms=1000) assert not isinstance(exc_info.value, GoogleScholarBlockedException) async def test_an_unrecognised_page_reports_what_it_actually_saw(page_factory): """The first attempt at this fix guessed at Google's challenge markup and matched none of it, and the resulting error - a bare selector timeout - said nothing about what was actually served. Whatever we fail to classify, the exception must carry enough of the page to identify it from a single live call, rather than needing another round of guesswork. """ page = await page_factory( "Totally Unexpected Page" "

a distinctive sentence that identifies this page

") with pytest.raises(GoogleScholarUnavailableError) as exc_info: await _extract_google_scholar_results(page, 10, timeout_ms=1000) message = str(exc_info.value) assert "Totally Unexpected Page" in message, "page title must be reported" assert "a distinctive sentence" in message, "page text excerpt must be reported" async def test_a_hidden_challenge_element_still_counts_as_a_block(page_factory): """reCAPTCHA renders inside an iframe and its container is commonly zero-height until it loads, so a challenge element can be present but never 'visible'. Detection has to key on presence, not visibility - the first version of this waited for visibility and so sat through the full timeout on exactly the pages it was written to catch. """ import time page = await page_factory( '') started = time.perf_counter() with pytest.raises(GoogleScholarBlockedException): await _extract_google_scholar_results(page, 10, timeout_ms=5000) elapsed = time.perf_counter() - started # Asserting the timing, not just the exception: the fallback checks # below catch a hidden element anyway, so a version that waits out the # whole timeout first still raises the right error - it just does it # slowly, which is precisely the bug this is guarding against. assert elapsed < 2, f"took {elapsed:.1f}s to notice a hidden challenge element" async def test_a_sorry_redirect_url_counts_as_a_block(page_factory): """Google redirects blocked clients to /sorry/index. The final URL is markup-independent evidence, unlike any selector we can guess at.""" page = await page_factory("

nothing recognisable here

") with pytest.raises(GoogleScholarBlockedException): await _extract_google_scholar_results( page, 10, timeout_ms=1000, final_url="https://scholar.google.com/sorry/index?continue=...") async def test_a_sorry_redirect_is_recognised_without_any_waiting(page_factory): """The URL is known before the page is even inspected, so a /sorry redirect must not cost a selector wait. Measured against the live deployment, this was the whole remaining problem: classification was correct but arrived only after the full 30s timeout, because the URL was checked in the timeout handler rather than up front. A long timeout is passed deliberately - the assertion is that it is never reached. """ import time page = await page_factory("

nothing recognisable here

") started = time.perf_counter() with pytest.raises(GoogleScholarBlockedException): await _extract_google_scholar_results( page, 10, timeout_ms=30_000, final_url="https://scholar.google.com/sorry/index?continue=...") elapsed = time.perf_counter() - started assert elapsed < 2, f"waited {elapsed:.1f}s on a URL that was known up front" async def test_a_block_reports_the_page_it_saw(page_factory): """Diagnostics were added only to the unclassified path, which is the path that did not fire in production. The block path needs them too - otherwise "blocked" is an assertion the reader cannot check, and there is no way to tell which signal fired. """ page = await page_factory("

nothing recognisable here

") with pytest.raises(GoogleScholarBlockedException) as exc_info: await _extract_google_scholar_results( page, 10, timeout_ms=1000, final_url="https://scholar.google.com/sorry/index?continue=xyz") assert "sorry" in str(exc_info.value), "the evidence must appear in the message" async def test_normal_scholar_results_are_unaffected(page_factory): """The block check must not cost the happy path anything.""" page = await page_factory(load_fixture("scholar_results.html")) results = await _extract_google_scholar_results(page, 10) assert [r["title"] for r in results] == ["Scholar Paper A", "Scholar Paper B"] # ------------------------------- Bing redirect links ------------------------------- async def test_bing_result_hrefs_are_decoded(page_factory): """Bing wraps result links in its /ck/a redirector; the extractor should hand back the destination, not the tracking link.""" page = await page_factory(load_fixture("bing_results_redirect.html")) results = await _extract_bing_results(page, 10) assert [r["href"] for r in results] == [ "https://docs.python.org/3/library/asyncio.html", "https://example.org/direct", ] async def test_a_challenge_page_with_only_wording_is_also_recognised_immediately(page_factory): """Don't leave the fast path depending on which signal happens to fire. A page with no results at all, whose wording matches Google's interstitial, is decidable the moment it loads - Scholar is server-rendered, so a genuine results page has div.gs_ri in the DOM as soon as navigation completes. Requiring "no results present" is what makes the wording check safe this early: a real search for "unusual traffic" returns papers, so it never reaches this branch. """ import time page = await page_factory( "

Our systems have detected unusual traffic from your " "computer network.

") started = time.perf_counter() with pytest.raises(GoogleScholarBlockedException): await _extract_google_scholar_results(page, 10, timeout_ms=30_000) elapsed = time.perf_counter() - started assert elapsed < 2, f"waited {elapsed:.1f}s on a page that was decidable up front" async def test_a_real_results_page_mentioning_unusual_traffic_is_not_a_block(page_factory): """The guard on the branch above, stated as a test: results win over wording, so a legitimate search whose snippets contain the interstitial phrasing is never misreported as a block.""" page = await page_factory( '

' "Detecting unusual traffic in networks

" '
Our systems have detected unusual traffic patterns.
' "
") results = await _extract_google_scholar_results(page, 10) assert [r["title"] for r in results] == ["Detecting unusual traffic in networks"]