SERPent / tests /test_serp_playwright.py
Claude
Claude Opus 5
Classify a blocked Scholar page before waiting, not after
8a62982 unverified
Raw History Blame Contribute Delete
15.9 kB
"""Tests for the DOM-extraction half of the Playwright-driven scrapers.
Each scraper in serp.py is three separable pieces: a URL builder (pure,
tested in test_serp_urls.py), a navigation step, and an `_extract_*`
function that reads an already-loaded page. These test the third piece
against a real headless Chromium - the same engine production uses, so
selector behaviour is genuinely exercised - with the fixture HTML loaded
via `set_content`.
This used to require a wrapper that replaced `page.goto` on a
purpose-built fake Browser, because fetch and parse weren't separated.
That wrapper accepted the navigation URL and discarded it, which is
precisely why URL bugs were invisible to the suite; splitting the scrapers
made it unnecessary and it has been deleted.
"""
import pytest
from helpers import load_fixture
from serp import (
BraveSearchBlockedException,
BrowserUnavailableError,
GoogleScholarBlockedException,
GoogleScholarUnavailableError,
_block_stylesheet_and_image_resources,
_extract_bing_results,
_extract_brave_results,
_extract_google_patents_results,
_extract_google_scholar_results,
playwright_open_page,
)
async def test_bing_results_are_extracted(page_factory):
page = await page_factory(load_fixture("bing_results.html"))
results = await _extract_bing_results(page, 10)
assert results == [
{"title": "Result A", "href": "https://example.org/a", "body": "Snippet A"},
{"title": "Result B", "href": "https://example.org/b", "body": "Snippet B"},
]
async def test_bing_extraction_honours_the_result_limit(page_factory):
page = await page_factory(load_fixture("bing_results.html"))
results = await _extract_bing_results(page, 1)
assert [r["title"] for r in results] == ["Result A"]
async def test_brave_results_are_extracted(page_factory):
page = await page_factory(load_fixture("brave_results.html"))
results = await _extract_brave_results(page, 10)
assert results == [
{"title": "Brave Result A", "body": "Brave snippet A", "href": "https://example.org/brave-a"},
{"title": "Brave Result B", "body": "Brave snippet B", "href": "https://example.org/brave-b"},
]
async def test_brave_extraction_honours_the_result_limit(page_factory):
page = await page_factory(load_fixture("brave_results.html"))
results = await _extract_brave_results(page, 1)
assert [r["title"] for r in results] == ["Brave Result A"]
async def test_brave_block_page_raises(page_factory):
"""Brave's anti-bot interstitial has no result cards, so it would
otherwise look like a legitimate zero-result search - and the
`/serp/search` fallback chain needs an exception to move to the next
backend.
"""
page = await page_factory(load_fixture("brave_blocked.html"))
with pytest.raises(BraveSearchBlockedException):
await _extract_brave_results(page, 10)
async def test_google_scholar_results_are_extracted(page_factory):
page = await page_factory(load_fixture("scholar_results.html"))
results = await _extract_google_scholar_results(page, 10)
assert results == [
{"title": "Scholar Paper A", "body": "Scholar snippet A", "href": "https://example.org/scholar-a"},
{"title": "Scholar Paper B", "body": "Scholar snippet B", "href": "https://example.org/scholar-b"},
]
async def test_google_scholar_extraction_honours_the_result_limit(page_factory):
page = await page_factory(load_fixture("scholar_results.html"))
results = await _extract_google_scholar_results(page, 1)
assert [r["title"] for r in results] == ["Scholar Paper A"]
async def test_google_patents_results_are_extracted(page_factory):
page = await page_factory(load_fixture("patents", "search_results.html"))
results = await _extract_google_patents_results(page, 10)
assert results == [
{
"id": "US11930446B2",
"href": "https://patents.google.com/patent/US11930446B2/en",
"title": "Widget apparatus",
"body": "An apparatus comprising a widget.",
},
{
"id": "EP4760514A1",
"href": "https://patents.google.com/patent/EP4760514A1/en",
"title": "Gadget system",
"body": "A system comprising a gadget.",
},
]
async def test_google_patents_extraction_honours_the_result_limit(page_factory):
page = await page_factory(load_fixture("patents", "search_results.html"))
results = await _extract_google_patents_results(page, 1)
assert [r["id"] for r in results] == ["US11930446B2"]
# --------------------------------- shared plumbing ---------------------------------
async def test_playwright_open_page_raises_when_browser_unavailable():
with pytest.raises(BrowserUnavailableError):
async with playwright_open_page(None):
pass
class _FakeRoute:
def __init__(self):
self.aborted = False
self.continued = False
async def abort(self):
self.aborted = True
async def continue_(self):
self.continued = True
class _FakeRequest:
def __init__(self, resource_type):
self.resource_type = resource_type
@pytest.mark.parametrize("resource_type", ["stylesheet", "image"])
async def test_decorative_resources_are_blocked(resource_type):
"""Only page text matters to these scrapers, and skipping stylesheets
and images meaningfully speeds up navigation."""
route = _FakeRoute()
await _block_stylesheet_and_image_resources(route, _FakeRequest(resource_type))
assert route.aborted and not route.continued
@pytest.mark.parametrize("resource_type", ["document", "script", "xhr", "fetch"])
async def test_everything_else_is_allowed_through(resource_type):
"""Blocking the main document (or the XHR that renders results) would
break the scrape entirely."""
route = _FakeRoute()
await _block_stylesheet_and_image_resources(route, _FakeRequest(resource_type))
assert route.continued and not route.aborted
# ----------------------------- Google Scholar blocking -----------------------------
# Scholar serves an anti-bot interstitial to datacenter IPs instead of
# results. Because that page simply has no `div.gs_ri`, the scraper used to
# sit in `wait_for_selector` for the full 30s timeout and then report a bare
# "Timeout 30000ms exceeded" - slow, and it reads like a broken selector
# rather than a blocked deployment. Observed live on the deployed Space.
async def test_google_scholar_block_page_raises(page_factory):
page = await page_factory(load_fixture("scholar_blocked.html"))
with pytest.raises(GoogleScholarBlockedException):
await _extract_google_scholar_results(page, 10)
async def test_google_scholar_block_is_detected_without_waiting_for_the_timeout(page_factory):
"""The whole point: a challenge page is recognised immediately rather
than costing every caller a full selector timeout."""
import time
page = await page_factory(load_fixture("scholar_blocked.html"))
started = time.perf_counter()
with pytest.raises(GoogleScholarBlockedException):
await _extract_google_scholar_results(page, 10)
elapsed = time.perf_counter() - started
assert elapsed < 5, f"took {elapsed:.1f}s to notice a block page"
async def test_google_scholar_block_is_detected_from_page_text_alone(page_factory):
"""Google varies its interstitial. When none of the known challenge
elements are present, fall back to the wording before giving up.
This path is deliberately reached only *after* the selector wait times
out, so the test passes a short timeout rather than sitting through the
real one. Sniffing the text up front would be faster but wrong: a
legitimate Scholar search for "unusual traffic" returns papers whose
snippets contain that phrase, and would be misreported as a block. The
fuzzy check is only safe once the precise one has already failed.
"""
page = await page_factory(load_fixture("scholar_blocked_no_markers.html"))
with pytest.raises(GoogleScholarBlockedException):
await _extract_google_scholar_results(page, 10, timeout_ms=1000)
async def test_a_page_with_neither_results_nor_a_block_is_not_called_a_block(page_factory):
"""An unrecognised page must not be reported as a block - that would
hide a genuine selector regression behind a plausible excuse."""
page = await page_factory("<html><body><p>something else entirely</p></body></html>")
with pytest.raises(GoogleScholarUnavailableError) as exc_info:
await _extract_google_scholar_results(page, 10, timeout_ms=1000)
assert not isinstance(exc_info.value, GoogleScholarBlockedException)
async def test_an_unrecognised_page_reports_what_it_actually_saw(page_factory):
"""The first attempt at this fix guessed at Google's challenge markup
and matched none of it, and the resulting error - a bare selector
timeout - said nothing about what was actually served. Whatever we
fail to classify, the exception must carry enough of the page to
identify it from a single live call, rather than needing another
round of guesswork.
"""
page = await page_factory(
"<html><head><title>Totally Unexpected Page</title></head>"
"<body><p>a distinctive sentence that identifies this page</p></body></html>")
with pytest.raises(GoogleScholarUnavailableError) as exc_info:
await _extract_google_scholar_results(page, 10, timeout_ms=1000)
message = str(exc_info.value)
assert "Totally Unexpected Page" in message, "page title must be reported"
assert "a distinctive sentence" in message, "page text excerpt must be reported"
async def test_a_hidden_challenge_element_still_counts_as_a_block(page_factory):
"""reCAPTCHA renders inside an iframe and its container is commonly
zero-height until it loads, so a challenge element can be present but
never 'visible'. Detection has to key on presence, not visibility -
the first version of this waited for visibility and so sat through the
full timeout on exactly the pages it was written to catch.
"""
import time
page = await page_factory(
'<html><body><div class="g-recaptcha" style="display:none"></div></body></html>')
started = time.perf_counter()
with pytest.raises(GoogleScholarBlockedException):
await _extract_google_scholar_results(page, 10, timeout_ms=5000)
elapsed = time.perf_counter() - started
# Asserting the timing, not just the exception: the fallback checks
# below catch a hidden element anyway, so a version that waits out the
# whole timeout first still raises the right error - it just does it
# slowly, which is precisely the bug this is guarding against.
assert elapsed < 2, f"took {elapsed:.1f}s to notice a hidden challenge element"
async def test_a_sorry_redirect_url_counts_as_a_block(page_factory):
"""Google redirects blocked clients to /sorry/index. The final URL is
markup-independent evidence, unlike any selector we can guess at."""
page = await page_factory("<html><body><p>nothing recognisable here</p></body></html>")
with pytest.raises(GoogleScholarBlockedException):
await _extract_google_scholar_results(
page, 10, timeout_ms=1000,
final_url="https://scholar.google.com/sorry/index?continue=...")
async def test_a_sorry_redirect_is_recognised_without_any_waiting(page_factory):
"""The URL is known before the page is even inspected, so a /sorry
redirect must not cost a selector wait.
Measured against the live deployment, this was the whole remaining
problem: classification was correct but arrived only after the full
30s timeout, because the URL was checked in the timeout handler rather
than up front. A long timeout is passed deliberately - the assertion
is that it is never reached.
"""
import time
page = await page_factory("<html><body><p>nothing recognisable here</p></body></html>")
started = time.perf_counter()
with pytest.raises(GoogleScholarBlockedException):
await _extract_google_scholar_results(
page, 10, timeout_ms=30_000,
final_url="https://scholar.google.com/sorry/index?continue=...")
elapsed = time.perf_counter() - started
assert elapsed < 2, f"waited {elapsed:.1f}s on a URL that was known up front"
async def test_a_block_reports_the_page_it_saw(page_factory):
"""Diagnostics were added only to the unclassified path, which is the
path that did not fire in production. The block path needs them too -
otherwise "blocked" is an assertion the reader cannot check, and there
is no way to tell which signal fired.
"""
page = await page_factory("<html><body><p>nothing recognisable here</p></body></html>")
with pytest.raises(GoogleScholarBlockedException) as exc_info:
await _extract_google_scholar_results(
page, 10, timeout_ms=1000,
final_url="https://scholar.google.com/sorry/index?continue=xyz")
assert "sorry" in str(exc_info.value), "the evidence must appear in the message"
async def test_normal_scholar_results_are_unaffected(page_factory):
"""The block check must not cost the happy path anything."""
page = await page_factory(load_fixture("scholar_results.html"))
results = await _extract_google_scholar_results(page, 10)
assert [r["title"] for r in results] == ["Scholar Paper A", "Scholar Paper B"]
# ------------------------------- Bing redirect links -------------------------------
async def test_bing_result_hrefs_are_decoded(page_factory):
"""Bing wraps result links in its /ck/a redirector; the extractor should
hand back the destination, not the tracking link."""
page = await page_factory(load_fixture("bing_results_redirect.html"))
results = await _extract_bing_results(page, 10)
assert [r["href"] for r in results] == [
"https://docs.python.org/3/library/asyncio.html",
"https://example.org/direct",
]
async def test_a_challenge_page_with_only_wording_is_also_recognised_immediately(page_factory):
"""Don't leave the fast path depending on which signal happens to fire.
A page with no results at all, whose wording matches Google's
interstitial, is decidable the moment it loads - Scholar is
server-rendered, so a genuine results page has div.gs_ri in the DOM as
soon as navigation completes. Requiring "no results present" is what
makes the wording check safe this early: a real search for "unusual
traffic" returns papers, so it never reaches this branch.
"""
import time
page = await page_factory(
"<html><body><p>Our systems have detected unusual traffic from your "
"computer network.</p></body></html>")
started = time.perf_counter()
with pytest.raises(GoogleScholarBlockedException):
await _extract_google_scholar_results(page, 10, timeout_ms=30_000)
elapsed = time.perf_counter() - started
assert elapsed < 2, f"waited {elapsed:.1f}s on a page that was decidable up front"
async def test_a_real_results_page_mentioning_unusual_traffic_is_not_a_block(page_factory):
"""The guard on the branch above, stated as a test: results win over
wording, so a legitimate search whose snippets contain the interstitial
phrasing is never misreported as a block."""
page = await page_factory(
'<html><body><div class="gs_ri"><h3><a href="https://example.org/p">'
"Detecting unusual traffic in networks</a></h3>"
'<div class="gs_rs">Our systems have detected unusual traffic patterns.</div>'
"</div></body></html>")
results = await _extract_google_scholar_results(page, 10)
assert [r["title"] for r in results] == ["Detecting unusual traffic in networks"]