Spaces:
Running
Running
File size: 15,913 Bytes
232eb80 5796881 232eb80 5796881 f493ed5 addbf67 232eb80 4e043b3 232eb80 5796881 232eb80 4e043b3 232eb80 4e043b3 232eb80 5796881 232eb80 5796881 232eb80 5796881 232eb80 5796881 232eb80 5796881 232eb80 5796881 232eb80 5796881 232eb80 5796881 232eb80 5796881 232eb80 5796881 232eb80 5796881 232eb80 f493ed5 addbf67 f493ed5 addbf67 f493ed5 addbf67 8a62982 addbf67 f493ed5 8a62982 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 | """Tests for the DOM-extraction half of the Playwright-driven scrapers.
Each scraper in serp.py is three separable pieces: a URL builder (pure,
tested in test_serp_urls.py), a navigation step, and an `_extract_*`
function that reads an already-loaded page. These test the third piece
against a real headless Chromium - the same engine production uses, so
selector behaviour is genuinely exercised - with the fixture HTML loaded
via `set_content`.
This used to require a wrapper that replaced `page.goto` on a
purpose-built fake Browser, because fetch and parse weren't separated.
That wrapper accepted the navigation URL and discarded it, which is
precisely why URL bugs were invisible to the suite; splitting the scrapers
made it unnecessary and it has been deleted.
"""
import pytest
from helpers import load_fixture
from serp import (
BraveSearchBlockedException,
BrowserUnavailableError,
GoogleScholarBlockedException,
GoogleScholarUnavailableError,
_block_stylesheet_and_image_resources,
_extract_bing_results,
_extract_brave_results,
_extract_google_patents_results,
_extract_google_scholar_results,
playwright_open_page,
)
async def test_bing_results_are_extracted(page_factory):
page = await page_factory(load_fixture("bing_results.html"))
results = await _extract_bing_results(page, 10)
assert results == [
{"title": "Result A", "href": "https://example.org/a", "body": "Snippet A"},
{"title": "Result B", "href": "https://example.org/b", "body": "Snippet B"},
]
async def test_bing_extraction_honours_the_result_limit(page_factory):
page = await page_factory(load_fixture("bing_results.html"))
results = await _extract_bing_results(page, 1)
assert [r["title"] for r in results] == ["Result A"]
async def test_brave_results_are_extracted(page_factory):
page = await page_factory(load_fixture("brave_results.html"))
results = await _extract_brave_results(page, 10)
assert results == [
{"title": "Brave Result A", "body": "Brave snippet A", "href": "https://example.org/brave-a"},
{"title": "Brave Result B", "body": "Brave snippet B", "href": "https://example.org/brave-b"},
]
async def test_brave_extraction_honours_the_result_limit(page_factory):
page = await page_factory(load_fixture("brave_results.html"))
results = await _extract_brave_results(page, 1)
assert [r["title"] for r in results] == ["Brave Result A"]
async def test_brave_block_page_raises(page_factory):
"""Brave's anti-bot interstitial has no result cards, so it would
otherwise look like a legitimate zero-result search - and the
`/serp/search` fallback chain needs an exception to move to the next
backend.
"""
page = await page_factory(load_fixture("brave_blocked.html"))
with pytest.raises(BraveSearchBlockedException):
await _extract_brave_results(page, 10)
async def test_google_scholar_results_are_extracted(page_factory):
page = await page_factory(load_fixture("scholar_results.html"))
results = await _extract_google_scholar_results(page, 10)
assert results == [
{"title": "Scholar Paper A", "body": "Scholar snippet A", "href": "https://example.org/scholar-a"},
{"title": "Scholar Paper B", "body": "Scholar snippet B", "href": "https://example.org/scholar-b"},
]
async def test_google_scholar_extraction_honours_the_result_limit(page_factory):
page = await page_factory(load_fixture("scholar_results.html"))
results = await _extract_google_scholar_results(page, 1)
assert [r["title"] for r in results] == ["Scholar Paper A"]
async def test_google_patents_results_are_extracted(page_factory):
page = await page_factory(load_fixture("patents", "search_results.html"))
results = await _extract_google_patents_results(page, 10)
assert results == [
{
"id": "US11930446B2",
"href": "https://patents.google.com/patent/US11930446B2/en",
"title": "Widget apparatus",
"body": "An apparatus comprising a widget.",
},
{
"id": "EP4760514A1",
"href": "https://patents.google.com/patent/EP4760514A1/en",
"title": "Gadget system",
"body": "A system comprising a gadget.",
},
]
async def test_google_patents_extraction_honours_the_result_limit(page_factory):
page = await page_factory(load_fixture("patents", "search_results.html"))
results = await _extract_google_patents_results(page, 1)
assert [r["id"] for r in results] == ["US11930446B2"]
# --------------------------------- shared plumbing ---------------------------------
async def test_playwright_open_page_raises_when_browser_unavailable():
with pytest.raises(BrowserUnavailableError):
async with playwright_open_page(None):
pass
class _FakeRoute:
def __init__(self):
self.aborted = False
self.continued = False
async def abort(self):
self.aborted = True
async def continue_(self):
self.continued = True
class _FakeRequest:
def __init__(self, resource_type):
self.resource_type = resource_type
@pytest.mark.parametrize("resource_type", ["stylesheet", "image"])
async def test_decorative_resources_are_blocked(resource_type):
"""Only page text matters to these scrapers, and skipping stylesheets
and images meaningfully speeds up navigation."""
route = _FakeRoute()
await _block_stylesheet_and_image_resources(route, _FakeRequest(resource_type))
assert route.aborted and not route.continued
@pytest.mark.parametrize("resource_type", ["document", "script", "xhr", "fetch"])
async def test_everything_else_is_allowed_through(resource_type):
"""Blocking the main document (or the XHR that renders results) would
break the scrape entirely."""
route = _FakeRoute()
await _block_stylesheet_and_image_resources(route, _FakeRequest(resource_type))
assert route.continued and not route.aborted
# ----------------------------- Google Scholar blocking -----------------------------
# Scholar serves an anti-bot interstitial to datacenter IPs instead of
# results. Because that page simply has no `div.gs_ri`, the scraper used to
# sit in `wait_for_selector` for the full 30s timeout and then report a bare
# "Timeout 30000ms exceeded" - slow, and it reads like a broken selector
# rather than a blocked deployment. Observed live on the deployed Space.
async def test_google_scholar_block_page_raises(page_factory):
page = await page_factory(load_fixture("scholar_blocked.html"))
with pytest.raises(GoogleScholarBlockedException):
await _extract_google_scholar_results(page, 10)
async def test_google_scholar_block_is_detected_without_waiting_for_the_timeout(page_factory):
"""The whole point: a challenge page is recognised immediately rather
than costing every caller a full selector timeout."""
import time
page = await page_factory(load_fixture("scholar_blocked.html"))
started = time.perf_counter()
with pytest.raises(GoogleScholarBlockedException):
await _extract_google_scholar_results(page, 10)
elapsed = time.perf_counter() - started
assert elapsed < 5, f"took {elapsed:.1f}s to notice a block page"
async def test_google_scholar_block_is_detected_from_page_text_alone(page_factory):
"""Google varies its interstitial. When none of the known challenge
elements are present, fall back to the wording before giving up.
This path is deliberately reached only *after* the selector wait times
out, so the test passes a short timeout rather than sitting through the
real one. Sniffing the text up front would be faster but wrong: a
legitimate Scholar search for "unusual traffic" returns papers whose
snippets contain that phrase, and would be misreported as a block. The
fuzzy check is only safe once the precise one has already failed.
"""
page = await page_factory(load_fixture("scholar_blocked_no_markers.html"))
with pytest.raises(GoogleScholarBlockedException):
await _extract_google_scholar_results(page, 10, timeout_ms=1000)
async def test_a_page_with_neither_results_nor_a_block_is_not_called_a_block(page_factory):
"""An unrecognised page must not be reported as a block - that would
hide a genuine selector regression behind a plausible excuse."""
page = await page_factory("<html><body><p>something else entirely</p></body></html>")
with pytest.raises(GoogleScholarUnavailableError) as exc_info:
await _extract_google_scholar_results(page, 10, timeout_ms=1000)
assert not isinstance(exc_info.value, GoogleScholarBlockedException)
async def test_an_unrecognised_page_reports_what_it_actually_saw(page_factory):
"""The first attempt at this fix guessed at Google's challenge markup
and matched none of it, and the resulting error - a bare selector
timeout - said nothing about what was actually served. Whatever we
fail to classify, the exception must carry enough of the page to
identify it from a single live call, rather than needing another
round of guesswork.
"""
page = await page_factory(
"<html><head><title>Totally Unexpected Page</title></head>"
"<body><p>a distinctive sentence that identifies this page</p></body></html>")
with pytest.raises(GoogleScholarUnavailableError) as exc_info:
await _extract_google_scholar_results(page, 10, timeout_ms=1000)
message = str(exc_info.value)
assert "Totally Unexpected Page" in message, "page title must be reported"
assert "a distinctive sentence" in message, "page text excerpt must be reported"
async def test_a_hidden_challenge_element_still_counts_as_a_block(page_factory):
"""reCAPTCHA renders inside an iframe and its container is commonly
zero-height until it loads, so a challenge element can be present but
never 'visible'. Detection has to key on presence, not visibility -
the first version of this waited for visibility and so sat through the
full timeout on exactly the pages it was written to catch.
"""
import time
page = await page_factory(
'<html><body><div class="g-recaptcha" style="display:none"></div></body></html>')
started = time.perf_counter()
with pytest.raises(GoogleScholarBlockedException):
await _extract_google_scholar_results(page, 10, timeout_ms=5000)
elapsed = time.perf_counter() - started
# Asserting the timing, not just the exception: the fallback checks
# below catch a hidden element anyway, so a version that waits out the
# whole timeout first still raises the right error - it just does it
# slowly, which is precisely the bug this is guarding against.
assert elapsed < 2, f"took {elapsed:.1f}s to notice a hidden challenge element"
async def test_a_sorry_redirect_url_counts_as_a_block(page_factory):
"""Google redirects blocked clients to /sorry/index. The final URL is
markup-independent evidence, unlike any selector we can guess at."""
page = await page_factory("<html><body><p>nothing recognisable here</p></body></html>")
with pytest.raises(GoogleScholarBlockedException):
await _extract_google_scholar_results(
page, 10, timeout_ms=1000,
final_url="https://scholar.google.com/sorry/index?continue=...")
async def test_a_sorry_redirect_is_recognised_without_any_waiting(page_factory):
"""The URL is known before the page is even inspected, so a /sorry
redirect must not cost a selector wait.
Measured against the live deployment, this was the whole remaining
problem: classification was correct but arrived only after the full
30s timeout, because the URL was checked in the timeout handler rather
than up front. A long timeout is passed deliberately - the assertion
is that it is never reached.
"""
import time
page = await page_factory("<html><body><p>nothing recognisable here</p></body></html>")
started = time.perf_counter()
with pytest.raises(GoogleScholarBlockedException):
await _extract_google_scholar_results(
page, 10, timeout_ms=30_000,
final_url="https://scholar.google.com/sorry/index?continue=...")
elapsed = time.perf_counter() - started
assert elapsed < 2, f"waited {elapsed:.1f}s on a URL that was known up front"
async def test_a_block_reports_the_page_it_saw(page_factory):
"""Diagnostics were added only to the unclassified path, which is the
path that did not fire in production. The block path needs them too -
otherwise "blocked" is an assertion the reader cannot check, and there
is no way to tell which signal fired.
"""
page = await page_factory("<html><body><p>nothing recognisable here</p></body></html>")
with pytest.raises(GoogleScholarBlockedException) as exc_info:
await _extract_google_scholar_results(
page, 10, timeout_ms=1000,
final_url="https://scholar.google.com/sorry/index?continue=xyz")
assert "sorry" in str(exc_info.value), "the evidence must appear in the message"
async def test_normal_scholar_results_are_unaffected(page_factory):
"""The block check must not cost the happy path anything."""
page = await page_factory(load_fixture("scholar_results.html"))
results = await _extract_google_scholar_results(page, 10)
assert [r["title"] for r in results] == ["Scholar Paper A", "Scholar Paper B"]
# ------------------------------- Bing redirect links -------------------------------
async def test_bing_result_hrefs_are_decoded(page_factory):
"""Bing wraps result links in its /ck/a redirector; the extractor should
hand back the destination, not the tracking link."""
page = await page_factory(load_fixture("bing_results_redirect.html"))
results = await _extract_bing_results(page, 10)
assert [r["href"] for r in results] == [
"https://docs.python.org/3/library/asyncio.html",
"https://example.org/direct",
]
async def test_a_challenge_page_with_only_wording_is_also_recognised_immediately(page_factory):
"""Don't leave the fast path depending on which signal happens to fire.
A page with no results at all, whose wording matches Google's
interstitial, is decidable the moment it loads - Scholar is
server-rendered, so a genuine results page has div.gs_ri in the DOM as
soon as navigation completes. Requiring "no results present" is what
makes the wording check safe this early: a real search for "unusual
traffic" returns papers, so it never reaches this branch.
"""
import time
page = await page_factory(
"<html><body><p>Our systems have detected unusual traffic from your "
"computer network.</p></body></html>")
started = time.perf_counter()
with pytest.raises(GoogleScholarBlockedException):
await _extract_google_scholar_results(page, 10, timeout_ms=30_000)
elapsed = time.perf_counter() - started
assert elapsed < 2, f"waited {elapsed:.1f}s on a page that was decidable up front"
async def test_a_real_results_page_mentioning_unusual_traffic_is_not_a_block(page_factory):
"""The guard on the branch above, stated as a test: results win over
wording, so a legitimate search whose snippets contain the interstitial
phrasing is never misreported as a block."""
page = await page_factory(
'<html><body><div class="gs_ri"><h3><a href="https://example.org/p">'
"Detecting unusual traffic in networks</a></h3>"
'<div class="gs_rs">Our systems have detected unusual traffic patterns.</div>'
"</div></body></html>")
results = await _extract_google_scholar_results(page, 10)
assert [r["title"] for r in results] == ["Detecting unusual traffic in networks"]
|