Spaces:
Running
Make a failed Scholar scrape say what it actually got
Browse filesThe block detection added in #7 did not work. It waited for challenge
elements I had guessed at without being able to observe the real page -
Scholar is unreachable from the development sandbox - and matched none of
them, so a blocked scrape still cost the full 30s timeout and still
reported a bare "Timeout 30000ms exceeded". Guessing again would repeat
the mistake, so this changes the shape of the failure instead.
An unclassifiable page now raises GoogleScholarUnavailableError carrying
the final URL, the page title and a text excerpt. Whatever Google is
actually serving, the next live call identifies it - no more
deploy-and-hope rounds. The error path is best-effort throughout: failing
to read the page must not replace the original problem with a new one.
Two real defects in the first attempt, both now covered:
- Challenge elements were matched by visibility. reCAPTCHA renders in an
iframe whose container is commonly zero-height until it loads, so the
element can be attached but never visible - meaning detection sat
through the entire timeout on exactly the pages it was written to catch.
Matched on presence now.
- Evidence was purely markup-based. A /sorry redirect is markup-
independent and survives Google restyling the page, so the final URL is
checked too.
The test for hidden elements originally asserted only that the right
exception came out, which a visibility-based version also satisfies - the
later fallback checks catch it anyway, just slowly. Since speed is the
entire point of racing the selectors, it now asserts the timing, and the
presence-vs-visibility mutant fails the suite.
Marker lists are widened, but they are no longer what the fix rests on.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01MSNSYnceqvdVz7Csis4K9e
- CONTRIBUTING.md +1 -1
- serp.py +77 -15
- tests/test_serp_playwright.py +62 -7
|
@@ -30,7 +30,7 @@ ruff check . # same lint CI runs
|
|
| 30 |
pytest
|
| 31 |
```
|
| 32 |
|
| 33 |
-
|
| 34 |
`respx`; the Playwright-driven scrapers (Bing, Brave, Google Scholar,
|
| 35 |
Google Patents search) run against a real headless Chromium but navigate to
|
| 36 |
local fixture HTML instead of the live sites — see `tests/helpers.py` for
|
|
|
|
| 30 |
pytest
|
| 31 |
```
|
| 32 |
|
| 33 |
+
260 tests, a few seconds, fully offline. All outbound HTTP is mocked with
|
| 34 |
`respx`; the Playwright-driven scrapers (Bing, Brave, Google Scholar,
|
| 35 |
Google Patents search) run against a real headless Chromium but navigate to
|
| 36 |
local fixture HTML instead of the live sites — see `tests/helpers.py` for
|
|
@@ -71,6 +71,17 @@ class GoogleScholarBlockedException(Exception):
|
|
| 71 |
"instead of results); the deployment's IP is likely rate-limited.")
|
| 72 |
|
| 73 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 74 |
class BrowserUnavailableError(Exception):
|
| 75 |
"""Raised when a Playwright-backed query is attempted but the browser failed to start."""
|
| 76 |
|
|
@@ -141,17 +152,28 @@ async def _block_stylesheet_and_image_resources(route, request):
|
|
| 141 |
SCHOLAR_SELECTOR_TIMEOUT_MS = 30_000
|
| 142 |
|
| 143 |
_SCHOLAR_RESULT_SELECTOR = "div.gs_ri"
|
| 144 |
-
# Elements Google's challenge
|
| 145 |
-
# the results selector is what
|
| 146 |
-
# full timeout.
|
| 147 |
-
|
| 148 |
-
#
|
| 149 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 150 |
_SCHOLAR_BLOCK_TEXT_MARKERS = (
|
| 151 |
"unusual traffic",
|
| 152 |
"not a robot",
|
| 153 |
"/sorry/index",
|
|
|
|
|
|
|
| 154 |
)
|
|
|
|
|
|
|
| 155 |
|
| 156 |
|
| 157 |
def _looks_like_scholar_block(page_content: str) -> bool:
|
|
@@ -159,24 +181,64 @@ def _looks_like_scholar_block(page_content: str) -> bool:
|
|
| 159 |
return any(marker in lowered for marker in _SCHOLAR_BLOCK_TEXT_MARKERS)
|
| 160 |
|
| 161 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 162 |
async def _extract_google_scholar_results(page: Page, n_results: int,
|
| 163 |
-
timeout_ms: int = SCHOLAR_SELECTOR_TIMEOUT_MS
|
|
|
|
| 164 |
"""Extract results from an already-loaded Google Scholar results page.
|
| 165 |
|
| 166 |
-
Raises GoogleScholarBlockedException when Google served
|
| 167 |
-
interstitial instead of results
|
| 168 |
-
|
| 169 |
-
|
|
|
|
| 170 |
"""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 171 |
try:
|
| 172 |
await page.wait_for_selector(
|
| 173 |
-
f"{_SCHOLAR_RESULT_SELECTOR}, {_SCHOLAR_BLOCK_SELECTOR}",
|
|
|
|
| 174 |
except PlaywrightTimeoutError:
|
| 175 |
-
if
|
| 176 |
raise GoogleScholarBlockedException() from None
|
| 177 |
-
raise
|
|
|
|
|
|
|
| 178 |
|
| 179 |
-
if await page.locator(
|
| 180 |
raise GoogleScholarBlockedException()
|
| 181 |
|
| 182 |
items = await page.locator(_SCHOLAR_RESULT_SELECTOR).all()
|
|
|
|
| 71 |
"instead of results); the deployment's IP is likely rate-limited.")
|
| 72 |
|
| 73 |
|
| 74 |
+
class GoogleScholarUnavailableError(Exception):
|
| 75 |
+
"""Scholar returned neither results nor anything we can classify.
|
| 76 |
+
|
| 77 |
+
Carries what the page actually was - final URL, title, a text excerpt -
|
| 78 |
+
because the alternative is a bare selector timeout that says nothing.
|
| 79 |
+
The first attempt at block detection here guessed at Google's challenge
|
| 80 |
+
markup and matched none of it; the error was useless for working out
|
| 81 |
+
why. Whatever we fail to classify, we report.
|
| 82 |
+
"""
|
| 83 |
+
|
| 84 |
+
|
| 85 |
class BrowserUnavailableError(Exception):
|
| 86 |
"""Raised when a Playwright-backed query is attempted but the browser failed to start."""
|
| 87 |
|
|
|
|
| 152 |
SCHOLAR_SELECTOR_TIMEOUT_MS = 30_000
|
| 153 |
|
| 154 |
_SCHOLAR_RESULT_SELECTOR = "div.gs_ri"
|
| 155 |
+
# Elements Google's challenge pages have carried. Waiting for these
|
| 156 |
+
# *alongside* the results selector is what lets a block fail fast instead
|
| 157 |
+
# of costing a full timeout. Matched on presence rather than visibility:
|
| 158 |
+
# reCAPTCHA renders in an iframe and its container is commonly zero-height
|
| 159 |
+
# until it loads, so a challenge element can be attached but never visible.
|
| 160 |
+
_SCHOLAR_BLOCK_SELECTOR = (
|
| 161 |
+
"form#captcha-form, #recaptcha, .g-recaptcha, #infoDiv, "
|
| 162 |
+
"form[action*='sorry'], #gs_captcha_f, img[src*='sorry'], "
|
| 163 |
+
"div#af-error-container, input[name='captcha']"
|
| 164 |
+
)
|
| 165 |
+
# Wording Google's interstitials have used. Only consulted once the
|
| 166 |
+
# selectors have failed - a legitimate Scholar search for "unusual traffic"
|
| 167 |
+
# returns papers whose snippets contain that phrase.
|
| 168 |
_SCHOLAR_BLOCK_TEXT_MARKERS = (
|
| 169 |
"unusual traffic",
|
| 170 |
"not a robot",
|
| 171 |
"/sorry/index",
|
| 172 |
+
"automated queries",
|
| 173 |
+
"your computer network",
|
| 174 |
)
|
| 175 |
+
# A /sorry redirect is markup-independent evidence, unlike any selector.
|
| 176 |
+
_SCHOLAR_BLOCK_URL_MARKERS = ("/sorry", "/challenge")
|
| 177 |
|
| 178 |
|
| 179 |
def _looks_like_scholar_block(page_content: str) -> bool:
|
|
|
|
| 181 |
return any(marker in lowered for marker in _SCHOLAR_BLOCK_TEXT_MARKERS)
|
| 182 |
|
| 183 |
|
| 184 |
+
def _url_looks_blocked(url: Optional[str]) -> bool:
|
| 185 |
+
return bool(url) and any(m in url.lower() for m in _SCHOLAR_BLOCK_URL_MARKERS)
|
| 186 |
+
|
| 187 |
+
|
| 188 |
+
async def _describe_page(page: Page, final_url: Optional[str]) -> str:
|
| 189 |
+
"""A short, human-readable summary of whatever we actually got.
|
| 190 |
+
|
| 191 |
+
Deliberately best-effort: this runs on an error path, so a failure to
|
| 192 |
+
read the page must not replace the original problem with a new one.
|
| 193 |
+
"""
|
| 194 |
+
parts = [f"url={final_url or getattr(page, 'url', 'unknown')!r}"]
|
| 195 |
+
try:
|
| 196 |
+
parts.append(f"title={await page.title()!r}")
|
| 197 |
+
except Exception:
|
| 198 |
+
parts.append("title=<unreadable>")
|
| 199 |
+
try:
|
| 200 |
+
text = " ".join((await page.inner_text("body")).split())
|
| 201 |
+
parts.append(f"text={text[:300]!r}")
|
| 202 |
+
except Exception:
|
| 203 |
+
parts.append("text=<unreadable>")
|
| 204 |
+
return ", ".join(parts)
|
| 205 |
+
|
| 206 |
+
|
| 207 |
async def _extract_google_scholar_results(page: Page, n_results: int,
|
| 208 |
+
timeout_ms: int = SCHOLAR_SELECTOR_TIMEOUT_MS,
|
| 209 |
+
final_url: Optional[str] = None) -> list[dict]:
|
| 210 |
"""Extract results from an already-loaded Google Scholar results page.
|
| 211 |
|
| 212 |
+
Raises GoogleScholarBlockedException when Google served an anti-bot
|
| 213 |
+
interstitial instead of results, and GoogleScholarUnavailableError -
|
| 214 |
+
carrying the page's URL, title and a text excerpt - when it served
|
| 215 |
+
something we can't classify. A bare selector timeout is never the
|
| 216 |
+
outcome, because it tells whoever reads the logs nothing about why.
|
| 217 |
"""
|
| 218 |
+
url = final_url if final_url is not None else getattr(page, "url", None)
|
| 219 |
+
|
| 220 |
+
async def _blocked() -> bool:
|
| 221 |
+
if _url_looks_blocked(url):
|
| 222 |
+
return True
|
| 223 |
+
try:
|
| 224 |
+
if await page.locator(_SCHOLAR_BLOCK_SELECTOR).count():
|
| 225 |
+
return True
|
| 226 |
+
except Exception:
|
| 227 |
+
pass
|
| 228 |
+
return _looks_like_scholar_block(await page.content())
|
| 229 |
+
|
| 230 |
try:
|
| 231 |
await page.wait_for_selector(
|
| 232 |
+
f"{_SCHOLAR_RESULT_SELECTOR}, {_SCHOLAR_BLOCK_SELECTOR}",
|
| 233 |
+
state="attached", timeout=timeout_ms)
|
| 234 |
except PlaywrightTimeoutError:
|
| 235 |
+
if await _blocked():
|
| 236 |
raise GoogleScholarBlockedException() from None
|
| 237 |
+
raise GoogleScholarUnavailableError(
|
| 238 |
+
"Google Scholar returned neither results nor a recognisable "
|
| 239 |
+
f"challenge page ({await _describe_page(page, url)})") from None
|
| 240 |
|
| 241 |
+
if await page.locator(_SCHOLAR_RESULT_SELECTOR).count() == 0 or await _blocked():
|
| 242 |
raise GoogleScholarBlockedException()
|
| 243 |
|
| 244 |
items = await page.locator(_SCHOLAR_RESULT_SELECTOR).all()
|
|
@@ -21,6 +21,7 @@ from serp import (
|
|
| 21 |
BraveSearchBlockedException,
|
| 22 |
BrowserUnavailableError,
|
| 23 |
GoogleScholarBlockedException,
|
|
|
|
| 24 |
_block_stylesheet_and_image_resources,
|
| 25 |
_extract_bing_results,
|
| 26 |
_extract_brave_results,
|
|
@@ -223,17 +224,71 @@ async def test_google_scholar_block_is_detected_from_page_text_alone(page_factor
|
|
| 223 |
await _extract_google_scholar_results(page, 10, timeout_ms=1000)
|
| 224 |
|
| 225 |
|
| 226 |
-
async def
|
| 227 |
-
"""An unrecognised
|
| 228 |
-
|
| 229 |
-
"""
|
| 230 |
-
from playwright.async_api import TimeoutError as PlaywrightTimeoutError
|
| 231 |
-
|
| 232 |
page = await page_factory("<html><body><p>something else entirely</p></body></html>")
|
| 233 |
|
| 234 |
-
with pytest.raises(
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 235 |
await _extract_google_scholar_results(page, 10, timeout_ms=1000)
|
| 236 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 237 |
|
| 238 |
async def test_normal_scholar_results_are_unaffected(page_factory):
|
| 239 |
"""The block check must not cost the happy path anything."""
|
|
|
|
| 21 |
BraveSearchBlockedException,
|
| 22 |
BrowserUnavailableError,
|
| 23 |
GoogleScholarBlockedException,
|
| 24 |
+
GoogleScholarUnavailableError,
|
| 25 |
_block_stylesheet_and_image_resources,
|
| 26 |
_extract_bing_results,
|
| 27 |
_extract_brave_results,
|
|
|
|
| 224 |
await _extract_google_scholar_results(page, 10, timeout_ms=1000)
|
| 225 |
|
| 226 |
|
| 227 |
+
async def test_a_page_with_neither_results_nor_a_block_is_not_called_a_block(page_factory):
|
| 228 |
+
"""An unrecognised page must not be reported as a block - that would
|
| 229 |
+
hide a genuine selector regression behind a plausible excuse."""
|
|
|
|
|
|
|
|
|
|
| 230 |
page = await page_factory("<html><body><p>something else entirely</p></body></html>")
|
| 231 |
|
| 232 |
+
with pytest.raises(GoogleScholarUnavailableError) as exc_info:
|
| 233 |
+
await _extract_google_scholar_results(page, 10, timeout_ms=1000)
|
| 234 |
+
|
| 235 |
+
assert not isinstance(exc_info.value, GoogleScholarBlockedException)
|
| 236 |
+
|
| 237 |
+
|
| 238 |
+
async def test_an_unrecognised_page_reports_what_it_actually_saw(page_factory):
|
| 239 |
+
"""The first attempt at this fix guessed at Google's challenge markup
|
| 240 |
+
and matched none of it, and the resulting error - a bare selector
|
| 241 |
+
timeout - said nothing about what was actually served. Whatever we
|
| 242 |
+
fail to classify, the exception must carry enough of the page to
|
| 243 |
+
identify it from a single live call, rather than needing another
|
| 244 |
+
round of guesswork.
|
| 245 |
+
"""
|
| 246 |
+
page = await page_factory(
|
| 247 |
+
"<html><head><title>Totally Unexpected Page</title></head>"
|
| 248 |
+
"<body><p>a distinctive sentence that identifies this page</p></body></html>")
|
| 249 |
+
|
| 250 |
+
with pytest.raises(GoogleScholarUnavailableError) as exc_info:
|
| 251 |
await _extract_google_scholar_results(page, 10, timeout_ms=1000)
|
| 252 |
|
| 253 |
+
message = str(exc_info.value)
|
| 254 |
+
assert "Totally Unexpected Page" in message, "page title must be reported"
|
| 255 |
+
assert "a distinctive sentence" in message, "page text excerpt must be reported"
|
| 256 |
+
|
| 257 |
+
|
| 258 |
+
async def test_a_hidden_challenge_element_still_counts_as_a_block(page_factory):
|
| 259 |
+
"""reCAPTCHA renders inside an iframe and its container is commonly
|
| 260 |
+
zero-height until it loads, so a challenge element can be present but
|
| 261 |
+
never 'visible'. Detection has to key on presence, not visibility -
|
| 262 |
+
the first version of this waited for visibility and so sat through the
|
| 263 |
+
full timeout on exactly the pages it was written to catch.
|
| 264 |
+
"""
|
| 265 |
+
import time
|
| 266 |
+
|
| 267 |
+
page = await page_factory(
|
| 268 |
+
'<html><body><div class="g-recaptcha" style="display:none"></div></body></html>')
|
| 269 |
+
|
| 270 |
+
started = time.perf_counter()
|
| 271 |
+
with pytest.raises(GoogleScholarBlockedException):
|
| 272 |
+
await _extract_google_scholar_results(page, 10, timeout_ms=5000)
|
| 273 |
+
elapsed = time.perf_counter() - started
|
| 274 |
+
|
| 275 |
+
# Asserting the timing, not just the exception: the fallback checks
|
| 276 |
+
# below catch a hidden element anyway, so a version that waits out the
|
| 277 |
+
# whole timeout first still raises the right error - it just does it
|
| 278 |
+
# slowly, which is precisely the bug this is guarding against.
|
| 279 |
+
assert elapsed < 2, f"took {elapsed:.1f}s to notice a hidden challenge element"
|
| 280 |
+
|
| 281 |
+
|
| 282 |
+
async def test_a_sorry_redirect_url_counts_as_a_block(page_factory):
|
| 283 |
+
"""Google redirects blocked clients to /sorry/index. The final URL is
|
| 284 |
+
markup-independent evidence, unlike any selector we can guess at."""
|
| 285 |
+
page = await page_factory("<html><body><p>nothing recognisable here</p></body></html>")
|
| 286 |
+
page.url_override = "https://scholar.google.com/sorry/index?continue=..."
|
| 287 |
+
|
| 288 |
+
with pytest.raises(GoogleScholarBlockedException):
|
| 289 |
+
await _extract_google_scholar_results(
|
| 290 |
+
page, 10, timeout_ms=1000, final_url=page.url_override)
|
| 291 |
+
|
| 292 |
|
| 293 |
async def test_normal_scholar_results_are_unaffected(page_factory):
|
| 294 |
"""The block check must not cost the happy path anything."""
|