File size: 15,913 Bytes
232eb80
 
 
 
 
 
 
 
 
 
 
 
 
 
5796881
 
 
 
232eb80
5796881
 
 
f493ed5
addbf67
232eb80
4e043b3
232eb80
 
 
5796881
 
 
 
232eb80
 
4e043b3
 
 
 
 
 
 
 
 
232eb80
 
4e043b3
232eb80
5796881
232eb80
5796881
 
232eb80
 
5796881
232eb80
5796881
 
 
 
 
 
 
232eb80
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5796881
 
232eb80
5796881
 
232eb80
 
5796881
232eb80
5796881
 
 
 
 
 
 
232eb80
 
 
 
 
 
5796881
232eb80
 
 
 
 
5796881
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
232eb80
 
 
 
 
 
 
 
 
 
 
5796881
 
 
 
232eb80
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f493ed5
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
addbf67
 
 
f493ed5
 
addbf67
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f493ed5
 
addbf67
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
8a62982
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
addbf67
f493ed5
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
8a62982
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
"""Tests for the DOM-extraction half of the Playwright-driven scrapers.

Each scraper in serp.py is three separable pieces: a URL builder (pure,
tested in test_serp_urls.py), a navigation step, and an `_extract_*`
function that reads an already-loaded page. These test the third piece
against a real headless Chromium - the same engine production uses, so
selector behaviour is genuinely exercised - with the fixture HTML loaded
via `set_content`.

This used to require a wrapper that replaced `page.goto` on a
purpose-built fake Browser, because fetch and parse weren't separated.
That wrapper accepted the navigation URL and discarded it, which is
precisely why URL bugs were invisible to the suite; splitting the scrapers
made it unnecessary and it has been deleted.
"""

import pytest

from helpers import load_fixture
from serp import (
    BraveSearchBlockedException,
    BrowserUnavailableError,
    GoogleScholarBlockedException,
    GoogleScholarUnavailableError,
    _block_stylesheet_and_image_resources,
    _extract_bing_results,
    _extract_brave_results,
    _extract_google_patents_results,
    _extract_google_scholar_results,
    playwright_open_page,
)


async def test_bing_results_are_extracted(page_factory):
    page = await page_factory(load_fixture("bing_results.html"))

    results = await _extract_bing_results(page, 10)

    assert results == [
        {"title": "Result A", "href": "https://example.org/a", "body": "Snippet A"},
        {"title": "Result B", "href": "https://example.org/b", "body": "Snippet B"},
    ]


async def test_bing_extraction_honours_the_result_limit(page_factory):
    page = await page_factory(load_fixture("bing_results.html"))

    results = await _extract_bing_results(page, 1)

    assert [r["title"] for r in results] == ["Result A"]


async def test_brave_results_are_extracted(page_factory):
    page = await page_factory(load_fixture("brave_results.html"))

    results = await _extract_brave_results(page, 10)

    assert results == [
        {"title": "Brave Result A", "body": "Brave snippet A", "href": "https://example.org/brave-a"},
        {"title": "Brave Result B", "body": "Brave snippet B", "href": "https://example.org/brave-b"},
    ]


async def test_brave_extraction_honours_the_result_limit(page_factory):
    page = await page_factory(load_fixture("brave_results.html"))

    results = await _extract_brave_results(page, 1)

    assert [r["title"] for r in results] == ["Brave Result A"]


async def test_brave_block_page_raises(page_factory):
    """Brave's anti-bot interstitial has no result cards, so it would
    otherwise look like a legitimate zero-result search - and the
    `/serp/search` fallback chain needs an exception to move to the next
    backend.
    """
    page = await page_factory(load_fixture("brave_blocked.html"))

    with pytest.raises(BraveSearchBlockedException):
        await _extract_brave_results(page, 10)


async def test_google_scholar_results_are_extracted(page_factory):
    page = await page_factory(load_fixture("scholar_results.html"))

    results = await _extract_google_scholar_results(page, 10)

    assert results == [
        {"title": "Scholar Paper A", "body": "Scholar snippet A", "href": "https://example.org/scholar-a"},
        {"title": "Scholar Paper B", "body": "Scholar snippet B", "href": "https://example.org/scholar-b"},
    ]


async def test_google_scholar_extraction_honours_the_result_limit(page_factory):
    page = await page_factory(load_fixture("scholar_results.html"))

    results = await _extract_google_scholar_results(page, 1)

    assert [r["title"] for r in results] == ["Scholar Paper A"]


async def test_google_patents_results_are_extracted(page_factory):
    page = await page_factory(load_fixture("patents", "search_results.html"))

    results = await _extract_google_patents_results(page, 10)

    assert results == [
        {
            "id": "US11930446B2",
            "href": "https://patents.google.com/patent/US11930446B2/en",
            "title": "Widget apparatus",
            "body": "An apparatus comprising a widget.",
        },
        {
            "id": "EP4760514A1",
            "href": "https://patents.google.com/patent/EP4760514A1/en",
            "title": "Gadget system",
            "body": "A system comprising a gadget.",
        },
    ]


async def test_google_patents_extraction_honours_the_result_limit(page_factory):
    page = await page_factory(load_fixture("patents", "search_results.html"))

    results = await _extract_google_patents_results(page, 1)

    assert [r["id"] for r in results] == ["US11930446B2"]


# --------------------------------- shared plumbing ---------------------------------


async def test_playwright_open_page_raises_when_browser_unavailable():
    with pytest.raises(BrowserUnavailableError):
        async with playwright_open_page(None):
            pass


class _FakeRoute:
    def __init__(self):
        self.aborted = False
        self.continued = False

    async def abort(self):
        self.aborted = True

    async def continue_(self):
        self.continued = True


class _FakeRequest:
    def __init__(self, resource_type):
        self.resource_type = resource_type


@pytest.mark.parametrize("resource_type", ["stylesheet", "image"])
async def test_decorative_resources_are_blocked(resource_type):
    """Only page text matters to these scrapers, and skipping stylesheets
    and images meaningfully speeds up navigation."""
    route = _FakeRoute()

    await _block_stylesheet_and_image_resources(route, _FakeRequest(resource_type))

    assert route.aborted and not route.continued


@pytest.mark.parametrize("resource_type", ["document", "script", "xhr", "fetch"])
async def test_everything_else_is_allowed_through(resource_type):
    """Blocking the main document (or the XHR that renders results) would
    break the scrape entirely."""
    route = _FakeRoute()

    await _block_stylesheet_and_image_resources(route, _FakeRequest(resource_type))

    assert route.continued and not route.aborted


# ----------------------------- Google Scholar blocking -----------------------------
# Scholar serves an anti-bot interstitial to datacenter IPs instead of
# results. Because that page simply has no `div.gs_ri`, the scraper used to
# sit in `wait_for_selector` for the full 30s timeout and then report a bare
# "Timeout 30000ms exceeded" - slow, and it reads like a broken selector
# rather than a blocked deployment. Observed live on the deployed Space.


async def test_google_scholar_block_page_raises(page_factory):
    page = await page_factory(load_fixture("scholar_blocked.html"))

    with pytest.raises(GoogleScholarBlockedException):
        await _extract_google_scholar_results(page, 10)


async def test_google_scholar_block_is_detected_without_waiting_for_the_timeout(page_factory):
    """The whole point: a challenge page is recognised immediately rather
    than costing every caller a full selector timeout."""
    import time

    page = await page_factory(load_fixture("scholar_blocked.html"))

    started = time.perf_counter()
    with pytest.raises(GoogleScholarBlockedException):
        await _extract_google_scholar_results(page, 10)
    elapsed = time.perf_counter() - started

    assert elapsed < 5, f"took {elapsed:.1f}s to notice a block page"


async def test_google_scholar_block_is_detected_from_page_text_alone(page_factory):
    """Google varies its interstitial. When none of the known challenge
    elements are present, fall back to the wording before giving up.

    This path is deliberately reached only *after* the selector wait times
    out, so the test passes a short timeout rather than sitting through the
    real one. Sniffing the text up front would be faster but wrong: a
    legitimate Scholar search for "unusual traffic" returns papers whose
    snippets contain that phrase, and would be misreported as a block. The
    fuzzy check is only safe once the precise one has already failed.
    """
    page = await page_factory(load_fixture("scholar_blocked_no_markers.html"))

    with pytest.raises(GoogleScholarBlockedException):
        await _extract_google_scholar_results(page, 10, timeout_ms=1000)


async def test_a_page_with_neither_results_nor_a_block_is_not_called_a_block(page_factory):
    """An unrecognised page must not be reported as a block - that would
    hide a genuine selector regression behind a plausible excuse."""
    page = await page_factory("<html><body><p>something else entirely</p></body></html>")

    with pytest.raises(GoogleScholarUnavailableError) as exc_info:
        await _extract_google_scholar_results(page, 10, timeout_ms=1000)

    assert not isinstance(exc_info.value, GoogleScholarBlockedException)


async def test_an_unrecognised_page_reports_what_it_actually_saw(page_factory):
    """The first attempt at this fix guessed at Google's challenge markup
    and matched none of it, and the resulting error - a bare selector
    timeout - said nothing about what was actually served. Whatever we
    fail to classify, the exception must carry enough of the page to
    identify it from a single live call, rather than needing another
    round of guesswork.
    """
    page = await page_factory(
        "<html><head><title>Totally Unexpected Page</title></head>"
        "<body><p>a distinctive sentence that identifies this page</p></body></html>")

    with pytest.raises(GoogleScholarUnavailableError) as exc_info:
        await _extract_google_scholar_results(page, 10, timeout_ms=1000)

    message = str(exc_info.value)
    assert "Totally Unexpected Page" in message, "page title must be reported"
    assert "a distinctive sentence" in message, "page text excerpt must be reported"


async def test_a_hidden_challenge_element_still_counts_as_a_block(page_factory):
    """reCAPTCHA renders inside an iframe and its container is commonly
    zero-height until it loads, so a challenge element can be present but
    never 'visible'. Detection has to key on presence, not visibility -
    the first version of this waited for visibility and so sat through the
    full timeout on exactly the pages it was written to catch.
    """
    import time

    page = await page_factory(
        '<html><body><div class="g-recaptcha" style="display:none"></div></body></html>')

    started = time.perf_counter()
    with pytest.raises(GoogleScholarBlockedException):
        await _extract_google_scholar_results(page, 10, timeout_ms=5000)
    elapsed = time.perf_counter() - started

    # Asserting the timing, not just the exception: the fallback checks
    # below catch a hidden element anyway, so a version that waits out the
    # whole timeout first still raises the right error - it just does it
    # slowly, which is precisely the bug this is guarding against.
    assert elapsed < 2, f"took {elapsed:.1f}s to notice a hidden challenge element"


async def test_a_sorry_redirect_url_counts_as_a_block(page_factory):
    """Google redirects blocked clients to /sorry/index. The final URL is
    markup-independent evidence, unlike any selector we can guess at."""
    page = await page_factory("<html><body><p>nothing recognisable here</p></body></html>")

    with pytest.raises(GoogleScholarBlockedException):
        await _extract_google_scholar_results(
            page, 10, timeout_ms=1000,
            final_url="https://scholar.google.com/sorry/index?continue=...")


async def test_a_sorry_redirect_is_recognised_without_any_waiting(page_factory):
    """The URL is known before the page is even inspected, so a /sorry
    redirect must not cost a selector wait.

    Measured against the live deployment, this was the whole remaining
    problem: classification was correct but arrived only after the full
    30s timeout, because the URL was checked in the timeout handler rather
    than up front. A long timeout is passed deliberately - the assertion
    is that it is never reached.
    """
    import time

    page = await page_factory("<html><body><p>nothing recognisable here</p></body></html>")

    started = time.perf_counter()
    with pytest.raises(GoogleScholarBlockedException):
        await _extract_google_scholar_results(
            page, 10, timeout_ms=30_000,
            final_url="https://scholar.google.com/sorry/index?continue=...")
    elapsed = time.perf_counter() - started

    assert elapsed < 2, f"waited {elapsed:.1f}s on a URL that was known up front"


async def test_a_block_reports_the_page_it_saw(page_factory):
    """Diagnostics were added only to the unclassified path, which is the
    path that did not fire in production. The block path needs them too -
    otherwise "blocked" is an assertion the reader cannot check, and there
    is no way to tell which signal fired.
    """
    page = await page_factory("<html><body><p>nothing recognisable here</p></body></html>")

    with pytest.raises(GoogleScholarBlockedException) as exc_info:
        await _extract_google_scholar_results(
            page, 10, timeout_ms=1000,
            final_url="https://scholar.google.com/sorry/index?continue=xyz")

    assert "sorry" in str(exc_info.value), "the evidence must appear in the message"


async def test_normal_scholar_results_are_unaffected(page_factory):
    """The block check must not cost the happy path anything."""
    page = await page_factory(load_fixture("scholar_results.html"))

    results = await _extract_google_scholar_results(page, 10)

    assert [r["title"] for r in results] == ["Scholar Paper A", "Scholar Paper B"]


# ------------------------------- Bing redirect links -------------------------------


async def test_bing_result_hrefs_are_decoded(page_factory):
    """Bing wraps result links in its /ck/a redirector; the extractor should
    hand back the destination, not the tracking link."""
    page = await page_factory(load_fixture("bing_results_redirect.html"))

    results = await _extract_bing_results(page, 10)

    assert [r["href"] for r in results] == [
        "https://docs.python.org/3/library/asyncio.html",
        "https://example.org/direct",
    ]


async def test_a_challenge_page_with_only_wording_is_also_recognised_immediately(page_factory):
    """Don't leave the fast path depending on which signal happens to fire.

    A page with no results at all, whose wording matches Google's
    interstitial, is decidable the moment it loads - Scholar is
    server-rendered, so a genuine results page has div.gs_ri in the DOM as
    soon as navigation completes. Requiring "no results present" is what
    makes the wording check safe this early: a real search for "unusual
    traffic" returns papers, so it never reaches this branch.
    """
    import time

    page = await page_factory(
        "<html><body><p>Our systems have detected unusual traffic from your "
        "computer network.</p></body></html>")

    started = time.perf_counter()
    with pytest.raises(GoogleScholarBlockedException):
        await _extract_google_scholar_results(page, 10, timeout_ms=30_000)
    elapsed = time.perf_counter() - started

    assert elapsed < 2, f"waited {elapsed:.1f}s on a page that was decidable up front"


async def test_a_real_results_page_mentioning_unusual_traffic_is_not_a_block(page_factory):
    """The guard on the branch above, stated as a test: results win over
    wording, so a legitimate search whose snippets contain the interstitial
    phrasing is never misreported as a block."""
    page = await page_factory(
        '<html><body><div class="gs_ri"><h3><a href="https://example.org/p">'
        "Detecting unusual traffic in networks</a></h3>"
        '<div class="gs_rs">Our systems have detected unusual traffic patterns.</div>'
        "</div></body></html>")

    results = await _extract_google_scholar_results(page, 10)

    assert [r["title"] for r in results] == ["Detecting unusual traffic in networks"]