Spaces:
Running
Running
Download tests/test_serp_playwright.py from OrganizedProgrammers/SERPent: direct link, hf CLI and curl.
- Browser
- Download file 15.9 kB
-
https://huggingface.co/spaces/OrganizedProgrammers/SERPent/resolve/main/tests/test_serp_playwright.py
- Command line
-
hf download hf://spaces/OrganizedProgrammers/SERPent/tests/test_serp_playwright.py
-
curl -L -o test_serp_playwright.py https://huggingface.co/spaces/OrganizedProgrammers/SERPent/resolve/main/tests/test_serp_playwright.py
15.9 kB
| """Tests for the DOM-extraction half of the Playwright-driven scrapers. | |
| Each scraper in serp.py is three separable pieces: a URL builder (pure, | |
| tested in test_serp_urls.py), a navigation step, and an `_extract_*` | |
| function that reads an already-loaded page. These test the third piece | |
| against a real headless Chromium - the same engine production uses, so | |
| selector behaviour is genuinely exercised - with the fixture HTML loaded | |
| via `set_content`. | |
| This used to require a wrapper that replaced `page.goto` on a | |
| purpose-built fake Browser, because fetch and parse weren't separated. | |
| That wrapper accepted the navigation URL and discarded it, which is | |
| precisely why URL bugs were invisible to the suite; splitting the scrapers | |
| made it unnecessary and it has been deleted. | |
| """ | |
| import pytest | |
| from helpers import load_fixture | |
| from serp import ( | |
| BraveSearchBlockedException, | |
| BrowserUnavailableError, | |
| GoogleScholarBlockedException, | |
| GoogleScholarUnavailableError, | |
| _block_stylesheet_and_image_resources, | |
| _extract_bing_results, | |
| _extract_brave_results, | |
| _extract_google_patents_results, | |
| _extract_google_scholar_results, | |
| playwright_open_page, | |
| ) | |
| async def test_bing_results_are_extracted(page_factory): | |
| page = await page_factory(load_fixture("bing_results.html")) | |
| results = await _extract_bing_results(page, 10) | |
| assert results == [ | |
| {"title": "Result A", "href": "https://example.org/a", "body": "Snippet A"}, | |
| {"title": "Result B", "href": "https://example.org/b", "body": "Snippet B"}, | |
| ] | |
| async def test_bing_extraction_honours_the_result_limit(page_factory): | |
| page = await page_factory(load_fixture("bing_results.html")) | |
| results = await _extract_bing_results(page, 1) | |
| assert [r["title"] for r in results] == ["Result A"] | |
| async def test_brave_results_are_extracted(page_factory): | |
| page = await page_factory(load_fixture("brave_results.html")) | |
| results = await _extract_brave_results(page, 10) | |
| assert results == [ | |
| {"title": "Brave Result A", "body": "Brave snippet A", "href": "https://example.org/brave-a"}, | |
| {"title": "Brave Result B", "body": "Brave snippet B", "href": "https://example.org/brave-b"}, | |
| ] | |
| async def test_brave_extraction_honours_the_result_limit(page_factory): | |
| page = await page_factory(load_fixture("brave_results.html")) | |
| results = await _extract_brave_results(page, 1) | |
| assert [r["title"] for r in results] == ["Brave Result A"] | |
| async def test_brave_block_page_raises(page_factory): | |
| """Brave's anti-bot interstitial has no result cards, so it would | |
| otherwise look like a legitimate zero-result search - and the | |
| `/serp/search` fallback chain needs an exception to move to the next | |
| backend. | |
| """ | |
| page = await page_factory(load_fixture("brave_blocked.html")) | |
| with pytest.raises(BraveSearchBlockedException): | |
| await _extract_brave_results(page, 10) | |
| async def test_google_scholar_results_are_extracted(page_factory): | |
| page = await page_factory(load_fixture("scholar_results.html")) | |
| results = await _extract_google_scholar_results(page, 10) | |
| assert results == [ | |
| {"title": "Scholar Paper A", "body": "Scholar snippet A", "href": "https://example.org/scholar-a"}, | |
| {"title": "Scholar Paper B", "body": "Scholar snippet B", "href": "https://example.org/scholar-b"}, | |
| ] | |
| async def test_google_scholar_extraction_honours_the_result_limit(page_factory): | |
| page = await page_factory(load_fixture("scholar_results.html")) | |
| results = await _extract_google_scholar_results(page, 1) | |
| assert [r["title"] for r in results] == ["Scholar Paper A"] | |
| async def test_google_patents_results_are_extracted(page_factory): | |
| page = await page_factory(load_fixture("patents", "search_results.html")) | |
| results = await _extract_google_patents_results(page, 10) | |
| assert results == [ | |
| { | |
| "id": "US11930446B2", | |
| "href": "https://patents.google.com/patent/US11930446B2/en", | |
| "title": "Widget apparatus", | |
| "body": "An apparatus comprising a widget.", | |
| }, | |
| { | |
| "id": "EP4760514A1", | |
| "href": "https://patents.google.com/patent/EP4760514A1/en", | |
| "title": "Gadget system", | |
| "body": "A system comprising a gadget.", | |
| }, | |
| ] | |
| async def test_google_patents_extraction_honours_the_result_limit(page_factory): | |
| page = await page_factory(load_fixture("patents", "search_results.html")) | |
| results = await _extract_google_patents_results(page, 1) | |
| assert [r["id"] for r in results] == ["US11930446B2"] | |
| # --------------------------------- shared plumbing --------------------------------- | |
| async def test_playwright_open_page_raises_when_browser_unavailable(): | |
| with pytest.raises(BrowserUnavailableError): | |
| async with playwright_open_page(None): | |
| pass | |
| class _FakeRoute: | |
| def __init__(self): | |
| self.aborted = False | |
| self.continued = False | |
| async def abort(self): | |
| self.aborted = True | |
| async def continue_(self): | |
| self.continued = True | |
| class _FakeRequest: | |
| def __init__(self, resource_type): | |
| self.resource_type = resource_type | |
| async def test_decorative_resources_are_blocked(resource_type): | |
| """Only page text matters to these scrapers, and skipping stylesheets | |
| and images meaningfully speeds up navigation.""" | |
| route = _FakeRoute() | |
| await _block_stylesheet_and_image_resources(route, _FakeRequest(resource_type)) | |
| assert route.aborted and not route.continued | |
| async def test_everything_else_is_allowed_through(resource_type): | |
| """Blocking the main document (or the XHR that renders results) would | |
| break the scrape entirely.""" | |
| route = _FakeRoute() | |
| await _block_stylesheet_and_image_resources(route, _FakeRequest(resource_type)) | |
| assert route.continued and not route.aborted | |
| # ----------------------------- Google Scholar blocking ----------------------------- | |
| # Scholar serves an anti-bot interstitial to datacenter IPs instead of | |
| # results. Because that page simply has no `div.gs_ri`, the scraper used to | |
| # sit in `wait_for_selector` for the full 30s timeout and then report a bare | |
| # "Timeout 30000ms exceeded" - slow, and it reads like a broken selector | |
| # rather than a blocked deployment. Observed live on the deployed Space. | |
| async def test_google_scholar_block_page_raises(page_factory): | |
| page = await page_factory(load_fixture("scholar_blocked.html")) | |
| with pytest.raises(GoogleScholarBlockedException): | |
| await _extract_google_scholar_results(page, 10) | |
| async def test_google_scholar_block_is_detected_without_waiting_for_the_timeout(page_factory): | |
| """The whole point: a challenge page is recognised immediately rather | |
| than costing every caller a full selector timeout.""" | |
| import time | |
| page = await page_factory(load_fixture("scholar_blocked.html")) | |
| started = time.perf_counter() | |
| with pytest.raises(GoogleScholarBlockedException): | |
| await _extract_google_scholar_results(page, 10) | |
| elapsed = time.perf_counter() - started | |
| assert elapsed < 5, f"took {elapsed:.1f}s to notice a block page" | |
| async def test_google_scholar_block_is_detected_from_page_text_alone(page_factory): | |
| """Google varies its interstitial. When none of the known challenge | |
| elements are present, fall back to the wording before giving up. | |
| This path is deliberately reached only *after* the selector wait times | |
| out, so the test passes a short timeout rather than sitting through the | |
| real one. Sniffing the text up front would be faster but wrong: a | |
| legitimate Scholar search for "unusual traffic" returns papers whose | |
| snippets contain that phrase, and would be misreported as a block. The | |
| fuzzy check is only safe once the precise one has already failed. | |
| """ | |
| page = await page_factory(load_fixture("scholar_blocked_no_markers.html")) | |
| with pytest.raises(GoogleScholarBlockedException): | |
| await _extract_google_scholar_results(page, 10, timeout_ms=1000) | |
| async def test_a_page_with_neither_results_nor_a_block_is_not_called_a_block(page_factory): | |
| """An unrecognised page must not be reported as a block - that would | |
| hide a genuine selector regression behind a plausible excuse.""" | |
| page = await page_factory("<html><body><p>something else entirely</p></body></html>") | |
| with pytest.raises(GoogleScholarUnavailableError) as exc_info: | |
| await _extract_google_scholar_results(page, 10, timeout_ms=1000) | |
| assert not isinstance(exc_info.value, GoogleScholarBlockedException) | |
| async def test_an_unrecognised_page_reports_what_it_actually_saw(page_factory): | |
| """The first attempt at this fix guessed at Google's challenge markup | |
| and matched none of it, and the resulting error - a bare selector | |
| timeout - said nothing about what was actually served. Whatever we | |
| fail to classify, the exception must carry enough of the page to | |
| identify it from a single live call, rather than needing another | |
| round of guesswork. | |
| """ | |
| page = await page_factory( | |
| "<html><head><title>Totally Unexpected Page</title></head>" | |
| "<body><p>a distinctive sentence that identifies this page</p></body></html>") | |
| with pytest.raises(GoogleScholarUnavailableError) as exc_info: | |
| await _extract_google_scholar_results(page, 10, timeout_ms=1000) | |
| message = str(exc_info.value) | |
| assert "Totally Unexpected Page" in message, "page title must be reported" | |
| assert "a distinctive sentence" in message, "page text excerpt must be reported" | |
| async def test_a_hidden_challenge_element_still_counts_as_a_block(page_factory): | |
| """reCAPTCHA renders inside an iframe and its container is commonly | |
| zero-height until it loads, so a challenge element can be present but | |
| never 'visible'. Detection has to key on presence, not visibility - | |
| the first version of this waited for visibility and so sat through the | |
| full timeout on exactly the pages it was written to catch. | |
| """ | |
| import time | |
| page = await page_factory( | |
| '<html><body><div class="g-recaptcha" style="display:none"></div></body></html>') | |
| started = time.perf_counter() | |
| with pytest.raises(GoogleScholarBlockedException): | |
| await _extract_google_scholar_results(page, 10, timeout_ms=5000) | |
| elapsed = time.perf_counter() - started | |
| # Asserting the timing, not just the exception: the fallback checks | |
| # below catch a hidden element anyway, so a version that waits out the | |
| # whole timeout first still raises the right error - it just does it | |
| # slowly, which is precisely the bug this is guarding against. | |
| assert elapsed < 2, f"took {elapsed:.1f}s to notice a hidden challenge element" | |
| async def test_a_sorry_redirect_url_counts_as_a_block(page_factory): | |
| """Google redirects blocked clients to /sorry/index. The final URL is | |
| markup-independent evidence, unlike any selector we can guess at.""" | |
| page = await page_factory("<html><body><p>nothing recognisable here</p></body></html>") | |
| with pytest.raises(GoogleScholarBlockedException): | |
| await _extract_google_scholar_results( | |
| page, 10, timeout_ms=1000, | |
| final_url="https://scholar.google.com/sorry/index?continue=...") | |
| async def test_a_sorry_redirect_is_recognised_without_any_waiting(page_factory): | |
| """The URL is known before the page is even inspected, so a /sorry | |
| redirect must not cost a selector wait. | |
| Measured against the live deployment, this was the whole remaining | |
| problem: classification was correct but arrived only after the full | |
| 30s timeout, because the URL was checked in the timeout handler rather | |
| than up front. A long timeout is passed deliberately - the assertion | |
| is that it is never reached. | |
| """ | |
| import time | |
| page = await page_factory("<html><body><p>nothing recognisable here</p></body></html>") | |
| started = time.perf_counter() | |
| with pytest.raises(GoogleScholarBlockedException): | |
| await _extract_google_scholar_results( | |
| page, 10, timeout_ms=30_000, | |
| final_url="https://scholar.google.com/sorry/index?continue=...") | |
| elapsed = time.perf_counter() - started | |
| assert elapsed < 2, f"waited {elapsed:.1f}s on a URL that was known up front" | |
| async def test_a_block_reports_the_page_it_saw(page_factory): | |
| """Diagnostics were added only to the unclassified path, which is the | |
| path that did not fire in production. The block path needs them too - | |
| otherwise "blocked" is an assertion the reader cannot check, and there | |
| is no way to tell which signal fired. | |
| """ | |
| page = await page_factory("<html><body><p>nothing recognisable here</p></body></html>") | |
| with pytest.raises(GoogleScholarBlockedException) as exc_info: | |
| await _extract_google_scholar_results( | |
| page, 10, timeout_ms=1000, | |
| final_url="https://scholar.google.com/sorry/index?continue=xyz") | |
| assert "sorry" in str(exc_info.value), "the evidence must appear in the message" | |
| async def test_normal_scholar_results_are_unaffected(page_factory): | |
| """The block check must not cost the happy path anything.""" | |
| page = await page_factory(load_fixture("scholar_results.html")) | |
| results = await _extract_google_scholar_results(page, 10) | |
| assert [r["title"] for r in results] == ["Scholar Paper A", "Scholar Paper B"] | |
| # ------------------------------- Bing redirect links ------------------------------- | |
| async def test_bing_result_hrefs_are_decoded(page_factory): | |
| """Bing wraps result links in its /ck/a redirector; the extractor should | |
| hand back the destination, not the tracking link.""" | |
| page = await page_factory(load_fixture("bing_results_redirect.html")) | |
| results = await _extract_bing_results(page, 10) | |
| assert [r["href"] for r in results] == [ | |
| "https://docs.python.org/3/library/asyncio.html", | |
| "https://example.org/direct", | |
| ] | |
| async def test_a_challenge_page_with_only_wording_is_also_recognised_immediately(page_factory): | |
| """Don't leave the fast path depending on which signal happens to fire. | |
| A page with no results at all, whose wording matches Google's | |
| interstitial, is decidable the moment it loads - Scholar is | |
| server-rendered, so a genuine results page has div.gs_ri in the DOM as | |
| soon as navigation completes. Requiring "no results present" is what | |
| makes the wording check safe this early: a real search for "unusual | |
| traffic" returns papers, so it never reaches this branch. | |
| """ | |
| import time | |
| page = await page_factory( | |
| "<html><body><p>Our systems have detected unusual traffic from your " | |
| "computer network.</p></body></html>") | |
| started = time.perf_counter() | |
| with pytest.raises(GoogleScholarBlockedException): | |
| await _extract_google_scholar_results(page, 10, timeout_ms=30_000) | |
| elapsed = time.perf_counter() - started | |
| assert elapsed < 2, f"waited {elapsed:.1f}s on a page that was decidable up front" | |
| async def test_a_real_results_page_mentioning_unusual_traffic_is_not_a_block(page_factory): | |
| """The guard on the branch above, stated as a test: results win over | |
| wording, so a legitimate search whose snippets contain the interstitial | |
| phrasing is never misreported as a block.""" | |
| page = await page_factory( | |
| '<html><body><div class="gs_ri"><h3><a href="https://example.org/p">' | |
| "Detecting unusual traffic in networks</a></h3>" | |
| '<div class="gs_rs">Our systems have detected unusual traffic patterns.</div>' | |
| "</div></body></html>") | |
| results = await _extract_google_scholar_results(page, 10) | |
| assert [r["title"] for r in results] == ["Detecting unusual traffic in networks"] | |