Spaces:
Sleeping
Sleeping
File size: 5,127 Bytes
e44fdef | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 | """The glue holding each scraper together.
Every Playwright scraper is now: build a URL, open a page, block
decorative resources, navigate, extract. The builders are tested in
test_serp_urls.py and the extractors in test_serp_playwright.py; this
covers the composition - specifically that each scraper navigates to the
URL its own builder produced.
That link is the one the old goto-patching harness could not test, because
the stand-in it installed accepted the navigation URL and discarded it.
These use a recording stand-in browser instead: no real Chromium, and the
URL is the thing being asserted on.
"""
import pytest
import serp
from serp import (bing_search_url, brave_search_url, google_patents_search_url,
google_scholar_url, playwright_open_page,
query_bing_search, query_brave_search, query_google_patents,
query_google_scholar)
SENTINEL = [{"title": "extracted"}]
class _RecordingPage:
def __init__(self):
self.goto_urls = []
self.route_patterns = []
self.closed = False
async def route(self, pattern, handler):
self.route_patterns.append(pattern)
async def goto(self, url, **kwargs):
self.goto_urls.append(url)
return None
async def close(self):
self.closed = True
class _RecordingContext:
def __init__(self, page):
self._page = page
self.closed = False
async def new_page(self):
return self._page
async def close(self):
self.closed = True
class _RecordingBrowser:
def __init__(self):
self.page = _RecordingPage()
self.context = _RecordingContext(self.page)
async def new_context(self, **kwargs):
return self.context
async def _sentinel_extractor(page, n_results):
return SENTINEL
SCRAPERS = [
("query_google_scholar", query_google_scholar,
"_extract_google_scholar_results", google_scholar_url),
("query_google_patents", query_google_patents,
"_extract_google_patents_results", google_patents_search_url),
("query_brave_search", query_brave_search,
"_extract_brave_results", brave_search_url),
("query_bing_search", query_bing_search,
"_extract_bing_results", bing_search_url),
]
@pytest.mark.parametrize("name, scraper, extractor_attr, builder",
SCRAPERS, ids=[s[0] for s in SCRAPERS])
async def test_scraper_navigates_to_the_url_its_builder_produced(
monkeypatch, name, scraper, extractor_attr, builder):
monkeypatch.setattr(serp, extractor_attr, _sentinel_extractor)
browser = _RecordingBrowser()
await scraper(browser, "agentic ai", 25)
assert browser.page.goto_urls == [builder("agentic ai", 25)]
@pytest.mark.parametrize("name, scraper, extractor_attr, builder",
SCRAPERS, ids=[s[0] for s in SCRAPERS])
async def test_scraper_returns_what_its_extractor_produced(
monkeypatch, name, scraper, extractor_attr, builder):
monkeypatch.setattr(serp, extractor_attr, _sentinel_extractor)
results = await scraper(_RecordingBrowser(), "widgets", 10)
assert results == SENTINEL
@pytest.mark.parametrize("name, scraper, extractor_attr, builder",
SCRAPERS, ids=[s[0] for s in SCRAPERS])
async def test_scraper_blocks_decorative_resources(
monkeypatch, name, scraper, extractor_attr, builder):
"""Skipping stylesheets and images meaningfully speeds up navigation,
and every scraper is supposed to do it."""
monkeypatch.setattr(serp, extractor_attr, _sentinel_extractor)
browser = _RecordingBrowser()
await scraper(browser, "widgets", 10)
assert browser.page.route_patterns == ["**/*"]
@pytest.mark.parametrize("name, scraper, extractor_attr, builder",
SCRAPERS, ids=[s[0] for s in SCRAPERS])
async def test_scraper_closes_its_page_and_context(
monkeypatch, name, scraper, extractor_attr, builder):
monkeypatch.setattr(serp, extractor_attr, _sentinel_extractor)
browser = _RecordingBrowser()
await scraper(browser, "widgets", 10)
assert browser.page.closed and browser.context.closed
async def test_page_and_context_are_closed_even_when_extraction_raises(monkeypatch):
async def exploding_extractor(page, n_results):
raise RuntimeError("selectors changed")
monkeypatch.setattr(serp, "_extract_bing_results", exploding_extractor)
browser = _RecordingBrowser()
with pytest.raises(RuntimeError):
await query_bing_search(browser, "widgets", 10)
assert browser.page.closed and browser.context.closed
async def test_context_is_closed_even_when_closing_the_page_raises():
"""A crashed renderer can make page.close() throw; the context still has
to be released or it leaks for the process's lifetime."""
browser = _RecordingBrowser()
async def exploding_close():
raise RuntimeError("renderer gone")
browser.page.close = exploding_close
with pytest.raises(RuntimeError):
async with playwright_open_page(browser):
pass
assert browser.context.closed
|