File size: 5,127 Bytes
e44fdef
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
"""The glue holding each scraper together.

Every Playwright scraper is now: build a URL, open a page, block
decorative resources, navigate, extract. The builders are tested in
test_serp_urls.py and the extractors in test_serp_playwright.py; this
covers the composition - specifically that each scraper navigates to the
URL its own builder produced.

That link is the one the old goto-patching harness could not test, because
the stand-in it installed accepted the navigation URL and discarded it.
These use a recording stand-in browser instead: no real Chromium, and the
URL is the thing being asserted on.
"""

import pytest

import serp
from serp import (bing_search_url, brave_search_url, google_patents_search_url,
                  google_scholar_url, playwright_open_page,
                  query_bing_search, query_brave_search, query_google_patents,
                  query_google_scholar)

SENTINEL = [{"title": "extracted"}]


class _RecordingPage:
    def __init__(self):
        self.goto_urls = []
        self.route_patterns = []
        self.closed = False

    async def route(self, pattern, handler):
        self.route_patterns.append(pattern)

    async def goto(self, url, **kwargs):
        self.goto_urls.append(url)
        return None

    async def close(self):
        self.closed = True


class _RecordingContext:
    def __init__(self, page):
        self._page = page
        self.closed = False

    async def new_page(self):
        return self._page

    async def close(self):
        self.closed = True


class _RecordingBrowser:
    def __init__(self):
        self.page = _RecordingPage()
        self.context = _RecordingContext(self.page)

    async def new_context(self, **kwargs):
        return self.context


async def _sentinel_extractor(page, n_results):
    return SENTINEL


SCRAPERS = [
    ("query_google_scholar", query_google_scholar,
     "_extract_google_scholar_results", google_scholar_url),
    ("query_google_patents", query_google_patents,
     "_extract_google_patents_results", google_patents_search_url),
    ("query_brave_search", query_brave_search,
     "_extract_brave_results", brave_search_url),
    ("query_bing_search", query_bing_search,
     "_extract_bing_results", bing_search_url),
]


@pytest.mark.parametrize("name, scraper, extractor_attr, builder",
                         SCRAPERS, ids=[s[0] for s in SCRAPERS])
async def test_scraper_navigates_to_the_url_its_builder_produced(
        monkeypatch, name, scraper, extractor_attr, builder):
    monkeypatch.setattr(serp, extractor_attr, _sentinel_extractor)
    browser = _RecordingBrowser()

    await scraper(browser, "agentic ai", 25)

    assert browser.page.goto_urls == [builder("agentic ai", 25)]


@pytest.mark.parametrize("name, scraper, extractor_attr, builder",
                         SCRAPERS, ids=[s[0] for s in SCRAPERS])
async def test_scraper_returns_what_its_extractor_produced(
        monkeypatch, name, scraper, extractor_attr, builder):
    monkeypatch.setattr(serp, extractor_attr, _sentinel_extractor)

    results = await scraper(_RecordingBrowser(), "widgets", 10)

    assert results == SENTINEL


@pytest.mark.parametrize("name, scraper, extractor_attr, builder",
                         SCRAPERS, ids=[s[0] for s in SCRAPERS])
async def test_scraper_blocks_decorative_resources(
        monkeypatch, name, scraper, extractor_attr, builder):
    """Skipping stylesheets and images meaningfully speeds up navigation,
    and every scraper is supposed to do it."""
    monkeypatch.setattr(serp, extractor_attr, _sentinel_extractor)
    browser = _RecordingBrowser()

    await scraper(browser, "widgets", 10)

    assert browser.page.route_patterns == ["**/*"]


@pytest.mark.parametrize("name, scraper, extractor_attr, builder",
                         SCRAPERS, ids=[s[0] for s in SCRAPERS])
async def test_scraper_closes_its_page_and_context(
        monkeypatch, name, scraper, extractor_attr, builder):
    monkeypatch.setattr(serp, extractor_attr, _sentinel_extractor)
    browser = _RecordingBrowser()

    await scraper(browser, "widgets", 10)

    assert browser.page.closed and browser.context.closed


async def test_page_and_context_are_closed_even_when_extraction_raises(monkeypatch):
    async def exploding_extractor(page, n_results):
        raise RuntimeError("selectors changed")

    monkeypatch.setattr(serp, "_extract_bing_results", exploding_extractor)
    browser = _RecordingBrowser()

    with pytest.raises(RuntimeError):
        await query_bing_search(browser, "widgets", 10)

    assert browser.page.closed and browser.context.closed


async def test_context_is_closed_even_when_closing_the_page_raises():
    """A crashed renderer can make page.close() throw; the context still has
    to be released or it leaks for the process's lifetime."""
    browser = _RecordingBrowser()

    async def exploding_close():
        raise RuntimeError("renderer gone")

    browser.page.close = exploding_close

    with pytest.raises(RuntimeError):
        async with playwright_open_page(browser):
            pass

    assert browser.context.closed