Claude commited on
Commit
4e043b3
·
unverified ·
1 Parent(s): 21042c7

Split fetch from parse in scrap.py and serp.py (item 2 from review)

Browse files

Both scrap_patent_async and the four Playwright scrapers previously fused
navigation/fetching with DOM parsing in one function, so there was no way
to unit-test "given this page, what does the current selector contract
extract" without also driving a real network request or browser
navigation through the exact same call.

- scrap.py: extracted parse_patent_html(html, patent_url) as a plain
synchronous function; scrap_patent_async is now just fetch + delegate.
- serp.py: extracted _extract_google_scholar_results,
_extract_google_patents_results, _extract_brave_results and
_extract_bing_results, each taking an already-loaded Page. The public
query_* functions are now just navigate + delegate. Also hoisted the
identical `_block_resources` closure (copy-pasted four times) to one
module-level _block_stylesheet_and_image_resources, and PATENT_ID_REGEX
out of query_google_patents's body.

No behavior change - all 67 existing characterization tests pass
unmodified, which is the point: this refactor happened under a green
safety net rather than being verified by hand. Added a few new tests
demonstrating the payoff: parse_patent_html and _extract_bing_results can
now be tested directly (a saved HTML string, or page.set_content()) with
no respx mocking or goto-patching wrapper needed.

Files changed (4) hide show
  1. scrap.py +12 -1
  2. serp.py +140 -134
  3. tests/test_scrap.py +24 -1
  4. tests/test_serp_playwright.py +21 -0
scrap.py CHANGED
@@ -38,7 +38,18 @@ async def scrap_patent_async(client: AsyncClient, patent_url: str) -> PatentScra
38
  response = await client.get(patent_url, headers=headers)
39
  response.raise_for_status()
40
 
41
- soup = BeautifulSoup(response.text, "html.parser")
 
 
 
 
 
 
 
 
 
 
 
42
 
43
  # Abstract
44
  abstract_div = soup.find("div", {"class": "abstract"})
 
38
  response = await client.get(patent_url, headers=headers)
39
  response.raise_for_status()
40
 
41
+ return parse_patent_html(response.text, patent_url)
42
+
43
+
44
+ def parse_patent_html(html: str, patent_url: str) -> PatentScrapResult:
45
+ """Parse a Google Patents patent page into a PatentScrapResult.
46
+
47
+ Pure function of the page's HTML - no network I/O - so it can be tested
48
+ directly against saved/fixture HTML without mocking a client.
49
+ `patent_url` is only used to identify the page in the error message
50
+ below.
51
+ """
52
+ soup = BeautifulSoup(html, "html.parser")
53
 
54
  # Abstract
55
  abstract_div = soup.find("div", {"class": "abstract"})
serp.py CHANGED
@@ -82,187 +82,193 @@ async def playwright_open_page(browser: Optional[Browser]):
82
  await context.close()
83
 
84
 
85
- async def query_google_scholar(browser: Browser, q: str, n_results: int = 10):
86
- """Queries google scholar for the specified query and number of results. Returns relevant papers"""
 
 
 
 
 
 
 
87
 
88
- async with playwright_open_page(browser) as page:
89
 
90
- async def _block_resources(route, request):
91
- if request.resource_type in ["stylesheet", "image"]:
92
- await route.abort()
93
- else:
94
- await route.continue_()
 
 
 
 
 
 
 
 
 
 
 
95
 
96
- await page.route("**/*", _block_resources)
 
 
 
 
 
 
97
 
98
  url = f"https://scholar.google.com/scholar?q={quote_plus(q)}&num={n_results}"
99
  await page.goto(url)
100
 
101
- await page.wait_for_selector("div.gs_ri")
102
 
103
- items = await page.locator("div.gs_ri").all()
104
- results = []
105
- for item in items[:n_results]:
106
- title = await item.locator("h3").inner_text(timeout=1000)
107
- body = await item.locator("div.gs_rs").inner_text(timeout=1000)
108
- href = await item.locator("h3 > a").get_attribute("href")
109
 
110
- results.append({
111
- "title": title,
112
- "body": body,
113
- "href": href
114
- })
115
 
116
- return results
117
 
 
 
 
 
 
 
 
 
118
 
119
- async def query_google_patents(browser: Browser, q: str, n_results: int = 10):
120
- """Queries google patents for the specified query and number of results. Returns relevant patents"""
 
 
 
 
 
121
 
122
- # regex to locate a patent id
123
- PATENT_ID_REGEX = r"\b[A-Z]{2}\d{6,}(?:[A-Z]\d?)?\b"
124
 
125
- async with playwright_open_page(browser) as page:
 
 
 
 
 
 
 
 
 
 
 
 
 
126
 
127
- async def _block_resources(route, request):
128
- if request.resource_type in ["stylesheet", "image"]:
129
- await route.abort()
130
- else:
131
- await route.continue_()
132
 
133
- await page.route("**/*", _block_resources)
 
 
 
134
 
135
  url = f"https://patents.google.com/?q={quote_plus(q)}&num={n_results}"
136
  await page.goto(url)
137
 
138
- # Wait for at least one search result item to appear
139
- # This ensures the page has loaded enough to start scraping
140
- await page.wait_for_function(
141
- f"""() => document.querySelectorAll('search-result-item').length >= 1""",
142
- timeout=30_000
143
- )
144
-
145
- items = await page.locator("search-result-item").all()
146
- results = []
147
- for item in items:
148
- text = " ".join(await item.locator("span").all_inner_texts())
149
- match = re.search(PATENT_ID_REGEX, text)
150
- if not match:
151
- continue
152
-
153
- patent_id = match.group()
154
-
155
- try:
156
- title = await item.locator("h3, h4").first.inner_text(timeout=1000)
157
- body = await item.locator("div.abstract, div.result-snippet, .snippet, .result-text").first.inner_text(timeout=1000)
158
- except Exception:
159
- continue # If we can't get title or body, skip this item
160
 
161
- results.append({
162
- "id": patent_id,
163
- "href": f"https://patents.google.com/patent/{patent_id}/en",
164
- "title": title,
165
- "body": body
166
- })
167
 
168
- return results[:n_results]
 
169
 
 
 
 
 
170
 
171
- async def query_brave_search(browser: Browser, q: str, n_results: int = 10):
172
- """Queries Brave Search for the specified query."""
173
 
174
- async with playwright_open_page(browser) as page:
 
175
 
176
- async def _block_resources(route, request):
177
- if request.resource_type in ["stylesheet", "image"]:
178
- await route.abort()
179
- else:
180
- await route.continue_()
181
 
182
- await page.route("**/*", _block_resources)
 
 
 
183
 
184
- url = f"https://search.brave.com/search?q={quote_plus(q)}"
185
- await page.goto(url)
 
186
 
187
- results_cards = await page.locator('.snippet').all()
 
 
 
 
188
 
189
- if len(results_cards) == 0:
190
- page_content = await page.content()
191
 
192
- if "suspicious" in page_content:
193
- raise BraveSearchBlockedException()
194
 
195
- results = []
196
 
197
- for result in results_cards:
198
- title = await result.locator('.title').all_inner_texts()
199
- description = await result.locator('.snippet-description').all_inner_texts()
200
- url = await result.locator('a').nth(0).get_attribute('href')
 
 
 
 
 
201
 
202
- # Filter out results with no URL or brave-specific URLs
203
- if url is None or url.startswith('/'):
204
- continue
205
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
206
  results.append({
207
- "title": title[0] if title else "",
208
- "body": description[0] if description else "",
209
- "href": url
210
  })
211
 
212
- if len(results) >= n_results:
213
- break
214
-
215
- return results
216
 
217
 
218
  async def query_bing_search(browser: Browser, q: str, n_results: int = 10):
219
  """Queries bing search for the specified query"""
220
  async with playwright_open_page(browser) as page:
221
- async def _block_resources(route, request):
222
- if request.resource_type in ["stylesheet", "image"]:
223
- await route.abort()
224
- else:
225
- await route.continue_()
226
-
227
- await page.route("**/*", _block_resources)
228
 
229
  url = f"https://www.bing.com/search?q={quote_plus(q)}"
230
  await page.goto(url)
231
 
232
- await page.wait_for_selector("li.b_algo")
233
-
234
- results = []
235
-
236
- items = await page.query_selector_all("li.b_algo")
237
- for item in items[:n_results]:
238
- title_el = await item.query_selector("h2 > a")
239
- url = await title_el.get_attribute("href") if title_el else None
240
- title = await title_el.inner_text() if title_el else ""
241
-
242
- snippet = ""
243
-
244
- # Try several fallback selectors
245
- for selector in [
246
- "div.b_caption p", # typical snippet
247
- "div.b_caption", # sometimes snippet is here
248
- "div.b_snippet", # used in some result types
249
- "div.b_text", # used in some panels
250
- "p" # fallback to any paragraph
251
- ]:
252
- snippet_el = await item.query_selector(selector)
253
- if snippet_el:
254
- snippet = await snippet_el.inner_text()
255
- if snippet.strip():
256
- break
257
-
258
- if title and url:
259
- results.append({
260
- "title": title.strip(),
261
- "href": url.strip(),
262
- "body": snippet.strip()
263
- })
264
-
265
- return results
266
 
267
 
268
  async def query_ddg_search(q: str, n_results: int = 10):
 
82
  await context.close()
83
 
84
 
85
+ async def _block_stylesheet_and_image_resources(route, request):
86
+ """Shared `page.route("**/*", ...)` handler for every scraper below: skip
87
+ fetching stylesheets/images, since only the page's text content matters
88
+ here and this meaningfully speeds up navigation.
89
+ """
90
+ if request.resource_type in ["stylesheet", "image"]:
91
+ await route.abort()
92
+ else:
93
+ await route.continue_()
94
 
 
95
 
96
+ async def _extract_google_scholar_results(page: Page, n_results: int) -> list[dict]:
97
+ """Extract results from an already-loaded Google Scholar results page."""
98
+ await page.wait_for_selector("div.gs_ri")
99
+
100
+ items = await page.locator("div.gs_ri").all()
101
+ results = []
102
+ for item in items[:n_results]:
103
+ title = await item.locator("h3").inner_text(timeout=1000)
104
+ body = await item.locator("div.gs_rs").inner_text(timeout=1000)
105
+ href = await item.locator("h3 > a").get_attribute("href")
106
+
107
+ results.append({
108
+ "title": title,
109
+ "body": body,
110
+ "href": href
111
+ })
112
 
113
+ return results
114
+
115
+
116
+ async def query_google_scholar(browser: Browser, q: str, n_results: int = 10):
117
+ """Queries google scholar for the specified query and number of results. Returns relevant papers"""
118
+ async with playwright_open_page(browser) as page:
119
+ await page.route("**/*", _block_stylesheet_and_image_resources)
120
 
121
  url = f"https://scholar.google.com/scholar?q={quote_plus(q)}&num={n_results}"
122
  await page.goto(url)
123
 
124
+ return await _extract_google_scholar_results(page, n_results)
125
 
 
 
 
 
 
 
126
 
127
+ # regex to locate a patent id, e.g. "US11930446B2" or "EP4760514A1"
128
+ PATENT_ID_REGEX = r"\b[A-Z]{2}\d{6,}(?:[A-Z]\d?)?\b"
 
 
 
129
 
 
130
 
131
+ async def _extract_google_patents_results(page: Page, n_results: int) -> list[dict]:
132
+ """Extract results from an already-loaded Google Patents search page."""
133
+ # Wait for at least one search result item to appear
134
+ # This ensures the page has loaded enough to start scraping
135
+ await page.wait_for_function(
136
+ f"""() => document.querySelectorAll('search-result-item').length >= 1""",
137
+ timeout=30_000
138
+ )
139
 
140
+ items = await page.locator("search-result-item").all()
141
+ results = []
142
+ for item in items:
143
+ text = " ".join(await item.locator("span").all_inner_texts())
144
+ match = re.search(PATENT_ID_REGEX, text)
145
+ if not match:
146
+ continue
147
 
148
+ patent_id = match.group()
 
149
 
150
+ try:
151
+ title = await item.locator("h3, h4").first.inner_text(timeout=1000)
152
+ body = await item.locator("div.abstract, div.result-snippet, .snippet, .result-text").first.inner_text(timeout=1000)
153
+ except Exception:
154
+ continue # If we can't get title or body, skip this item
155
+
156
+ results.append({
157
+ "id": patent_id,
158
+ "href": f"https://patents.google.com/patent/{patent_id}/en",
159
+ "title": title,
160
+ "body": body
161
+ })
162
+
163
+ return results[:n_results]
164
 
 
 
 
 
 
165
 
166
+ async def query_google_patents(browser: Browser, q: str, n_results: int = 10):
167
+ """Queries google patents for the specified query and number of results. Returns relevant patents"""
168
+ async with playwright_open_page(browser) as page:
169
+ await page.route("**/*", _block_stylesheet_and_image_resources)
170
 
171
  url = f"https://patents.google.com/?q={quote_plus(q)}&num={n_results}"
172
  await page.goto(url)
173
 
174
+ return await _extract_google_patents_results(page, n_results)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
175
 
 
 
 
 
 
 
176
 
177
+ async def _extract_brave_results(page: Page, n_results: int) -> list[dict]:
178
+ """Extract results from an already-loaded Brave Search results page.
179
 
180
+ Raises BraveSearchBlockedException if the page looks like Brave's
181
+ anti-bot interstitial rather than a results page.
182
+ """
183
+ results_cards = await page.locator('.snippet').all()
184
 
185
+ if len(results_cards) == 0:
186
+ page_content = await page.content()
187
 
188
+ if "suspicious" in page_content:
189
+ raise BraveSearchBlockedException()
190
 
191
+ results = []
 
 
 
 
192
 
193
+ for result in results_cards:
194
+ title = await result.locator('.title').all_inner_texts()
195
+ description = await result.locator('.snippet-description').all_inner_texts()
196
+ url = await result.locator('a').nth(0).get_attribute('href')
197
 
198
+ # Filter out results with no URL or brave-specific URLs
199
+ if url is None or url.startswith('/'):
200
+ continue
201
 
202
+ results.append({
203
+ "title": title[0] if title else "",
204
+ "body": description[0] if description else "",
205
+ "href": url
206
+ })
207
 
208
+ if len(results) >= n_results:
209
+ break
210
 
211
+ return results
 
212
 
 
213
 
214
+ async def query_brave_search(browser: Browser, q: str, n_results: int = 10):
215
+ """Queries Brave Search for the specified query."""
216
+ async with playwright_open_page(browser) as page:
217
+ await page.route("**/*", _block_stylesheet_and_image_resources)
218
+
219
+ url = f"https://search.brave.com/search?q={quote_plus(q)}"
220
+ await page.goto(url)
221
+
222
+ return await _extract_brave_results(page, n_results)
223
 
 
 
 
224
 
225
+ async def _extract_bing_results(page: Page, n_results: int) -> list[dict]:
226
+ """Extract results from an already-loaded Bing results page."""
227
+ await page.wait_for_selector("li.b_algo")
228
+
229
+ results = []
230
+
231
+ items = await page.query_selector_all("li.b_algo")
232
+ for item in items[:n_results]:
233
+ title_el = await item.query_selector("h2 > a")
234
+ url = await title_el.get_attribute("href") if title_el else None
235
+ title = await title_el.inner_text() if title_el else ""
236
+
237
+ snippet = ""
238
+
239
+ # Try several fallback selectors
240
+ for selector in [
241
+ "div.b_caption p", # typical snippet
242
+ "div.b_caption", # sometimes snippet is here
243
+ "div.b_snippet", # used in some result types
244
+ "div.b_text", # used in some panels
245
+ "p" # fallback to any paragraph
246
+ ]:
247
+ snippet_el = await item.query_selector(selector)
248
+ if snippet_el:
249
+ snippet = await snippet_el.inner_text()
250
+ if snippet.strip():
251
+ break
252
+
253
+ if title and url:
254
  results.append({
255
+ "title": title.strip(),
256
+ "href": url.strip(),
257
+ "body": snippet.strip()
258
  })
259
 
260
+ return results
 
 
 
261
 
262
 
263
  async def query_bing_search(browser: Browser, q: str, n_results: int = 10):
264
  """Queries bing search for the specified query"""
265
  async with playwright_open_page(browser) as page:
266
+ await page.route("**/*", _block_stylesheet_and_image_resources)
 
 
 
 
 
 
267
 
268
  url = f"https://www.bing.com/search?q={quote_plus(q)}"
269
  await page.goto(url)
270
 
271
+ return await _extract_bing_results(page, n_results)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
272
 
273
 
274
  async def query_ddg_search(q: str, n_results: int = 10):
tests/test_scrap.py CHANGED
@@ -3,12 +3,35 @@ import pytest
3
  import respx
4
 
5
  from helpers import load_fixture
6
- from scrap import scrap_patent_async, scrap_patent_bulk_async
7
 
8
  FULL_PATENT_HTML = load_fixture("patents", "full_patent.html")
9
  MISSING_TITLE_HTML = load_fixture("patents", "missing_title.html")
10
 
11
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
12
  async def test_scrap_patent_async_extracts_all_fields():
13
  url = "https://patents.google.com/patent/US11930446B2/en"
14
  with respx.mock:
 
3
  import respx
4
 
5
  from helpers import load_fixture
6
+ from scrap import parse_patent_html, scrap_patent_async, scrap_patent_bulk_async
7
 
8
  FULL_PATENT_HTML = load_fixture("patents", "full_patent.html")
9
  MISSING_TITLE_HTML = load_fixture("patents", "missing_title.html")
10
 
11
 
12
+ def test_parse_patent_html_extracts_all_fields_with_no_network_involved():
13
+ """`scrap_patent_async` is now a thin fetch that delegates to this - a
14
+ plain, synchronous function of the HTML string, so the parsing logic
15
+ (including the regex-based section splitting, which is the part most
16
+ likely to break when Google Patents changes its markup) can be tested
17
+ directly against a saved page with no client/respx/event loop needed.
18
+ """
19
+ result = parse_patent_html(FULL_PATENT_HTML, "https://patents.google.com/patent/US11930446B2/en")
20
+
21
+ assert result.title == "Widget with improved gadget mechanism"
22
+ codes = {c.code: c.description for c in result.classifications}
23
+ assert codes == {
24
+ "G06F17/30": "Database structures therefor",
25
+ "G06F17/50": "Other database related",
26
+ "H04L9/00": "Cryptographic mechanisms",
27
+ }
28
+
29
+
30
+ def test_parse_patent_html_raises_when_page_has_no_title():
31
+ with pytest.raises(ValueError):
32
+ parse_patent_html(MISSING_TITLE_HTML, "https://patents.google.com/patent/BOGUS/en")
33
+
34
+
35
  async def test_scrap_patent_async_extracts_all_fields():
36
  url = "https://patents.google.com/patent/US11930446B2/en"
37
  with respx.mock:
tests/test_serp_playwright.py CHANGED
@@ -13,6 +13,7 @@ from helpers import load_fixture, make_fixture_browser
13
  from serp import (
14
  BraveSearchBlockedException,
15
  BrowserUnavailableError,
 
16
  playwright_open_page,
17
  query_bing_search,
18
  query_brave_search,
@@ -21,6 +22,26 @@ from serp import (
21
  )
22
 
23
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
24
  async def test_bing_search_parses_results(browser):
25
  fixture_browser = make_fixture_browser(browser, load_fixture("bing_results.html"))
26
 
 
13
  from serp import (
14
  BraveSearchBlockedException,
15
  BrowserUnavailableError,
16
+ _extract_bing_results,
17
  playwright_open_page,
18
  query_bing_search,
19
  query_brave_search,
 
22
  )
23
 
24
 
25
+ async def test_extract_bing_results_works_on_an_already_loaded_page(browser):
26
+ """Demonstrates the payoff of splitting fetch (navigate) from parse
27
+ (extract): `_extract_bing_results` takes a Page that's already loaded,
28
+ so this needs nothing beyond `set_content` - no goto-patching wrapper,
29
+ no browser-as-a-parameter indirection.
30
+ """
31
+ context = await browser.new_context()
32
+ page = await context.new_page()
33
+ await page.set_content(load_fixture("bing_results.html"))
34
+
35
+ results = await _extract_bing_results(page, 10)
36
+
37
+ assert results == [
38
+ {"title": "Result A", "href": "https://example.org/a", "body": "Snippet A"},
39
+ {"title": "Result B", "href": "https://example.org/b", "body": "Snippet B"},
40
+ ]
41
+
42
+ await context.close()
43
+
44
+
45
  async def test_bing_search_parses_results(browser):
46
  fixture_browser = make_fixture_browser(browser, load_fixture("bing_results.html"))
47