Spaces:
Sleeping
Split fetch from parse in scrap.py and serp.py (item 2 from review)
Browse filesBoth scrap_patent_async and the four Playwright scrapers previously fused
navigation/fetching with DOM parsing in one function, so there was no way
to unit-test "given this page, what does the current selector contract
extract" without also driving a real network request or browser
navigation through the exact same call.
- scrap.py: extracted parse_patent_html(html, patent_url) as a plain
synchronous function; scrap_patent_async is now just fetch + delegate.
- serp.py: extracted _extract_google_scholar_results,
_extract_google_patents_results, _extract_brave_results and
_extract_bing_results, each taking an already-loaded Page. The public
query_* functions are now just navigate + delegate. Also hoisted the
identical `_block_resources` closure (copy-pasted four times) to one
module-level _block_stylesheet_and_image_resources, and PATENT_ID_REGEX
out of query_google_patents's body.
No behavior change - all 67 existing characterization tests pass
unmodified, which is the point: this refactor happened under a green
safety net rather than being verified by hand. Added a few new tests
demonstrating the payoff: parse_patent_html and _extract_bing_results can
now be tested directly (a saved HTML string, or page.set_content()) with
no respx mocking or goto-patching wrapper needed.
- scrap.py +12 -1
- serp.py +140 -134
- tests/test_scrap.py +24 -1
- tests/test_serp_playwright.py +21 -0
|
@@ -38,7 +38,18 @@ async def scrap_patent_async(client: AsyncClient, patent_url: str) -> PatentScra
|
|
| 38 |
response = await client.get(patent_url, headers=headers)
|
| 39 |
response.raise_for_status()
|
| 40 |
|
| 41 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 42 |
|
| 43 |
# Abstract
|
| 44 |
abstract_div = soup.find("div", {"class": "abstract"})
|
|
|
|
| 38 |
response = await client.get(patent_url, headers=headers)
|
| 39 |
response.raise_for_status()
|
| 40 |
|
| 41 |
+
return parse_patent_html(response.text, patent_url)
|
| 42 |
+
|
| 43 |
+
|
| 44 |
+
def parse_patent_html(html: str, patent_url: str) -> PatentScrapResult:
|
| 45 |
+
"""Parse a Google Patents patent page into a PatentScrapResult.
|
| 46 |
+
|
| 47 |
+
Pure function of the page's HTML - no network I/O - so it can be tested
|
| 48 |
+
directly against saved/fixture HTML without mocking a client.
|
| 49 |
+
`patent_url` is only used to identify the page in the error message
|
| 50 |
+
below.
|
| 51 |
+
"""
|
| 52 |
+
soup = BeautifulSoup(html, "html.parser")
|
| 53 |
|
| 54 |
# Abstract
|
| 55 |
abstract_div = soup.find("div", {"class": "abstract"})
|
|
@@ -82,187 +82,193 @@ async def playwright_open_page(browser: Optional[Browser]):
|
|
| 82 |
await context.close()
|
| 83 |
|
| 84 |
|
| 85 |
-
async def
|
| 86 |
-
"""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 87 |
|
| 88 |
-
async with playwright_open_page(browser) as page:
|
| 89 |
|
| 90 |
-
|
| 91 |
-
|
| 92 |
-
|
| 93 |
-
|
| 94 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 95 |
|
| 96 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 97 |
|
| 98 |
url = f"https://scholar.google.com/scholar?q={quote_plus(q)}&num={n_results}"
|
| 99 |
await page.goto(url)
|
| 100 |
|
| 101 |
-
await
|
| 102 |
|
| 103 |
-
items = await page.locator("div.gs_ri").all()
|
| 104 |
-
results = []
|
| 105 |
-
for item in items[:n_results]:
|
| 106 |
-
title = await item.locator("h3").inner_text(timeout=1000)
|
| 107 |
-
body = await item.locator("div.gs_rs").inner_text(timeout=1000)
|
| 108 |
-
href = await item.locator("h3 > a").get_attribute("href")
|
| 109 |
|
| 110 |
-
|
| 111 |
-
|
| 112 |
-
"body": body,
|
| 113 |
-
"href": href
|
| 114 |
-
})
|
| 115 |
|
| 116 |
-
return results
|
| 117 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 118 |
|
| 119 |
-
|
| 120 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 121 |
|
| 122 |
-
|
| 123 |
-
PATENT_ID_REGEX = r"\b[A-Z]{2}\d{6,}(?:[A-Z]\d?)?\b"
|
| 124 |
|
| 125 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 126 |
|
| 127 |
-
async def _block_resources(route, request):
|
| 128 |
-
if request.resource_type in ["stylesheet", "image"]:
|
| 129 |
-
await route.abort()
|
| 130 |
-
else:
|
| 131 |
-
await route.continue_()
|
| 132 |
|
| 133 |
-
|
|
|
|
|
|
|
|
|
|
| 134 |
|
| 135 |
url = f"https://patents.google.com/?q={quote_plus(q)}&num={n_results}"
|
| 136 |
await page.goto(url)
|
| 137 |
|
| 138 |
-
|
| 139 |
-
# This ensures the page has loaded enough to start scraping
|
| 140 |
-
await page.wait_for_function(
|
| 141 |
-
f"""() => document.querySelectorAll('search-result-item').length >= 1""",
|
| 142 |
-
timeout=30_000
|
| 143 |
-
)
|
| 144 |
-
|
| 145 |
-
items = await page.locator("search-result-item").all()
|
| 146 |
-
results = []
|
| 147 |
-
for item in items:
|
| 148 |
-
text = " ".join(await item.locator("span").all_inner_texts())
|
| 149 |
-
match = re.search(PATENT_ID_REGEX, text)
|
| 150 |
-
if not match:
|
| 151 |
-
continue
|
| 152 |
-
|
| 153 |
-
patent_id = match.group()
|
| 154 |
-
|
| 155 |
-
try:
|
| 156 |
-
title = await item.locator("h3, h4").first.inner_text(timeout=1000)
|
| 157 |
-
body = await item.locator("div.abstract, div.result-snippet, .snippet, .result-text").first.inner_text(timeout=1000)
|
| 158 |
-
except Exception:
|
| 159 |
-
continue # If we can't get title or body, skip this item
|
| 160 |
|
| 161 |
-
results.append({
|
| 162 |
-
"id": patent_id,
|
| 163 |
-
"href": f"https://patents.google.com/patent/{patent_id}/en",
|
| 164 |
-
"title": title,
|
| 165 |
-
"body": body
|
| 166 |
-
})
|
| 167 |
|
| 168 |
-
|
|
|
|
| 169 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 170 |
|
| 171 |
-
|
| 172 |
-
|
| 173 |
|
| 174 |
-
|
|
|
|
| 175 |
|
| 176 |
-
|
| 177 |
-
if request.resource_type in ["stylesheet", "image"]:
|
| 178 |
-
await route.abort()
|
| 179 |
-
else:
|
| 180 |
-
await route.continue_()
|
| 181 |
|
| 182 |
-
|
|
|
|
|
|
|
|
|
|
| 183 |
|
| 184 |
-
|
| 185 |
-
|
|
|
|
| 186 |
|
| 187 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 188 |
|
| 189 |
-
if len(
|
| 190 |
-
|
| 191 |
|
| 192 |
-
|
| 193 |
-
raise BraveSearchBlockedException()
|
| 194 |
|
| 195 |
-
results = []
|
| 196 |
|
| 197 |
-
|
| 198 |
-
|
| 199 |
-
|
| 200 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 201 |
|
| 202 |
-
# Filter out results with no URL or brave-specific URLs
|
| 203 |
-
if url is None or url.startswith('/'):
|
| 204 |
-
continue
|
| 205 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 206 |
results.append({
|
| 207 |
-
"title": title
|
| 208 |
-
"
|
| 209 |
-
"
|
| 210 |
})
|
| 211 |
|
| 212 |
-
|
| 213 |
-
break
|
| 214 |
-
|
| 215 |
-
return results
|
| 216 |
|
| 217 |
|
| 218 |
async def query_bing_search(browser: Browser, q: str, n_results: int = 10):
|
| 219 |
"""Queries bing search for the specified query"""
|
| 220 |
async with playwright_open_page(browser) as page:
|
| 221 |
-
|
| 222 |
-
if request.resource_type in ["stylesheet", "image"]:
|
| 223 |
-
await route.abort()
|
| 224 |
-
else:
|
| 225 |
-
await route.continue_()
|
| 226 |
-
|
| 227 |
-
await page.route("**/*", _block_resources)
|
| 228 |
|
| 229 |
url = f"https://www.bing.com/search?q={quote_plus(q)}"
|
| 230 |
await page.goto(url)
|
| 231 |
|
| 232 |
-
await
|
| 233 |
-
|
| 234 |
-
results = []
|
| 235 |
-
|
| 236 |
-
items = await page.query_selector_all("li.b_algo")
|
| 237 |
-
for item in items[:n_results]:
|
| 238 |
-
title_el = await item.query_selector("h2 > a")
|
| 239 |
-
url = await title_el.get_attribute("href") if title_el else None
|
| 240 |
-
title = await title_el.inner_text() if title_el else ""
|
| 241 |
-
|
| 242 |
-
snippet = ""
|
| 243 |
-
|
| 244 |
-
# Try several fallback selectors
|
| 245 |
-
for selector in [
|
| 246 |
-
"div.b_caption p", # typical snippet
|
| 247 |
-
"div.b_caption", # sometimes snippet is here
|
| 248 |
-
"div.b_snippet", # used in some result types
|
| 249 |
-
"div.b_text", # used in some panels
|
| 250 |
-
"p" # fallback to any paragraph
|
| 251 |
-
]:
|
| 252 |
-
snippet_el = await item.query_selector(selector)
|
| 253 |
-
if snippet_el:
|
| 254 |
-
snippet = await snippet_el.inner_text()
|
| 255 |
-
if snippet.strip():
|
| 256 |
-
break
|
| 257 |
-
|
| 258 |
-
if title and url:
|
| 259 |
-
results.append({
|
| 260 |
-
"title": title.strip(),
|
| 261 |
-
"href": url.strip(),
|
| 262 |
-
"body": snippet.strip()
|
| 263 |
-
})
|
| 264 |
-
|
| 265 |
-
return results
|
| 266 |
|
| 267 |
|
| 268 |
async def query_ddg_search(q: str, n_results: int = 10):
|
|
|
|
| 82 |
await context.close()
|
| 83 |
|
| 84 |
|
| 85 |
+
async def _block_stylesheet_and_image_resources(route, request):
|
| 86 |
+
"""Shared `page.route("**/*", ...)` handler for every scraper below: skip
|
| 87 |
+
fetching stylesheets/images, since only the page's text content matters
|
| 88 |
+
here and this meaningfully speeds up navigation.
|
| 89 |
+
"""
|
| 90 |
+
if request.resource_type in ["stylesheet", "image"]:
|
| 91 |
+
await route.abort()
|
| 92 |
+
else:
|
| 93 |
+
await route.continue_()
|
| 94 |
|
|
|
|
| 95 |
|
| 96 |
+
async def _extract_google_scholar_results(page: Page, n_results: int) -> list[dict]:
|
| 97 |
+
"""Extract results from an already-loaded Google Scholar results page."""
|
| 98 |
+
await page.wait_for_selector("div.gs_ri")
|
| 99 |
+
|
| 100 |
+
items = await page.locator("div.gs_ri").all()
|
| 101 |
+
results = []
|
| 102 |
+
for item in items[:n_results]:
|
| 103 |
+
title = await item.locator("h3").inner_text(timeout=1000)
|
| 104 |
+
body = await item.locator("div.gs_rs").inner_text(timeout=1000)
|
| 105 |
+
href = await item.locator("h3 > a").get_attribute("href")
|
| 106 |
+
|
| 107 |
+
results.append({
|
| 108 |
+
"title": title,
|
| 109 |
+
"body": body,
|
| 110 |
+
"href": href
|
| 111 |
+
})
|
| 112 |
|
| 113 |
+
return results
|
| 114 |
+
|
| 115 |
+
|
| 116 |
+
async def query_google_scholar(browser: Browser, q: str, n_results: int = 10):
|
| 117 |
+
"""Queries google scholar for the specified query and number of results. Returns relevant papers"""
|
| 118 |
+
async with playwright_open_page(browser) as page:
|
| 119 |
+
await page.route("**/*", _block_stylesheet_and_image_resources)
|
| 120 |
|
| 121 |
url = f"https://scholar.google.com/scholar?q={quote_plus(q)}&num={n_results}"
|
| 122 |
await page.goto(url)
|
| 123 |
|
| 124 |
+
return await _extract_google_scholar_results(page, n_results)
|
| 125 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 126 |
|
| 127 |
+
# regex to locate a patent id, e.g. "US11930446B2" or "EP4760514A1"
|
| 128 |
+
PATENT_ID_REGEX = r"\b[A-Z]{2}\d{6,}(?:[A-Z]\d?)?\b"
|
|
|
|
|
|
|
|
|
|
| 129 |
|
|
|
|
| 130 |
|
| 131 |
+
async def _extract_google_patents_results(page: Page, n_results: int) -> list[dict]:
|
| 132 |
+
"""Extract results from an already-loaded Google Patents search page."""
|
| 133 |
+
# Wait for at least one search result item to appear
|
| 134 |
+
# This ensures the page has loaded enough to start scraping
|
| 135 |
+
await page.wait_for_function(
|
| 136 |
+
f"""() => document.querySelectorAll('search-result-item').length >= 1""",
|
| 137 |
+
timeout=30_000
|
| 138 |
+
)
|
| 139 |
|
| 140 |
+
items = await page.locator("search-result-item").all()
|
| 141 |
+
results = []
|
| 142 |
+
for item in items:
|
| 143 |
+
text = " ".join(await item.locator("span").all_inner_texts())
|
| 144 |
+
match = re.search(PATENT_ID_REGEX, text)
|
| 145 |
+
if not match:
|
| 146 |
+
continue
|
| 147 |
|
| 148 |
+
patent_id = match.group()
|
|
|
|
| 149 |
|
| 150 |
+
try:
|
| 151 |
+
title = await item.locator("h3, h4").first.inner_text(timeout=1000)
|
| 152 |
+
body = await item.locator("div.abstract, div.result-snippet, .snippet, .result-text").first.inner_text(timeout=1000)
|
| 153 |
+
except Exception:
|
| 154 |
+
continue # If we can't get title or body, skip this item
|
| 155 |
+
|
| 156 |
+
results.append({
|
| 157 |
+
"id": patent_id,
|
| 158 |
+
"href": f"https://patents.google.com/patent/{patent_id}/en",
|
| 159 |
+
"title": title,
|
| 160 |
+
"body": body
|
| 161 |
+
})
|
| 162 |
+
|
| 163 |
+
return results[:n_results]
|
| 164 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 165 |
|
| 166 |
+
async def query_google_patents(browser: Browser, q: str, n_results: int = 10):
|
| 167 |
+
"""Queries google patents for the specified query and number of results. Returns relevant patents"""
|
| 168 |
+
async with playwright_open_page(browser) as page:
|
| 169 |
+
await page.route("**/*", _block_stylesheet_and_image_resources)
|
| 170 |
|
| 171 |
url = f"https://patents.google.com/?q={quote_plus(q)}&num={n_results}"
|
| 172 |
await page.goto(url)
|
| 173 |
|
| 174 |
+
return await _extract_google_patents_results(page, n_results)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 175 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 176 |
|
| 177 |
+
async def _extract_brave_results(page: Page, n_results: int) -> list[dict]:
|
| 178 |
+
"""Extract results from an already-loaded Brave Search results page.
|
| 179 |
|
| 180 |
+
Raises BraveSearchBlockedException if the page looks like Brave's
|
| 181 |
+
anti-bot interstitial rather than a results page.
|
| 182 |
+
"""
|
| 183 |
+
results_cards = await page.locator('.snippet').all()
|
| 184 |
|
| 185 |
+
if len(results_cards) == 0:
|
| 186 |
+
page_content = await page.content()
|
| 187 |
|
| 188 |
+
if "suspicious" in page_content:
|
| 189 |
+
raise BraveSearchBlockedException()
|
| 190 |
|
| 191 |
+
results = []
|
|
|
|
|
|
|
|
|
|
|
|
|
| 192 |
|
| 193 |
+
for result in results_cards:
|
| 194 |
+
title = await result.locator('.title').all_inner_texts()
|
| 195 |
+
description = await result.locator('.snippet-description').all_inner_texts()
|
| 196 |
+
url = await result.locator('a').nth(0).get_attribute('href')
|
| 197 |
|
| 198 |
+
# Filter out results with no URL or brave-specific URLs
|
| 199 |
+
if url is None or url.startswith('/'):
|
| 200 |
+
continue
|
| 201 |
|
| 202 |
+
results.append({
|
| 203 |
+
"title": title[0] if title else "",
|
| 204 |
+
"body": description[0] if description else "",
|
| 205 |
+
"href": url
|
| 206 |
+
})
|
| 207 |
|
| 208 |
+
if len(results) >= n_results:
|
| 209 |
+
break
|
| 210 |
|
| 211 |
+
return results
|
|
|
|
| 212 |
|
|
|
|
| 213 |
|
| 214 |
+
async def query_brave_search(browser: Browser, q: str, n_results: int = 10):
|
| 215 |
+
"""Queries Brave Search for the specified query."""
|
| 216 |
+
async with playwright_open_page(browser) as page:
|
| 217 |
+
await page.route("**/*", _block_stylesheet_and_image_resources)
|
| 218 |
+
|
| 219 |
+
url = f"https://search.brave.com/search?q={quote_plus(q)}"
|
| 220 |
+
await page.goto(url)
|
| 221 |
+
|
| 222 |
+
return await _extract_brave_results(page, n_results)
|
| 223 |
|
|
|
|
|
|
|
|
|
|
| 224 |
|
| 225 |
+
async def _extract_bing_results(page: Page, n_results: int) -> list[dict]:
|
| 226 |
+
"""Extract results from an already-loaded Bing results page."""
|
| 227 |
+
await page.wait_for_selector("li.b_algo")
|
| 228 |
+
|
| 229 |
+
results = []
|
| 230 |
+
|
| 231 |
+
items = await page.query_selector_all("li.b_algo")
|
| 232 |
+
for item in items[:n_results]:
|
| 233 |
+
title_el = await item.query_selector("h2 > a")
|
| 234 |
+
url = await title_el.get_attribute("href") if title_el else None
|
| 235 |
+
title = await title_el.inner_text() if title_el else ""
|
| 236 |
+
|
| 237 |
+
snippet = ""
|
| 238 |
+
|
| 239 |
+
# Try several fallback selectors
|
| 240 |
+
for selector in [
|
| 241 |
+
"div.b_caption p", # typical snippet
|
| 242 |
+
"div.b_caption", # sometimes snippet is here
|
| 243 |
+
"div.b_snippet", # used in some result types
|
| 244 |
+
"div.b_text", # used in some panels
|
| 245 |
+
"p" # fallback to any paragraph
|
| 246 |
+
]:
|
| 247 |
+
snippet_el = await item.query_selector(selector)
|
| 248 |
+
if snippet_el:
|
| 249 |
+
snippet = await snippet_el.inner_text()
|
| 250 |
+
if snippet.strip():
|
| 251 |
+
break
|
| 252 |
+
|
| 253 |
+
if title and url:
|
| 254 |
results.append({
|
| 255 |
+
"title": title.strip(),
|
| 256 |
+
"href": url.strip(),
|
| 257 |
+
"body": snippet.strip()
|
| 258 |
})
|
| 259 |
|
| 260 |
+
return results
|
|
|
|
|
|
|
|
|
|
| 261 |
|
| 262 |
|
| 263 |
async def query_bing_search(browser: Browser, q: str, n_results: int = 10):
|
| 264 |
"""Queries bing search for the specified query"""
|
| 265 |
async with playwright_open_page(browser) as page:
|
| 266 |
+
await page.route("**/*", _block_stylesheet_and_image_resources)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 267 |
|
| 268 |
url = f"https://www.bing.com/search?q={quote_plus(q)}"
|
| 269 |
await page.goto(url)
|
| 270 |
|
| 271 |
+
return await _extract_bing_results(page, n_results)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 272 |
|
| 273 |
|
| 274 |
async def query_ddg_search(q: str, n_results: int = 10):
|
|
@@ -3,12 +3,35 @@ import pytest
|
|
| 3 |
import respx
|
| 4 |
|
| 5 |
from helpers import load_fixture
|
| 6 |
-
from scrap import scrap_patent_async, scrap_patent_bulk_async
|
| 7 |
|
| 8 |
FULL_PATENT_HTML = load_fixture("patents", "full_patent.html")
|
| 9 |
MISSING_TITLE_HTML = load_fixture("patents", "missing_title.html")
|
| 10 |
|
| 11 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 12 |
async def test_scrap_patent_async_extracts_all_fields():
|
| 13 |
url = "https://patents.google.com/patent/US11930446B2/en"
|
| 14 |
with respx.mock:
|
|
|
|
| 3 |
import respx
|
| 4 |
|
| 5 |
from helpers import load_fixture
|
| 6 |
+
from scrap import parse_patent_html, scrap_patent_async, scrap_patent_bulk_async
|
| 7 |
|
| 8 |
FULL_PATENT_HTML = load_fixture("patents", "full_patent.html")
|
| 9 |
MISSING_TITLE_HTML = load_fixture("patents", "missing_title.html")
|
| 10 |
|
| 11 |
|
| 12 |
+
def test_parse_patent_html_extracts_all_fields_with_no_network_involved():
|
| 13 |
+
"""`scrap_patent_async` is now a thin fetch that delegates to this - a
|
| 14 |
+
plain, synchronous function of the HTML string, so the parsing logic
|
| 15 |
+
(including the regex-based section splitting, which is the part most
|
| 16 |
+
likely to break when Google Patents changes its markup) can be tested
|
| 17 |
+
directly against a saved page with no client/respx/event loop needed.
|
| 18 |
+
"""
|
| 19 |
+
result = parse_patent_html(FULL_PATENT_HTML, "https://patents.google.com/patent/US11930446B2/en")
|
| 20 |
+
|
| 21 |
+
assert result.title == "Widget with improved gadget mechanism"
|
| 22 |
+
codes = {c.code: c.description for c in result.classifications}
|
| 23 |
+
assert codes == {
|
| 24 |
+
"G06F17/30": "Database structures therefor",
|
| 25 |
+
"G06F17/50": "Other database related",
|
| 26 |
+
"H04L9/00": "Cryptographic mechanisms",
|
| 27 |
+
}
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
def test_parse_patent_html_raises_when_page_has_no_title():
|
| 31 |
+
with pytest.raises(ValueError):
|
| 32 |
+
parse_patent_html(MISSING_TITLE_HTML, "https://patents.google.com/patent/BOGUS/en")
|
| 33 |
+
|
| 34 |
+
|
| 35 |
async def test_scrap_patent_async_extracts_all_fields():
|
| 36 |
url = "https://patents.google.com/patent/US11930446B2/en"
|
| 37 |
with respx.mock:
|
|
@@ -13,6 +13,7 @@ from helpers import load_fixture, make_fixture_browser
|
|
| 13 |
from serp import (
|
| 14 |
BraveSearchBlockedException,
|
| 15 |
BrowserUnavailableError,
|
|
|
|
| 16 |
playwright_open_page,
|
| 17 |
query_bing_search,
|
| 18 |
query_brave_search,
|
|
@@ -21,6 +22,26 @@ from serp import (
|
|
| 21 |
)
|
| 22 |
|
| 23 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 24 |
async def test_bing_search_parses_results(browser):
|
| 25 |
fixture_browser = make_fixture_browser(browser, load_fixture("bing_results.html"))
|
| 26 |
|
|
|
|
| 13 |
from serp import (
|
| 14 |
BraveSearchBlockedException,
|
| 15 |
BrowserUnavailableError,
|
| 16 |
+
_extract_bing_results,
|
| 17 |
playwright_open_page,
|
| 18 |
query_bing_search,
|
| 19 |
query_brave_search,
|
|
|
|
| 22 |
)
|
| 23 |
|
| 24 |
|
| 25 |
+
async def test_extract_bing_results_works_on_an_already_loaded_page(browser):
|
| 26 |
+
"""Demonstrates the payoff of splitting fetch (navigate) from parse
|
| 27 |
+
(extract): `_extract_bing_results` takes a Page that's already loaded,
|
| 28 |
+
so this needs nothing beyond `set_content` - no goto-patching wrapper,
|
| 29 |
+
no browser-as-a-parameter indirection.
|
| 30 |
+
"""
|
| 31 |
+
context = await browser.new_context()
|
| 32 |
+
page = await context.new_page()
|
| 33 |
+
await page.set_content(load_fixture("bing_results.html"))
|
| 34 |
+
|
| 35 |
+
results = await _extract_bing_results(page, 10)
|
| 36 |
+
|
| 37 |
+
assert results == [
|
| 38 |
+
{"title": "Result A", "href": "https://example.org/a", "body": "Snippet A"},
|
| 39 |
+
{"title": "Result B", "href": "https://example.org/b", "body": "Snippet B"},
|
| 40 |
+
]
|
| 41 |
+
|
| 42 |
+
await context.close()
|
| 43 |
+
|
| 44 |
+
|
| 45 |
async def test_bing_search_parses_results(browser):
|
| 46 |
fixture_browser = make_fixture_browser(browser, load_fixture("bing_results.html"))
|
| 47 |
|