File size: 5,082 Bytes
c38bd79
 
cf1c265
c38bd79
 
 
 
 
 
681fc53
 
 
 
 
 
c38bd79
 
cf1c265
c38bd79
cf1c265
c38bd79
cf1c265
c38bd79
cf1c265
c38bd79
cf1c265
 
 
 
681fc53
 
c38bd79
 
 
 
 
 
66641c2
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f27076f
 
 
 
681fc53
 
f27076f
 
 
 
 
 
 
 
 
 
 
 
681fc53
66641c2
 
 
 
 
 
 
681fc53
 
66641c2
 
 
 
 
 
 
 
 
 
c38bd79
66641c2
 
 
c38bd79
66641c2
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
import asyncio
import logging
import re
from typing import Optional
from httpx import AsyncClient
from bs4 import BeautifulSoup
from pydantic import BaseModel


class ClassificationCode(BaseModel):
    """A single CPC or IPC classification code with its description."""
    code: str
    description: str


class PatentScrapResult(BaseModel):
    """Schema for the result of scraping a google patents page."""
    # The title of the patent.
    title: str
    # The abstract of the patent, if available.
    abstract: Optional[str] = None
    # The full description of the patent containing the field of the invention, background, summary, etc.
    description: Optional[str] = None
    # The full claims of the patent.
    claims: Optional[str] = None
    # The field of the invention, if available.
    field_of_invention: Optional[str] = None
    # The background of the invention, if available.
    background: Optional[str] = None
    # CPC and IPC classification codes with descriptions.
    classifications: Optional[list[ClassificationCode]] = None


async def scrap_patent_async(client: AsyncClient, patent_url: str) -> PatentScrapResult:
    headers = {
        "User-Agent": "Mozilla/5.0 (compatible; GPTBot/1.0; +https://openai.com/gptbot)"
    }
    response = await client.get(patent_url, headers=headers)
    response.raise_for_status()

    soup = BeautifulSoup(response.text, "html.parser")

    # Abstract
    abstract_div = soup.find("div", {"class": "abstract"})
    abstract = abstract_div.get_text(
        strip=True) if abstract_div else None

    # Description
    description_section = soup.find("section", itemprop="description")
    description = description_section.get_text(
        separator="\n", strip=True) if description_section else None

    # Field of the Invention
    invention_field_match = re.findall(
        r"(FIELD OF THE INVENTION|TECHNICAL FIELD)(.*?)(?:(BACKGROUND|BACKGROUND OF THE INVENTION|SUMMARY|BRIEF SUMMARY|DETAILED DESCRIPTION|DESCRIPTION OF THE RELATED ART))", description, re.IGNORECASE | re.DOTALL) if description_section else None
    invention_field = invention_field_match[0][1].strip(
    ) if invention_field_match else None

    # Background of the Invention
    invention_background_match = re.findall(
        r"(BACKGROUND OF THE INVENTION|BACKGROUND)(.*?)(?:(SUMMARY|BRIEF SUMMARY|DETAILED DESCRIPTION|DESCRIPTION OF THE PREFERRED EMBODIMENTS|DESCRIPTION))", description, re.IGNORECASE | re.DOTALL) if description_section else None
    invention_background = invention_background_match[0][1].strip(
    ) if invention_background_match else None

    # Claims
    claims_section = soup.find("section", itemprop="claims")
    claims = claims_section.get_text(
        separator="\n", strip=True) if claims_section else None

    # Patent Title
    meta_title = soup.find("meta", {"name": "DC.title"}).get(
        "content").strip()

    # Patent publication number
    # pub_num = soup.select_one("h2#pubnum").get_text(strip=True)
    # get the h2 with id ="pubnum" and extract the text

    # Classification codes (CPC + IPC, flat list, deduplicated by code).
    # Google Patents renders each code in a <span> inside a <li>. Leaf entries
    # (no nested <ul>) carry exactly one code; parent entries are breadcrumbs.
    leaf_code_re = re.compile(r'^[A-Z]\d{2}[A-Z]\d+/\d+$')
    classifications = []
    seen_codes: set[str] = set()
    for li in soup.find_all("li"):
        if li.find("ul"):
            continue
        span = li.find("span")
        if not span:
            continue
        code = span.get_text(strip=True)
        if leaf_code_re.match(code) and code not in seen_codes:
            seen_codes.add(code)
            full_text = li.get_text(separator=" ", strip=True)
            desc = full_text.replace(code, "", 1).strip().lstrip("—").lstrip("-").strip()
            classifications.append(ClassificationCode(code=code, description=desc))

    return PatentScrapResult(
        # publication_number=pub_num,
        abstract=abstract,
        description=description,
        claims=claims,
        title=meta_title,
        field_of_invention=invention_field,
        background=invention_background,
        classifications=classifications or None
    )


class PatentScrapBulkResponse(BaseModel):
    """Response model for bulk patent scraping."""
    patents: list[PatentScrapResult]
    failed_ids: list[str]


async def scrap_patent_bulk_async(client: AsyncClient, patent_ids: list[int]) -> PatentScrapBulkResponse:
    """Scrape multiple patents asynchronously."""
    urls = [
        f"https://patents.google.com/patent/{pid}/en" for pid in patent_ids]
    results = await asyncio.gather(*[scrap_patent_async(client, url) for url in urls], return_exceptions=True)

    filtered_results = [
        res for res in results if not isinstance(res, Exception)]

    failed_ids = [
        patent_ids[i] for i, res in enumerate(results) if isinstance(res, Exception)
    ]

    return PatentScrapBulkResponse(
        patents=filtered_results,
        failed_ids=failed_ids
    )