File size: 5,012 Bytes
3e0f245
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
#!/usr/bin/env python
"""Robust page fetch + readable-text extraction for the Web Research Agent (#13).

The real web is messy, so this handles what a demo skips:
  - resolves redirects (Google News RSS hands back news.google.com redirect links, not the article)
  - SSRF guard (refuses internal/loopback/private hosts — safe for a public demo)
  - trafilatura main-content extraction, with a BeautifulSoup fallback
  - citation-marker / whitespace cleanup

Reused patterns: text-summarizer's trafilatura→bs4 fetch, companion's SSRF guard.
"""
import ipaddress
import re
import socket
import urllib.parse

import requests
from bs4 import BeautifulSoup

_UA = ("Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
       "(KHTML, like Gecko) Chrome/124.0 Safari/537.36")
TIMEOUT = 20
_CITE = re.compile(r"\[\d+\]|\[citation needed\]|\[edit\]", re.I)


def _blocked_host(url: str) -> bool:
    """True if the URL points at a non-public address (SSRF guard)."""
    try:
        host = urllib.parse.urlparse(url).hostname
        if not host:
            return True
        if host.lower() in ("localhost",) or host.endswith(".local"):
            return True
        # resolve and check every A record
        for fam, _, _, _, sockaddr in socket.getaddrinfo(host, None):
            ip = ipaddress.ip_address(sockaddr[0])
            if ip.is_private or ip.is_loopback or ip.is_link_local or ip.is_reserved:
                return True
        return False
    except Exception:
        return True   # fail closed


def resolve_url(url: str) -> str:
    """Follow redirects to the real destination. Google News RSS links (news.google.com/rss/...)
    bounce to the actual article — we want that final URL for clean extraction + honest citations."""
    try:
        r = requests.get(url, headers={"User-Agent": _UA}, timeout=TIMEOUT, allow_redirects=True)
        final = r.url
        # Google News sometimes lands on an interstitial with the real link in a <a> or data-attr.
        if "news.google.com" in final:
            m = re.search(r'data-n-au="(https?://[^"]+)"', r.text) or \
                re.search(r'<a[^>]+href="(https?://(?!news\.google\.com)[^"]+)"', r.text)
            if m:
                return m.group(1)
        return final
    except Exception:
        return url


def _clean(text: str) -> str:
    text = _CITE.sub("", text)
    lines = [ln.strip() for ln in text.splitlines()]
    return "\n".join(ln for ln in lines if ln)


def fetch(url: str, max_chars: int = 6000) -> dict:
    """Fetch a URL and return {url, final_url, title, text, error}. text is readable main content."""
    final = resolve_url(url)
    if _blocked_host(final):
        return {"url": url, "final_url": final, "title": "", "text": "", "error": "blocked (non-public host)"}
    try:
        r = requests.get(final, headers={"User-Agent": _UA}, timeout=TIMEOUT)
        doc = r.text
    except Exception as e:
        return {"url": url, "final_url": final, "title": "", "text": "", "error": f"fetch failed ({type(e).__name__})"}

    title, text = "", ""
    try:
        import trafilatura
        text = trafilatura.extract(doc, include_comments=False, include_tables=False,
                                   favor_precision=True) or ""
        md = trafilatura.extract_metadata(doc)
        if md and md.title:
            title = md.title
    except Exception:
        text = ""

    soup = None
    if len(text) < 200:   # fallback: strip chrome, take main/article
        try:
            soup = BeautifulSoup(doc, "html.parser")
            for tag in soup(["script", "style", "nav", "header", "footer", "aside", "noscript", "form"]):
                tag.decompose()
            main = (soup.find("main") or soup.find("article")
                    or soup.find(id="mw-content-text") or soup.body or soup)
            text = main.get_text("\n")
        except Exception:
            pass
    if not title:
        try:
            soup = soup or BeautifulSoup(doc, "html.parser")
            if soup.title and soup.title.string:
                title = soup.title.string.strip()
        except Exception:
            pass

    text = _clean(text)
    if len(text) < 120:
        return {"url": url, "final_url": final, "title": title, "text": "",
                "error": "could not extract readable text (JS-only / paywall / PDF)"}
    return {"url": url, "final_url": final, "title": title[:200], "text": text[:max_chars], "error": None}


if __name__ == "__main__":
    import sys
    from search import search
    q = " ".join(sys.argv[1:]) or "retrieval augmented generation"
    print(f"search: {q}")
    for res in search(q, 3):
        f = fetch(res["url"], max_chars=500)
        status = f["error"] or f"OK · {len(f['text'])} chars"
        print(f"\n[{res['source']}] {res['title'][:70]}")
        print(f"  → final: {f['final_url'][:90]}")
        print(f"  → {status}")
        if f["text"]:
            print(f"  → {f['text'][:200].replace(chr(10),' ')}...")