#!/usr/bin/env python """Robust page fetch + readable-text extraction for the Web Research Agent (#13). The real web is messy, so this handles what a demo skips: - resolves redirects (Google News RSS hands back news.google.com redirect links, not the article) - SSRF guard (refuses internal/loopback/private hosts — safe for a public demo) - trafilatura main-content extraction, with a BeautifulSoup fallback - citation-marker / whitespace cleanup Reused patterns: text-summarizer's trafilatura→bs4 fetch, companion's SSRF guard. """ import ipaddress import re import socket import urllib.parse import requests from bs4 import BeautifulSoup _UA = ("Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " "(KHTML, like Gecko) Chrome/124.0 Safari/537.36") TIMEOUT = 20 _CITE = re.compile(r"\[\d+\]|\[citation needed\]|\[edit\]", re.I) def _blocked_host(url: str) -> bool: """True if the URL points at a non-public address (SSRF guard).""" try: host = urllib.parse.urlparse(url).hostname if not host: return True if host.lower() in ("localhost",) or host.endswith(".local"): return True # resolve and check every A record for fam, _, _, _, sockaddr in socket.getaddrinfo(host, None): ip = ipaddress.ip_address(sockaddr[0]) if ip.is_private or ip.is_loopback or ip.is_link_local or ip.is_reserved: return True return False except Exception: return True # fail closed def resolve_url(url: str) -> str: """Follow redirects to the real destination. Google News RSS links (news.google.com/rss/...) bounce to the actual article — we want that final URL for clean extraction + honest citations.""" try: r = requests.get(url, headers={"User-Agent": _UA}, timeout=TIMEOUT, allow_redirects=True) final = r.url # Google News sometimes lands on an interstitial with the real link in a or data-attr. if "news.google.com" in final: m = re.search(r'data-n-au="(https?://[^"]+)"', r.text) or \ re.search(r']+href="(https?://(?!news\.google\.com)[^"]+)"', r.text) if m: return m.group(1) return final except Exception: return url def _clean(text: str) -> str: text = _CITE.sub("", text) lines = [ln.strip() for ln in text.splitlines()] return "\n".join(ln for ln in lines if ln) def fetch(url: str, max_chars: int = 6000) -> dict: """Fetch a URL and return {url, final_url, title, text, error}. text is readable main content.""" final = resolve_url(url) if _blocked_host(final): return {"url": url, "final_url": final, "title": "", "text": "", "error": "blocked (non-public host)"} try: r = requests.get(final, headers={"User-Agent": _UA}, timeout=TIMEOUT) doc = r.text except Exception as e: return {"url": url, "final_url": final, "title": "", "text": "", "error": f"fetch failed ({type(e).__name__})"} title, text = "", "" try: import trafilatura text = trafilatura.extract(doc, include_comments=False, include_tables=False, favor_precision=True) or "" md = trafilatura.extract_metadata(doc) if md and md.title: title = md.title except Exception: text = "" soup = None if len(text) < 200: # fallback: strip chrome, take main/article try: soup = BeautifulSoup(doc, "html.parser") for tag in soup(["script", "style", "nav", "header", "footer", "aside", "noscript", "form"]): tag.decompose() main = (soup.find("main") or soup.find("article") or soup.find(id="mw-content-text") or soup.body or soup) text = main.get_text("\n") except Exception: pass if not title: try: soup = soup or BeautifulSoup(doc, "html.parser") if soup.title and soup.title.string: title = soup.title.string.strip() except Exception: pass text = _clean(text) if len(text) < 120: return {"url": url, "final_url": final, "title": title, "text": "", "error": "could not extract readable text (JS-only / paywall / PDF)"} return {"url": url, "final_url": final, "title": title[:200], "text": text[:max_chars], "error": None} if __name__ == "__main__": import sys from search import search q = " ".join(sys.argv[1:]) or "retrieval augmented generation" print(f"search: {q}") for res in search(q, 3): f = fetch(res["url"], max_chars=500) status = f["error"] or f"OK · {len(f['text'])} chars" print(f"\n[{res['source']}] {res['title'][:70]}") print(f" → final: {f['final_url'][:90]}") print(f" → {status}") if f["text"]: print(f" → {f['text'][:200].replace(chr(10),' ')}...")