#!/usr/bin/env python
"""Robust page fetch + readable-text extraction for the Web Research Agent (#13).
The real web is messy, so this handles what a demo skips:
- resolves redirects (Google News RSS hands back news.google.com redirect links, not the article)
- SSRF guard (refuses internal/loopback/private hosts — safe for a public demo)
- trafilatura main-content extraction, with a BeautifulSoup fallback
- citation-marker / whitespace cleanup
Reused patterns: text-summarizer's trafilatura→bs4 fetch, companion's SSRF guard.
"""
import ipaddress
import re
import socket
import urllib.parse
import requests
from bs4 import BeautifulSoup
_UA = ("Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/124.0 Safari/537.36")
TIMEOUT = 20
_CITE = re.compile(r"\[\d+\]|\[citation needed\]|\[edit\]", re.I)
def _blocked_host(url: str) -> bool:
"""True if the URL points at a non-public address (SSRF guard)."""
try:
host = urllib.parse.urlparse(url).hostname
if not host:
return True
if host.lower() in ("localhost",) or host.endswith(".local"):
return True
# resolve and check every A record
for fam, _, _, _, sockaddr in socket.getaddrinfo(host, None):
ip = ipaddress.ip_address(sockaddr[0])
if ip.is_private or ip.is_loopback or ip.is_link_local or ip.is_reserved:
return True
return False
except Exception:
return True # fail closed
def resolve_url(url: str) -> str:
"""Follow redirects to the real destination. Google News RSS links (news.google.com/rss/...)
bounce to the actual article — we want that final URL for clean extraction + honest citations."""
try:
r = requests.get(url, headers={"User-Agent": _UA}, timeout=TIMEOUT, allow_redirects=True)
final = r.url
# Google News sometimes lands on an interstitial with the real link in a or data-attr.
if "news.google.com" in final:
m = re.search(r'data-n-au="(https?://[^"]+)"', r.text) or \
re.search(r']+href="(https?://(?!news\.google\.com)[^"]+)"', r.text)
if m:
return m.group(1)
return final
except Exception:
return url
def _clean(text: str) -> str:
text = _CITE.sub("", text)
lines = [ln.strip() for ln in text.splitlines()]
return "\n".join(ln for ln in lines if ln)
def fetch(url: str, max_chars: int = 6000) -> dict:
"""Fetch a URL and return {url, final_url, title, text, error}. text is readable main content."""
final = resolve_url(url)
if _blocked_host(final):
return {"url": url, "final_url": final, "title": "", "text": "", "error": "blocked (non-public host)"}
try:
r = requests.get(final, headers={"User-Agent": _UA}, timeout=TIMEOUT)
doc = r.text
except Exception as e:
return {"url": url, "final_url": final, "title": "", "text": "", "error": f"fetch failed ({type(e).__name__})"}
title, text = "", ""
try:
import trafilatura
text = trafilatura.extract(doc, include_comments=False, include_tables=False,
favor_precision=True) or ""
md = trafilatura.extract_metadata(doc)
if md and md.title:
title = md.title
except Exception:
text = ""
soup = None
if len(text) < 200: # fallback: strip chrome, take main/article
try:
soup = BeautifulSoup(doc, "html.parser")
for tag in soup(["script", "style", "nav", "header", "footer", "aside", "noscript", "form"]):
tag.decompose()
main = (soup.find("main") or soup.find("article")
or soup.find(id="mw-content-text") or soup.body or soup)
text = main.get_text("\n")
except Exception:
pass
if not title:
try:
soup = soup or BeautifulSoup(doc, "html.parser")
if soup.title and soup.title.string:
title = soup.title.string.strip()
except Exception:
pass
text = _clean(text)
if len(text) < 120:
return {"url": url, "final_url": final, "title": title, "text": "",
"error": "could not extract readable text (JS-only / paywall / PDF)"}
return {"url": url, "final_url": final, "title": title[:200], "text": text[:max_chars], "error": None}
if __name__ == "__main__":
import sys
from search import search
q = " ".join(sys.argv[1:]) or "retrieval augmented generation"
print(f"search: {q}")
for res in search(q, 3):
f = fetch(res["url"], max_chars=500)
status = f["error"] or f"OK · {len(f['text'])} chars"
print(f"\n[{res['source']}] {res['title'][:70]}")
print(f" → final: {f['final_url'][:90]}")
print(f" → {status}")
if f["text"]:
print(f" → {f['text'][:200].replace(chr(10),' ')}...")