Spaces:
Sleeping
Sleeping
Download fetch.py from BonusLockSMith/web-research-agent: direct link, hf CLI and curl.
- Browser
- Download file 5.01 kB
-
https://huggingface.co/spaces/BonusLockSMith/web-research-agent/resolve/main/fetch.py
- Command line
-
hf download hf://spaces/BonusLockSMith/web-research-agent/fetch.py
-
curl -L -o fetch.py https://huggingface.co/spaces/BonusLockSMith/web-research-agent/resolve/main/fetch.py
5.01 kB
| #!/usr/bin/env python | |
| """Robust page fetch + readable-text extraction for the Web Research Agent (#13). | |
| The real web is messy, so this handles what a demo skips: | |
| - resolves redirects (Google News RSS hands back news.google.com redirect links, not the article) | |
| - SSRF guard (refuses internal/loopback/private hosts — safe for a public demo) | |
| - trafilatura main-content extraction, with a BeautifulSoup fallback | |
| - citation-marker / whitespace cleanup | |
| Reused patterns: text-summarizer's trafilatura→bs4 fetch, companion's SSRF guard. | |
| """ | |
| import ipaddress | |
| import re | |
| import socket | |
| import urllib.parse | |
| import requests | |
| from bs4 import BeautifulSoup | |
| _UA = ("Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " | |
| "(KHTML, like Gecko) Chrome/124.0 Safari/537.36") | |
| TIMEOUT = 20 | |
| _CITE = re.compile(r"\[\d+\]|\[citation needed\]|\[edit\]", re.I) | |
| def _blocked_host(url: str) -> bool: | |
| """True if the URL points at a non-public address (SSRF guard).""" | |
| try: | |
| host = urllib.parse.urlparse(url).hostname | |
| if not host: | |
| return True | |
| if host.lower() in ("localhost",) or host.endswith(".local"): | |
| return True | |
| # resolve and check every A record | |
| for fam, _, _, _, sockaddr in socket.getaddrinfo(host, None): | |
| ip = ipaddress.ip_address(sockaddr[0]) | |
| if ip.is_private or ip.is_loopback or ip.is_link_local or ip.is_reserved: | |
| return True | |
| return False | |
| except Exception: | |
| return True # fail closed | |
| def resolve_url(url: str) -> str: | |
| """Follow redirects to the real destination. Google News RSS links (news.google.com/rss/...) | |
| bounce to the actual article — we want that final URL for clean extraction + honest citations.""" | |
| try: | |
| r = requests.get(url, headers={"User-Agent": _UA}, timeout=TIMEOUT, allow_redirects=True) | |
| final = r.url | |
| # Google News sometimes lands on an interstitial with the real link in a <a> or data-attr. | |
| if "news.google.com" in final: | |
| m = re.search(r'data-n-au="(https?://[^"]+)"', r.text) or \ | |
| re.search(r'<a[^>]+href="(https?://(?!news\.google\.com)[^"]+)"', r.text) | |
| if m: | |
| return m.group(1) | |
| return final | |
| except Exception: | |
| return url | |
| def _clean(text: str) -> str: | |
| text = _CITE.sub("", text) | |
| lines = [ln.strip() for ln in text.splitlines()] | |
| return "\n".join(ln for ln in lines if ln) | |
| def fetch(url: str, max_chars: int = 6000) -> dict: | |
| """Fetch a URL and return {url, final_url, title, text, error}. text is readable main content.""" | |
| final = resolve_url(url) | |
| if _blocked_host(final): | |
| return {"url": url, "final_url": final, "title": "", "text": "", "error": "blocked (non-public host)"} | |
| try: | |
| r = requests.get(final, headers={"User-Agent": _UA}, timeout=TIMEOUT) | |
| doc = r.text | |
| except Exception as e: | |
| return {"url": url, "final_url": final, "title": "", "text": "", "error": f"fetch failed ({type(e).__name__})"} | |
| title, text = "", "" | |
| try: | |
| import trafilatura | |
| text = trafilatura.extract(doc, include_comments=False, include_tables=False, | |
| favor_precision=True) or "" | |
| md = trafilatura.extract_metadata(doc) | |
| if md and md.title: | |
| title = md.title | |
| except Exception: | |
| text = "" | |
| soup = None | |
| if len(text) < 200: # fallback: strip chrome, take main/article | |
| try: | |
| soup = BeautifulSoup(doc, "html.parser") | |
| for tag in soup(["script", "style", "nav", "header", "footer", "aside", "noscript", "form"]): | |
| tag.decompose() | |
| main = (soup.find("main") or soup.find("article") | |
| or soup.find(id="mw-content-text") or soup.body or soup) | |
| text = main.get_text("\n") | |
| except Exception: | |
| pass | |
| if not title: | |
| try: | |
| soup = soup or BeautifulSoup(doc, "html.parser") | |
| if soup.title and soup.title.string: | |
| title = soup.title.string.strip() | |
| except Exception: | |
| pass | |
| text = _clean(text) | |
| if len(text) < 120: | |
| return {"url": url, "final_url": final, "title": title, "text": "", | |
| "error": "could not extract readable text (JS-only / paywall / PDF)"} | |
| return {"url": url, "final_url": final, "title": title[:200], "text": text[:max_chars], "error": None} | |
| if __name__ == "__main__": | |
| import sys | |
| from search import search | |
| q = " ".join(sys.argv[1:]) or "retrieval augmented generation" | |
| print(f"search: {q}") | |
| for res in search(q, 3): | |
| f = fetch(res["url"], max_chars=500) | |
| status = f["error"] or f"OK · {len(f['text'])} chars" | |
| print(f"\n[{res['source']}] {res['title'][:70]}") | |
| print(f" → final: {f['final_url'][:90]}") | |
| print(f" → {status}") | |
| if f["text"]: | |
| print(f" → {f['text'][:200].replace(chr(10),' ')}...") | |