#!/usr/bin/env python """Text Summarizer — paste text or a URL, get a clean TL;DR + key points. Streams the summary live. Backend auto-selects: Anthropic Claude API if ANTHROPIC_API_KEY is set, otherwise a local Ollama model. pip install flask requests anthropic beautifulsoup4 python app.py # http://127.0.0.1:8500 Project #2 of the "30 Projects in 15 Days" challenge — GritAI. """ import os, re, json, html from flask import Flask, request, Response PORT = int(os.environ.get("PORT", "8500")) ANTHROPIC_KEY = os.environ.get("ANTHROPIC_API_KEY") ANTHROPIC_MODEL = os.environ.get("ANTHROPIC_MODEL", "claude-sonnet-5") OLLAMA_URL = os.environ.get("OLLAMA_URL", "http://127.0.0.1:11434") OLLAMA_MODEL = os.environ.get("OLLAMA_MODEL", "qwen2.5:7b") MAX_CHARS = 40000 # cap input to control tokens/cost LENGTHS = { "brief": "a tight 2-4 sentence TL;DR that captures the essence. No bullet points, no headings.", "standard": "first a 2-3 sentence overview paragraph, then a blank line, then 4-6 key points, each on its own line starting with '- '.", "detailed": "first a short overview paragraph, then a blank line, then 7-10 key points each on its own line starting with '- ', preserving important specifics (names, numbers, dates, outcomes).", } def system_prompt(length): return ( "You are an expert editor who writes clear, coherent summaries.\n" "OUTPUT FORMAT: " + LENGTHS.get(length, LENGTHS["standard"]) + "\n" "RULES:\n" "- Be faithful to the source; never add facts, opinions, or details not present.\n" "- Write clean, natural plain text. Do NOT use markdown (**bold**, # headings). " "Strip any citation markers like [1], footnote symbols, or wiki artifacts.\n" "- Bullets (when used) start with '- ', are clear parallel phrases or sentences, and are non-redundant.\n" "- Lead with the single most important point. Be specific, not vague. Do not restate these instructions." ) _CITE = re.compile(r'\[(?:\d+|[a-z]{1,3}|note \d+|citation needed|clarification needed|edit|update|when\?|who\?|why\?)\]', re.I) def _clean(text): text = _CITE.sub("", text or "") return "\n".join(l.strip() for l in text.splitlines() if l.strip()) def fetch_url_text(url): """Fetch a page and extract the main article text (trafilatura, bs4 fallback).""" import requests r = requests.get(url, timeout=20, headers={"User-Agent": "Mozilla/5.0 (compatible; GritAI-Summarizer/1.0)"}) r.raise_for_status() doc = r.text text = "" try: import trafilatura text = trafilatura.extract(doc, include_comments=False, include_tables=False, favor_precision=True) or "" except Exception: text = "" if len(text) < 200: # fallback: main-content via bs4 from bs4 import BeautifulSoup soup = BeautifulSoup(doc, "html.parser") for tag in soup(["script", "style", "nav", "header", "footer", "aside", "noscript", "form"]): tag.decompose() main = (soup.find("main") or soup.find("article") or soup.find(id="mw-content-text") or soup.body or soup) text = main.get_text("\n") text = _clean(text) if len(text) < 200: raise ValueError("Couldn't extract readable text from that page (it may be JavaScript-only or blocked).") return text def summarize_stream(text, length): sys = system_prompt(length) user = "Summarize the following:\n\n" + text[:MAX_CHARS] if ANTHROPIC_KEY: import anthropic client = anthropic.Anthropic(api_key=ANTHROPIC_KEY) with client.messages.stream(model=ANTHROPIC_MODEL, max_tokens=800, system=sys, messages=[{"role": "user", "content": user}]) as s: for t in s.text_stream: yield t return import requests with requests.post(f"{OLLAMA_URL}/api/chat", stream=True, timeout=120, json={ "model": OLLAMA_MODEL, "stream": True, "messages": [{"role": "system", "content": sys}, {"role": "user", "content": user}], "options": {"temperature": 0.2}}) as r: r.raise_for_status() for line in r.iter_lines(): if not line: continue d = json.loads(line) if d.get("message", {}).get("content"): yield d["message"]["content"] if d.get("done"): break app = Flask(__name__) @app.route("/") def home(): return Response(PAGE, mimetype="text/html") @app.route("/api/summarize", methods=["POST"]) def api_summarize(): body = request.get_json(force=True) text = (body.get("text") or "").strip() url = (body.get("url") or "").strip() length = body.get("length", "standard") def gen(): import sys as _sys try: src = text if url: yield "data: " + json.dumps({"t": "Fetching the page…\n\n"}) + "\n\n" src = fetch_url_text(url) if not src: yield "data: " + json.dumps({"t": "Please paste some text or a URL to summarize."}) + "\n\n" yield "data: [DONE]\n\n" return if url: yield "data: " + json.dumps({"reset": True}) + "\n\n" # clear the "fetching" note for chunk in summarize_stream(src, length): yield "data: " + json.dumps({"t": chunk}) + "\n\n" except Exception as e: print("summarize error:", type(e).__name__, file=_sys.stderr, flush=True) yield "data: " + json.dumps({"t": "Sorry — something went wrong. Please try again."}) + "\n\n" yield "data: [DONE]\n\n" return Response(gen(), mimetype="text/event-stream", headers={"Cache-Control": "no-cache", "X-Accel-Buffering": "no"}) PAGE = """ Text Summarizer — GritAI
GRITAIText Summarizer ▸ LiveSDVOSBLawton OK

Get the gist in seconds.

Paste an article, email, or notes — or drop in a link — and get a faithful TL;DR with the key points, streamed live.

01Source
›
02Length
03Summary
""" if __name__ == "__main__": backend = "Claude API" if ANTHROPIC_KEY else f"Ollama ({OLLAMA_MODEL} @ {OLLAMA_URL})" print(f"Text Summarizer on http://127.0.0.1:{PORT} [backend: {backend}]", flush=True) app.run(host="0.0.0.0", port=PORT, threaded=True)