#!/usr/bin/env python """Structured Data Extractor — paste messy text (an email, receipt, job post…) and get clean, schema-valid JSON. Uses Claude's forced tool-use so the output always matches the schema (Ollama fallback uses constrained JSON decoding). pip install flask requests anthropic python app.py # http://127.0.0.1:8500 Project #4 of the "30 Projects in 15 Days" challenge — GritAI. """ import os, json from flask import Flask, request, Response, jsonify PORT = int(os.environ.get("PORT", "8500")) ANTHROPIC_KEY = os.environ.get("ANTHROPIC_API_KEY") ANTHROPIC_MODEL = os.environ.get("ANTHROPIC_MODEL", "claude-sonnet-5") OLLAMA_URL = os.environ.get("OLLAMA_URL", "http://127.0.0.1:11434") OLLAMA_MODEL = os.environ.get("OLLAMA_MODEL", "qwen2.5:7b") S = lambda: {"type": ["string", "null"]} N = lambda: {"type": ["number", "null"]} SCHEMAS = { "contact": {"label": "Contact", "emoji": "👤", "schema": {"type": "object", "properties": { "name": S(), "title": S(), "company": S(), "email": S(), "phone": S(), "website": S(), "address": S()}, "required": ["name"]}}, "invoice": {"label": "Invoice / Receipt", "emoji": "🧾", "schema": {"type": "object", "properties": { "vendor": S(), "date": S(), "currency": S(), "subtotal": N(), "tax": N(), "total": N(), "line_items": {"type": "array", "items": {"type": "object", "properties": { "description": {"type": "string"}, "qty": N(), "unit_price": N(), "amount": N()}}}}, "required": ["vendor", "total"]}}, "event": {"label": "Event", "emoji": "📅", "schema": {"type": "object", "properties": { "title": S(), "date": S(), "start_time": S(), "end_time": S(), "location": S(), "organizer": S(), "description": S()}, "required": ["title"]}}, "job": {"label": "Job posting", "emoji": "💼", "schema": {"type": "object", "properties": { "title": S(), "company": S(), "location": S(), "employment_type": S(), "salary_range": S(), "responsibilities": {"type": "array", "items": {"type": "string"}}, "requirements": {"type": "array", "items": {"type": "string"}}, "apply_url": S()}, "required": ["title"]}}, } SAMPLES = { "contact": ("👤", "Email signature", "thanks again! — Maya\n\nMaya R. Fitzgerald\nSr. Solutions Architect, Northwind Cloud\nmaya.fitzgerald@northwind.example | cell (415) 555-0182\n1200 Harbor Blvd, Suite 400, San Diego CA 92101\nnorthwindcloud.example"), "invoice": ("🧾", "Receipt", "RIVERSIDE HARDWARE\nInvoice #4821 — Mar 14, 2026\n\n2x Cordless drill @ $89.99 = $179.98\n1x Drill bit set @ $24.50 = $24.50\n3x Work gloves @ $12.00 = $36.00\n\nSubtotal: $240.48\nTax (8.25%): $19.84\nTOTAL: $260.32\nPaid: Visa ****4417"), "job": ("💼", "Job posting", "We're hiring a Senior Backend Engineer (Remote, US) at Lumina Labs. Comp: $150k-$185k + equity. You'll design and scale our API platform, own service reliability, and mentor two junior engineers. Must have: 5+ yrs backend, strong Python or Go, experience with distributed systems and Postgres. Nice to have: Kafka, k8s. Apply at lumina.example/careers/be-senior"), } def build_schema(key, custom_fields): if key in SCHEMAS: return SCHEMAS[key]["schema"] if key == "custom": props = {f.strip(): S() for f in (custom_fields or "").split(",") if f.strip()} return {"type": "object", "properties": props} return SCHEMAS["contact"]["schema"] def extract(text, schema): prompt = ("Extract structured data from the text below into the required schema. " "Use null for anything not present — never invent values.\n\nTEXT:\n" + text[:12000]) if ANTHROPIC_KEY: import anthropic client = anthropic.Anthropic(api_key=ANTHROPIC_KEY) r = client.messages.create(model=ANTHROPIC_MODEL, max_tokens=1500, tools=[{"name": "extract", "description": "Return the structured data from the text.", "input_schema": schema}], tool_choice={"type": "tool", "name": "extract"}, messages=[{"role": "user", "content": prompt}]) for b in r.content: if b.type == "tool_use": return b.input return {} import requests r = requests.post(f"{OLLAMA_URL}/api/chat", timeout=90, json={ "model": OLLAMA_MODEL, "stream": False, "format": schema, "messages": [{"role": "user", "content": prompt}], "options": {"temperature": 0}}) r.raise_for_status() return json.loads(r.json()["message"]["content"]) app = Flask(__name__) @app.route("/") def home(): return Response(PAGE, mimetype="text/html") @app.route("/api/extract", methods=["POST"]) def api_extract(): b = request.get_json(force=True) text = (b.get("text") or "").strip() key = b.get("schema", "contact") fields = b.get("fields", "") if not text: return jsonify(error="Paste some text to extract from."), 400 schema = build_schema(key, fields) if not schema.get("properties"): return jsonify(error="No fields to extract — add some custom fields (comma-separated)."), 400 try: return jsonify(data=extract(text, schema)) except Exception as e: import sys; print("extract error:", type(e).__name__, file=sys.stderr, flush=True) return jsonify(error="Extraction failed — please try again."), 200 PAGE = """ Data Extractor — GritAI
GRITAIData Extractor
▸ LiveSDVOSBLawton OK

Messy text in.
Schema-valid JSON out.

Paste an email signature, a receipt, a job post, or an event invite — get back clean, structured JSON via forced tool-use. It never invents values.

01Source Text
02Target Schema
$
03Extracted JSON
""" @app.route("/api/sample_text") def sample_text(): name = request.args.get("name", "") if name in SAMPLES: return jsonify(text=SAMPLES[name][2]) return jsonify(error="unknown"), 404 if __name__ == "__main__": backend = "Claude API" if ANTHROPIC_KEY else f"Ollama ({OLLAMA_MODEL} @ {OLLAMA_URL})" print(f"Structured Data Extractor on http://127.0.0.1:{PORT} [backend: {backend}]", flush=True) app.run(host="0.0.0.0", port=PORT, threaded=True)