Download space_app.py from graziul/differance-engine: direct link, hf CLI and curl.
- Browser
- Download file 44 kB
-
https://huggingface.co/graziul/differance-engine/resolve/main/space_app.py
- Command line
-
hf download hf://graziul/differance-engine/space_app.py
-
curl -L -o space_app.py https://huggingface.co/graziul/differance-engine/resolve/main/space_app.py
44 kB
| """ | |
| HF Space entry point — FastAPI app serving the Différance Engine. | |
| Copy of site/app.py but with paths adjusted for the Space root layout. | |
| The Space root is under-erasure/ with: | |
| - pipeline/ Python pipeline code | |
| - data/ SQLite database + benchmarks | |
| - site/dist/ Astro-built static files | |
| - site/src/ Astro source (not needed at runtime) | |
| """ | |
| from __future__ import annotations | |
| import json | |
| import os | |
| import sys | |
| from pathlib import Path | |
| # Ensure pipeline is importable from Space root | |
| _space_root = Path(__file__).resolve().parent | |
| if str(_space_root) not in sys.path: | |
| sys.path.insert(0, str(_space_root)) | |
| from fastapi import FastAPI, HTTPException, Query | |
| from fastapi.responses import FileResponse, HTMLResponse, RedirectResponse | |
| from fastapi.staticfiles import StaticFiles | |
| app = FastAPI(title="Différance Engine", version="0.2.0") | |
| DIST_DIR = _space_root / "site" / "dist" | |
| # Serve Astro static assets (CSS, JS, fonts) | |
| if (DIST_DIR / "_astro").exists(): | |
| app.mount("/_astro", StaticFiles(directory=str(DIST_DIR / "_astro")), name="astro_assets") | |
| async def index(): | |
| """Dynamic index — queries live DB for stats and papers at render time.""" | |
| try: | |
| from pipeline.db import get_db | |
| db = get_db() | |
| db.connect() | |
| stats = db.stats() | |
| papers = db.get_displayable_papers(limit=30) | |
| db.close() | |
| return HTMLResponse(_render_index_html(stats, papers)) | |
| except Exception: | |
| return HTMLResponse(_fallback_html()) | |
| async def discourse_page(): | |
| """Translation trace: papers sorted by reduction rate, descending and ascending. | |
| NOT a quality leaderboard. Translation efficiency against a declared taxonomy | |
| is descriptive, not evaluative. High rate = vocabulary overlap. Low rate = | |
| where to look for genuine novelty OR taxonomy gaps OR bullshit. Reader decides. | |
| """ | |
| try: | |
| from pipeline.db import get_db | |
| db = get_db(); db.connect() | |
| papers = db.get_displayable_papers(limit=50) | |
| # Enrich with per-paper reduction rate | |
| enriched = [] | |
| for p in papers: | |
| reds = db.get_reductions_for_paper(p["arxiv_id"]) | |
| total = len(reds) | |
| mapped = sum(1 for r in reds if r["result_type"] in ("identity", "compositional")) | |
| rate = mapped / total if total > 0 else 0 | |
| unknown = total - mapped | |
| p["_reduction_rate"] = rate | |
| p["_mapped"] = mapped | |
| p["_unknown"] = unknown | |
| p["_total"] = total | |
| enriched.append(p) | |
| db.close() | |
| # Sort descending (most translated) and ascending (least translated) | |
| desc = sorted(enriched, key=lambda p: p["_reduction_rate"], reverse=True) | |
| asc = sorted(enriched, key=lambda p: p["_reduction_rate"]) | |
| return HTMLResponse(_render_discourse_html(desc, asc)) | |
| except Exception as e: | |
| return HTMLResponse(_fallback_html(str(e))) | |
| async def lookup_page(): | |
| lookup_path = DIST_DIR / "lookup" / "index.html" | |
| if lookup_path.exists(): | |
| return HTMLResponse(lookup_path.read_text()) | |
| return HTMLResponse(_fallback_html()) | |
| async def paper_page(arxiv_id: str): | |
| paper_path = DIST_DIR / "paper" / arxiv_id / "index.html" | |
| if paper_path.exists(): | |
| return HTMLResponse(paper_path.read_text()) | |
| try: | |
| from pipeline.db import get_db | |
| db = get_db() | |
| db.connect() | |
| paper = db.find_paper(arxiv_id) | |
| if paper: | |
| canonical_id = paper["arxiv_id"] | |
| reductions = db.get_reductions_for_paper(canonical_id) | |
| db.close() | |
| return HTMLResponse(_render_paper_html(paper, reductions)) | |
| db.close() | |
| except Exception: | |
| pass | |
| return HTMLResponse(_fallback_html(f"Paper {arxiv_id} not found."), status_code=404) | |
| # --- API --- | |
| async def api_stats(): | |
| try: | |
| from pipeline.db import get_db | |
| db = get_db() | |
| db.connect() | |
| stats = db.stats() | |
| db.close() | |
| return stats | |
| except Exception as e: | |
| return {"error": str(e)} | |
| async def api_paper(arxiv_id: str): | |
| try: | |
| from pipeline.db import get_db | |
| db = get_db() | |
| db.connect() | |
| paper = db.find_paper(arxiv_id) | |
| if not paper: | |
| db.close() | |
| raise HTTPException(status_code=404) | |
| canonical_id = paper["arxiv_id"] | |
| reductions = db.get_reductions_for_paper(canonical_id) | |
| db.close() | |
| return {"paper": paper, "reductions": reductions} | |
| except HTTPException: | |
| raise | |
| except Exception as e: | |
| raise HTTPException(status_code=500, detail=str(e)) | |
| async def api_lookup(q: str = Query(...)): | |
| try: | |
| from pipeline.db import get_db | |
| db = get_db() | |
| db.connect() | |
| paper = db.find_paper(q.strip()) | |
| db.close() | |
| if paper: | |
| return RedirectResponse(f"/paper/{paper['arxiv_id']}") | |
| raise HTTPException(status_code=404, detail=f"Not found: {q}") | |
| except HTTPException: | |
| raise | |
| async def api_index_analogs(): | |
| """All canonical analogs cited, with paper counts (cross-reference index).""" | |
| try: | |
| from pipeline.db import get_db | |
| db = get_db(); db.connect() | |
| analogs = db.get_canonical_analogs() | |
| db.close() | |
| return analogs | |
| except Exception as e: | |
| return {"error": str(e)} | |
| async def api_index_moves(): | |
| """Deconstructive move counts across all concepts.""" | |
| try: | |
| from pipeline.db import get_db | |
| db = get_db(); db.connect() | |
| moves = db.get_move_counts() | |
| db.close() | |
| return moves | |
| except Exception as e: | |
| return {"error": str(e)} | |
| async def api_crossref(arxiv_id: str): | |
| """Papers that share canonical analogs with this paper.""" | |
| try: | |
| from pipeline.db import get_db | |
| db = get_db(); db.connect() | |
| paper = db.find_paper(arxiv_id) | |
| if not paper: | |
| db.close() | |
| raise HTTPException(status_code=404) | |
| reductions = db.get_reductions_for_paper(paper["arxiv_id"]) | |
| # Collect unique canonical analogs for this paper | |
| analogs = list({r["canonical_analog"] for r in reductions if r.get("canonical_analog")}) | |
| # Find related papers for each analog | |
| related: dict[str, list] = {} | |
| for analog in analogs: | |
| papers = db.get_papers_by_analog(analog, limit=10) | |
| # Exclude self | |
| papers = [p for p in papers if p["arxiv_id"] != paper["arxiv_id"]] | |
| if papers: | |
| related[analog] = papers | |
| db.close() | |
| return { | |
| "paper_id": paper["arxiv_id"], | |
| "paper_title": paper["title"], | |
| "canonical_analogs": analogs, | |
| "related_papers": related, | |
| } | |
| except HTTPException: | |
| raise | |
| except Exception as e: | |
| raise HTTPException(status_code=500, detail=str(e)) | |
| async def canonical_page(formalism_id: str): | |
| """Page for a canonical formalism, with its orienting paper deconstructed.""" | |
| try: | |
| from pipeline.match import load_engine | |
| from pipeline.db import get_db | |
| engine = load_engine() | |
| fm = engine._by_id.get(formalism_id) | |
| if not fm: | |
| raise HTTPException(status_code=404, detail=f"Unknown formalism: {formalism_id}") | |
| db = get_db(); db.connect() | |
| # Papers that cite this formalism as a canonical analog | |
| citing_papers = db.get_papers_by_analog(fm.name, limit=30) | |
| # Check if canonical paper is already deconstructed | |
| canonical_arxiv = getattr(fm, 'canonical_arxiv_id', None) | |
| # Also load from raw YAML for canonical_arxiv_id | |
| import yaml | |
| from pathlib import Path | |
| kb_path = Path(__file__).resolve().parent / "pipeline" / "kb" / "formalisms.yaml" | |
| with open(kb_path) as f: | |
| raw = yaml.safe_load(f) | |
| for entry in raw.get("formalisms", []): | |
| if entry.get("id") == formalism_id: | |
| canonical_arxiv = entry.get("canonical_arxiv_id") | |
| break | |
| canonical_paper = None | |
| canonical_reductions = [] | |
| if canonical_arxiv: | |
| paper = db.find_paper(canonical_arxiv) | |
| if paper and paper.get("status") == "matched": | |
| canonical_paper = paper | |
| canonical_reductions = db.get_reductions_for_paper(paper["arxiv_id"]) | |
| db.close() | |
| return HTMLResponse(_render_canonical_html( | |
| fm, citing_papers, canonical_arxiv, | |
| canonical_paper, canonical_reductions | |
| )) | |
| except HTTPException: | |
| raise | |
| except Exception as e: | |
| return HTMLResponse(_fallback_html(str(e))) | |
| async def api_canonical(formalism_id: str): | |
| """JSON API for a canonical formalism.""" | |
| try: | |
| from pipeline.match import load_engine | |
| from pipeline.db import get_db | |
| engine = load_engine() | |
| fm = engine._by_id.get(formalism_id) | |
| if not fm: | |
| raise HTTPException(status_code=404) | |
| # Get canonical_arxiv_id | |
| import yaml | |
| from pathlib import Path | |
| kb_path = Path(__file__).resolve().parent / "pipeline" / "kb" / "formalisms.yaml" | |
| with open(kb_path) as f: | |
| raw = yaml.safe_load(f) | |
| canonical_arxiv = None | |
| for entry in raw.get("formalisms", []): | |
| if entry.get("id") == formalism_id: | |
| canonical_arxiv = entry.get("canonical_arxiv_id") | |
| break | |
| db = get_db(); db.connect() | |
| citing = db.get_papers_by_analog(fm.name, limit=30) | |
| db.close() | |
| return { | |
| "id": fm.id, "name": fm.name, "year": fm.year, | |
| "origin": fm.origin, | |
| "signature": { | |
| "operation": fm.signature.operation, | |
| "domain": fm.signature.domain, | |
| "codomain": fm.signature.codomain, | |
| "objective_family": fm.signature.objective_family, | |
| }, | |
| "meso_type": fm.meso_type, "macro_type": fm.macro_type, | |
| "canonical_reference": fm.canonical_reference, | |
| "canonical_arxiv_id": canonical_arxiv, | |
| "citing_papers_count": len(citing), | |
| "citing_papers": citing[:10], | |
| } | |
| except HTTPException: | |
| raise | |
| except Exception as e: | |
| raise HTTPException(status_code=500, detail=str(e)) | |
| async def api_cite(arxiv_id: str): | |
| """Look up citation count from Semantic Scholar.""" | |
| import urllib.request, json as _json | |
| url = f"https://api.semanticscholar.org/graph/v1/paper/ArXiv:{arxiv_id}?fields=citationCount,influentialCitationCount,title,year" | |
| try: | |
| req = urllib.request.Request(url, headers={"User-Agent": "DifferanceEngine/1.0"}) | |
| with urllib.request.urlopen(req, timeout=10) as resp: | |
| data = _json.loads(resp.read()) | |
| return { | |
| "arxiv_id": arxiv_id, | |
| "title": data.get("title", ""), | |
| "year": data.get("year"), | |
| "citation_count": data.get("citationCount", 0), | |
| "influential_citation_count": data.get("influentialCitationCount", 0), | |
| } | |
| except Exception as e: | |
| return {"arxiv_id": arxiv_id, "error": str(e), "citation_count": 0} | |
| async def api_models(): | |
| try: | |
| from pipeline.extract import ( | |
| select_extraction_model, | |
| _benchmark_cache, | |
| EXTRACTION_MODEL_CANDIDATES, | |
| ) | |
| api_key = os.environ.get("HF_API_KEY", "") | |
| selection = select_extraction_model(api_key) | |
| benchmarks = {} | |
| for mid, bench in _benchmark_cache.items(): | |
| benchmarks[mid] = { | |
| "passed": bench.passed, | |
| "correctness_score": bench.correctness_score, | |
| "completeness_score": bench.completeness_score, | |
| "latency_sec": bench.latency_sec, | |
| "cost_est": bench.cost_est, | |
| "error": bench.error, | |
| } | |
| return { | |
| "selected_model": selection.model_id, | |
| "selected_reason": selection.reason, | |
| "selected_at": selection.selected_at, | |
| "candidates": [ | |
| {"id": c["id"], "cost_per_1k_tokens": c["cost_per_1k_tokens"]} | |
| for c in EXTRACTION_MODEL_CANDIDATES | |
| ], | |
| "benchmarks": benchmarks, | |
| } | |
| except Exception as e: | |
| return {"error": str(e)} | |
| async def api_trigger(arxiv_id: str | None = None, retroactive: bool = False): | |
| api_key = os.environ.get("HF_API_KEY", "") | |
| if not api_key: | |
| raise HTTPException(status_code=401, detail="HF_API_KEY not configured") | |
| try: | |
| from pipeline.db import get_db | |
| from pipeline.ingest import ingest_single | |
| from pipeline.match import load_engine | |
| from pipeline.extract import extract_paper as _extract | |
| db = get_db() | |
| engine = load_engine() | |
| if arxiv_id: | |
| paper = ingest_single(arxiv_id.strip(), db=db) | |
| if not paper: | |
| raise HTTPException(status_code=404, detail=f"Could not fetch {arxiv_id}") | |
| # Use the canonical arXiv ID from the DB/API (may include version suffix) | |
| canonical_id = paper["arxiv_id"] | |
| import traceback | |
| try: | |
| extraction = _extract(title=paper["title"], abstract=paper["abstract"], api_key=api_key) | |
| except Exception as exc: | |
| return {"status": "extraction_error", "error": str(exc), "traceback": traceback.format_exc()} | |
| if extraction: | |
| db.delete_concepts_for_paper(canonical_id) | |
| db.delete_reductions_for_paper(canonical_id) | |
| for concept in extraction.get("concepts", []): | |
| cid = db.insert_concept(canonical_id, concept) | |
| match = engine.match_concept(concept) | |
| db.insert_reduction(cid, canonical_id, engine._result_to_dict(match)) | |
| db.update_status(canonical_id, "matched") | |
| return {"status": "matched", "concepts": len(extraction.get("concepts", []))} | |
| return {"status": "extraction_failed", "debug": "extract_paper returned None — check Space logs for details"} | |
| if retroactive: | |
| papers = db.get_papers_with_unknown(limit=30) | |
| for p in papers: | |
| pid = p["arxiv_id"] | |
| db.delete_reductions_for_paper(pid) | |
| return {"status": "retroactive_queued", "papers": len(papers)} | |
| return {"status": "no_action", "message": "Specify arxiv_id or retroactive=true"} | |
| except HTTPException: | |
| raise | |
| except Exception as e: | |
| raise HTTPException(status_code=500, detail=str(e)) | |
| # --- Helpers --- | |
| def _render_discourse_html(desc: list[dict], asc: list[dict]) -> str: | |
| """Render the discourse page: translation trace, not quality leaderboard.""" | |
| def _paper_row(p: dict) -> str: | |
| rate = p["_reduction_rate"] | |
| pct = int(rate * 100) | |
| bar_color = ( | |
| "#4ecdc4" if pct >= 67 else "#ffe66d" if pct >= 34 else "#ff6b35" | |
| ) | |
| title = (p.get("title") or "Untitled")[:80] | |
| aid = p.get("arxiv_id", "?") | |
| return f"""<tr> | |
| <td style="text-align:right;padding-right:1rem;color:#888;font-size:0.8rem">{pct}%</td> | |
| <td style="width:120px"><div style="height:6px;background:#2a2a2a;border-radius:3px;overflow:hidden"><div style="height:100%;width:{pct}%;background:{bar_color};border-radius:3px"></div></div></td> | |
| <td><a style="color:var(--fg);text-decoration:none" href="/paper/{aid}">{title}</a></td> | |
| <td style="color:#888;font-size:0.75rem;text-align:right">{p['_mapped']}/{p['_total']} mapped</td> | |
| </tr>""" | |
| desc_rows = "\n".join(_paper_row(p) for p in desc if p["_total"] > 0) | |
| asc_rows = "\n".join(_paper_row(p) for p in asc if p["_total"] > 0) | |
| return f"""<!doctype html> | |
| <html lang="en"> | |
| <head> | |
| <meta charset="utf-8"><meta name="viewport" content="width=device-width,initial-scale=1"> | |
| <title>Translation Trace — Différance Engine</title> | |
| <style> | |
| :root {{ | |
| --bg: #0d0d0d; --fg: #e0e0e0; --muted: #888; --accent: #ff6b35; | |
| --math: #4ecdc4; --border: #2a2a2a; --card-bg: #141414; | |
| }} | |
| * {{ box-sizing: border-box; margin: 0; padding: 0; }} | |
| body {{ | |
| font-family: "IBM Plex Mono","SF Mono","Cascadia Code",monospace; | |
| background: var(--bg); color: var(--fg); line-height: 1.6; | |
| max-width: 900px; margin: 0 auto; padding: 2rem 1.5rem; | |
| }} | |
| h1 {{ font-size: 1.5rem; font-weight: 700; }} | |
| h1 span {{ color: var(--accent); }} | |
| h2 {{ font-size: 1rem; margin: 1.5rem 0 0.75rem; color: var(--muted); }} | |
| .back {{ color: var(--math); font-size: 0.85rem; text-decoration: none; }} | |
| .back:hover {{ text-decoration: underline; }} | |
| table {{ width: 100%; border-collapse: collapse; }} | |
| td {{ padding: 0.5rem 0; border-bottom: 1px solid #1a1a1a; }} | |
| tr:hover td {{ background: #111; }} | |
| .disclaimer {{ | |
| background: #141414; border: 1px solid #2a2a2a; border-radius: 6px; | |
| padding: 1rem; margin: 1.5rem 0; font-size: 0.85rem; color: var(--muted); | |
| }} | |
| .disclaimer strong {{ color: var(--fg); }} | |
| .section-label {{ | |
| display: inline-block; padding: 0.15rem 0.6rem; border-radius: 3px; | |
| font-size: 0.7rem; font-weight: 600; text-transform: uppercase; letter-spacing: 0.05em; | |
| margin-bottom: 0.75rem; | |
| }} | |
| .section-desc {{ color: var(--math); }} | |
| .section-asc {{ color: var(--accent); }} | |
| </style> | |
| </head> | |
| <body> | |
| <p><a class="back" href="/">← The Différance Engine</a></p> | |
| <h1>Translation <span>Trace</span></h1> | |
| <div class="disclaimer"> | |
| <strong>This is not a quality leaderboard.</strong> It describes translation | |
| efficiency — what fraction of a paper's extracted concepts successfully | |
| mapped to a declared formal vocabulary (taxonomy v1.0). A high rate means | |
| the paper's vocabulary overlapped heavily with ours. A low rate means it | |
| didn't. The low-rate papers are where you should look — for genuine novelty, | |
| for taxonomy gaps, or for bullshit. The tool doesn't adjudicate. It shows | |
| the trace and lets you decide. | |
| </div> | |
| <h2><span class="section-label section-desc">↓ Descending</span> Most Translated</h2> | |
| <p style="color:var(--muted);font-size:0.85rem;margin-bottom:1rem"> | |
| These papers' vocabulary aligned heavily with our formal taxonomy. | |
| An invitation to understand the mathematical primitives they rest on — | |
| and to branch out from them. | |
| </p> | |
| <table>{desc_rows}</table> | |
| <h2><span class="section-label section-asc">↑ Ascending</span> Least Translated</h2> | |
| <p style="color:var(--muted);font-size:0.85rem;margin-bottom:1rem"> | |
| These papers resisted translation. The gaps might be genuine novelty, | |
| taxonomy inadequacy, or underspecified language. The unknowns are the | |
| interesting part — not an error condition, but an invitation to look closer. | |
| </p> | |
| <table>{asc_rows}</table> | |
| <p style="margin-top:2rem;font-size:0.75rem;color:var(--muted);text-align:center"> | |
| Translation trace against the Différance Engine taxonomy v1.0.0 | |
| (<a style="color:var(--accent)" href="/api/index/analogs">canonical analogs</a> · | |
| <a style="color:var(--accent)" href="/api/index/moves">deconstructive moves</a>) | |
| </p> | |
| </body> | |
| </html>""" | |
| def _render_index_html(stats: dict, papers: list[dict]) -> str: | |
| """Render the main feed page with live DB data. | |
| Uses the same CSS as the Astro build so styling stays consistent. | |
| """ | |
| total = stats.get("total_papers", 0) | |
| matched = stats.get("by_status", {}).get("matched", 0) | |
| reduction_rate = stats.get("reduction_rate", 0) | |
| total_reductions = stats.get("total_reductions", 0) | |
| reduction_counts = stats.get("reductions_by_type", {}) | |
| paper_cards = "" | |
| if papers: | |
| for paper in papers: | |
| aid = paper.get("arxiv_id", "?") | |
| title = paper.get("title", "Untitled") | |
| updated = (paper.get("updated") or "")[:10] | |
| categories = paper.get("categories", []) | |
| if isinstance(categories, str): | |
| try: | |
| import json; categories = json.loads(categories) | |
| except Exception: | |
| categories = [] | |
| cat_tags = "".join( | |
| f'<span class="badge" style="background:var(--border);color:var(--muted)">{c}</span>' | |
| for c in categories[:2] | |
| ) | |
| # Fetch reductions for this paper | |
| reductions_html = "" | |
| try: | |
| from pipeline.db import get_db | |
| db = get_db() | |
| db.connect() | |
| reds = db.get_reductions_for_paper(aid) | |
| db.close() | |
| if reds: | |
| red_items = "" | |
| for r in reds: | |
| rtype = r.get("result_type", "unknown") | |
| display = r.get("display", "") | |
| delta = r.get("genuine_delta", "") | |
| # Parse display: ~~Term~~ ≡/≈/→ Reduction | |
| red_items += ( | |
| f'<div class="reduction-item">' | |
| f'<span class="badge badge-{rtype}">{rtype}</span> ' | |
| f'{_format_reduction_html(display, delta)}' | |
| f'</div>' | |
| ) | |
| reductions_html = f'<div class="reduction-list">{red_items}</div>' | |
| except Exception: | |
| pass | |
| paper_cards += f""" | |
| <article class="paper-card"> | |
| <div class="paper-meta"> | |
| <span>{updated}</span> | |
| <a href="https://arxiv.org/abs/{aid}" target="_blank" rel="noopener">{aid}</a> | |
| {cat_tags} | |
| </div> | |
| <h2 class="paper-title"> | |
| <a href="/paper/{aid}">{title}</a> | |
| </h2> | |
| {reductions_html} | |
| </article>""" | |
| paper_list_html = ( | |
| f'<div class="paper-list">{paper_cards}</div>' if papers | |
| else f"""<div class="empty-state"> | |
| <h2>No papers deconstructed yet</h2> | |
| <p>Trigger a pipeline run to populate the feed:</p> | |
| <p style="margin-top:1rem"> | |
| <code>POST /api/trigger?arxiv_id=<id></code> | |
| </p> | |
| <p style="margin-top:0.5rem;font-size:0.85rem"> | |
| Set <code>HF_API_KEY</code> as a Space secret for LLM extraction. | |
| </p> | |
| </div>""" | |
| ) | |
| return f"""<!doctype html> | |
| <html lang="en"> | |
| <head> | |
| <meta charset="utf-8" /> | |
| <meta name="viewport" content="width=device-width, initial-scale=1" /> | |
| <title>The Différance Engine — Under Erasure</title> | |
| <meta name="description" content="Productive deconstruction — what creative destruction looks like under erasure." /> | |
| <style> | |
| :root {{ | |
| --bg: #0d0d0d; --fg: #e0e0e0; --muted: #888; --accent: #ff6b35; | |
| --strike: #c44; --math: #4ecdc4; --border: #2a2a2a; --card-bg: #141414; | |
| --identity: #4ecdc4; --compositional: #ffe66d; --analogy: #a29bfe; | |
| --unknown: #636e72; --confused: #e17055; | |
| }} | |
| * {{ box-sizing: border-box; margin: 0; padding: 0; }} | |
| body {{ | |
| font-family: "IBM Plex Mono", "SF Mono", "Cascadia Code", monospace; | |
| background: var(--bg); color: var(--fg); line-height: 1.6; | |
| max-width: 900px; margin: 0 auto; padding: 2rem 1.5rem; | |
| }} | |
| header {{ border-bottom: 2px solid var(--accent); padding-bottom: 1.5rem; margin-bottom: 2rem; }} | |
| h1 {{ font-size: 2rem; font-weight: 700; letter-spacing: -0.02em; }} | |
| h1 span {{ color: var(--accent); }} | |
| .tagline {{ color: var(--muted); font-style: italic; margin-top: 0.25rem; font-size: 0.95rem; }} | |
| .stats-bar {{ display: flex; gap: 2rem; margin-top: 1rem; font-size: 0.85rem; color: var(--muted); flex-wrap: wrap; }} | |
| .stats-bar strong {{ color: var(--fg); }} | |
| .nav {{ display: flex; gap: 1.5rem; margin-top: 1rem; font-size: 0.9rem; }} | |
| .nav a {{ color: var(--accent); text-decoration: none; }} | |
| .nav a:hover {{ text-decoration: underline; }} | |
| .paper-list {{ display: flex; flex-direction: column; gap: 1.5rem; }} | |
| .paper-card {{ | |
| background: var(--card-bg); border: 1px solid var(--border); | |
| border-radius: 6px; padding: 1.25rem; transition: border-color 0.2s; | |
| }} | |
| .paper-card:hover {{ border-color: var(--accent); }} | |
| .paper-meta {{ font-size: 0.8rem; color: var(--muted); margin-bottom: 0.5rem; display: flex; gap: 1rem; flex-wrap: wrap; }} | |
| .paper-meta a {{ color: var(--accent); text-decoration: none; }} | |
| .paper-title {{ font-size: 1.1rem; font-weight: 600; margin-bottom: 0.75rem; }} | |
| .paper-title a {{ color: var(--fg); text-decoration: none; }} | |
| .paper-title a:hover {{ color: var(--accent); }} | |
| .reduction-list {{ display: flex; flex-direction: column; gap: 0.5rem; margin-top: 0.75rem; }} | |
| .reduction-item {{ font-size: 0.9rem; }} | |
| .reduction-item .ai-term {{ text-decoration: line-through; color: var(--strike); margin-right: 0.5rem; }} | |
| .reduction-item .math-term {{ color: var(--math); font-weight: 600; }} | |
| .reduction-item .connector {{ color: var(--muted); }} | |
| .reduction-item .delta {{ color: var(--accent); font-size: 0.8rem; }} | |
| .badge {{ | |
| display: inline-block; padding: 0.1rem 0.5rem; border-radius: 3px; | |
| font-size: 0.7rem; font-weight: 600; text-transform: uppercase; letter-spacing: 0.05em; | |
| }} | |
| .badge-identity {{ background: #1a3a3a; color: var(--identity); }} | |
| .badge-compositional {{ background: #3a3a1a; color: var(--compositional); }} | |
| .badge-analogy {{ background: #2a2a3a; color: var(--analogy); }} | |
| .badge-unknown {{ background: #1a1a1a; color: var(--unknown); }} | |
| .badge-confused {{ background: #3a1a1a; color: var(--confused); }} | |
| .expand-toggle {{ display: inline-block; margin-left: 0.3rem; }} | |
| .expand-toggle:hover {{ color: #fff !important; }} | |
| .expansion-chain {{ display: block; padding: 0.25rem 0 0.25rem 0; word-break: break-all; }} | |
| footer {{ margin-top: 3rem; padding-top: 1.5rem; border-top: 1px solid var(--border); font-size: 0.8rem; color: var(--muted); text-align: center; }} | |
| footer a {{ color: var(--accent); }} | |
| .empty-state {{ text-align: center; padding: 4rem 1rem; color: var(--muted); }} | |
| .empty-state h2 {{ font-size: 1.5rem; margin-bottom: 0.5rem; color: var(--fg); }} | |
| .empty-state code {{ background: var(--card-bg); padding: 0.2rem 0.5rem; border-radius: 3px; font-size: 0.85rem; }} | |
| </style> | |
| </head> | |
| <body> | |
| <header> | |
| <h1>The Différ<span>a</span>nce Engine</h1> | |
| <p class="tagline">Productive deconstruction — what creative destruction looks like under erasure.</p> | |
| <div class="stats-bar"> | |
| <span><strong>{total}</strong> papers ingested</span> | |
| <span><strong>{matched}</strong> deconstructed</span> | |
| <span><strong>{int(reduction_rate * 100)}%</strong> reduction rate</span> | |
| <span><strong>{total_reductions}</strong> reductions</span> | |
| </div> | |
| <nav class="nav"> | |
| <a href="/">Feed</a> | |
| <a href="/lookup">Lookup</a> | |
| <a href="/discourse">Translation Trace</a> | |
| <a href="https://arxiv.org/list/cs.LG/recent" target="_blank" rel="noopener">arXiv cs.LG →</a> | |
| </nav> | |
| </header> | |
| <main> | |
| {paper_list_html} | |
| </main> | |
| <footer> | |
| <p>The Différance Engine — <a href="https://under-erasure.graziul.io">under-erasure.graziul.io</a></p> | |
| <p style="margin-top:0.25rem"> | |
| Powered by a formalism KB of 59 canonical mathematical operations. | |
| All reductions are <em>sous rature</em>: the AI term is crossed out; | |
| the math beneath is what the paper actually computes. | |
| </p> | |
| </footer> | |
| </body> | |
| </html>""" | |
| def _format_reduction_html(display: str, delta: str) -> str: | |
| """Parse a sous-rature display string like '~~GLIGEN~~ ≡ Diffusion Process ∘ diffuse' | |
| into formatted HTML spans. Handles expandable canonical decomposition chains.""" | |
| import re | |
| import html as _html | |
| # Check for embedded expansion data (NUL-delimited sentinels from match.py) | |
| expand_html = "" | |
| if "\x00EXPAND\x00" in display: | |
| parts = display.split("\x00EXPAND\x00", 1) | |
| display = parts[0] | |
| expanded = parts[1].split("\x00/EXPAND\x00", 1)[0] if "\x00/EXPAND\x00" in parts[1] else "" | |
| if expanded: | |
| expanded = _html.escape(expanded) | |
| uid = abs(hash(expanded)) % 100000 | |
| expand_html = ( | |
| f' <span class="expand-toggle" onclick="' | |
| f'var e=document.getElementById(\'exp-{uid}\');' | |
| f'var t=this;' | |
| f'if(e.style.display==\'none\'){{e.style.display=\'block\';t.textContent=\'[−]\'}}' | |
| f'else{{e.style.display=\'none\';t.textContent=\'[+]\'}}' | |
| f'" style="cursor:pointer;color:#ffe66d;font-size:0.8rem;user-select:none">[+]</span>' | |
| f'<span id="exp-{uid}" class="expansion-chain" style="display:none;font-size:0.8rem;color:#ffe66d;margin-left:1.5rem">' | |
| f'└ {expanded}</span>' | |
| ) | |
| # Match ~~struck term~~ then connector (=/~/->) then math term | |
| m = re.match(r"~~(.+?)~~\s*(≡|≈|→)\s*(.+?)(?:\s*\(Δ:\s*(.+?)\))?$", display) | |
| if m: | |
| struck = m.group(1) | |
| connector = m.group(2) | |
| math_term = m.group(3) | |
| inner_delta = m.group(4) or delta | |
| delta_html = f' <span class="delta">(Δ: {inner_delta})</span>' if inner_delta else "" | |
| # Link recognized formalism names to canonical pages | |
| math_term = _link_formalism_names(math_term) | |
| return ( | |
| f'<span class="ai-term">{struck}</span>' | |
| f'<span class="connector">{connector}</span> ' | |
| f'<span class="math-term">{math_term}</span>' | |
| f'{delta_html}' | |
| f'{expand_html}' | |
| ) | |
| # Fallback: just escape and display (still check for expansion) | |
| return _html.escape(display) + expand_html | |
| # Cache for formalism name -> ID mapping (populated lazily) | |
| _formalism_name_to_id: dict[str, str] | None = None | |
| def _get_formalism_name_map() -> dict[str, str]: | |
| """Return a mapping from lowercased formalism names to their KB IDs. | |
| Used for linking formalism names in reductions to canonical pages.""" | |
| global _formalism_name_to_id | |
| if _formalism_name_to_id is not None: | |
| return _formalism_name_to_id | |
| try: | |
| import yaml | |
| from pathlib import Path | |
| kb_path = Path(__file__).resolve().parent / "pipeline" / "kb" / "formalisms.yaml" | |
| with open(kb_path) as f: | |
| data = yaml.safe_load(f) | |
| _formalism_name_to_id = {} | |
| for entry in data.get("formalisms", []): | |
| fid = entry.get("id", "") | |
| name = (entry.get("name", "") or "").lower() | |
| if name and fid: | |
| _formalism_name_to_id[name] = fid | |
| return _formalism_name_to_id | |
| except Exception: | |
| return {} | |
| def _link_formalism_names(text: str) -> str: | |
| """Wrap known formalism names in links to their canonical pages. | |
| Returns HTML with <a> tags for recognized formalisms.""" | |
| import re as _re | |
| fm_map = _get_formalism_name_map() | |
| if not fm_map: | |
| return text | |
| # Sort by length descending to match longest names first | |
| for name in sorted(fm_map.keys(), key=len, reverse=True): | |
| fid = fm_map[name] | |
| # Case-insensitive replacement, but only whole-word-ish | |
| pattern = _re.compile(_re.escape(name), _re.IGNORECASE) | |
| text = pattern.sub( | |
| f'<a style="color:#4ecdc4;text-decoration:none" ' | |
| f'href="/canonical/{fid}" title="Canonical formalism page">{name}</a>', | |
| text | |
| ) | |
| return text | |
| def _fallback_html(msg: str = "") -> str: | |
| message = msg or ( | |
| "The Différance Engine is live but no papers have been deconstructed yet. " | |
| "Set <code>HF_API_KEY</code> as a Space secret and trigger a pipeline run." | |
| ) | |
| return f"""<!doctype html> | |
| <html lang="en"><head><meta charset="utf-8"><meta name="viewport" content="width=device-width,initial-scale=1"> | |
| <title>The Différance Engine</title> | |
| <style> | |
| body {{ font-family:"IBM Plex Mono",monospace; background:#0d0d0d; color:#e0e0e0; | |
| max-width:700px; margin:4rem auto; padding:2rem; text-align:center; }} | |
| h1 {{ color:#ff6b35; }} a {{ color:#4ecdc4; }} | |
| code {{ background:#141414; padding:0.2rem 0.5rem; border-radius:3px; }} | |
| </style></head><body> | |
| <h1>The Différance Engine</h1> | |
| <p>Productive deconstruction — what creative destruction looks like under erasure.</p> | |
| <p style="margin-top:2rem;color:#888">{message}</p> | |
| <p style="margin-top:2rem"> | |
| <a href="/api/stats">Stats</a> · <a href="/api/models">Models</a> · <a href="/lookup">Lookup</a> | |
| </p> | |
| </body></html>""" | |
| def _render_paper_html(paper: dict, reductions: list[dict]) -> str: | |
| title = paper.get("title", "Untitled") | |
| arxiv_id = paper.get("arxiv_id", "?") | |
| abstract = paper.get("abstract", "") | |
| # --- Reductions --- | |
| reds = "" | |
| for r in reductions: | |
| disp = r.get("display", f"~~{r.get('concept_name','')}~~ → {r.get('reduction','')}") | |
| delta = r.get("genuine_delta", "") | |
| rtype = r.get("result_type", "unknown") | |
| analog = r.get("canonical_analog", "") | |
| reds += f"""<div style="background:#141414;border:1px solid #2a2a2a;border-radius:6px;padding:1rem;margin:0.75rem 0"> | |
| <p style="font-size:1.1rem;font-weight:600">{_format_reduction_html(disp, delta)}</p> | |
| <p style="font-size:0.8rem;color:#888"> | |
| <span style="display:inline-block;padding:0.1rem 0.5rem;border-radius:3px;font-size:0.7rem;font-weight:600;text-transform:uppercase;background:{'#1a3a3a' if rtype=='identity' else '#3a3a1a' if rtype=='compositional' else '#2a2a3a' if rtype=='analogy' else '#1a1a1a'};color:{'#4ecdc4' if rtype=='identity' else '#ffe66d' if rtype=='compositional' else '#a29bfe' if rtype=='analogy' else '#636e72'}">{rtype}</span> | |
| micro: {r.get('micro','?')} | meso: {r.get('meso','?')} | macro: {r.get('macro','?')} | |
| </p> | |
| </div>""" | |
| # --- Cross-references: canonical analogs shared with other papers --- | |
| from pipeline.db import get_db | |
| xref_html = "" | |
| try: | |
| db = get_db(); db.connect() | |
| analogs_seen = set() | |
| for r in reductions: | |
| analog = r.get("canonical_analog", "") | |
| if analog and analog not in analogs_seen: | |
| analogs_seen.add(analog) | |
| related = db.get_papers_by_analog(analog, limit=8) | |
| related = [p for p in related if p["arxiv_id"] != arxiv_id] | |
| if related: | |
| short_analog = analog[:80] + ("…" if len(analog) > 80 else "") | |
| items = "".join( | |
| f'<li style="margin:0.2rem 0;font-size:0.85rem"><a style="color:#4ecdc4" href="/paper/{p["arxiv_id"]}">{p["title"]}</a> <span style="color:#888;font-size:0.75rem">({p["arxiv_id"]})</span></li>' | |
| for p in related[:5] | |
| ) | |
| xref_html += f"""<div style="background:#141414;border:1px solid #2a2a2a;border-radius:6px;padding:0.75rem;margin:0.5rem 0"> | |
| <p style="font-size:0.8rem;color:#888">Also reduces to <span style="color:#ffe66d">{short_analog}</span>:</p> | |
| <ul style="list-style:none;padding:0">{items}</ul> | |
| </div>""" | |
| db.close() | |
| except Exception: | |
| pass | |
| return f"""<!doctype html><html lang="en"><head><meta charset="utf-8"> | |
| <meta name="viewport" content="width=device-width,initial-scale=1"> | |
| <title>{title} — Différance Engine</title> | |
| <style> | |
| body {{ font-family:"IBM Plex Mono",monospace; background:#0d0d0d; color:#e0e0e0; | |
| max-width:900px; margin:2rem auto; padding:1.5rem; line-height:1.6; }} | |
| h1 {{ font-size:1.3rem; }} h1 a {{ color:#ff6b35; text-decoration:none; }} | |
| .abstract {{ background:#141414; border:1px solid #2a2a2a; border-radius:6px; padding:1rem; margin:1rem 0; font-size:0.9rem; }} | |
| .back {{ color:#4ecdc4; font-size:0.85rem; }} | |
| .ai-term {{ text-decoration:line-through; color:#c44; }} | |
| .math-term {{ color:#4ecdc4; font-weight:600; }} | |
| .connector {{ color:#888; }} | |
| .delta {{ color:#ff6b35; font-size:0.85rem; }} | |
| .expand-toggle {{ display:inline-block; margin-left:0.3rem; }} | |
| .expand-toggle:hover {{ color:#fff !important; }} | |
| .expansion-chain {{ display:block; padding:0.25rem 0; word-break:break-all; }} | |
| </style></head><body> | |
| <p><a class="back" href="/">← The Différance Engine</a></p> | |
| <h1><a href="https://arxiv.org/abs/{arxiv_id}">{title}</a></h1> | |
| <p style="color:#888;font-size:0.8rem">{arxiv_id} · {paper.get('updated','?')[:10]}</p> | |
| <details class="abstract"><summary>Abstract</summary>{abstract}</details> | |
| <h2 style="font-size:1rem;margin-top:1.5rem;color:#888">Reductions</h2> | |
| {reds} | |
| {f'<h2 style="font-size:1rem;margin-top:1.5rem;color:#888">Cross-References</h2>{xref_html}' if xref_html else ''} | |
| </body></html>""" | |
| def _render_canonical_html(fm, citing_papers, canonical_arxiv, canonical_paper, canonical_reductions) -> str: | |
| """Render a page for a canonical formalism, optionally with its orienting paper.""" | |
| import html as _html | |
| sig = fm.signature | |
| sig_str = f"{sig.operation or '?'}({sig.domain or '?'} -> {sig.codomain or '?'})" | |
| if sig.objective_family: | |
| sig_str += f" objective={sig.objective_family}" | |
| # Canonical paper section | |
| canonical_html = "" | |
| if canonical_arxiv: | |
| if canonical_paper: | |
| cp_title = _html.escape(canonical_paper.get("title", "Untitled")) | |
| cp_id = canonical_paper["arxiv_id"] | |
| reds_html = "" | |
| for r in canonical_reductions: | |
| disp = r.get("display", "") | |
| delta = r.get("genuine_delta", "") | |
| rtype = r.get("result_type", "unknown") | |
| reds_html += ( | |
| f'<div style="background:#111;border:1px solid #2a2a2a;' | |
| f'border-radius:4px;padding:0.75rem;margin:0.5rem 0;font-size:0.9rem">' | |
| f'{_format_reduction_html(disp, delta)}' | |
| f'<span style="display:inline-block;margin-left:0.5rem;padding:0.1rem 0.4rem;' | |
| f'border-radius:2px;font-size:0.65rem;background:#1a3a1a;color:#ffe66d;' | |
| f'vertical-align:middle">{rtype}</span></div>' | |
| ) | |
| canonical_html = f""" | |
| <div style="background:#141414;border:1px solid #ffe66d;border-radius:6px;padding:1rem;margin:1rem 0"> | |
| <p style="color:#ffe66d;font-weight:600;margin-bottom:0.5rem">Canonical Paper (Orienting Document)</p> | |
| <p><a style="color:#4ecdc4;font-size:1.1rem" href="/paper/{cp_id}">{cp_title}</a></p> | |
| <p style="color:#888;font-size:0.8rem">{cp_id}{' — ' + str(canonical_paper.get('citation_count','')) + ' citations' if canonical_paper.get('citation_count') else ''}</p> | |
| <details style="margin-top:0.5rem"><summary style="color:#888;font-size:0.85rem;cursor:pointer">Deconstruction ({len(canonical_reductions)} concepts)</summary> | |
| {reds_html} | |
| </details> | |
| </div>""" | |
| else: | |
| canonical_html = f""" | |
| <div style="background:#141414;border:1px dashed #ffe66d;border-radius:6px;padding:1rem;margin:1rem 0"> | |
| <p style="color:#ffe66d;font-weight:600;margin-bottom:0.5rem">Canonical Paper: <span style="font-family:monospace">{canonical_arxiv}</span></p> | |
| <p style="color:#888;font-size:0.85rem">Not yet ingested. Trigger deconstruction to see the raw formalism beneath the vocabulary.</p> | |
| <form method="POST" action="/api/trigger?arxiv_id={canonical_arxiv}" style="margin-top:0.75rem" onsubmit="var b=this.querySelector('button');b.disabled=true;b.textContent='Deconstructing...';return true"> | |
| <button style="background:#ffe66d;color:#0d0d0d;border:none;padding:0.4rem 1rem;border-radius:4px;cursor:pointer;font-family:inherit;font-weight:600">Deconstruct Canonical Paper</button> | |
| </form> | |
| </div>""" | |
| # Citing papers table | |
| citing_rows = "" | |
| if citing_papers: | |
| for p in citing_papers[:20]: | |
| aid = p.get("arxiv_id", "?") | |
| title = (p.get("title") or "Untitled")[:80] | |
| citing_rows += ( | |
| f'<tr><td><a style="color:var(--fg);text-decoration:none" ' | |
| f'href="/paper/{aid}">{_html.escape(title)}</a></td>' | |
| f'<td style="color:#888;font-size:0.75rem;text-align:right">{aid}</td></tr>' | |
| ) | |
| citing_section = ( | |
| f'<h2 style="font-size:1rem;margin-top:1.5rem;color:#888">' | |
| f'Papers Reducing to This Formalism ({len(citing_papers)})</h2>' | |
| f'<table style="width:100%;border-collapse:collapse">{citing_rows}</table>' | |
| ) if citing_papers else "" | |
| return f"""<!doctype html><html lang="en"><head><meta charset="utf-8"> | |
| <meta name="viewport" content="width=device-width,initial-scale=1"> | |
| <title>{fm.name} — Canonical Formalism — Différance Engine</title> | |
| <style> | |
| body {{ font-family:"IBM Plex Mono",monospace; background:#0d0d0d; color:#e0e0e0; | |
| max-width:900px; margin:2rem auto; padding:1.5rem; line-height:1.6; }} | |
| h1 {{ font-size:1.3rem; }} .back {{ color:#4ecdc4; font-size:0.85rem; text-decoration:none; }} | |
| .back:hover {{ text-decoration:underline; }} | |
| .prop-table {{ width:100%; border-collapse:collapse; font-size:0.9rem; }} | |
| .prop-table td {{ padding:0.35rem 0.5rem; border-bottom:1px solid #1a1a1a; }} | |
| .prop-table td:first-child {{ color:#888; width:160px; }} | |
| .prop-table td:last-child {{ color:#e0e0e0; }} | |
| .ai-term {{ text-decoration:line-through; color:#c44; }} | |
| .math-term {{ color:#4ecdc4; font-weight:600; }} | |
| .connector {{ color:#888; }} | |
| .delta {{ color:#ff6b35; font-size:0.85rem; }} | |
| .expand-toggle {{ display:inline-block; margin-left:0.3rem; }} | |
| .expand-toggle:hover {{ color:#fff !important; }} | |
| .expansion-chain {{ display:block; padding:0.25rem 0; word-break:break-all; }} | |
| table {{ width:100%; border-collapse:collapse; }} | |
| td {{ padding:0.35rem 0; border-bottom:1px solid #1a1a1a; }} | |
| tr:hover td {{ background:#111; }} | |
| </style></head><body> | |
| <p><a class="back" href="/">← The Différance Engine</a></p> | |
| <h1 style="color:#ffe66d">{_html.escape(fm.name)}</h1> | |
| <p style="color:#888;font-size:0.85rem"> | |
| {fm.year or '?'} · {_html.escape(fm.origin or '')} · {fm.status} | |
| </p> | |
| <table class="prop-table" style="margin-top:1rem"> | |
| <tr><td>Operation</td><td>{sig.operation or '?'}</td></tr> | |
| <tr><td>Domain</td><td>{sig.domain or '?'}</td></tr> | |
| <tr><td>Codomain</td><td>{sig.codomain or '?'}</td></tr> | |
| <tr><td>Objective</td><td>{sig.objective_family or '?'}</td></tr> | |
| <tr><td>Meso-type</td><td>{fm.meso_type or 'none'}</td></tr> | |
| <tr><td>Macro-type</td><td>{fm.macro_type or 'none'}</td></tr> | |
| <tr><td>Reference</td><td style="font-size:0.85rem">{_html.escape(fm.canonical_reference or '')}</td></tr> | |
| </table> | |
| {canonical_html} | |
| {citing_section} | |
| <p style="margin-top:2rem;font-size:0.75rem;color:#888;text-align:center"> | |
| Canonical formalism in the Différance Engine taxonomy. · | |
| <a style="color:#ff6b35" href="/api/canonical/{fm.id}">JSON</a> | |
| </p> | |
| </body></html>""" | |