"""Knowledge Graph — the whole database as one graph, always current. agent/kg.py reads every property row (rows in the review queue included, and marked as such) and every harvested figure and builds the graph; this page shows it in an explorer (agent/kg_explorer.html) and offers the same data for download. Nothing is cached beyond a fingerprint check: when the tables change (a cycle ingests papers, a reviewer promotes a row, routes are added), the next view or the scheduler's refresh job rebuilds the graph. The explorer loads its data from /app/static/kg/ when Streamlit's static file serving is on and the files could be written (the Dockerfile makes that directory writable): the graph, the per-value detail files (quoted sentence, flag reason, date, model) it fetches when a value is opened, and the figure pictures. Otherwise the data is embedded in the page, so the page works either way; only the figure pictures need the static files. """ import os import streamlit as st from agent import kg, ui_theme as T from agent.ui_common import status_banner EXPLORER_HEIGHT = 880 INLINE_DETAILS_MAX = 6_000_000 # bytes of compressed value details embedded when there are no static files # How often the open page asks whether the graph was rebuilt (kg-meta.json). try: POLL_SECONDS = int(os.environ.get("AGENT_KG_POLL_SECONDS", "") or 60) except ValueError: POLL_SECONDS = 60 # The explorer is a three-pane workbench: give this page more width than the # 1360 px the other pages use. st.html("""""") T.page_header("Knowledge Graph", "Every material, property, source document, figure and processing route " "in the database as one graph, the review queue included. It is rebuilt " "from the live tables whenever they change.") status_banner() try: snap = kg.current() except Exception as exc: # the database answered the banner but the build failed st.error(f"The knowledge graph could not be built: {exc}") st.stop() c = snap.counts if not c.get("rows"): T.empty_state("🕸️", "The database is empty so far", "The graph appears here as soon as the agent has ingested its " "first document, and grows with every cycle.") st.stop() @st.fragment(run_every=POLL_SECONDS) def _tiles() -> None: """The counts, re-read on a timer so they follow the database like the explorer below does (which swaps in a new graph on its own).""" try: now = kg.current() except Exception: now = snap k = now.counts n_proc = int(k.get("rows_with_process") or 0) T.kpi_row([ {"label": "Materials", "value": k["materials"], "help": "Distinct materials across Polymers, Fibers and Composites"}, {"label": "Properties", "value": k["properties"], "help": "Property names after merging case, spelling and listed synonyms"}, {"label": "Source documents", "value": k["sources"], "delta": (f"{k['figures']:,} figures · {k.get('figures_with_values', 0):,} with values" if k.get("figures") else None), "delta_kind": "flat", "help": "Figures harvested from the documents. Values read off a figure, and " "values whose sentence cites one, link to it; the picture opens with it."}, {"label": "Measured values", "value": k["rows"], "delta": f"{k['verified']:,} verified · {k['flagged']:,} in review", "delta_kind": "flat", "help": "Every property row in the database is in the graph. Rows in the review " "queue are marked with the reason they were flagged; the switch above " "the graph shows all values, the verified ones, or only the queue."}, {"label": "With processing route", "value": n_proc, "delta": (f"{100 * n_proc / k['rows']:.0f}% of values" if k["rows"] else None), "delta_kind": "flat", "help": "Values that carry the process type and conditions of their " "specimen (extracted since prompt 2.1)."}, {"label": "Rebuilt", "value": now.built_at.strftime("%H:%M UTC"), "delta": now.built_at.strftime("%d %b %Y") + f" · {now.seconds:.1f} s", "delta_kind": "flat", "help": "The graph is rebuilt from the live tables whenever their row " "counts, verified counts or newest extraction time change: right " "after each cycle, and on a timer in between."}, ]) _tiles() def _static_served() -> bool: try: return bool(snap.static) and bool(st.get_option("server.enableStaticServing")) except Exception: return False def _embed(html: str, height: int) -> None: """st.iframe where it exists (Streamlit >= 1.50); components.v1.html before.""" if hasattr(st, "iframe"): try: st.iframe(html, height=height) return except Exception: pass import streamlit.components.v1 as components components.html(html, height=height, scrolling=True) served = _static_served() if served: ver = "?v=" + snap.version page = kg.explorer_html(data_url=snap.static["json"] + ver, gz_url=snap.static["gz"] + ver, meta_url=snap.static["meta"], detail_base=snap.static["base"], fig_base=snap.static["fig"], poll_seconds=POLL_SECONDS, theme="dark" if T.is_dark() else "light", live=True, embedded=True) else: # No static files: embed the graph, and the value details too while they are # small enough to send with the page. Figure pictures are not available then. small = sum(len(b) for b in snap.details) <= INLINE_DETAILS_MAX page = kg.explorer_html(inline_gz=snap.gz, inline_details_gz=snap.details_gz() if small else None, theme="dark" if T.is_dark() else "light", live=True, embedded=True) _embed(page, EXPLORER_HEIGHT) with st.container(border=True): T.card_title("Use the graph elsewhere", "regenerated with every rebuild") stamp = snap.built_at.strftime("%Y%m%d_%H%M") host = os.environ.get("SPACE_HOST", "").strip() base = f"https://{host}" if host else "" left, right = st.columns([3, 2]) with left: if served: url_json, url_gz = base + snap.static["json"], base + snap.static["gz"] st.html( '
The explorer data is published as a file that always ' 'matches the database:
' f'' f'{T.esc(url_json)} ({len(snap.json) / 1e6:.1f} MB) · ' f'' f'compressed ({len(snap.gz) / 1e6:.1f} MB) · ' f'' f'counts and build time
The quoted sentence, flag reason, model and ' f'prompt of each value are in {len(snap.details):,} detail files next to it ' f'(the folder is named in the file, under detail.dir), and the ' f'figure pictures are at {T.esc(base + snap.static["fig"])}' f'<figure id>.png.
') else: st.download_button("Graph data (JSON)", data=snap.json, file_name=f"aim_kg_{stamp}.json", mime="application/json", key="kg_dl_json") st.html('
Materials, properties, sources and values come from the ' 'database as stored. Polymer and fiber families are assigned here by ' 'pattern rules, and property names are merged only for case, spelling and a ' 'short synonym list (agent/kg.py); expect some wrong assignments at the edges. ' 'The map is computed from names and counts, not simulated: the same database ' 'always gives the same picture, and it shifts only where families grow.
') with right: if st.session_state.get("kg_neo4j_for") == snap.version: st.download_button("Download nodes.csv + relationships.csv (zip)", data=snap.neo4j_zip(), file_name=f"aim_kg_neo4j_{stamp}.zip", mime="application/zip", key="kg_dl_neo4j") elif st.button("Prepare the Neo4j export", key="kg_prep_neo4j", help="One Measurement node per value, linked to its material, " "property, source, figure and processing route, with its quoted " "sentence, date and review-queue mark; with a load script."): st.session_state["kg_neo4j_for"] = snap.version st.rerun() st.html('
For Neo4j, Memgraph, NetworkX or pandas: the full graph ' 'with one node per measured value.
')