Spaces:
Running on Zero
Running on Zero
Download knowledge_graph/sjs_knowledge_graph.py from cmuchancel/patent2sysml: direct link, hf CLI and curl.
- Browser
- Download file 15.7 kB
-
https://huggingface.co/spaces/cmuchancel/patent2sysml/resolve/main/knowledge_graph/sjs_knowledge_graph.py
- Command line
-
hf download hf://spaces/cmuchancel/patent2sysml/knowledge_graph/sjs_knowledge_graph.py
-
curl -L -o sjs_knowledge_graph.py https://huggingface.co/spaces/cmuchancel/patent2sysml/resolve/main/knowledge_graph/sjs_knowledge_graph.py
15.7 kB
| """Human-readable SJS-driven patent knowledge graphs. | |
| Import these functions from a notebook. SJS is the semantic contract; the | |
| large OMG JSON schema is not consulted by this module. | |
| """ | |
| from __future__ import annotations | |
| from copy import deepcopy | |
| from datetime import date | |
| import inspect | |
| import json | |
| from pathlib import Path | |
| import re | |
| from typing import Callable | |
| from .schema_knowledge_graph import ROOT, SchemaKnowledgeGraph | |
| SJS_SCHEMA = ROOT / "sjs.kg.schema.json" | |
| def _identifier(text: str, fallback: str) -> str: | |
| value = re.sub(r"[^A-Za-z0-9_]+", "_", text or "").strip("_") or fallback | |
| return "_" + value if value[0].isdigit() else value | |
| class SJSKnowledgeGraph(SchemaKnowledgeGraph): | |
| """Incremental extraction whose definitions and properties are SJS concepts.""" | |
| def __init__(self, text: str, schema_path: str | Path = SJS_SCHEMA): | |
| super().__init__(text, schema_path=schema_path) | |
| def load(cls, path: str | Path, schema_path: str | Path = SJS_SCHEMA): | |
| return super().load(path, schema_path=schema_path) | |
| def _anchors(self, definition: str) -> list[dict]: | |
| result = self.passes.get(definition, {}) | |
| found: dict[str, dict] = {} | |
| for mention in result.get("mentions", []): | |
| if mention.get("property") is None: | |
| found.setdefault(mention["text"], mention) | |
| return sorted(found.values(), key=lambda row: (row["start"], row["end"], row["text"])) | |
| def _edges(self, definition: str, prop: str | None = None) -> list[dict]: | |
| edges = self.passes.get(definition, {}).get("edges", []) | |
| return [edge for edge in edges if prop is None or edge["property"] == prop] | |
| def to_sjs(self, *, system_name: str | None = None) -> dict: | |
| """Build deterministic, reviewable SJS candidate JSON from completed passes. | |
| Only model predictions are represented. Missing SJS fields are omitted, | |
| and all inferred relationships remain explicit candidates with evidence in | |
| the graph checkpoint. | |
| """ | |
| source = self.source_document or {} | |
| title = system_name or source.get("title") or Path(source.get("filename", "PatentSystem")).stem | |
| title = title or "Patent System" | |
| anchors = {name: self._anchors(name) for name in self.definitions} | |
| def candidates(definition: str, *roles: str) -> list[dict]: | |
| rows = list(anchors.get(definition, [])) | |
| wanted = set(roles) | |
| for result in self.passes.values(): | |
| rows.extend(mention for mention in result.get("mentions", []) | |
| if mention.get("property") in wanted) | |
| unique: dict[str, dict] = {} | |
| for row in sorted(rows, key=lambda item: (item["start"], item["end"], item["text"])): | |
| unique.setdefault(row["text"], row) | |
| return list(unique.values()) | |
| subsystem_mentions = anchors.get("Subsystem", []) | |
| if not subsystem_mentions: | |
| subsystem_mentions = [{"text": title, "start": 0, "end": 0, "score": 1.0}] | |
| subsystem_ids = {row["text"]: f"SS-{i:03d}" for i, row in enumerate(subsystem_mentions, 1)} | |
| part_ids = {row["text"]: f"P-{i:03d}" for i, row in enumerate(candidates("Part", "parts"), 1)} | |
| port_ids = {row["text"]: f"PT-{i:03d}" for i, row in enumerate( | |
| candidates("Port", "ports", "port_this", "port_mate"), 1)} | |
| interface_ids = {row["text"]: f"IF-{i:03d}" for i, row in enumerate(candidates("Interface", "interfaces"), 1)} | |
| requirement_ids = {row["text"]: f"REQ-{i:03d}" for i, row in enumerate( | |
| candidates("Requirement", "satisfies_requirements", "requirements"), 1)} | |
| action_ids = {row["text"]: f"ACT-{i:03d}" for i, row in enumerate( | |
| candidates("Action", "allocated_functions", "transitions"), 1)} | |
| part_records = { | |
| text: {"item_no": str(i), "part_id": identifier, "description": text, | |
| "quantity": 1, "source": "patent extraction", "notes": "Model-extracted candidate."} | |
| for i, (text, identifier) in enumerate(part_ids.items(), 1) | |
| } | |
| port_records = { | |
| text: {"port_id": identifier, "name": text, "direction": "inout", | |
| "flow_type": "unspecified", "notes": "Direction requires review."} | |
| for text, identifier in port_ids.items() | |
| } | |
| interface_records = { | |
| text: {"interface_id": identifier, "mating_subsystem": "unresolved", | |
| "interface_type": "unspecified", "notes": text} | |
| for text, identifier in interface_ids.items() | |
| } | |
| assignments = {text: {"parts": [], "ports": [], "interfaces": []} for text in subsystem_ids} | |
| record_maps = {"parts": part_records, "ports": port_records, "interfaces": interface_records} | |
| for prop, records in record_maps.items(): | |
| for edge in self._edges("Subsystem", prop): | |
| if edge["source"] in assignments and edge["target"] in records: | |
| assignments[edge["source"]][prop].append(edge["target"]) | |
| # Preserve unlinked candidates in the first system boundary rather than dropping them. | |
| first_system = next(iter(assignments)) | |
| for prop, records in record_maps.items(): | |
| linked = {name for values in assignments.values() for name in values[prop]} | |
| assignments[first_system][prop].extend(name for name in records if name not in linked) | |
| subsystems = [] | |
| for mention in subsystem_mentions: | |
| text = mention["text"] | |
| allocation = assignments[text] | |
| subsystem = { | |
| "subsystem_id": subsystem_ids[text], | |
| "subsystem_name": text, | |
| "sysml_equivalent": "part def", | |
| "description": f"Model-extracted system candidate: {text}", | |
| "domain": "systems", | |
| "maturity": "candidate", | |
| } | |
| for prop, records in record_maps.items(): | |
| values = [deepcopy(records[name]) for name in allocation[prop]] | |
| if values: | |
| subsystem[prop] = values | |
| satisfied = [requirement_ids[e["target"]] for e in self._edges("Subsystem", "satisfies_requirements") | |
| if e["source"] == text and e["target"] in requirement_ids] | |
| functions = [action_ids[e["target"]] for e in self._edges("Subsystem", "allocated_functions") | |
| if e["source"] == text and e["target"] in action_ids] | |
| if satisfied: | |
| subsystem["satisfies_requirements"] = list(dict.fromkeys(satisfied)) | |
| if functions: | |
| subsystem["allocated_functions"] = list(dict.fromkeys(functions)) | |
| subsystems.append(subsystem) | |
| model: dict = { | |
| "$schema": "sjs/1.0", | |
| "model_meta": { | |
| "system_name": title, | |
| "system_id": _identifier(source.get("filename", title), "PATENT_SYSTEM").upper(), | |
| "description": f"Candidate system model extracted from {source.get('filename', 'cleaned text')}.", | |
| "version": "0.1.0", | |
| "lifecycle_stage": "concept", | |
| "domain": "systems", | |
| "maturity": "machine-extracted-candidate", | |
| "date_modified": date.today().isoformat(), | |
| }, | |
| "subsystems": subsystems, | |
| } | |
| flows = [{"flow_id": f"FL-{i:03d}", "name": row["text"], "flow_type": "unspecified", | |
| "notes": "Model-extracted candidate."} | |
| for i, row in enumerate(candidates("ItemFlow", "flow_ref"), 1)] | |
| if flows: | |
| model["item_flows"] = flows | |
| values = [{"value_id": f"VAL-{i:03d}", "name": row["text"], "expression": row["text"], | |
| "source": "patent extraction"} | |
| for i, row in enumerate(candidates("Value", "attributes", "parameters", "variables"), 1)] | |
| if values: | |
| model["values"] = values | |
| requirements = [{"req_id": identifier, "category": "extracted_candidate", | |
| "statement": text, "status": "candidate"} | |
| for text, identifier in requirement_ids.items()] | |
| if requirements: | |
| model["requirements"] = requirements | |
| actions = [{"action_id": identifier, "name": text, | |
| "steps": [{"step_no": 1, "description": text}]} | |
| for text, identifier in action_ids.items()] | |
| states = [{"sm_id": f"SM-{i:03d}", "name": row["text"], "states": [], | |
| "owner": "unresolved"} | |
| for i, row in enumerate(anchors.get("StateMachine", []), 1)] | |
| if actions or states: | |
| model["behaviour"] = {} | |
| if states: | |
| model["behaviour"]["state_machines"] = states | |
| if actions: | |
| model["behaviour"]["actions"] = actions | |
| constraints = [{"constraint_id": f"CON-{i:03d}", "name": row["text"], | |
| "expression": row["text"], "type": "extracted_candidate", "status": "candidate"} | |
| for i, row in enumerate(anchors.get("Constraint", []), 1)] | |
| if constraints: | |
| model["constraints"] = constraints | |
| cases = [{"case_id": f"VER-{i:03d}", "name": row["text"], "type": "verification", | |
| "subject": "unresolved", "method": row["text"], "status": "candidate"} | |
| for i, row in enumerate(anchors.get("VerificationCase", []), 1)] | |
| if cases: | |
| model["verification"] = {"verification_cases": cases} | |
| relationships = [] | |
| for definition, result in self.passes.items(): | |
| for edge in result.get("edges", []): | |
| relationships.append({ | |
| "relationship_id": f"REL-{len(relationships) + 1:04d}", | |
| "type": edge["property"], "source": edge["source"], "target": edge["target"], | |
| "context": definition, "source_file": source.get("filename"), | |
| "notes": f"Model score {edge['score']:.4f}; review against graph evidence.", | |
| }) | |
| if relationships: | |
| model["relationships"] = relationships | |
| model["provenance"] = { | |
| "translator": "SJSKnowledgeGraph", | |
| "translator_version": "1.0", | |
| "schema_version": "sjs/1.0", | |
| "source_files": [{"path": source.get("source", "inline text"), | |
| "name": source.get("filename", "inline text")}], | |
| "mapping_notes": [ | |
| "All records are machine-extracted candidates.", | |
| "Graph evidence and scores remain authoritative for review.", | |
| "Unlinked part, port, and interface candidates are retained under the first system boundary.", | |
| ], | |
| } | |
| return model | |
| def export_sjs(self, path: str | Path, *, system_name: str | None = None) -> Path: | |
| path = Path(path) | |
| path.parent.mkdir(parents=True, exist_ok=True) | |
| path.write_text(json.dumps(self.to_sjs(system_name=system_name), ensure_ascii=False, indent=2) + "\n", | |
| encoding="utf-8") | |
| return path | |
| def export_sysml(self, path: str | Path, translator: Callable[[dict], str], | |
| *, system_name: str | None = None) -> Path: | |
| """Export through the supplied custom translator's ``sysml_from_sjs`` function.""" | |
| text = translator(self.to_sjs(system_name=system_name)) | |
| path = Path(path) | |
| path.parent.mkdir(parents=True, exist_ok=True) | |
| path.write_text(text, encoding="utf-8") | |
| return path | |
| def graph_from_patent(document: dict) -> SJSKnowledgeGraph: | |
| graph = SJSKnowledgeGraph(document["text"]) | |
| graph.source_document = {key: value for key, value in document.items() if key not in {"text", "sections"}} | |
| graph.source_document["sections"] = [ | |
| {key: value for key, value in section.items() if key != "text"} for section in document["sections"] | |
| ] | |
| return graph | |
| def run_patent_directory(model, *, directory=ROOT / "GT-Patents", | |
| output_dir=ROOT / "sjs_patent_graphs", definition="Subsystem", | |
| sections=("abstract", "background", "description", "claims"), | |
| resume=True, progress=None, **pass_options): | |
| """Run one SJS definition over every patent and yield per-file summaries.""" | |
| from patent_html import parse_patent_directory | |
| documents = parse_patent_directory(directory, sections=sections) | |
| output = Path(output_dir) | |
| output.mkdir(parents=True, exist_ok=True) | |
| summaries = [] | |
| for position, document in enumerate(documents, 1): | |
| folder = output / Path(document["filename"]).stem | |
| checkpoint = folder / "graph.json" | |
| graph = graph_from_patent(document) | |
| if resume and checkpoint.exists(): | |
| restored = SJSKnowledgeGraph.load(checkpoint) | |
| if restored.text != graph.text or restored.source_document != graph.source_document: | |
| raise ValueError(f"Source or selected sections changed for {document['filename']}; " | |
| "use a new output_dir or resume=False.") | |
| graph = restored | |
| bound = inspect.signature(graph.run_pass).bind(model, definition, **pass_options) | |
| bound.apply_defaults() | |
| options = bound.arguments | |
| plan = graph.plan(definition, **{key: options[key] for key in ("properties", "prompts", "anchor")}) | |
| settings = {key: options[key] for key in ("threshold", "relation_threshold", "adjacency_threshold", | |
| "properties_per_batch", "max_tokens", "overlap_tokens")} | |
| settings["max_tokens"] = min(settings["max_tokens"], int( | |
| getattr(model.config, "max_len", settings["max_tokens"]) or settings["max_tokens"])) | |
| previous = graph.passes.get(definition) | |
| reusable = bool(previous and previous["plan"] == plan and previous["settings"] == settings and | |
| previous["model"] == getattr(model, "knowledge_graph_checkpoint", type(model).__name__)) | |
| prefix = f"[{position}/{len(documents)}] {document['filename']}" | |
| if progress: | |
| progress(prefix + (" — resuming saved SJS pass" if reusable else " — " + definition)) | |
| if not reusable: | |
| graph.run_pass(model, definition, | |
| progress=(lambda message, p=prefix: progress(p + " — " + message)) if progress else None, | |
| **pass_options) | |
| folder.mkdir(parents=True, exist_ok=True) | |
| (folder / "text.txt").write_text(document["text"], encoding="utf-8") | |
| (folder / "parsed_patent.json").write_text( | |
| json.dumps(document, ensure_ascii=False, indent=2), encoding="utf-8") | |
| graph.save(checkpoint) | |
| graph.export_html(folder / "graph.html") | |
| graph.export_sjs(folder / "model.sjs.json") | |
| result = graph.passes[definition] | |
| summary = { | |
| "filename": document["filename"], "folder": folder.name, "title": document["title"], | |
| "definition": definition, "characters": len(document["text"]), | |
| "mentions": len(result["mentions"]), "connections": len(result["edges"]), | |
| "status": "reused" if reusable else "completed", "warnings": document["warnings"], | |
| } | |
| summaries.append(summary) | |
| (output / "manifest.json").write_text( | |
| json.dumps(summaries, ensure_ascii=False, indent=2), encoding="utf-8") | |
| yield summary | |