Download src/knowledge_extraction/queue/review_queue.py from DataEyond/Agentic-Service-Data-Eyond-Catalog: direct link, hf CLI and curl.
- Browser
- Download file 9.67 kB
-
https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_extraction/queue/review_queue.py
- Command line
-
hf download hf://spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/src/knowledge_extraction/queue/review_queue.py
-
curl -L -o review_queue.py https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_extraction/queue/review_queue.py
9.67 kB
| """The frequency-sorted review queue β the pipeline's actual product. | |
| Ordering is the product decision here, and it targets the bottleneck directly: | |
| the expert is the scarce resource, so they should hit the terms whose definition | |
| propagates furthest first. Conflicts are promoted above frequency regardless, | |
| because a contradiction is a decision only they can make. | |
| Each row carries page, section and the verbatim span so review is a matter of | |
| checking a quote against a page, not a claim against memory. That is what makes | |
| the queue finishable. | |
| Rows carry BOTH `page` (0-based, as the parser reports) and `page_no` (1-based, | |
| what a human reads). A review UI must render `page_no`: an off-by-one here is | |
| invisible until an expert opens the wrong page and concludes the provenance is | |
| wrong. | |
| v2 dropped `extraction_status`, `diff_status`, `definition_conflict` and | |
| `conflict_variants` from the row: all four are already expressed by | |
| `review_reason`, which is the field a reviewer actually reads, and they remain | |
| on the entry for a consumer that wants them. | |
| """ | |
| from __future__ import annotations | |
| import re | |
| import textwrap | |
| # At most one unresolved key parameter per this many term rows. See | |
| # `_interleave` for why a cap replaced the promotion tier it used to have. | |
| KEY_PARAMETER_INTERLEAVE = 5 | |
| # A key parameter NAME longer than this is not a name. The branch that produces | |
| # them has no span check and no length guard, and on a textbook it returns the | |
| # whole defining paragraph as the surface β see `_parameter_name`. | |
| KEY_PARAMETER_NAME_MAX = 60 | |
| # The first sentence/label boundary in a returned surface. Applied only past | |
| # `KEY_PARAMETER_NAME_MAX`, which is what keeps "No. of units" intact. | |
| _HEAD_SPLIT = re.compile(r"[.:\n]") | |
| def build_queue(entries: list[dict], domain: dict | None = None) -> list[dict]: | |
| """Rows an expert works through, most-decision-worthy first. | |
| **The queue is no longer glossary-only.** Every row carries `row_kind`, and | |
| a consumer must switch on it rather than assume `term`: an unresolved key | |
| parameter has a surface and no term, no definition and no mention count. | |
| """ | |
| def sort_key(entry: dict): | |
| conflicting = ( | |
| entry.get("diff_status") == "conflicting" | |
| or entry.get("definition_conflict") is True | |
| ) | |
| return (0 if conflicting else 1, -int(entry.get("mention_count", 0) or 0)) | |
| term_rows = [] | |
| for entry in sorted(entries, key=sort_key): | |
| provenance = entry.get("provenance") or {} | |
| term_rows.append( | |
| { | |
| "row_kind": "term", | |
| # What an approve/edit/reject decision attaches to. Without it a | |
| # re-run loses which rows the expert already cleared. | |
| "term_id": entry.get("term_id"), | |
| "term": entry.get("term"), | |
| "definition": entry.get("definition"), | |
| "source_wording": entry.get("source_wording"), | |
| "mention_count": entry.get("mention_count", 0), | |
| "page": provenance.get("page"), | |
| # 1-based, the number a reviewer reads. `page` stays 0-based. | |
| "page_no": provenance.get("page_no"), | |
| "section_no": provenance.get("section_no"), | |
| "span": provenance.get("span"), | |
| "review_reason": _reason(entry), | |
| "_conflict": sort_key(entry)[0] == 0, | |
| } | |
| ) | |
| queue = _interleave(term_rows, _key_parameter_rows(domain)) | |
| for rank, row in enumerate(queue, start=1): | |
| row["rank"] = rank | |
| row.pop("_conflict", None) | |
| return queue | |
| def _interleave(term_rows: list[dict], parameter_rows: list[dict]) -> list[dict]: | |
| """Conflicts, then terms by frequency with key parameters spliced in. | |
| Unresolved key parameters used to occupy their own tier directly beneath | |
| conflicts, which meant EVERY one of them outranked EVERY routine term no | |
| matter how many there were. On the Komatsu shop manual that is 24 junk rows | |
| standing in front of `Engine` (108 mentions) β the branch produces them in | |
| bulk on exactly the documents where it understands least, so the promotion | |
| fired hardest where it was least deserved and the expert met a wall of | |
| noise before reaching anything real. | |
| The original reason for promoting them is still sound and is preserved: the | |
| key-parameter branch has no span check, so an unresolvable surface is the | |
| one hallucination signal it can raise, and it must not sink beneath sixty | |
| routine rows where nobody reaches it. A CAP expresses that without letting | |
| the tail take the queue over β one parameter per five terms, so they are | |
| always visible early and never crowd out frequency. | |
| Conflicts stay strictly on top: a contradiction is a decision only the | |
| expert can make, and nothing interleaves above it. | |
| """ | |
| conflicts = [row for row in term_rows if row.get("_conflict")] | |
| routine = [row for row in term_rows if not row.get("_conflict")] | |
| queue = list(conflicts) | |
| pending = list(parameter_rows) | |
| for index, row in enumerate(routine, start=1): | |
| queue.append(row) | |
| if pending and index % KEY_PARAMETER_INTERLEAVE == 0: | |
| queue.append(pending.pop(0)) | |
| # Whatever the interleave could not place β a document with more unresolved | |
| # parameters than terms, which is itself a signal β rather than dropped. | |
| queue.extend(pending) | |
| return queue | |
| def _key_parameter_rows(domain: dict | None) -> list[dict]: | |
| """One row per key parameter that resolved to no extracted term. | |
| Resolved parameters produce no row β they are corroborated and need nothing | |
| from a human. Only the dangles are worth an expert's attention, and they are | |
| worth it precisely because this branch has no span check: an unresolvable | |
| surface is the one hallucination signal it can produce. | |
| """ | |
| provenance = (domain or {}).get("provenance") or {} | |
| rows = [] | |
| for parameter in (domain or {}).get("key_parameters") or []: | |
| if parameter.get("term_id"): | |
| continue | |
| name, verbatim = _parameter_name(parameter.get("surface")) | |
| rows.append( | |
| { | |
| "row_kind": "key_parameter", | |
| # The decision attaches to the domain entry plus the surface β | |
| # there is no term_id, which is the entire point of the row. | |
| "term_id": None, | |
| "brief_id": (domain or {}).get("brief_id"), | |
| "term": name, | |
| # Null unless the surface had to be shortened. The full text is | |
| # never discarded: it is what the expert judges "is this real?" | |
| # against, and the branch picks the right CONCEPTS even when it | |
| # returns them at paragraph length. | |
| "surface_full": verbatim, | |
| "definition": None, | |
| "source_wording": None, | |
| "mention_count": 0, | |
| "page": provenance.get("page"), | |
| "page_no": provenance.get("page_no"), | |
| "section_no": provenance.get("section_no"), | |
| "span": provenance.get("span"), | |
| "review_reason": ( | |
| "named as a key parameter but no extracted term corroborates it " | |
| "β confirm it is real" | |
| ), | |
| } | |
| ) | |
| return rows | |
| def _parameter_name(surface: str | None) -> tuple[str, str | None]: | |
| """A key parameter's displayable name, and its full text when shortened. | |
| The branch returns a `surface` that is supposed to be a parameter name, and | |
| on the Open Pit textbook it returns the right concept wrapped in its whole | |
| defining sentence. Nothing downstream caps it, so the queue renders a | |
| paragraph where a name belongs and the row is unreadable at a glance. | |
| Shortening is applied ONLY past `KEY_PARAMETER_NAME_MAX`, which is what | |
| keeps a legitimate short name containing a period ("No. of units") intact β | |
| splitting unconditionally on the first `.` is the obvious implementation and | |
| it silently truncates that class of name. | |
| """ | |
| text = (surface or "").strip() | |
| if not text: | |
| return "", None | |
| if len(text) <= KEY_PARAMETER_NAME_MAX and "\n" not in text: | |
| return text, None | |
| head = _HEAD_SPLIT.split(text, 1)[0].strip() | |
| if not head or len(head) > KEY_PARAMETER_NAME_MAX: | |
| # No usable boundary β a run-on clause. Cut on a word boundary rather | |
| # than mid-token so the row still reads as language. | |
| head = textwrap.shorten(text, width=KEY_PARAMETER_NAME_MAX, placeholder="β¦") | |
| return head, text | |
| def _reason(entry: dict) -> str: | |
| if entry.get("definition_conflict") or entry.get("diff_status") == "conflicting": | |
| return "conflicting definitions β expert decision required" | |
| if _wording_differs(entry): | |
| return "source wording differs from the expanded name β confirm which is correct" | |
| if entry.get("extraction_status") == "no_definition_found": | |
| return "term found but no definition in document" | |
| if not entry.get("definition"): | |
| return "definition rejected by span check or absent" | |
| return "routine confirmation" | |
| def _wording_differs(entry: dict) -> bool: | |
| """The document says "Physical of Availability"; the expansion says | |
| "Physical Availability". Surfacing that to the expert is a locked | |
| requirement, so it earns its own review reason.""" | |
| source = (entry.get("source_wording") or "").strip().casefold() | |
| full = (entry.get("full_name") or "").strip().casefold() | |
| return bool(source and full) and full not in source | |