ishaq101's picture
/fix parsing and term extract (#21)
f07443e
Raw History Blame Contribute Delete
9.67 kB
"""The frequency-sorted review queue β€” the pipeline's actual product.
Ordering is the product decision here, and it targets the bottleneck directly:
the expert is the scarce resource, so they should hit the terms whose definition
propagates furthest first. Conflicts are promoted above frequency regardless,
because a contradiction is a decision only they can make.
Each row carries page, section and the verbatim span so review is a matter of
checking a quote against a page, not a claim against memory. That is what makes
the queue finishable.
Rows carry BOTH `page` (0-based, as the parser reports) and `page_no` (1-based,
what a human reads). A review UI must render `page_no`: an off-by-one here is
invisible until an expert opens the wrong page and concludes the provenance is
wrong.
v2 dropped `extraction_status`, `diff_status`, `definition_conflict` and
`conflict_variants` from the row: all four are already expressed by
`review_reason`, which is the field a reviewer actually reads, and they remain
on the entry for a consumer that wants them.
"""
from __future__ import annotations
import re
import textwrap
# At most one unresolved key parameter per this many term rows. See
# `_interleave` for why a cap replaced the promotion tier it used to have.
KEY_PARAMETER_INTERLEAVE = 5
# A key parameter NAME longer than this is not a name. The branch that produces
# them has no span check and no length guard, and on a textbook it returns the
# whole defining paragraph as the surface β€” see `_parameter_name`.
KEY_PARAMETER_NAME_MAX = 60
# The first sentence/label boundary in a returned surface. Applied only past
# `KEY_PARAMETER_NAME_MAX`, which is what keeps "No. of units" intact.
_HEAD_SPLIT = re.compile(r"[.:\n]")
def build_queue(entries: list[dict], domain: dict | None = None) -> list[dict]:
"""Rows an expert works through, most-decision-worthy first.
**The queue is no longer glossary-only.** Every row carries `row_kind`, and
a consumer must switch on it rather than assume `term`: an unresolved key
parameter has a surface and no term, no definition and no mention count.
"""
def sort_key(entry: dict):
conflicting = (
entry.get("diff_status") == "conflicting"
or entry.get("definition_conflict") is True
)
return (0 if conflicting else 1, -int(entry.get("mention_count", 0) or 0))
term_rows = []
for entry in sorted(entries, key=sort_key):
provenance = entry.get("provenance") or {}
term_rows.append(
{
"row_kind": "term",
# What an approve/edit/reject decision attaches to. Without it a
# re-run loses which rows the expert already cleared.
"term_id": entry.get("term_id"),
"term": entry.get("term"),
"definition": entry.get("definition"),
"source_wording": entry.get("source_wording"),
"mention_count": entry.get("mention_count", 0),
"page": provenance.get("page"),
# 1-based, the number a reviewer reads. `page` stays 0-based.
"page_no": provenance.get("page_no"),
"section_no": provenance.get("section_no"),
"span": provenance.get("span"),
"review_reason": _reason(entry),
"_conflict": sort_key(entry)[0] == 0,
}
)
queue = _interleave(term_rows, _key_parameter_rows(domain))
for rank, row in enumerate(queue, start=1):
row["rank"] = rank
row.pop("_conflict", None)
return queue
def _interleave(term_rows: list[dict], parameter_rows: list[dict]) -> list[dict]:
"""Conflicts, then terms by frequency with key parameters spliced in.
Unresolved key parameters used to occupy their own tier directly beneath
conflicts, which meant EVERY one of them outranked EVERY routine term no
matter how many there were. On the Komatsu shop manual that is 24 junk rows
standing in front of `Engine` (108 mentions) β€” the branch produces them in
bulk on exactly the documents where it understands least, so the promotion
fired hardest where it was least deserved and the expert met a wall of
noise before reaching anything real.
The original reason for promoting them is still sound and is preserved: the
key-parameter branch has no span check, so an unresolvable surface is the
one hallucination signal it can raise, and it must not sink beneath sixty
routine rows where nobody reaches it. A CAP expresses that without letting
the tail take the queue over β€” one parameter per five terms, so they are
always visible early and never crowd out frequency.
Conflicts stay strictly on top: a contradiction is a decision only the
expert can make, and nothing interleaves above it.
"""
conflicts = [row for row in term_rows if row.get("_conflict")]
routine = [row for row in term_rows if not row.get("_conflict")]
queue = list(conflicts)
pending = list(parameter_rows)
for index, row in enumerate(routine, start=1):
queue.append(row)
if pending and index % KEY_PARAMETER_INTERLEAVE == 0:
queue.append(pending.pop(0))
# Whatever the interleave could not place β€” a document with more unresolved
# parameters than terms, which is itself a signal β€” rather than dropped.
queue.extend(pending)
return queue
def _key_parameter_rows(domain: dict | None) -> list[dict]:
"""One row per key parameter that resolved to no extracted term.
Resolved parameters produce no row β€” they are corroborated and need nothing
from a human. Only the dangles are worth an expert's attention, and they are
worth it precisely because this branch has no span check: an unresolvable
surface is the one hallucination signal it can produce.
"""
provenance = (domain or {}).get("provenance") or {}
rows = []
for parameter in (domain or {}).get("key_parameters") or []:
if parameter.get("term_id"):
continue
name, verbatim = _parameter_name(parameter.get("surface"))
rows.append(
{
"row_kind": "key_parameter",
# The decision attaches to the domain entry plus the surface β€”
# there is no term_id, which is the entire point of the row.
"term_id": None,
"brief_id": (domain or {}).get("brief_id"),
"term": name,
# Null unless the surface had to be shortened. The full text is
# never discarded: it is what the expert judges "is this real?"
# against, and the branch picks the right CONCEPTS even when it
# returns them at paragraph length.
"surface_full": verbatim,
"definition": None,
"source_wording": None,
"mention_count": 0,
"page": provenance.get("page"),
"page_no": provenance.get("page_no"),
"section_no": provenance.get("section_no"),
"span": provenance.get("span"),
"review_reason": (
"named as a key parameter but no extracted term corroborates it "
"β€” confirm it is real"
),
}
)
return rows
def _parameter_name(surface: str | None) -> tuple[str, str | None]:
"""A key parameter's displayable name, and its full text when shortened.
The branch returns a `surface` that is supposed to be a parameter name, and
on the Open Pit textbook it returns the right concept wrapped in its whole
defining sentence. Nothing downstream caps it, so the queue renders a
paragraph where a name belongs and the row is unreadable at a glance.
Shortening is applied ONLY past `KEY_PARAMETER_NAME_MAX`, which is what
keeps a legitimate short name containing a period ("No. of units") intact β€”
splitting unconditionally on the first `.` is the obvious implementation and
it silently truncates that class of name.
"""
text = (surface or "").strip()
if not text:
return "", None
if len(text) <= KEY_PARAMETER_NAME_MAX and "\n" not in text:
return text, None
head = _HEAD_SPLIT.split(text, 1)[0].strip()
if not head or len(head) > KEY_PARAMETER_NAME_MAX:
# No usable boundary β€” a run-on clause. Cut on a word boundary rather
# than mid-token so the row still reads as language.
head = textwrap.shorten(text, width=KEY_PARAMETER_NAME_MAX, placeholder="…")
return head, text
def _reason(entry: dict) -> str:
if entry.get("definition_conflict") or entry.get("diff_status") == "conflicting":
return "conflicting definitions β€” expert decision required"
if _wording_differs(entry):
return "source wording differs from the expanded name β€” confirm which is correct"
if entry.get("extraction_status") == "no_definition_found":
return "term found but no definition in document"
if not entry.get("definition"):
return "definition rejected by span check or absent"
return "routine confirmation"
def _wording_differs(entry: dict) -> bool:
"""The document says "Physical of Availability"; the expansion says
"Physical Availability". Surfacing that to the expert is a locked
requirement, so it earns its own review reason."""
source = (entry.get("source_wording") or "").strip().casefold()
full = (entry.get("full_name") or "").strip().casefold()
return bool(source and full) and full not in source