ishaq101's picture
/fix parsing and term extract (#21)
f07443e
Raw History Blame Contribute Delete
4.43 kB
"""Has an entry changed in a way that invalidates an expert's ruling? (i3)
From the 2026-09-08 discussion: *"kalau dia sudah termutate harusnya dia perlu
review lagi… kalau tidak berubah, tetap approved."* The whole difficulty is in
the second half. Flipping everything a new document touches is easy and useless
— a document that merely mentions `PA` again moves `mention_count`, and if that
counts as a mutation the expert comes back to forty rows where nothing has
actually changed, finds that out one row at a time, and stops trusting the queue.
So the comparison is deliberately narrow: **only the fields an expert ruled on**
(decided 2026-09-14). Everything the pipeline computes about an entry — how
often it was mentioned, how it was classified, where it was found — can move
freely without disturbing an approval, because none of it is what the expert was
looking at when they approved.
Whitespace is collapsed before comparing, so re-extraction that reflows a
definition does not re-open it. Any other difference counts: this stays
conservative, because missing a real change means serving knowledge an expert
never agreed to, and that is the failure mode this exists to prevent.
"""
from __future__ import annotations
from collections.abc import Mapping
from typing import Any
# What a reviewer is actually judging, per kind — the fields the review queue
# puts in front of them. Everything else on the entry is machinery.
RULED_FIELDS: dict[str, tuple[str, ...]] = {
"glossary": ("term", "full_name", "definition", "source_wording"),
"rule": ("statement", "condition", "consequence"),
"formula": ("name", "formula_latex", "unit"),
"document": ("title", "purpose_verbatim"),
"domain": ("domain_name", "purpose_verbatim"),
}
# Pipeline bookkeeping. Named explicitly so the fallback below cannot mistake a
# recomputed count for a changed claim.
VOLATILE_FIELDS: frozenset[str] = frozenset(
{
"mention_count",
"diff_status",
"extraction_status",
"definition_conflict",
"conflict_variants",
"provenance",
"doc_id",
"doc_ids",
}
)
def _norm(value: Any) -> Any:
"""Collapse whitespace in strings; recurse into lists and dicts.
Reflowed prose is not a changed claim. Nothing else is normalised — each
further normalisation is a difference this would stop noticing.
"""
if isinstance(value, str):
return " ".join(value.split())
if isinstance(value, list | tuple):
return [_norm(v) for v in value]
if isinstance(value, Mapping):
return {k: _norm(v) for k, v in sorted(value.items())}
return value
def ruled_content(kind: str, payload: Mapping[str, Any] | None) -> dict[str, Any]:
"""The subset of an entry that an approval actually covers.
An unregistered kind falls back to the whole payload minus the volatile
fields — conservative on purpose: a new kind should over-report changes
until someone decides what its reviewer is looking at, never under-report.
"""
payload = payload or {}
fields = RULED_FIELDS.get(kind)
if fields is not None:
return {field: _norm(payload.get(field)) for field in fields}
return {
key: _norm(value)
for key, value in sorted(payload.items())
if key not in VOLATILE_FIELDS
}
def changed_fields(
kind: str,
reviewed: Mapping[str, Any] | None,
current: Mapping[str, Any] | None,
) -> list[str]:
"""Which ruled-on fields moved — the before/after a reviewer is shown (r4).
Same comparison as `has_mutated`, reported field by field so the queue can
say *what* changed instead of making the expert re-read the whole entry to
find out. That was the ask: show before / after, do not send them back to
the start.
"""
before = ruled_content(kind, reviewed)
after = ruled_content(kind, current)
return [field for field in before if before[field] != after.get(field)]
def has_mutated(
kind: str,
reviewed: Mapping[str, Any] | None,
current: Mapping[str, Any] | None,
) -> bool:
"""True when `current` differs from what was approved, in ways that matter.
`reviewed` is what the expert had in front of them — for an `edited`
decision that is their corrected payload, not the model's original claim.
"""
return ruled_content(kind, reviewed) != ruled_content(kind, current)