Download src/knowledge_extraction/mutation.py from DataEyond/Agentic-Service-Data-Eyond-Catalog: direct link, hf CLI and curl.
- Browser
- Download file 4.43 kB
-
https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_extraction/mutation.py
- Command line
-
hf download hf://spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/src/knowledge_extraction/mutation.py
-
curl -L -o mutation.py https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_extraction/mutation.py
4.43 kB
| """Has an entry changed in a way that invalidates an expert's ruling? (i3) | |
| From the 2026-09-08 discussion: *"kalau dia sudah termutate harusnya dia perlu | |
| review lagi… kalau tidak berubah, tetap approved."* The whole difficulty is in | |
| the second half. Flipping everything a new document touches is easy and useless | |
| — a document that merely mentions `PA` again moves `mention_count`, and if that | |
| counts as a mutation the expert comes back to forty rows where nothing has | |
| actually changed, finds that out one row at a time, and stops trusting the queue. | |
| So the comparison is deliberately narrow: **only the fields an expert ruled on** | |
| (decided 2026-09-14). Everything the pipeline computes about an entry — how | |
| often it was mentioned, how it was classified, where it was found — can move | |
| freely without disturbing an approval, because none of it is what the expert was | |
| looking at when they approved. | |
| Whitespace is collapsed before comparing, so re-extraction that reflows a | |
| definition does not re-open it. Any other difference counts: this stays | |
| conservative, because missing a real change means serving knowledge an expert | |
| never agreed to, and that is the failure mode this exists to prevent. | |
| """ | |
| from __future__ import annotations | |
| from collections.abc import Mapping | |
| from typing import Any | |
| # What a reviewer is actually judging, per kind — the fields the review queue | |
| # puts in front of them. Everything else on the entry is machinery. | |
| RULED_FIELDS: dict[str, tuple[str, ...]] = { | |
| "glossary": ("term", "full_name", "definition", "source_wording"), | |
| "rule": ("statement", "condition", "consequence"), | |
| "formula": ("name", "formula_latex", "unit"), | |
| "document": ("title", "purpose_verbatim"), | |
| "domain": ("domain_name", "purpose_verbatim"), | |
| } | |
| # Pipeline bookkeeping. Named explicitly so the fallback below cannot mistake a | |
| # recomputed count for a changed claim. | |
| VOLATILE_FIELDS: frozenset[str] = frozenset( | |
| { | |
| "mention_count", | |
| "diff_status", | |
| "extraction_status", | |
| "definition_conflict", | |
| "conflict_variants", | |
| "provenance", | |
| "doc_id", | |
| "doc_ids", | |
| } | |
| ) | |
| def _norm(value: Any) -> Any: | |
| """Collapse whitespace in strings; recurse into lists and dicts. | |
| Reflowed prose is not a changed claim. Nothing else is normalised — each | |
| further normalisation is a difference this would stop noticing. | |
| """ | |
| if isinstance(value, str): | |
| return " ".join(value.split()) | |
| if isinstance(value, list | tuple): | |
| return [_norm(v) for v in value] | |
| if isinstance(value, Mapping): | |
| return {k: _norm(v) for k, v in sorted(value.items())} | |
| return value | |
| def ruled_content(kind: str, payload: Mapping[str, Any] | None) -> dict[str, Any]: | |
| """The subset of an entry that an approval actually covers. | |
| An unregistered kind falls back to the whole payload minus the volatile | |
| fields — conservative on purpose: a new kind should over-report changes | |
| until someone decides what its reviewer is looking at, never under-report. | |
| """ | |
| payload = payload or {} | |
| fields = RULED_FIELDS.get(kind) | |
| if fields is not None: | |
| return {field: _norm(payload.get(field)) for field in fields} | |
| return { | |
| key: _norm(value) | |
| for key, value in sorted(payload.items()) | |
| if key not in VOLATILE_FIELDS | |
| } | |
| def changed_fields( | |
| kind: str, | |
| reviewed: Mapping[str, Any] | None, | |
| current: Mapping[str, Any] | None, | |
| ) -> list[str]: | |
| """Which ruled-on fields moved — the before/after a reviewer is shown (r4). | |
| Same comparison as `has_mutated`, reported field by field so the queue can | |
| say *what* changed instead of making the expert re-read the whole entry to | |
| find out. That was the ask: show before / after, do not send them back to | |
| the start. | |
| """ | |
| before = ruled_content(kind, reviewed) | |
| after = ruled_content(kind, current) | |
| return [field for field in before if before[field] != after.get(field)] | |
| def has_mutated( | |
| kind: str, | |
| reviewed: Mapping[str, Any] | None, | |
| current: Mapping[str, Any] | None, | |
| ) -> bool: | |
| """True when `current` differs from what was approved, in ways that matter. | |
| `reviewed` is what the expert had in front of them — for an `edited` | |
| decision that is their corrected payload, not the model's original claim. | |
| """ | |
| return ruled_content(kind, reviewed) != ruled_content(kind, current) | |