File size: 9,673 Bytes
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f07443e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
 
f07443e
b68816f
f07443e
b68816f
 
f07443e
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f07443e
b68816f
 
 
f07443e
b68816f
 
f07443e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f07443e
b68816f
 
 
 
 
 
 
f07443e
 
 
 
 
 
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f07443e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
"""The frequency-sorted review queue β€” the pipeline's actual product.

Ordering is the product decision here, and it targets the bottleneck directly:
the expert is the scarce resource, so they should hit the terms whose definition
propagates furthest first. Conflicts are promoted above frequency regardless,
because a contradiction is a decision only they can make.

Each row carries page, section and the verbatim span so review is a matter of
checking a quote against a page, not a claim against memory. That is what makes
the queue finishable.

Rows carry BOTH `page` (0-based, as the parser reports) and `page_no` (1-based,
what a human reads). A review UI must render `page_no`: an off-by-one here is
invisible until an expert opens the wrong page and concludes the provenance is
wrong.

v2 dropped `extraction_status`, `diff_status`, `definition_conflict` and
`conflict_variants` from the row: all four are already expressed by
`review_reason`, which is the field a reviewer actually reads, and they remain
on the entry for a consumer that wants them.
"""

from __future__ import annotations

import re
import textwrap

# At most one unresolved key parameter per this many term rows. See
# `_interleave` for why a cap replaced the promotion tier it used to have.
KEY_PARAMETER_INTERLEAVE = 5

# A key parameter NAME longer than this is not a name. The branch that produces
# them has no span check and no length guard, and on a textbook it returns the
# whole defining paragraph as the surface β€” see `_parameter_name`.
KEY_PARAMETER_NAME_MAX = 60

# The first sentence/label boundary in a returned surface. Applied only past
# `KEY_PARAMETER_NAME_MAX`, which is what keeps "No. of units" intact.
_HEAD_SPLIT = re.compile(r"[.:\n]")


def build_queue(entries: list[dict], domain: dict | None = None) -> list[dict]:
    """Rows an expert works through, most-decision-worthy first.

    **The queue is no longer glossary-only.** Every row carries `row_kind`, and
    a consumer must switch on it rather than assume `term`: an unresolved key
    parameter has a surface and no term, no definition and no mention count.
    """

    def sort_key(entry: dict):
        conflicting = (
            entry.get("diff_status") == "conflicting"
            or entry.get("definition_conflict") is True
        )
        return (0 if conflicting else 1, -int(entry.get("mention_count", 0) or 0))

    term_rows = []
    for entry in sorted(entries, key=sort_key):
        provenance = entry.get("provenance") or {}
        term_rows.append(
            {
                "row_kind": "term",
                # What an approve/edit/reject decision attaches to. Without it a
                # re-run loses which rows the expert already cleared.
                "term_id": entry.get("term_id"),
                "term": entry.get("term"),
                "definition": entry.get("definition"),
                "source_wording": entry.get("source_wording"),
                "mention_count": entry.get("mention_count", 0),
                "page": provenance.get("page"),
                # 1-based, the number a reviewer reads. `page` stays 0-based.
                "page_no": provenance.get("page_no"),
                "section_no": provenance.get("section_no"),
                "span": provenance.get("span"),
                "review_reason": _reason(entry),
                "_conflict": sort_key(entry)[0] == 0,
            }
        )

    queue = _interleave(term_rows, _key_parameter_rows(domain))
    for rank, row in enumerate(queue, start=1):
        row["rank"] = rank
        row.pop("_conflict", None)
    return queue


def _interleave(term_rows: list[dict], parameter_rows: list[dict]) -> list[dict]:
    """Conflicts, then terms by frequency with key parameters spliced in.

    Unresolved key parameters used to occupy their own tier directly beneath
    conflicts, which meant EVERY one of them outranked EVERY routine term no
    matter how many there were. On the Komatsu shop manual that is 24 junk rows
    standing in front of `Engine` (108 mentions) β€” the branch produces them in
    bulk on exactly the documents where it understands least, so the promotion
    fired hardest where it was least deserved and the expert met a wall of
    noise before reaching anything real.

    The original reason for promoting them is still sound and is preserved: the
    key-parameter branch has no span check, so an unresolvable surface is the
    one hallucination signal it can raise, and it must not sink beneath sixty
    routine rows where nobody reaches it. A CAP expresses that without letting
    the tail take the queue over β€” one parameter per five terms, so they are
    always visible early and never crowd out frequency.

    Conflicts stay strictly on top: a contradiction is a decision only the
    expert can make, and nothing interleaves above it.
    """
    conflicts = [row for row in term_rows if row.get("_conflict")]
    routine = [row for row in term_rows if not row.get("_conflict")]

    queue = list(conflicts)
    pending = list(parameter_rows)
    for index, row in enumerate(routine, start=1):
        queue.append(row)
        if pending and index % KEY_PARAMETER_INTERLEAVE == 0:
            queue.append(pending.pop(0))

    # Whatever the interleave could not place β€” a document with more unresolved
    # parameters than terms, which is itself a signal β€” rather than dropped.
    queue.extend(pending)
    return queue


def _key_parameter_rows(domain: dict | None) -> list[dict]:
    """One row per key parameter that resolved to no extracted term.

    Resolved parameters produce no row β€” they are corroborated and need nothing
    from a human. Only the dangles are worth an expert's attention, and they are
    worth it precisely because this branch has no span check: an unresolvable
    surface is the one hallucination signal it can produce.
    """
    provenance = (domain or {}).get("provenance") or {}
    rows = []
    for parameter in (domain or {}).get("key_parameters") or []:
        if parameter.get("term_id"):
            continue
        name, verbatim = _parameter_name(parameter.get("surface"))
        rows.append(
            {
                "row_kind": "key_parameter",
                # The decision attaches to the domain entry plus the surface β€”
                # there is no term_id, which is the entire point of the row.
                "term_id": None,
                "brief_id": (domain or {}).get("brief_id"),
                "term": name,
                # Null unless the surface had to be shortened. The full text is
                # never discarded: it is what the expert judges "is this real?"
                # against, and the branch picks the right CONCEPTS even when it
                # returns them at paragraph length.
                "surface_full": verbatim,
                "definition": None,
                "source_wording": None,
                "mention_count": 0,
                "page": provenance.get("page"),
                "page_no": provenance.get("page_no"),
                "section_no": provenance.get("section_no"),
                "span": provenance.get("span"),
                "review_reason": (
                    "named as a key parameter but no extracted term corroborates it "
                    "β€” confirm it is real"
                ),
            }
        )
    return rows


def _parameter_name(surface: str | None) -> tuple[str, str | None]:
    """A key parameter's displayable name, and its full text when shortened.

    The branch returns a `surface` that is supposed to be a parameter name, and
    on the Open Pit textbook it returns the right concept wrapped in its whole
    defining sentence. Nothing downstream caps it, so the queue renders a
    paragraph where a name belongs and the row is unreadable at a glance.

    Shortening is applied ONLY past `KEY_PARAMETER_NAME_MAX`, which is what
    keeps a legitimate short name containing a period ("No. of units") intact β€”
    splitting unconditionally on the first `.` is the obvious implementation and
    it silently truncates that class of name.
    """
    text = (surface or "").strip()
    if not text:
        return "", None
    if len(text) <= KEY_PARAMETER_NAME_MAX and "\n" not in text:
        return text, None

    head = _HEAD_SPLIT.split(text, 1)[0].strip()
    if not head or len(head) > KEY_PARAMETER_NAME_MAX:
        # No usable boundary β€” a run-on clause. Cut on a word boundary rather
        # than mid-token so the row still reads as language.
        head = textwrap.shorten(text, width=KEY_PARAMETER_NAME_MAX, placeholder="…")
    return head, text


def _reason(entry: dict) -> str:
    if entry.get("definition_conflict") or entry.get("diff_status") == "conflicting":
        return "conflicting definitions β€” expert decision required"
    if _wording_differs(entry):
        return "source wording differs from the expanded name β€” confirm which is correct"
    if entry.get("extraction_status") == "no_definition_found":
        return "term found but no definition in document"
    if not entry.get("definition"):
        return "definition rejected by span check or absent"
    return "routine confirmation"


def _wording_differs(entry: dict) -> bool:
    """The document says "Physical of Availability"; the expansion says
    "Physical Availability". Surfacing that to the expert is a locked
    requirement, so it earns its own review reason."""
    source = (entry.get("source_wording") or "").strip().casefold()
    full = (entry.get("full_name") or "").strip().casefold()
    return bool(source and full) and full not in source