File size: 7,705 Bytes
e425536
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
from __future__ import annotations

import hashlib
import re
from collections import defaultdict
from decimal import Decimal, InvalidOperation
from typing import Any

from .schema import Category, Characteristic, ReviewStatus

TECHNICAL_LINE = re.compile(
    r"(?:[⌀Ø∅⌓⌖⏥⌯∥⊥±°]|\bM\s*\d|\bR\s*\d|\bRa\s*\d|"
    r"\d+(?:\.\d+)?\s*[±+\-/]|H\d\b|6H\b|THRU|TYP|MAX|MIN)",
    re.IGNORECASE,
)


def enrich_characteristic(item: Characteristic) -> Characteristic:
    text = " ".join(part for part in (item.requirement, item.source_text) if part).strip()
    if not item.requirement:
        item.requirement = item.source_text
    if item.category == Category.OTHER:
        item.category = infer_category(text)
    if item.quantity is None:
        quantity = re.search(r"(?<!\d)(\d{1,3})\s*[Xx](?!\w)", text)
        if quantity:
            item.quantity = int(quantity.group(1))
    if not item.thread:
        thread = re.search(
            r"\bM\s*\d+(?:\.\d+)?\s*[×xX]\s*\d+(?:\.\d+)?(?:\s*[-–]\s*[0-9A-Za-z]+)?",
            text,
        )
        if thread:
            item.thread = re.sub(r"\s+", "", thread.group(0)).replace("x", "×").replace("X", "×")
    if not item.surface_finish:
        finish = re.search(r"\bR[azq]\s*\d+(?:\.\d+)?(?:\s*µ?m)?", text, re.I)
        if finish:
            item.surface_finish = finish.group(0)
    if not item.nominal:
        nominal = _extract_nominal(text)
        if nominal is not None:
            item.nominal = nominal
    _extract_tolerances(item, text)
    _calculate_limits(item)
    if not item.record_id:
        seed = f"{item.page}|{item.balloon_id}|{item.requirement}|{item.source_text}"
        item.record_id = hashlib.sha1(seed.encode("utf-8")).hexdigest()[:12]
    return item


def infer_category(text: str) -> Category:
    normalized = text.strip()
    if re.search(r"\bM\s*\d+(?:\.\d+)?\s*[×xX]", normalized):
        return Category.THREAD
    if re.search(r"\bR[azq]\s*\d", normalized, re.I):
        return Category.SURFACE_FINISH
    if any(symbol in normalized for symbol in ("⌖", "⏥", "⌯", "∥", "⊥", "◎", "○")):
        return Category.GD_AND_T
    if re.search(r"[⌀Ø∅]\s*\d", normalized):
        return Category.DIAMETER
    if re.search(r"(?:^|\s)R\s*\d", normalized, re.I):
        return Category.RADIUS
    if "°" in normalized:
        return Category.ANGLE
    if re.search(r"\d", normalized) and any(
        token in normalized for token in ("±", "+", "-", "/", "MAX", "MIN")
    ):
        return Category.LINEAR_DIMENSION
    if re.search(r"MATERIAL|GRADE|HARDNESS|HBW|HRC", normalized, re.I):
        return Category.MATERIAL
    if normalized:
        return Category.NOTE
    return Category.OTHER


def records_from_ocr_text(text: str, page: int = 1) -> list[Characteristic]:
    """Conservative fallback for OCR output; unknown associations remain unassigned."""
    records: list[Characteristic] = []
    counter = 1
    for raw_line in text.splitlines():
        line = re.sub(r"\s+", " ", raw_line).strip(" |\t")
        if len(line) < 2 or not TECHNICAL_LINE.search(line):
            continue
        balloon = ""
        requirement = line
        matched = re.match(
            r"^(?:BALLOON\s*)?[#(]?\s*(\d{1,4})\s*[)\]:.\-]\s*(.+)$",
            line,
            re.I,
        )
        if matched:
            balloon, requirement = matched.group(1), matched.group(2)
        else:
            balloon = f"UNASSIGNED-{counter:03d}"
            counter += 1
        record = Characteristic(
            balloon_id=balloon,
            page=page,
            requirement=requirement,
            source_text=line,
            confidence=0.35 if balloon.startswith("UNASSIGNED") else 0.5,
            status=(
                ReviewStatus.NEEDS_MAPPING
                if balloon.startswith("UNASSIGNED")
                else ReviewStatus.NEEDS_REVIEW
            ),
        )
        records.append(enrich_characteristic(record))
    return records


def validate_characteristics(records: list[Characteristic]) -> list[Characteristic]:
    groups: dict[tuple[int, str], list[Characteristic]] = defaultdict(list)
    for record in records:
        enrich_characteristic(record)
        record.validation_issues = list(dict.fromkeys(record.validation_issues))
        if not record.balloon_id:
            record.validation_issues.append("Balloon identifier was not read.")
            record.status = ReviewStatus.NEEDS_MAPPING
        if record.balloon_id.upper().startswith("UNASSIGNED"):
            record.validation_issues.append("Requirement is not mapped to a balloon.")
            record.status = ReviewStatus.NEEDS_MAPPING
        if not record.requirement:
            record.validation_issues.append("Requirement text is empty or unreadable.")
        if record.category in {
            Category.LINEAR_DIMENSION,
            Category.DIAMETER,
            Category.RADIUS,
            Category.ANGLE,
        } and not record.nominal:
            record.validation_issues.append("Nominal value requires confirmation.")
        if not record.source_text:
            record.validation_issues.append("No verbatim source text was preserved.")
        # AI results never become approved merely because confidence is high.
        if record.status == ReviewStatus.VERIFIED:
            record.status = ReviewStatus.NEEDS_REVIEW
        groups[(record.page, record.balloon_id)].append(record)

    for grouped in groups.values():
        unique_requirements = {
            _key(record.requirement) for record in grouped if record.requirement.strip()
        }
        if len(unique_requirements) > 1:
            for record in grouped:
                record.validation_issues.append(
                    "Conflicting requirements were extracted for the same balloon."
                )
    return records


def _extract_nominal(text: str) -> str | None:
    patterns = [
        r"[⌀Ø∅]\s*([+-]?\d+(?:\.\d+)?)",
        r"(?:^|\s)R\s*([+-]?\d+(?:\.\d+)?)",
        r"(?<![A-Za-z\d])([+-]?\d+(?:\.\d+)?)(?=\s*(?:±|\+|\-|°|$))",
    ]
    for pattern in patterns:
        match = re.search(pattern, text, re.I)
        if match:
            return match.group(1)
    return None


def _extract_tolerances(item: Characteristic, text: str) -> None:
    if item.upper_tolerance or item.lower_tolerance:
        return
    bilateral = re.search(r"±\s*(\d+(?:\.\d+)?)", text)
    if bilateral:
        item.upper_tolerance = bilateral.group(1)
        item.lower_tolerance = f"-{bilateral.group(1)}"
        return
    unilateral = re.search(
        r"\+\s*(\d+(?:\.\d+)?)\s*(?:/|\s)\s*-\s*(\d+(?:\.\d+)?)",
        text,
    )
    if unilateral:
        item.upper_tolerance = unilateral.group(1)
        item.lower_tolerance = f"-{unilateral.group(2)}"


def _calculate_limits(item: Characteristic) -> None:
    if item.upper_limit or item.lower_limit:
        return
    nominal = _decimal(item.nominal)
    upper = _decimal(item.upper_tolerance)
    lower = _decimal(item.lower_tolerance)
    if nominal is None or upper is None or lower is None:
        return
    item.upper_limit = _decimal_text(nominal + upper)
    item.lower_limit = _decimal_text(nominal + lower)


def _decimal(value: Any) -> Decimal | None:
    if value in (None, ""):
        return None
    text = str(value).strip().replace("+", "")
    if not re.fullmatch(r"-?\d+(?:\.\d+)?", text):
        return None
    try:
        return Decimal(text)
    except InvalidOperation:
        return None


def _decimal_text(value: Decimal) -> str:
    return format(value.normalize(), "f")


def _key(value: str) -> str:
    return re.sub(r"\W+", "", value.casefold())