File size: 2,890 Bytes
896a559
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
"""
Deterministic, lexicon-driven attribute extractor.

This is intentionally simple: for each attribute type, scan the (lowercased)
text for known surface forms and collect the canonical labels found, in the
order compound/longer phrases are checked before shorter substrings so we
don't e.g. match "neck" inside "v neckline" twice or "blue" inside "dusty
blue".

This module is used two ways in the pipeline:
  1. As a standalone baseline extractor (fast, explainable, zero training data
     needed beyond the lexicon itself).
  2. As a feature/ensemble signal that boosts the ML model's precision on
     attribute values it has seen described in the lexicon but not necessarily
     enough times in the 61-row training set to learn reliably.
"""
import re
from lexicon import ATTRIBUTE_LEXICON, COLOR_VOCAB


def _find_labels(text_lower: str, mapping: dict) -> list:
    found = []
    for canonical, surface_forms in mapping.items():
        for form in surface_forms:
            if form in text_lower:
                found.append(canonical)
                break
    return found


def extract_category(text_lower: str) -> list:
    # Category is hierarchical in our vocab (e.g. "wedding dress" also
    # contains "dress"); prefer the most specific match and only fall back
    # to the generic "dress"/"gown" if nothing specific hit.
    specific_order = [
        "bridesmaid dress", "prom gown", "wedding dress", "cocktail dress",
        "evening gown", "formal dress", "quinceanera dress", "party dress",
        "maternity dress",
    ]
    for canonical in specific_order:
        for form in ATTRIBUTE_LEXICON["category"][canonical]:
            if form in text_lower:
                return [canonical]
    for canonical in ["gown", "dress"]:
        for form in ATTRIBUTE_LEXICON["category"][canonical]:
            if form in text_lower:
                return [canonical]
    return []


def extract_colors(text_lower: str) -> list:
    found = []
    remaining = text_lower
    for color in COLOR_VOCAB:  # already ordered compound-first
        if color in remaining:
            found.append(color)
            # blank it out so we don't re-match its substring later
            remaining = remaining.replace(color, " " * len(color))
    return found


def extract_attributes_rules(text: str) -> dict:
    text_lower = text.lower()
    result = {}
    for attr in ["silhouette", "fabric", "neckline", "sleeve", "length", "embellishment"]:
        result[attr] = _find_labels(text_lower, ATTRIBUTE_LEXICON[attr])
    result["category"] = extract_category(text_lower)
    result["color"] = extract_colors(text_lower)
    return result


if __name__ == "__main__":
    sample = "Floor length chiffon bridesmaid dress with pleated bodice and V neckline available in sage and dusty blue"
    import json
    print(json.dumps(extract_attributes_rules(sample), indent=2))