""" Deterministic, lexicon-driven attribute extractor. This is intentionally simple: for each attribute type, scan the (lowercased) text for known surface forms and collect the canonical labels found, in the order compound/longer phrases are checked before shorter substrings so we don't e.g. match "neck" inside "v neckline" twice or "blue" inside "dusty blue". This module is used two ways in the pipeline: 1. As a standalone baseline extractor (fast, explainable, zero training data needed beyond the lexicon itself). 2. As a feature/ensemble signal that boosts the ML model's precision on attribute values it has seen described in the lexicon but not necessarily enough times in the 61-row training set to learn reliably. """ import re from lexicon import ATTRIBUTE_LEXICON, COLOR_VOCAB def _find_labels(text_lower: str, mapping: dict) -> list: found = [] for canonical, surface_forms in mapping.items(): for form in surface_forms: if form in text_lower: found.append(canonical) break return found def extract_category(text_lower: str) -> list: # Category is hierarchical in our vocab (e.g. "wedding dress" also # contains "dress"); prefer the most specific match and only fall back # to the generic "dress"/"gown" if nothing specific hit. specific_order = [ "bridesmaid dress", "prom gown", "wedding dress", "cocktail dress", "evening gown", "formal dress", "quinceanera dress", "party dress", "maternity dress", ] for canonical in specific_order: for form in ATTRIBUTE_LEXICON["category"][canonical]: if form in text_lower: return [canonical] for canonical in ["gown", "dress"]: for form in ATTRIBUTE_LEXICON["category"][canonical]: if form in text_lower: return [canonical] return [] def extract_colors(text_lower: str) -> list: found = [] remaining = text_lower for color in COLOR_VOCAB: # already ordered compound-first if color in remaining: found.append(color) # blank it out so we don't re-match its substring later remaining = remaining.replace(color, " " * len(color)) return found def extract_attributes_rules(text: str) -> dict: text_lower = text.lower() result = {} for attr in ["silhouette", "fabric", "neckline", "sleeve", "length", "embellishment"]: result[attr] = _find_labels(text_lower, ATTRIBUTE_LEXICON[attr]) result["category"] = extract_category(text_lower) result["color"] = extract_colors(text_lower) return result if __name__ == "__main__": sample = "Floor length chiffon bridesmaid dress with pleated bodice and V neckline available in sage and dusty blue" import json print(json.dumps(extract_attributes_rules(sample), indent=2))