product_attribute / src /rules_extractor.py
kshitiz14's picture
Fix Dockerfile for HF Spaces (port 7860), stop tracking model artifact
896a559
Raw History Blame Contribute Delete
2.89 kB
"""
Deterministic, lexicon-driven attribute extractor.
This is intentionally simple: for each attribute type, scan the (lowercased)
text for known surface forms and collect the canonical labels found, in the
order compound/longer phrases are checked before shorter substrings so we
don't e.g. match "neck" inside "v neckline" twice or "blue" inside "dusty
blue".
This module is used two ways in the pipeline:
1. As a standalone baseline extractor (fast, explainable, zero training data
needed beyond the lexicon itself).
2. As a feature/ensemble signal that boosts the ML model's precision on
attribute values it has seen described in the lexicon but not necessarily
enough times in the 61-row training set to learn reliably.
"""
import re
from lexicon import ATTRIBUTE_LEXICON, COLOR_VOCAB
def _find_labels(text_lower: str, mapping: dict) -> list:
found = []
for canonical, surface_forms in mapping.items():
for form in surface_forms:
if form in text_lower:
found.append(canonical)
break
return found
def extract_category(text_lower: str) -> list:
# Category is hierarchical in our vocab (e.g. "wedding dress" also
# contains "dress"); prefer the most specific match and only fall back
# to the generic "dress"/"gown" if nothing specific hit.
specific_order = [
"bridesmaid dress", "prom gown", "wedding dress", "cocktail dress",
"evening gown", "formal dress", "quinceanera dress", "party dress",
"maternity dress",
]
for canonical in specific_order:
for form in ATTRIBUTE_LEXICON["category"][canonical]:
if form in text_lower:
return [canonical]
for canonical in ["gown", "dress"]:
for form in ATTRIBUTE_LEXICON["category"][canonical]:
if form in text_lower:
return [canonical]
return []
def extract_colors(text_lower: str) -> list:
found = []
remaining = text_lower
for color in COLOR_VOCAB: # already ordered compound-first
if color in remaining:
found.append(color)
# blank it out so we don't re-match its substring later
remaining = remaining.replace(color, " " * len(color))
return found
def extract_attributes_rules(text: str) -> dict:
text_lower = text.lower()
result = {}
for attr in ["silhouette", "fabric", "neckline", "sleeve", "length", "embellishment"]:
result[attr] = _find_labels(text_lower, ATTRIBUTE_LEXICON[attr])
result["category"] = extract_category(text_lower)
result["color"] = extract_colors(text_lower)
return result
if __name__ == "__main__":
sample = "Floor length chiffon bridesmaid dress with pleated bodice and V neckline available in sage and dusty blue"
import json
print(json.dumps(extract_attributes_rules(sample), indent=2))