Spaces:
Sleeping
Sleeping
Download src/rules_extractor.py from kshitiz14/product_attribute: direct link, hf CLI and curl.
- Browser
- Download file 2.89 kB
-
https://huggingface.co/spaces/kshitiz14/product_attribute/resolve/main/src/rules_extractor.py
- Command line
-
hf download hf://spaces/kshitiz14/product_attribute/src/rules_extractor.py
-
curl -L -o rules_extractor.py https://huggingface.co/spaces/kshitiz14/product_attribute/resolve/main/src/rules_extractor.py
2.89 kB
| """ | |
| Deterministic, lexicon-driven attribute extractor. | |
| This is intentionally simple: for each attribute type, scan the (lowercased) | |
| text for known surface forms and collect the canonical labels found, in the | |
| order compound/longer phrases are checked before shorter substrings so we | |
| don't e.g. match "neck" inside "v neckline" twice or "blue" inside "dusty | |
| blue". | |
| This module is used two ways in the pipeline: | |
| 1. As a standalone baseline extractor (fast, explainable, zero training data | |
| needed beyond the lexicon itself). | |
| 2. As a feature/ensemble signal that boosts the ML model's precision on | |
| attribute values it has seen described in the lexicon but not necessarily | |
| enough times in the 61-row training set to learn reliably. | |
| """ | |
| import re | |
| from lexicon import ATTRIBUTE_LEXICON, COLOR_VOCAB | |
| def _find_labels(text_lower: str, mapping: dict) -> list: | |
| found = [] | |
| for canonical, surface_forms in mapping.items(): | |
| for form in surface_forms: | |
| if form in text_lower: | |
| found.append(canonical) | |
| break | |
| return found | |
| def extract_category(text_lower: str) -> list: | |
| # Category is hierarchical in our vocab (e.g. "wedding dress" also | |
| # contains "dress"); prefer the most specific match and only fall back | |
| # to the generic "dress"/"gown" if nothing specific hit. | |
| specific_order = [ | |
| "bridesmaid dress", "prom gown", "wedding dress", "cocktail dress", | |
| "evening gown", "formal dress", "quinceanera dress", "party dress", | |
| "maternity dress", | |
| ] | |
| for canonical in specific_order: | |
| for form in ATTRIBUTE_LEXICON["category"][canonical]: | |
| if form in text_lower: | |
| return [canonical] | |
| for canonical in ["gown", "dress"]: | |
| for form in ATTRIBUTE_LEXICON["category"][canonical]: | |
| if form in text_lower: | |
| return [canonical] | |
| return [] | |
| def extract_colors(text_lower: str) -> list: | |
| found = [] | |
| remaining = text_lower | |
| for color in COLOR_VOCAB: # already ordered compound-first | |
| if color in remaining: | |
| found.append(color) | |
| # blank it out so we don't re-match its substring later | |
| remaining = remaining.replace(color, " " * len(color)) | |
| return found | |
| def extract_attributes_rules(text: str) -> dict: | |
| text_lower = text.lower() | |
| result = {} | |
| for attr in ["silhouette", "fabric", "neckline", "sleeve", "length", "embellishment"]: | |
| result[attr] = _find_labels(text_lower, ATTRIBUTE_LEXICON[attr]) | |
| result["category"] = extract_category(text_lower) | |
| result["color"] = extract_colors(text_lower) | |
| return result | |
| if __name__ == "__main__": | |
| sample = "Floor length chiffon bridesmaid dress with pleated bodice and V neckline available in sage and dusty blue" | |
| import json | |
| print(json.dumps(extract_attributes_rules(sample), indent=2)) | |