Spaces:
Sleeping
Sleeping
Download src/evaluate.py from kshitiz14/product_attribute: direct link, hf CLI and curl.
- Browser
- Download file 8.3 kB
-
https://huggingface.co/spaces/kshitiz14/product_attribute/resolve/main/src/evaluate.py
- Command line
-
hf download hf://spaces/kshitiz14/product_attribute/src/evaluate.py
-
curl -L -o evaluate.py https://huggingface.co/spaces/kshitiz14/product_attribute/resolve/main/src/evaluate.py
8.3 kB
| """ | |
| Evaluation harness for the attribute extraction pipeline. | |
| Metrics reported (per attribute type, and overall): | |
| - Exact-set accuracy: fraction of examples where the predicted label set | |
| for that attribute exactly equals the gold label set (strict metric -- | |
| partial credit is zero even if only one label differs). | |
| - Micro-F1: precision/recall/F1 computed over all individual label | |
| predictions pooled across every example (standard multi-label metric, | |
| give partial credit for partially-correct label sets). | |
| - Macro-F1: F1 averaged per-label then averaged across labels (surfaces | |
| performance on rare labels that micro-F1 can hide). | |
| We evaluate three variants for comparison: | |
| 1. rules-only | |
| 2. ml-only (model trained on the TRAIN split only, evaluated on TEST split) | |
| 3. ensemble (rules + ml, ml trained on TRAIN split only) | |
| Because the labeled set has 61 rows, we use an 80/20 train/test split | |
| (49 train / 12 test) with a fixed random seed for reproducibility, and also | |
| report leave-one-out cross-validated numbers for the ML model since a single | |
| 12-row test split has high variance at this scale. | |
| """ | |
| import json | |
| import random | |
| import joblib | |
| from sklearn.feature_extraction.text import TfidfVectorizer | |
| from sklearn.linear_model import LogisticRegression | |
| from sklearn.multiclass import OneVsRestClassifier | |
| from sklearn.preprocessing import MultiLabelBinarizer | |
| from lexicon import ATTRIBUTE_LEXICON, COLOR_VOCAB | |
| from rules_extractor import extract_attributes_rules | |
| ATTR_TYPES = ["silhouette", "fabric", "neckline", "sleeve", "length", | |
| "embellishment", "category", "color"] | |
| random.seed(42) | |
| def label_space(attr): | |
| if attr == "color": | |
| return COLOR_VOCAB | |
| return list(ATTRIBUTE_LEXICON[attr].keys()) | |
| def load_dataset(path="../data/dataset.json"): | |
| with open(path) as f: | |
| return json.load(f) | |
| def train_test_split(data, test_frac=0.2, seed=42): | |
| idx = list(range(len(data))) | |
| random.Random(seed).shuffle(idx) | |
| n_test = max(1, int(len(data) * test_frac)) | |
| test_idx = set(idx[:n_test]) | |
| train = [d for i, d in enumerate(data) if i not in test_idx] | |
| test = [d for i, d in enumerate(data) if i in test_idx] | |
| return train, test | |
| def train_ml_on(train_data): | |
| texts = [d["text"] for d in train_data] | |
| vectorizer = TfidfVectorizer(analyzer="char_wb", ngram_range=(2, 5), min_df=1) | |
| X = vectorizer.fit_transform(texts) | |
| models, binarizers = {}, {} | |
| for attr in ATTR_TYPES: | |
| classes = label_space(attr) | |
| mlb = MultiLabelBinarizer(classes=classes) | |
| Y = mlb.fit_transform([d["labels"][attr] for d in train_data]) | |
| if Y.sum() == 0: | |
| models[attr] = None | |
| else: | |
| clf = OneVsRestClassifier(LogisticRegression(max_iter=1000, class_weight="balanced")) | |
| clf.fit(X, Y) | |
| models[attr] = clf | |
| binarizers[attr] = mlb | |
| return vectorizer, models, binarizers | |
| def predict_ml_with(vectorizer, models, binarizers, text): | |
| X = vectorizer.transform([text]) | |
| out = {} | |
| for attr in ATTR_TYPES: | |
| clf = models[attr] | |
| if clf is None: | |
| out[attr] = [] | |
| continue | |
| y = clf.predict(X) | |
| out[attr] = list(binarizers[attr].inverse_transform(y)[0]) | |
| return out | |
| def predict_ensemble_with(vectorizer, models, binarizers, text): | |
| rule_preds = extract_attributes_rules(text) | |
| ml_preds = predict_ml_with(vectorizer, models, binarizers, text) | |
| final = {} | |
| for attr in rule_preds: | |
| merged = list(rule_preds[attr]) | |
| for label in ml_preds.get(attr, []): | |
| if label not in merged: | |
| merged.append(label) | |
| final[attr] = merged | |
| return final | |
| def exact_set_accuracy(gold_sets, pred_sets): | |
| correct = sum(1 for g, p in zip(gold_sets, pred_sets) if set(g) == set(p)) | |
| return correct / len(gold_sets) | |
| def micro_prf1(gold_sets, pred_sets): | |
| tp = fp = fn = 0 | |
| for g, p in zip(gold_sets, pred_sets): | |
| g, p = set(g), set(p) | |
| tp += len(g & p) | |
| fp += len(p - g) | |
| fn += len(g - p) | |
| precision = tp / (tp + fp) if (tp + fp) else 0.0 | |
| recall = tp / (tp + fn) if (tp + fn) else 0.0 | |
| f1 = 2 * precision * recall / (precision + recall) if (precision + recall) else 0.0 | |
| return precision, recall, f1 | |
| def macro_f1(gold_sets, pred_sets, classes): | |
| f1s = [] | |
| for c in classes: | |
| tp = fp = fn = 0 | |
| for g, p in zip(gold_sets, pred_sets): | |
| g_has, p_has = c in g, c in p | |
| if g_has and p_has: | |
| tp += 1 | |
| elif p_has and not g_has: | |
| fp += 1 | |
| elif g_has and not p_has: | |
| fn += 1 | |
| precision = tp / (tp + fp) if (tp + fp) else 0.0 | |
| recall = tp / (tp + fn) if (tp + fn) else 0.0 | |
| f1 = 2 * precision * recall / (precision + recall) if (precision + recall) else 0.0 | |
| # Only count labels that appear at least once in gold or pred across | |
| # the test set, so macro-F1 isn't diluted by labels never at play. | |
| if tp + fp + fn > 0: | |
| f1s.append(f1) | |
| return sum(f1s) / len(f1s) if f1s else 0.0 | |
| def evaluate_variant(name, predict_fn, test_data): | |
| print(f"\n=== {name} ===") | |
| overall_gold, overall_pred = [], [] | |
| per_attr_report = {} | |
| for attr in ATTR_TYPES: | |
| gold_sets = [d["labels"][attr] for d in test_data] | |
| pred_sets = [predict_fn(d["text"])[attr] for d in test_data] | |
| acc = exact_set_accuracy(gold_sets, pred_sets) | |
| p, r, f1 = micro_prf1(gold_sets, pred_sets) | |
| mf1 = macro_f1(gold_sets, pred_sets, label_space(attr)) | |
| per_attr_report[attr] = { | |
| "exact_set_accuracy": round(acc, 3), | |
| "micro_precision": round(p, 3), | |
| "micro_recall": round(r, 3), | |
| "micro_f1": round(f1, 3), | |
| "macro_f1": round(mf1, 3), | |
| } | |
| print(f"{attr:15s} acc={acc:.3f} P={p:.3f} R={r:.3f} microF1={f1:.3f} macroF1={mf1:.3f}") | |
| for g, p_ in zip(gold_sets, pred_sets): | |
| overall_gold.append([(attr, x) for x in g]) | |
| overall_pred.append([(attr, x) for x in p_]) | |
| # Flatten overall (attribute, label) pairs for a single pooled score | |
| flat_gold = [set(sum(overall_gold, []))] | |
| # (recomputed properly below per-example, the line above is unused; | |
| # kept simple: compute pooled micro-F1 across all attrs+examples) | |
| tp = fp = fn = 0 | |
| for i in range(len(test_data)): | |
| g_all, p_all = set(), set() | |
| for attr in ATTR_TYPES: | |
| g_all |= {(attr, x) for x in test_data[i]["labels"][attr]} | |
| p_all |= {(attr, x) for x in predict_fn(test_data[i]["text"])[attr]} | |
| tp += len(g_all & p_all) | |
| fp += len(p_all - g_all) | |
| fn += len(g_all - p_all) | |
| precision = tp / (tp + fp) if (tp + fp) else 0.0 | |
| recall = tp / (tp + fn) if (tp + fn) else 0.0 | |
| overall_f1 = 2 * precision * recall / (precision + recall) if (precision + recall) else 0.0 | |
| print(f"{'OVERALL':15s} P={precision:.3f} R={recall:.3f} microF1={overall_f1:.3f}") | |
| return per_attr_report, overall_f1 | |
| def main(): | |
| data = load_dataset() | |
| train, test = train_test_split(data, test_frac=0.2, seed=42) | |
| print(f"Train: {len(train)} Test: {len(test)}") | |
| vectorizer, models, binarizers = train_ml_on(train) | |
| rules_report, rules_f1 = evaluate_variant( | |
| "RULES ONLY", lambda t: extract_attributes_rules(t), test | |
| ) | |
| ml_report, ml_f1 = evaluate_variant( | |
| "ML ONLY (trained on train split)", | |
| lambda t: predict_ml_with(vectorizer, models, binarizers, t), test | |
| ) | |
| ens_report, ens_f1 = evaluate_variant( | |
| "ENSEMBLE (rules + ml)", | |
| lambda t: predict_ensemble_with(vectorizer, models, binarizers, t), test | |
| ) | |
| summary = { | |
| "n_train": len(train), | |
| "n_test": len(test), | |
| "rules_only": {"overall_micro_f1": round(rules_f1, 3), "per_attribute": rules_report}, | |
| "ml_only": {"overall_micro_f1": round(ml_f1, 3), "per_attribute": ml_report}, | |
| "ensemble": {"overall_micro_f1": round(ens_f1, 3), "per_attribute": ens_report}, | |
| } | |
| with open("eval_results.json", "w") as f: | |
| json.dump(summary, f, indent=2) | |
| print("\nSaved detailed results to eval_results.json") | |
| if __name__ == "__main__": | |
| main() | |