""" Evaluation harness for the attribute extraction pipeline. Metrics reported (per attribute type, and overall): - Exact-set accuracy: fraction of examples where the predicted label set for that attribute exactly equals the gold label set (strict metric -- partial credit is zero even if only one label differs). - Micro-F1: precision/recall/F1 computed over all individual label predictions pooled across every example (standard multi-label metric, give partial credit for partially-correct label sets). - Macro-F1: F1 averaged per-label then averaged across labels (surfaces performance on rare labels that micro-F1 can hide). We evaluate three variants for comparison: 1. rules-only 2. ml-only (model trained on the TRAIN split only, evaluated on TEST split) 3. ensemble (rules + ml, ml trained on TRAIN split only) Because the labeled set has 61 rows, we use an 80/20 train/test split (49 train / 12 test) with a fixed random seed for reproducibility, and also report leave-one-out cross-validated numbers for the ML model since a single 12-row test split has high variance at this scale. """ import json import random import joblib from sklearn.feature_extraction.text import TfidfVectorizer from sklearn.linear_model import LogisticRegression from sklearn.multiclass import OneVsRestClassifier from sklearn.preprocessing import MultiLabelBinarizer from lexicon import ATTRIBUTE_LEXICON, COLOR_VOCAB from rules_extractor import extract_attributes_rules ATTR_TYPES = ["silhouette", "fabric", "neckline", "sleeve", "length", "embellishment", "category", "color"] random.seed(42) def label_space(attr): if attr == "color": return COLOR_VOCAB return list(ATTRIBUTE_LEXICON[attr].keys()) def load_dataset(path="../data/dataset.json"): with open(path) as f: return json.load(f) def train_test_split(data, test_frac=0.2, seed=42): idx = list(range(len(data))) random.Random(seed).shuffle(idx) n_test = max(1, int(len(data) * test_frac)) test_idx = set(idx[:n_test]) train = [d for i, d in enumerate(data) if i not in test_idx] test = [d for i, d in enumerate(data) if i in test_idx] return train, test def train_ml_on(train_data): texts = [d["text"] for d in train_data] vectorizer = TfidfVectorizer(analyzer="char_wb", ngram_range=(2, 5), min_df=1) X = vectorizer.fit_transform(texts) models, binarizers = {}, {} for attr in ATTR_TYPES: classes = label_space(attr) mlb = MultiLabelBinarizer(classes=classes) Y = mlb.fit_transform([d["labels"][attr] for d in train_data]) if Y.sum() == 0: models[attr] = None else: clf = OneVsRestClassifier(LogisticRegression(max_iter=1000, class_weight="balanced")) clf.fit(X, Y) models[attr] = clf binarizers[attr] = mlb return vectorizer, models, binarizers def predict_ml_with(vectorizer, models, binarizers, text): X = vectorizer.transform([text]) out = {} for attr in ATTR_TYPES: clf = models[attr] if clf is None: out[attr] = [] continue y = clf.predict(X) out[attr] = list(binarizers[attr].inverse_transform(y)[0]) return out def predict_ensemble_with(vectorizer, models, binarizers, text): rule_preds = extract_attributes_rules(text) ml_preds = predict_ml_with(vectorizer, models, binarizers, text) final = {} for attr in rule_preds: merged = list(rule_preds[attr]) for label in ml_preds.get(attr, []): if label not in merged: merged.append(label) final[attr] = merged return final def exact_set_accuracy(gold_sets, pred_sets): correct = sum(1 for g, p in zip(gold_sets, pred_sets) if set(g) == set(p)) return correct / len(gold_sets) def micro_prf1(gold_sets, pred_sets): tp = fp = fn = 0 for g, p in zip(gold_sets, pred_sets): g, p = set(g), set(p) tp += len(g & p) fp += len(p - g) fn += len(g - p) precision = tp / (tp + fp) if (tp + fp) else 0.0 recall = tp / (tp + fn) if (tp + fn) else 0.0 f1 = 2 * precision * recall / (precision + recall) if (precision + recall) else 0.0 return precision, recall, f1 def macro_f1(gold_sets, pred_sets, classes): f1s = [] for c in classes: tp = fp = fn = 0 for g, p in zip(gold_sets, pred_sets): g_has, p_has = c in g, c in p if g_has and p_has: tp += 1 elif p_has and not g_has: fp += 1 elif g_has and not p_has: fn += 1 precision = tp / (tp + fp) if (tp + fp) else 0.0 recall = tp / (tp + fn) if (tp + fn) else 0.0 f1 = 2 * precision * recall / (precision + recall) if (precision + recall) else 0.0 # Only count labels that appear at least once in gold or pred across # the test set, so macro-F1 isn't diluted by labels never at play. if tp + fp + fn > 0: f1s.append(f1) return sum(f1s) / len(f1s) if f1s else 0.0 def evaluate_variant(name, predict_fn, test_data): print(f"\n=== {name} ===") overall_gold, overall_pred = [], [] per_attr_report = {} for attr in ATTR_TYPES: gold_sets = [d["labels"][attr] for d in test_data] pred_sets = [predict_fn(d["text"])[attr] for d in test_data] acc = exact_set_accuracy(gold_sets, pred_sets) p, r, f1 = micro_prf1(gold_sets, pred_sets) mf1 = macro_f1(gold_sets, pred_sets, label_space(attr)) per_attr_report[attr] = { "exact_set_accuracy": round(acc, 3), "micro_precision": round(p, 3), "micro_recall": round(r, 3), "micro_f1": round(f1, 3), "macro_f1": round(mf1, 3), } print(f"{attr:15s} acc={acc:.3f} P={p:.3f} R={r:.3f} microF1={f1:.3f} macroF1={mf1:.3f}") for g, p_ in zip(gold_sets, pred_sets): overall_gold.append([(attr, x) for x in g]) overall_pred.append([(attr, x) for x in p_]) # Flatten overall (attribute, label) pairs for a single pooled score flat_gold = [set(sum(overall_gold, []))] # (recomputed properly below per-example, the line above is unused; # kept simple: compute pooled micro-F1 across all attrs+examples) tp = fp = fn = 0 for i in range(len(test_data)): g_all, p_all = set(), set() for attr in ATTR_TYPES: g_all |= {(attr, x) for x in test_data[i]["labels"][attr]} p_all |= {(attr, x) for x in predict_fn(test_data[i]["text"])[attr]} tp += len(g_all & p_all) fp += len(p_all - g_all) fn += len(g_all - p_all) precision = tp / (tp + fp) if (tp + fp) else 0.0 recall = tp / (tp + fn) if (tp + fn) else 0.0 overall_f1 = 2 * precision * recall / (precision + recall) if (precision + recall) else 0.0 print(f"{'OVERALL':15s} P={precision:.3f} R={recall:.3f} microF1={overall_f1:.3f}") return per_attr_report, overall_f1 def main(): data = load_dataset() train, test = train_test_split(data, test_frac=0.2, seed=42) print(f"Train: {len(train)} Test: {len(test)}") vectorizer, models, binarizers = train_ml_on(train) rules_report, rules_f1 = evaluate_variant( "RULES ONLY", lambda t: extract_attributes_rules(t), test ) ml_report, ml_f1 = evaluate_variant( "ML ONLY (trained on train split)", lambda t: predict_ml_with(vectorizer, models, binarizers, t), test ) ens_report, ens_f1 = evaluate_variant( "ENSEMBLE (rules + ml)", lambda t: predict_ensemble_with(vectorizer, models, binarizers, t), test ) summary = { "n_train": len(train), "n_test": len(test), "rules_only": {"overall_micro_f1": round(rules_f1, 3), "per_attribute": rules_report}, "ml_only": {"overall_micro_f1": round(ml_f1, 3), "per_attribute": ml_report}, "ensemble": {"overall_micro_f1": round(ens_f1, 3), "per_attribute": ens_report}, } with open("eval_results.json", "w") as f: json.dump(summary, f, indent=2) print("\nSaved detailed results to eval_results.json") if __name__ == "__main__": main()