""" evaluate.py - measure what each component of SentimentDetector actually adds. pip install datasets scikit-learn python evaluate.py # hand-built suite + TweetEval irony/sentiment python evaluate.py --n 1000 # TweetEval sentiment sample size Thresholds were NOT tuned on the TweetEval test sets. Paste your own measured numbers into README.md - never numbers you have not run. """ import argparse import numpy as np from sklearn.metrics import accuracy_score, f1_score from detector import SARCASM_THRESHOLD, SentimentDetector, label_from_score CASES = [ ("Oh great, another Monday morning meeting. Just what I needed 🙄", "negative"), ("I absolutely love waiting 3 hours at the DMV.", "negative"), ("Thanks a lot for telling me at the last minute 🙃", "negative"), ("Yeah right, because that always works /s", "negative"), ("Wow, what a brilliant idea. Nobody has ever thought of that.", "negative"), ("Great, my laptop died right before the deadline.", "negative"), ("Nothing beats debugging at 3 AM, truly living the dream 🙃", "negative"), ("Perfect. Just perfect. My flight got cancelled.", "negative"), ("Sure, because 'customer service' is exactly what this was.", "negative"), ("I can't believe how terrible this update is 😡", "negative"), ("Ugh, the wifi is down again.", "negative"), ("Just got my acceptance letter!!! 🎉😭", "positive"), ("I'm so proud of my little brother 🥰", "positive"), ("Best pizza I've had in years 🍕😍", "positive"), ("Not bad at all, honestly impressed.", "positive"), ("lol I'm dead 💀 that was hilarious", "positive"), ("So happy for you!! Congrats 🎊", "positive"), ("I actually enjoyed the workshop, learned a lot!", "positive"), ("Wow, you finished the whole project in one night? Amazing work!", "positive"), ("I hate how much I love this song 😍", "positive"), ("The package arrived on Tuesday.", "neutral"), ("It's a meeting at 3 PM in room 204.", "neutral"), ] def run_suite(det): print(f"\n== Hand-built suite ({len(CASES)} cases; small and written by the author - indicative only) ==") configs = {"baseline (text only)": (False, False), "+ emoji": (False, True), "+ emoji + sarcasm": (True, True)} y = [c[1] for c in CASES] for name, (sarc, emo) in configs.items(): pred = [label_from_score(det.analyze(t, sarcasm_aware=sarc, emoji_aware=emo).mean) for t, _ in CASES] print(f"{name:24s} accuracy = {accuracy_score(y, pred):.2%}") if sarc and emo: for (t, gold), p in zip(CASES, pred): if gold != p: print(f" miss: {t!r} gold={gold} pred={p}") def run_tweeteval(det, n): from datasets import load_dataset # hub id: cardiffnlp/tweet_eval print("\n== TweetEval irony (test) - F1 on the irony class ==") ds = load_dataset("cardiffnlp/tweet_eval", "irony", split="test") texts, y = ds["text"], np.array(ds["label"]) base = det.irony_probs(texts) >= 0.5 fused = np.array([max(r.sarcasm for r in det.analyze(t).sentences) >= SARCASM_THRESHOLD for t in texts]) print(f"irony model alone : F1 = {f1_score(y, base):.3f}") print(f"fused sarcasm : F1 = {f1_score(y, fused):.3f}") print(f"\n== TweetEval sentiment (test sample, n={n}) - macro-F1 ==") ds = load_dataset("cardiffnlp/tweet_eval", "sentiment", split="test").shuffle(seed=0).select(range(n)) texts, y = ds["text"], np.array(ds["label"]) # 0 neg, 1 neu, 2 pos base = det.sentiment_probs(texts).argmax(1) ids = {"negative": 0, "neutral": 1, "mixed": 1, "positive": 2} full = np.array([ids[label_from_score(det.analyze(t).mean)] for t in texts]) print(f"baseline model : macro-F1 = {f1_score(y, base, average='macro'):.3f}") print(f"full pipeline : macro-F1 = {f1_score(y, full, average='macro'):.3f}") if __name__ == "__main__": ap = argparse.ArgumentParser() ap.add_argument("--n", type=int, default=1000) ap.add_argument("--skip-tweeteval", action="store_true") a = ap.parse_args() det = SentimentDetector() run_suite(det) if not a.skip_tweeteval: run_tweeteval(det, a.n)