Download evaluate.py from spoortishankar/SentimentDetector: direct link, hf CLI and curl.
- Browser
- Download file 4.23 kB
-
https://huggingface.co/spaces/spoortishankar/SentimentDetector/resolve/main/evaluate.py
- Command line
-
hf download hf://spaces/spoortishankar/SentimentDetector/evaluate.py
-
curl -L -o evaluate.py https://huggingface.co/spaces/spoortishankar/SentimentDetector/resolve/main/evaluate.py
4.23 kB
| """ | |
| evaluate.py - measure what each component of SentimentDetector actually adds. | |
| pip install datasets scikit-learn | |
| python evaluate.py # hand-built suite + TweetEval irony/sentiment | |
| python evaluate.py --n 1000 # TweetEval sentiment sample size | |
| Thresholds were NOT tuned on the TweetEval test sets. Paste your own measured numbers | |
| into README.md - never numbers you have not run. | |
| """ | |
| import argparse | |
| import numpy as np | |
| from sklearn.metrics import accuracy_score, f1_score | |
| from detector import SARCASM_THRESHOLD, SentimentDetector, label_from_score | |
| CASES = [ | |
| ("Oh great, another Monday morning meeting. Just what I needed 🙄", "negative"), | |
| ("I absolutely love waiting 3 hours at the DMV.", "negative"), | |
| ("Thanks a lot for telling me at the last minute 🙃", "negative"), | |
| ("Yeah right, because that always works /s", "negative"), | |
| ("Wow, what a brilliant idea. Nobody has ever thought of that.", "negative"), | |
| ("Great, my laptop died right before the deadline.", "negative"), | |
| ("Nothing beats debugging at 3 AM, truly living the dream 🙃", "negative"), | |
| ("Perfect. Just perfect. My flight got cancelled.", "negative"), | |
| ("Sure, because 'customer service' is exactly what this was.", "negative"), | |
| ("I can't believe how terrible this update is 😡", "negative"), | |
| ("Ugh, the wifi is down again.", "negative"), | |
| ("Just got my acceptance letter!!! 🎉😭", "positive"), | |
| ("I'm so proud of my little brother 🥰", "positive"), | |
| ("Best pizza I've had in years 🍕😍", "positive"), | |
| ("Not bad at all, honestly impressed.", "positive"), | |
| ("lol I'm dead 💀 that was hilarious", "positive"), | |
| ("So happy for you!! Congrats 🎊", "positive"), | |
| ("I actually enjoyed the workshop, learned a lot!", "positive"), | |
| ("Wow, you finished the whole project in one night? Amazing work!", "positive"), | |
| ("I hate how much I love this song 😍", "positive"), | |
| ("The package arrived on Tuesday.", "neutral"), | |
| ("It's a meeting at 3 PM in room 204.", "neutral"), | |
| ] | |
| def run_suite(det): | |
| print(f"\n== Hand-built suite ({len(CASES)} cases; small and written by the author - indicative only) ==") | |
| configs = {"baseline (text only)": (False, False), "+ emoji": (False, True), "+ emoji + sarcasm": (True, True)} | |
| y = [c[1] for c in CASES] | |
| for name, (sarc, emo) in configs.items(): | |
| pred = [label_from_score(det.analyze(t, sarcasm_aware=sarc, emoji_aware=emo).mean) for t, _ in CASES] | |
| print(f"{name:24s} accuracy = {accuracy_score(y, pred):.2%}") | |
| if sarc and emo: | |
| for (t, gold), p in zip(CASES, pred): | |
| if gold != p: | |
| print(f" miss: {t!r} gold={gold} pred={p}") | |
| def run_tweeteval(det, n): | |
| from datasets import load_dataset # hub id: cardiffnlp/tweet_eval | |
| print("\n== TweetEval irony (test) - F1 on the irony class ==") | |
| ds = load_dataset("cardiffnlp/tweet_eval", "irony", split="test") | |
| texts, y = ds["text"], np.array(ds["label"]) | |
| base = det.irony_probs(texts) >= 0.5 | |
| fused = np.array([max(r.sarcasm for r in det.analyze(t).sentences) >= SARCASM_THRESHOLD for t in texts]) | |
| print(f"irony model alone : F1 = {f1_score(y, base):.3f}") | |
| print(f"fused sarcasm : F1 = {f1_score(y, fused):.3f}") | |
| print(f"\n== TweetEval sentiment (test sample, n={n}) - macro-F1 ==") | |
| ds = load_dataset("cardiffnlp/tweet_eval", "sentiment", split="test").shuffle(seed=0).select(range(n)) | |
| texts, y = ds["text"], np.array(ds["label"]) # 0 neg, 1 neu, 2 pos | |
| base = det.sentiment_probs(texts).argmax(1) | |
| ids = {"negative": 0, "neutral": 1, "mixed": 1, "positive": 2} | |
| full = np.array([ids[label_from_score(det.analyze(t).mean)] for t in texts]) | |
| print(f"baseline model : macro-F1 = {f1_score(y, base, average='macro'):.3f}") | |
| print(f"full pipeline : macro-F1 = {f1_score(y, full, average='macro'):.3f}") | |
| if __name__ == "__main__": | |
| ap = argparse.ArgumentParser() | |
| ap.add_argument("--n", type=int, default=1000) | |
| ap.add_argument("--skip-tweeteval", action="store_true") | |
| a = ap.parse_args() | |
| det = SentimentDetector() | |
| run_suite(det) | |
| if not a.skip_tweeteval: | |
| run_tweeteval(det, a.n) |