import gradio as gr import pandas as pd import torch from transformers import AutoTokenizer, AutoModelForSequenceClassification from sklearn.metrics import f1_score import numpy as np import os # ---------------------------- # Load Hugging Face Model # ---------------------------- MODEL_NAME = "cardiffnlp/twitter-roberta-base-sentiment" tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME) model = AutoModelForSequenceClassification.from_pretrained(MODEL_NAME) labels = ["negative", "neutral", "positive"] # ---------------------------- # Sentiment Prediction Function # ---------------------------- def predict_sentiment(text): inputs = tokenizer(text, return_tensors="pt", truncation=True) outputs = model(**inputs) probs = torch.nn.functional.softmax(outputs.logits, dim=-1) prediction = torch.argmax(probs).item() return labels[prediction] # ---------------------------- # Main Processing Function # ---------------------------- def process_file(file): df = pd.read_csv(file.name) if "Review" not in df.columns: return "CSV must contain a 'text' column.", None df["predicted_sentiment"] = df["Review"].astype(str).apply(predict_sentiment) # Optional evaluation if ground truth exists f1 = None if "sentiment" in df.columns: f1 = f1_score( df["sentiment"], df["predicted_sentiment"], average="weighted" ) output_path = "results.csv" df.to_csv(output_path, index=False) if f1: return f"Processing complete! F1 Score: {round(f1,4)}", output_path else: return "Processing complete! No ground truth column found.", output_path # ---------------------------- # Gradio Interface # ---------------------------- interface = gr.Interface( fn=process_file, inputs=gr.File(label="Upload CSV File"), outputs=[ gr.Textbox(label="Status"), gr.File(label="Download Results") ], title="Aspect-Based Sentiment Analysis App", description="Upload a CSV file with a 'text' column. Optionally include a 'sentiment' column for evaluation." ) if __name__ == "__main__": interface.launch()