Spaces:
Runtime error
Runtime error
Download modules/preprocessing.py from Pro-Coder/Sentiment_Analysis: direct link, hf CLI and curl.
- Browser
- Download file 935 Bytes
-
https://huggingface.co/spaces/Pro-Coder/Sentiment_Analysis/resolve/b47eaba69821c52088edbfbf449d3e367298ea24/modules/preprocessing.py
- Command line
-
hf download hf://spaces/Pro-Coder/Sentiment_Analysis@b47eaba69821c52088edbfbf449d3e367298ea24/modules/preprocessing.py
-
curl -L -o preprocessing.py https://huggingface.co/spaces/Pro-Coder/Sentiment_Analysis/resolve/b47eaba69821c52088edbfbf449d3e367298ea24/modules/preprocessing.py
935 Bytes
| # modules/preprocessing.py | |
| import re | |
| import nltk | |
| # Download stopwords if not already present | |
| nltk.download("stopwords", quiet=True) | |
| from nltk.corpus import stopwords | |
| STOPWORDS = set(stopwords.words("english")) | |
| def clean_text(text: str) -> str: | |
| """ | |
| Basic text cleaning: | |
| - Lowercase | |
| - Remove URLs, mentions, hashtags | |
| - Remove numbers, punctuation, and stopwords | |
| """ | |
| text = text.lower() | |
| # Remove URLs | |
| text = re.sub(r"http\S+|www\S+|https\S+", "", text) | |
| # Remove mentions and hashtags | |
| text = re.sub(r"@\w+|#\w+", "", text) | |
| # Remove numbers and special characters | |
| text = re.sub(r"[^a-z\s]", "", text) | |
| # Remove stopwords | |
| tokens = [word for word in text.split() if word not in STOPWORDS] | |
| return " ".join(tokens) | |
| def preprocess_texts(texts: list) -> list: | |
| """ | |
| Clean and preprocess a list of texts. | |
| """ | |
| return [clean_text(t) for t in texts if t.strip()] | |