Download preprocessing_new.py from viethung21IT/toxic_comment_classification_NLP: direct link, hf CLI and curl.
- Browser
- Download file 1.57 kB
-
https://huggingface.co/spaces/viethung21IT/toxic_comment_classification_NLP/resolve/main/preprocessing_new.py
- Command line
-
hf download hf://spaces/viethung21IT/toxic_comment_classification_NLP/preprocessing_new.py
-
curl -L -o preprocessing_new.py https://huggingface.co/spaces/viethung21IT/toxic_comment_classification_NLP/resolve/main/preprocessing_new.py
1.57 kB
| import pandas as pd | |
| import re | |
| import string | |
| import emoji | |
| import nltk | |
| from nltk.corpus import stopwords | |
| from nltk.stem import SnowballStemmer | |
| nltk.download('stopwords') | |
| stop_words_set = set(stopwords.words('english')) | |
| stemmer = SnowballStemmer("english") | |
| def preprocessing_clean_text(text): | |
| if pd.isnull(text): return "" | |
| # 0. Xử lý xuống dòng | |
| text = text.replace('\n', ' ') | |
| text = text.replace('\t', ' ') | |
| # 1. Xử lý Emoji (Dịch sang tiếng Anh) | |
| text = emoji.demojize(text, delimiters=(" ", " ")) | |
| # 2. Lowercase | |
| text = str(text).lower() | |
| # 3. Xóa IP, URLS, Username | |
| text = re.sub(r'\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3}', ' ', text) | |
| text = re.sub(r'http\S+|www\S+', ' ', text) | |
| text = re.sub(r'@[A-Za-z0-9]+', ' ', text) | |
| # 4. Chuẩn hóa ký tự lặp (hateeeee -> hatee) | |
| text = re.sub(r'(.)\1{2,}', r'\1\1', text) | |
| # 5. Xử lý dấu câu: Tách ! và ? ra (What? -> What ?) | |
| text = re.sub(r'([!?])', r' \1 ', text) | |
| # 6. Xóa số | |
| text = re.sub(r'[0-9]', ' ', text) | |
| # 7. Xóa ký tự lạ (Giữ lại chữ cái, ! và ?), giữ lại _ do đây là từ sau khi xử lý emoji tạo ra emoji -> A_B_C | |
| text = re.sub(r"[^a-zA-Z!?_']", ' ', text) | |
| # 8. Xử lý khoảng trắng thừa | |
| text = re.sub(r'\s+', ' ', text).strip() | |
| word = text.split() | |
| word = [w for w in word if w not in stop_words_set] | |
| text = ' '.join(word) | |
| word = text.split() | |
| stemmed_words = [stemmer.stem(w) for w in word] | |
| text = ' '.join(stemmed_words) | |
| return text |