Spaces:
Running on Zero
Running on Zero
Download preprocessing/clean.py from heatherlead/comment_moderation: direct link, hf CLI and curl.
- Browser
- Download file 5.69 kB
-
https://huggingface.co/spaces/heatherlead/comment_moderation/resolve/main/preprocessing/clean.py
- Command line
-
hf download hf://spaces/heatherlead/comment_moderation/preprocessing/clean.py
-
curl -L -o clean.py https://huggingface.co/spaces/heatherlead/comment_moderation/resolve/main/preprocessing/clean.py
5.69 kB
| """ | |
| clean.py | |
| ======== | |
| Text cleaning pipeline for comment moderation. | |
| Handles: | |
| - Lowercasing | |
| - URL removal | |
| - Username/mention removal | |
| - HTML entity decoding | |
| - Repeated character normalization | |
| - Emoji conversion to text | |
| - Whitespace normalization | |
| - Non-printable character removal | |
| Usage: | |
| python preprocessing/clean.py # Clean all raw CSVs | |
| python -c "from preprocessing.clean import clean_text; print(clean_text('Fuuuuck @user http://x.com 🤡'))" | |
| """ | |
| import re | |
| import html | |
| import unicodedata | |
| from typing import Optional | |
| import emoji | |
| # --------------------------------------------------------------------------- | |
| # Compiled regex patterns (precompiled for speed) | |
| # --------------------------------------------------------------------------- | |
| # URLs: http(s), www, or bare domain patterns | |
| RE_URL = re.compile( | |
| r"https?://\S+|www\.\S+|[\w.-]+\.(?:com|org|net|io|co|me|gov|edu)\S*", | |
| re.IGNORECASE, | |
| ) | |
| # @mentions and #hashtags | |
| RE_MENTION = re.compile(r"@\w+") | |
| RE_HASHTAG_SYMBOL = re.compile(r"#(\w+)") # Keep the word, remove # | |
| # HTML tags | |
| RE_HTML_TAG = re.compile(r"<[^>]+>") | |
| # Repeated characters: 3+ of the same char → 2 | |
| RE_REPEATED_CHARS = re.compile(r"(.)\1{2,}") | |
| # Repeated words: "haha haha haha" → "haha" | |
| RE_REPEATED_WORDS = re.compile(r"\b(\w+)(?:\s+\1){2,}\b", re.IGNORECASE) | |
| # Multiple spaces / tabs | |
| RE_MULTI_SPACE = re.compile(r"\s+") | |
| # Newlines (normalize to space for single-line processing) | |
| RE_NEWLINES = re.compile(r"[\r\n]+") | |
| # Non-printable / control characters (keep newlines, tabs) | |
| RE_NON_PRINTABLE = re.compile(r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f-\x9f]") | |
| # --------------------------------------------------------------------------- | |
| # Core cleaning function | |
| # --------------------------------------------------------------------------- | |
| def clean_text(text: Optional[str]) -> str: | |
| """ | |
| Apply the full cleaning pipeline to a single text string. | |
| Args: | |
| text: Raw input text. | |
| Returns: | |
| Cleaned text string. | |
| """ | |
| if not text or not isinstance(text, str): | |
| return "" | |
| # 1. Decode HTML entities: & → &, < → <, etc. | |
| text = html.unescape(text) | |
| # 2. Remove HTML tags | |
| text = RE_HTML_TAG.sub(" ", text) | |
| # 3. Normalize Unicode (NFC form) | |
| text = unicodedata.normalize("NFC", text) | |
| # 4. Remove non-printable characters | |
| text = RE_NON_PRINTABLE.sub("", text) | |
| # 5. Remove URLs | |
| text = RE_URL.sub(" ", text) | |
| # 6. Remove @mentions | |
| text = RE_MENTION.sub(" ", text) | |
| # 7. Convert #hashtags → keep word | |
| text = RE_HASHTAG_SYMBOL.sub(r"\1", text) | |
| # 8. Convert emojis to text descriptors | |
| text = emoji.demojize(text, delimiters=(" ", " ")) | |
| # 9. Lowercase | |
| text = text.lower() | |
| # 10. Normalize repeated characters: "fuuuuck" → "fuuck" | |
| text = RE_REPEATED_CHARS.sub(r"\1\1", text) | |
| # 11. Normalize repeated words: "ha ha ha ha" → "ha ha" | |
| text = RE_REPEATED_WORDS.sub(r"\1", text) | |
| # 12. Normalize newlines to spaces | |
| text = RE_NEWLINES.sub(" ", text) | |
| # 13. Normalize whitespace | |
| text = RE_MULTI_SPACE.sub(" ", text) | |
| # 14. Strip | |
| text = text.strip() | |
| return text | |
| # --------------------------------------------------------------------------- | |
| # Batch processing (apply to CSV files) | |
| # --------------------------------------------------------------------------- | |
| def clean_dataframe(df, text_column: str = "text"): | |
| """ | |
| Apply clean_text to a DataFrame's text column in-place. | |
| Args: | |
| df: pandas DataFrame. | |
| text_column: Name of the column containing text. | |
| Returns: | |
| DataFrame with cleaned text. | |
| """ | |
| import pandas as pd | |
| df = df.copy() | |
| df[text_column] = df[text_column].apply(clean_text) | |
| # Remove empty texts after cleaning | |
| df = df[df[text_column].str.len() > 0] | |
| # Remove exact duplicates | |
| df = df.drop_duplicates(subset=[text_column], keep="first") | |
| return df.reset_index(drop=True) | |
| # --------------------------------------------------------------------------- | |
| # CLI entry point | |
| # --------------------------------------------------------------------------- | |
| def main(): | |
| """Clean all raw CSV files and save to data/processed/.""" | |
| import logging | |
| from pathlib import Path | |
| import pandas as pd | |
| logging.basicConfig(level=logging.INFO, format="%(asctime)s | %(levelname)s | %(message)s") | |
| logger = logging.getLogger(__name__) | |
| project_root = Path(__file__).resolve().parent.parent | |
| raw_dir = project_root / "data" / "raw" | |
| processed_dir = project_root / "data" / "processed" | |
| processed_dir.mkdir(parents=True, exist_ok=True) | |
| for csv_file in raw_dir.glob("*.csv"): | |
| logger.info(f"Cleaning {csv_file.name}...") | |
| df = pd.read_csv(csv_file) | |
| # Determine text column | |
| text_col = "text" | |
| if text_col not in df.columns: | |
| # Try common alternatives | |
| for alt in ["comment_text", "tweet", "content", "post"]: | |
| if alt in df.columns: | |
| df = df.rename(columns={alt: "text"}) | |
| text_col = "text" | |
| break | |
| if text_col not in df.columns: | |
| logger.warning(f" No text column found in {csv_file.name}, skipping.") | |
| continue | |
| before = len(df) | |
| df = clean_dataframe(df, text_column=text_col) | |
| after = len(df) | |
| output_path = processed_dir / f"cleaned_{csv_file.name}" | |
| df.to_csv(output_path, index=False) | |
| logger.info(f" {before:,} → {after:,} rows ({before - after:,} removed)") | |
| logger.info(f" Saved to {output_path}") | |
| if __name__ == "__main__": | |
| main() | |