""" clean.py ======== Text cleaning pipeline for comment moderation. Handles: - Lowercasing - URL removal - Username/mention removal - HTML entity decoding - Repeated character normalization - Emoji conversion to text - Whitespace normalization - Non-printable character removal Usage: python preprocessing/clean.py # Clean all raw CSVs python -c "from preprocessing.clean import clean_text; print(clean_text('Fuuuuck @user http://x.com 🤡'))" """ import re import html import unicodedata from typing import Optional import emoji # --------------------------------------------------------------------------- # Compiled regex patterns (precompiled for speed) # --------------------------------------------------------------------------- # URLs: http(s), www, or bare domain patterns RE_URL = re.compile( r"https?://\S+|www\.\S+|[\w.-]+\.(?:com|org|net|io|co|me|gov|edu)\S*", re.IGNORECASE, ) # @mentions and #hashtags RE_MENTION = re.compile(r"@\w+") RE_HASHTAG_SYMBOL = re.compile(r"#(\w+)") # Keep the word, remove # # HTML tags RE_HTML_TAG = re.compile(r"<[^>]+>") # Repeated characters: 3+ of the same char → 2 RE_REPEATED_CHARS = re.compile(r"(.)\1{2,}") # Repeated words: "haha haha haha" → "haha" RE_REPEATED_WORDS = re.compile(r"\b(\w+)(?:\s+\1){2,}\b", re.IGNORECASE) # Multiple spaces / tabs RE_MULTI_SPACE = re.compile(r"\s+") # Newlines (normalize to space for single-line processing) RE_NEWLINES = re.compile(r"[\r\n]+") # Non-printable / control characters (keep newlines, tabs) RE_NON_PRINTABLE = re.compile(r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f-\x9f]") # --------------------------------------------------------------------------- # Core cleaning function # --------------------------------------------------------------------------- def clean_text(text: Optional[str]) -> str: """ Apply the full cleaning pipeline to a single text string. Args: text: Raw input text. Returns: Cleaned text string. """ if not text or not isinstance(text, str): return "" # 1. Decode HTML entities: & → &, < → <, etc. text = html.unescape(text) # 2. Remove HTML tags text = RE_HTML_TAG.sub(" ", text) # 3. Normalize Unicode (NFC form) text = unicodedata.normalize("NFC", text) # 4. Remove non-printable characters text = RE_NON_PRINTABLE.sub("", text) # 5. Remove URLs text = RE_URL.sub(" ", text) # 6. Remove @mentions text = RE_MENTION.sub(" ", text) # 7. Convert #hashtags → keep word text = RE_HASHTAG_SYMBOL.sub(r"\1", text) # 8. Convert emojis to text descriptors text = emoji.demojize(text, delimiters=(" ", " ")) # 9. Lowercase text = text.lower() # 10. Normalize repeated characters: "fuuuuck" → "fuuck" text = RE_REPEATED_CHARS.sub(r"\1\1", text) # 11. Normalize repeated words: "ha ha ha ha" → "ha ha" text = RE_REPEATED_WORDS.sub(r"\1", text) # 12. Normalize newlines to spaces text = RE_NEWLINES.sub(" ", text) # 13. Normalize whitespace text = RE_MULTI_SPACE.sub(" ", text) # 14. Strip text = text.strip() return text # --------------------------------------------------------------------------- # Batch processing (apply to CSV files) # --------------------------------------------------------------------------- def clean_dataframe(df, text_column: str = "text"): """ Apply clean_text to a DataFrame's text column in-place. Args: df: pandas DataFrame. text_column: Name of the column containing text. Returns: DataFrame with cleaned text. """ import pandas as pd df = df.copy() df[text_column] = df[text_column].apply(clean_text) # Remove empty texts after cleaning df = df[df[text_column].str.len() > 0] # Remove exact duplicates df = df.drop_duplicates(subset=[text_column], keep="first") return df.reset_index(drop=True) # --------------------------------------------------------------------------- # CLI entry point # --------------------------------------------------------------------------- def main(): """Clean all raw CSV files and save to data/processed/.""" import logging from pathlib import Path import pandas as pd logging.basicConfig(level=logging.INFO, format="%(asctime)s | %(levelname)s | %(message)s") logger = logging.getLogger(__name__) project_root = Path(__file__).resolve().parent.parent raw_dir = project_root / "data" / "raw" processed_dir = project_root / "data" / "processed" processed_dir.mkdir(parents=True, exist_ok=True) for csv_file in raw_dir.glob("*.csv"): logger.info(f"Cleaning {csv_file.name}...") df = pd.read_csv(csv_file) # Determine text column text_col = "text" if text_col not in df.columns: # Try common alternatives for alt in ["comment_text", "tweet", "content", "post"]: if alt in df.columns: df = df.rename(columns={alt: "text"}) text_col = "text" break if text_col not in df.columns: logger.warning(f" No text column found in {csv_file.name}, skipping.") continue before = len(df) df = clean_dataframe(df, text_column=text_col) after = len(df) output_path = processed_dir / f"cleaned_{csv_file.name}" df.to_csv(output_path, index=False) logger.info(f" {before:,} → {after:,} rows ({before - after:,} removed)") logger.info(f" Saved to {output_path}") if __name__ == "__main__": main()