heatherlead's picture
Add moderation backend with LFS tracking for large files
26c8f44
Raw History Blame Contribute Delete
5.69 kB
"""
clean.py
========
Text cleaning pipeline for comment moderation.
Handles:
- Lowercasing
- URL removal
- Username/mention removal
- HTML entity decoding
- Repeated character normalization
- Emoji conversion to text
- Whitespace normalization
- Non-printable character removal
Usage:
python preprocessing/clean.py # Clean all raw CSVs
python -c "from preprocessing.clean import clean_text; print(clean_text('Fuuuuck @user http://x.com 🤡'))"
"""
import re
import html
import unicodedata
from typing import Optional
import emoji
# ---------------------------------------------------------------------------
# Compiled regex patterns (precompiled for speed)
# ---------------------------------------------------------------------------
# URLs: http(s), www, or bare domain patterns
RE_URL = re.compile(
r"https?://\S+|www\.\S+|[\w.-]+\.(?:com|org|net|io|co|me|gov|edu)\S*",
re.IGNORECASE,
)
# @mentions and #hashtags
RE_MENTION = re.compile(r"@\w+")
RE_HASHTAG_SYMBOL = re.compile(r"#(\w+)") # Keep the word, remove #
# HTML tags
RE_HTML_TAG = re.compile(r"<[^>]+>")
# Repeated characters: 3+ of the same char → 2
RE_REPEATED_CHARS = re.compile(r"(.)\1{2,}")
# Repeated words: "haha haha haha" → "haha"
RE_REPEATED_WORDS = re.compile(r"\b(\w+)(?:\s+\1){2,}\b", re.IGNORECASE)
# Multiple spaces / tabs
RE_MULTI_SPACE = re.compile(r"\s+")
# Newlines (normalize to space for single-line processing)
RE_NEWLINES = re.compile(r"[\r\n]+")
# Non-printable / control characters (keep newlines, tabs)
RE_NON_PRINTABLE = re.compile(r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f-\x9f]")
# ---------------------------------------------------------------------------
# Core cleaning function
# ---------------------------------------------------------------------------
def clean_text(text: Optional[str]) -> str:
"""
Apply the full cleaning pipeline to a single text string.
Args:
text: Raw input text.
Returns:
Cleaned text string.
"""
if not text or not isinstance(text, str):
return ""
# 1. Decode HTML entities: &amp; → &, &lt; → <, etc.
text = html.unescape(text)
# 2. Remove HTML tags
text = RE_HTML_TAG.sub(" ", text)
# 3. Normalize Unicode (NFC form)
text = unicodedata.normalize("NFC", text)
# 4. Remove non-printable characters
text = RE_NON_PRINTABLE.sub("", text)
# 5. Remove URLs
text = RE_URL.sub(" ", text)
# 6. Remove @mentions
text = RE_MENTION.sub(" ", text)
# 7. Convert #hashtags → keep word
text = RE_HASHTAG_SYMBOL.sub(r"\1", text)
# 8. Convert emojis to text descriptors
text = emoji.demojize(text, delimiters=(" ", " "))
# 9. Lowercase
text = text.lower()
# 10. Normalize repeated characters: "fuuuuck" → "fuuck"
text = RE_REPEATED_CHARS.sub(r"\1\1", text)
# 11. Normalize repeated words: "ha ha ha ha" → "ha ha"
text = RE_REPEATED_WORDS.sub(r"\1", text)
# 12. Normalize newlines to spaces
text = RE_NEWLINES.sub(" ", text)
# 13. Normalize whitespace
text = RE_MULTI_SPACE.sub(" ", text)
# 14. Strip
text = text.strip()
return text
# ---------------------------------------------------------------------------
# Batch processing (apply to CSV files)
# ---------------------------------------------------------------------------
def clean_dataframe(df, text_column: str = "text"):
"""
Apply clean_text to a DataFrame's text column in-place.
Args:
df: pandas DataFrame.
text_column: Name of the column containing text.
Returns:
DataFrame with cleaned text.
"""
import pandas as pd
df = df.copy()
df[text_column] = df[text_column].apply(clean_text)
# Remove empty texts after cleaning
df = df[df[text_column].str.len() > 0]
# Remove exact duplicates
df = df.drop_duplicates(subset=[text_column], keep="first")
return df.reset_index(drop=True)
# ---------------------------------------------------------------------------
# CLI entry point
# ---------------------------------------------------------------------------
def main():
"""Clean all raw CSV files and save to data/processed/."""
import logging
from pathlib import Path
import pandas as pd
logging.basicConfig(level=logging.INFO, format="%(asctime)s | %(levelname)s | %(message)s")
logger = logging.getLogger(__name__)
project_root = Path(__file__).resolve().parent.parent
raw_dir = project_root / "data" / "raw"
processed_dir = project_root / "data" / "processed"
processed_dir.mkdir(parents=True, exist_ok=True)
for csv_file in raw_dir.glob("*.csv"):
logger.info(f"Cleaning {csv_file.name}...")
df = pd.read_csv(csv_file)
# Determine text column
text_col = "text"
if text_col not in df.columns:
# Try common alternatives
for alt in ["comment_text", "tweet", "content", "post"]:
if alt in df.columns:
df = df.rename(columns={alt: "text"})
text_col = "text"
break
if text_col not in df.columns:
logger.warning(f" No text column found in {csv_file.name}, skipping.")
continue
before = len(df)
df = clean_dataframe(df, text_column=text_col)
after = len(df)
output_path = processed_dir / f"cleaned_{csv_file.name}"
df.to_csv(output_path, index=False)
logger.info(f" {before:,} → {after:,} rows ({before - after:,} removed)")
logger.info(f" Saved to {output_path}")
if __name__ == "__main__":
main()