File size: 5,689 Bytes
26c8f44
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
"""
clean.py
========
Text cleaning pipeline for comment moderation.

Handles:
  - Lowercasing
  - URL removal
  - Username/mention removal
  - HTML entity decoding
  - Repeated character normalization
  - Emoji conversion to text
  - Whitespace normalization
  - Non-printable character removal

Usage:
    python preprocessing/clean.py              # Clean all raw CSVs
    python -c "from preprocessing.clean import clean_text; print(clean_text('Fuuuuck @user http://x.com 🀑'))"
"""

import re
import html
import unicodedata
from typing import Optional

import emoji


# ---------------------------------------------------------------------------
# Compiled regex patterns (precompiled for speed)
# ---------------------------------------------------------------------------

# URLs: http(s), www, or bare domain patterns
RE_URL = re.compile(
    r"https?://\S+|www\.\S+|[\w.-]+\.(?:com|org|net|io|co|me|gov|edu)\S*",
    re.IGNORECASE,
)

# @mentions and #hashtags
RE_MENTION = re.compile(r"@\w+")
RE_HASHTAG_SYMBOL = re.compile(r"#(\w+)")  # Keep the word, remove #

# HTML tags
RE_HTML_TAG = re.compile(r"<[^>]+>")

# Repeated characters: 3+ of the same char β†’ 2
RE_REPEATED_CHARS = re.compile(r"(.)\1{2,}")

# Repeated words: "haha haha haha" β†’ "haha"
RE_REPEATED_WORDS = re.compile(r"\b(\w+)(?:\s+\1){2,}\b", re.IGNORECASE)

# Multiple spaces / tabs
RE_MULTI_SPACE = re.compile(r"\s+")

# Newlines (normalize to space for single-line processing)
RE_NEWLINES = re.compile(r"[\r\n]+")

# Non-printable / control characters (keep newlines, tabs)
RE_NON_PRINTABLE = re.compile(r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f-\x9f]")


# ---------------------------------------------------------------------------
# Core cleaning function
# ---------------------------------------------------------------------------

def clean_text(text: Optional[str]) -> str:
    """
    Apply the full cleaning pipeline to a single text string.

    Args:
        text: Raw input text.

    Returns:
        Cleaned text string.
    """
    if not text or not isinstance(text, str):
        return ""

    # 1. Decode HTML entities: &amp; β†’ &, &lt; β†’ <, etc.
    text = html.unescape(text)

    # 2. Remove HTML tags
    text = RE_HTML_TAG.sub(" ", text)

    # 3. Normalize Unicode (NFC form)
    text = unicodedata.normalize("NFC", text)

    # 4. Remove non-printable characters
    text = RE_NON_PRINTABLE.sub("", text)

    # 5. Remove URLs
    text = RE_URL.sub(" ", text)

    # 6. Remove @mentions
    text = RE_MENTION.sub(" ", text)

    # 7. Convert #hashtags β†’ keep word
    text = RE_HASHTAG_SYMBOL.sub(r"\1", text)

    # 8. Convert emojis to text descriptors
    text = emoji.demojize(text, delimiters=(" ", " "))

    # 9. Lowercase
    text = text.lower()

    # 10. Normalize repeated characters: "fuuuuck" β†’ "fuuck"
    text = RE_REPEATED_CHARS.sub(r"\1\1", text)

    # 11. Normalize repeated words: "ha ha ha ha" β†’ "ha ha"
    text = RE_REPEATED_WORDS.sub(r"\1", text)

    # 12. Normalize newlines to spaces
    text = RE_NEWLINES.sub(" ", text)

    # 13. Normalize whitespace
    text = RE_MULTI_SPACE.sub(" ", text)

    # 14. Strip
    text = text.strip()

    return text


# ---------------------------------------------------------------------------
# Batch processing (apply to CSV files)
# ---------------------------------------------------------------------------

def clean_dataframe(df, text_column: str = "text"):
    """
    Apply clean_text to a DataFrame's text column in-place.

    Args:
        df: pandas DataFrame.
        text_column: Name of the column containing text.

    Returns:
        DataFrame with cleaned text.
    """
    import pandas as pd

    df = df.copy()
    df[text_column] = df[text_column].apply(clean_text)

    # Remove empty texts after cleaning
    df = df[df[text_column].str.len() > 0]

    # Remove exact duplicates
    df = df.drop_duplicates(subset=[text_column], keep="first")

    return df.reset_index(drop=True)


# ---------------------------------------------------------------------------
# CLI entry point
# ---------------------------------------------------------------------------

def main():
    """Clean all raw CSV files and save to data/processed/."""
    import logging
    from pathlib import Path
    import pandas as pd

    logging.basicConfig(level=logging.INFO, format="%(asctime)s | %(levelname)s | %(message)s")
    logger = logging.getLogger(__name__)

    project_root = Path(__file__).resolve().parent.parent
    raw_dir = project_root / "data" / "raw"
    processed_dir = project_root / "data" / "processed"
    processed_dir.mkdir(parents=True, exist_ok=True)

    for csv_file in raw_dir.glob("*.csv"):
        logger.info(f"Cleaning {csv_file.name}...")
        df = pd.read_csv(csv_file)

        # Determine text column
        text_col = "text"
        if text_col not in df.columns:
            # Try common alternatives
            for alt in ["comment_text", "tweet", "content", "post"]:
                if alt in df.columns:
                    df = df.rename(columns={alt: "text"})
                    text_col = "text"
                    break

        if text_col not in df.columns:
            logger.warning(f"  No text column found in {csv_file.name}, skipping.")
            continue

        before = len(df)
        df = clean_dataframe(df, text_column=text_col)
        after = len(df)

        output_path = processed_dir / f"cleaned_{csv_file.name}"
        df.to_csv(output_path, index=False)
        logger.info(f"  {before:,} β†’ {after:,} rows ({before - after:,} removed)")
        logger.info(f"  Saved to {output_path}")


if __name__ == "__main__":
    main()