BioIncidentRNN / preprocessing.py
wittyicon's picture
Upload 9 files
539d30b verified
Raw
History Blame Contribute Delete
1.91 kB
"""
Preprocessing pipeline — must mirror 'Data Preprocessing 3.ipynb' exactly,
so that inference-time tokenization matches what the model was trained on.
"""
import re
import pickle
import spacy
MAX_SENTENCES = 15
MAX_WORDS_SENT = 50
_nlp = None
def get_nlp():
"""Lazily load and cache the spaCy pipeline (sentence splitting only)."""
global _nlp
if _nlp is None:
_nlp = spacy.load("en_core_web_sm", disable=["ner", "parser"])
_nlp.add_pipe("sentencizer")
return _nlp
def clean_text(text: str) -> str:
t = str(text).lower()
t = re.sub(r"<[^>]+>", " ", t)
t = re.sub(r"http\S+|www\.\S+", " ", t)
t = re.sub(r"\b\d{5,}\b", " NUM ", t)
t = re.sub(r"[^a-z0-9\s\-\./]", " ", t)
t = re.sub(r"\s+", " ", t).strip()
return t
def tokenize(text: str, max_sentences=MAX_SENTENCES, max_words=MAX_WORDS_SENT):
"""Clean -> sentence split -> word tokenize. Returns list[list[str]]."""
cleaned = clean_text(text)
nlp = get_nlp()
doc = nlp(cleaned)
sents = []
for sent in doc.sents:
words = [t.text for t in sent if not t.is_space and len(t.text) > 1]
if words:
sents.append(words[:max_words])
return sents[:max_sentences]
def numericalize(tokenized_doc, word2idx, max_sentences=MAX_SENTENCES, max_words=MAX_WORDS_SENT):
"""Convert tokenized doc (list[list[str]]) -> list[list[int]] using word2idx (UNK=1)."""
return [
[word2idx.get(w, 1) for w in sent[:max_words]]
for sent in tokenized_doc[:max_sentences]
]
def load_artifacts(model_dir="."):
with open(f"{model_dir}/word2idx.pkl", "rb") as f:
word2idx = pickle.load(f)
with open(f"{model_dir}/idx2word.pkl", "rb") as f:
idx2word = pickle.load(f)
with open(f"{model_dir}/label_encoder.pkl", "rb") as f:
label_encoder = pickle.load(f)
return word2idx, idx2word, label_encoder