import re import numpy as np from sklearn.feature_extraction.text import TfidfVectorizer from sklearn.metrics.pairwise import cosine_similarity import nltk # Ensure punkt is downloaded for sentence tokenization try: nltk.data.find('tokenizers/punkt') except LookupError: nltk.download('punkt') nltk.download('punkt_tab') class FeatureExtractor: def __init__(self): pass def split_sentences(self, text): """ Splits text into sentences. Using nltk's sent_tokenize which handles standard punctuation better. """ # Balinese text might use standard punctuation. if not text: return [] return nltk.sent_tokenize(text) def calculate_features(self, title, article_text): """ Calculates the 6 features for each sentence in the article. Features: 1. f1_lead: 1 if first sentence, 0 otherwise. 2. f2_judul_sim: Cosine similarity between Sentence and Title. 3. f3_freq_word: Sum of TF-IDF scores of words in sentence. 4. f4_sim_sent: Average cosine similarity to all other sentences. 5. f5_len_norm: Sentence length / Longest sentence length. 6. f6_overlap: Number of common words between Sentence and Title. Returns: list of dicts (one per sentence) containing features and the sentence text. """ sentences = self.split_sentences(article_text) if not sentences: return [] # Precompute TF-IDF for the whole document (sentences + title) # We treat the Title as one document and Sentences as others for similarity calc. corpus = [title] + sentences vectorizer = TfidfVectorizer() try: tfidf_matrix = vectorizer.fit_transform(corpus) except ValueError: # Handle empty vocabulary or stop words issues return [] # tfidf_matrix[0] is Title # tfidf_matrix[1:] are Sentences title_vector = tfidf_matrix[0] sentence_vectors = tfidf_matrix[1:] # 1. f1_lead f1_lead = [1 if i == 0 else 0 for i in range(len(sentences))] # 2. f2_judul_sim # Compute cosine similarity between each sentence and title # cosine_similarity returns shape (n_samples_X, n_samples_Y) f2_judul_sim = cosine_similarity(sentence_vectors, title_vector).flatten() # 3. f3_freq_word # Sum of TF-IDF scores for words in the sentence # sentence_vectors is a sparse matrix. sum(axis=1) gives sum of each row. f3_freq_word = np.array(sentence_vectors.sum(axis=1)).flatten() # 4. f4_sim_sent # Average cosine similarity to all other sentences # Compute similarity matrix between sentences if len(sentences) > 1: sim_matrix = cosine_similarity(sentence_vectors) # We want average similarity to OTHERS. # sum of similarities minus self-similarity (which is 1.0) / (n-1) f4_sim_sent = (sim_matrix.sum(axis=1) - 1) / (len(sentences) - 1) else: f4_sim_sent = [0.0] * len(sentences) # 5. f5_len_norm # Number of words in sentence / Length of longest sentence in doc # We use simple whitespace splitting for word count sent_word_counts = [len(re.findall(r'\w+', s)) for s in sentences] max_len = max(sent_word_counts) if sent_word_counts else 1 f5_len_norm = [count / max_len for count in sent_word_counts] # 6. f6_overlap # Number of common words between sentence and Title title_words = set(re.findall(r'\w+', title.lower())) f6_overlap = [] for s in sentences: s_words = set(re.findall(r'\w+', s.lower())) overlap_count = len(title_words.intersection(s_words)) f6_overlap.append(overlap_count) # Combine all features features_list = [] for i in range(len(sentences)): features_list.append({ "teks_kalimat": sentences[i], "f1_lead": f1_lead[i], "f2_judul_sim": f2_judul_sim[i], "f3_freq_word": f3_freq_word[i], "f4_sim_sent": f4_sim_sent[i], "f5_len_norm": f5_len_norm[i], "f6_overlap": f6_overlap[i] }) return features_list