BalineseLanguageSummarization / src /feature_extraction.py
Brian045's picture
Upload feature_extraction.py
82c2526 verified
Raw History Blame Contribute Delete
4.4 kB
import re
import numpy as np
from sklearn.feature_extraction.text import TfidfVectorizer
from sklearn.metrics.pairwise import cosine_similarity
import nltk
# Ensure punkt is downloaded for sentence tokenization
try:
nltk.data.find('tokenizers/punkt')
except LookupError:
nltk.download('punkt')
nltk.download('punkt_tab')
class FeatureExtractor:
def __init__(self):
pass
def split_sentences(self, text):
"""
Splits text into sentences.
Using nltk's sent_tokenize which handles standard punctuation better.
"""
# Balinese text might use standard punctuation.
if not text:
return []
return nltk.sent_tokenize(text)
def calculate_features(self, title, article_text):
"""
Calculates the 6 features for each sentence in the article.
Features:
1. f1_lead: 1 if first sentence, 0 otherwise.
2. f2_judul_sim: Cosine similarity between Sentence and Title.
3. f3_freq_word: Sum of TF-IDF scores of words in sentence.
4. f4_sim_sent: Average cosine similarity to all other sentences.
5. f5_len_norm: Sentence length / Longest sentence length.
6. f6_overlap: Number of common words between Sentence and Title.
Returns:
list of dicts (one per sentence) containing features and the sentence text.
"""
sentences = self.split_sentences(article_text)
if not sentences:
return []
# Precompute TF-IDF for the whole document (sentences + title)
# We treat the Title as one document and Sentences as others for similarity calc.
corpus = [title] + sentences
vectorizer = TfidfVectorizer()
try:
tfidf_matrix = vectorizer.fit_transform(corpus)
except ValueError:
# Handle empty vocabulary or stop words issues
return []
# tfidf_matrix[0] is Title
# tfidf_matrix[1:] are Sentences
title_vector = tfidf_matrix[0]
sentence_vectors = tfidf_matrix[1:]
# 1. f1_lead
f1_lead = [1 if i == 0 else 0 for i in range(len(sentences))]
# 2. f2_judul_sim
# Compute cosine similarity between each sentence and title
# cosine_similarity returns shape (n_samples_X, n_samples_Y)
f2_judul_sim = cosine_similarity(sentence_vectors, title_vector).flatten()
# 3. f3_freq_word
# Sum of TF-IDF scores for words in the sentence
# sentence_vectors is a sparse matrix. sum(axis=1) gives sum of each row.
f3_freq_word = np.array(sentence_vectors.sum(axis=1)).flatten()
# 4. f4_sim_sent
# Average cosine similarity to all other sentences
# Compute similarity matrix between sentences
if len(sentences) > 1:
sim_matrix = cosine_similarity(sentence_vectors)
# We want average similarity to OTHERS.
# sum of similarities minus self-similarity (which is 1.0) / (n-1)
f4_sim_sent = (sim_matrix.sum(axis=1) - 1) / (len(sentences) - 1)
else:
f4_sim_sent = [0.0] * len(sentences)
# 5. f5_len_norm
# Number of words in sentence / Length of longest sentence in doc
# We use simple whitespace splitting for word count
sent_word_counts = [len(re.findall(r'\w+', s)) for s in sentences]
max_len = max(sent_word_counts) if sent_word_counts else 1
f5_len_norm = [count / max_len for count in sent_word_counts]
# 6. f6_overlap
# Number of common words between sentence and Title
title_words = set(re.findall(r'\w+', title.lower()))
f6_overlap = []
for s in sentences:
s_words = set(re.findall(r'\w+', s.lower()))
overlap_count = len(title_words.intersection(s_words))
f6_overlap.append(overlap_count)
# Combine all features
features_list = []
for i in range(len(sentences)):
features_list.append({
"teks_kalimat": sentences[i],
"f1_lead": f1_lead[i],
"f2_judul_sim": f2_judul_sim[i],
"f3_freq_word": f3_freq_word[i],
"f4_sim_sent": f4_sim_sent[i],
"f5_len_norm": f5_len_norm[i],
"f6_overlap": f6_overlap[i]
})
return features_list