Download src/feature_extraction.py from Brian045/BalineseLanguageSummarization: direct link, hf CLI and curl.
- Browser
- Download file 4.4 kB
-
https://huggingface.co/spaces/Brian045/BalineseLanguageSummarization/resolve/main/src/feature_extraction.py
- Command line
-
hf download hf://spaces/Brian045/BalineseLanguageSummarization/src/feature_extraction.py
-
curl -L -o feature_extraction.py https://huggingface.co/spaces/Brian045/BalineseLanguageSummarization/resolve/main/src/feature_extraction.py
4.4 kB
| import re | |
| import numpy as np | |
| from sklearn.feature_extraction.text import TfidfVectorizer | |
| from sklearn.metrics.pairwise import cosine_similarity | |
| import nltk | |
| # Ensure punkt is downloaded for sentence tokenization | |
| try: | |
| nltk.data.find('tokenizers/punkt') | |
| except LookupError: | |
| nltk.download('punkt') | |
| nltk.download('punkt_tab') | |
| class FeatureExtractor: | |
| def __init__(self): | |
| pass | |
| def split_sentences(self, text): | |
| """ | |
| Splits text into sentences. | |
| Using nltk's sent_tokenize which handles standard punctuation better. | |
| """ | |
| # Balinese text might use standard punctuation. | |
| if not text: | |
| return [] | |
| return nltk.sent_tokenize(text) | |
| def calculate_features(self, title, article_text): | |
| """ | |
| Calculates the 6 features for each sentence in the article. | |
| Features: | |
| 1. f1_lead: 1 if first sentence, 0 otherwise. | |
| 2. f2_judul_sim: Cosine similarity between Sentence and Title. | |
| 3. f3_freq_word: Sum of TF-IDF scores of words in sentence. | |
| 4. f4_sim_sent: Average cosine similarity to all other sentences. | |
| 5. f5_len_norm: Sentence length / Longest sentence length. | |
| 6. f6_overlap: Number of common words between Sentence and Title. | |
| Returns: | |
| list of dicts (one per sentence) containing features and the sentence text. | |
| """ | |
| sentences = self.split_sentences(article_text) | |
| if not sentences: | |
| return [] | |
| # Precompute TF-IDF for the whole document (sentences + title) | |
| # We treat the Title as one document and Sentences as others for similarity calc. | |
| corpus = [title] + sentences | |
| vectorizer = TfidfVectorizer() | |
| try: | |
| tfidf_matrix = vectorizer.fit_transform(corpus) | |
| except ValueError: | |
| # Handle empty vocabulary or stop words issues | |
| return [] | |
| # tfidf_matrix[0] is Title | |
| # tfidf_matrix[1:] are Sentences | |
| title_vector = tfidf_matrix[0] | |
| sentence_vectors = tfidf_matrix[1:] | |
| # 1. f1_lead | |
| f1_lead = [1 if i == 0 else 0 for i in range(len(sentences))] | |
| # 2. f2_judul_sim | |
| # Compute cosine similarity between each sentence and title | |
| # cosine_similarity returns shape (n_samples_X, n_samples_Y) | |
| f2_judul_sim = cosine_similarity(sentence_vectors, title_vector).flatten() | |
| # 3. f3_freq_word | |
| # Sum of TF-IDF scores for words in the sentence | |
| # sentence_vectors is a sparse matrix. sum(axis=1) gives sum of each row. | |
| f3_freq_word = np.array(sentence_vectors.sum(axis=1)).flatten() | |
| # 4. f4_sim_sent | |
| # Average cosine similarity to all other sentences | |
| # Compute similarity matrix between sentences | |
| if len(sentences) > 1: | |
| sim_matrix = cosine_similarity(sentence_vectors) | |
| # We want average similarity to OTHERS. | |
| # sum of similarities minus self-similarity (which is 1.0) / (n-1) | |
| f4_sim_sent = (sim_matrix.sum(axis=1) - 1) / (len(sentences) - 1) | |
| else: | |
| f4_sim_sent = [0.0] * len(sentences) | |
| # 5. f5_len_norm | |
| # Number of words in sentence / Length of longest sentence in doc | |
| # We use simple whitespace splitting for word count | |
| sent_word_counts = [len(re.findall(r'\w+', s)) for s in sentences] | |
| max_len = max(sent_word_counts) if sent_word_counts else 1 | |
| f5_len_norm = [count / max_len for count in sent_word_counts] | |
| # 6. f6_overlap | |
| # Number of common words between sentence and Title | |
| title_words = set(re.findall(r'\w+', title.lower())) | |
| f6_overlap = [] | |
| for s in sentences: | |
| s_words = set(re.findall(r'\w+', s.lower())) | |
| overlap_count = len(title_words.intersection(s_words)) | |
| f6_overlap.append(overlap_count) | |
| # Combine all features | |
| features_list = [] | |
| for i in range(len(sentences)): | |
| features_list.append({ | |
| "teks_kalimat": sentences[i], | |
| "f1_lead": f1_lead[i], | |
| "f2_judul_sim": f2_judul_sim[i], | |
| "f3_freq_word": f3_freq_word[i], | |
| "f4_sim_sent": f4_sim_sent[i], | |
| "f5_len_norm": f5_len_norm[i], | |
| "f6_overlap": f6_overlap[i] | |
| }) | |
| return features_list | |