File size: 4,395 Bytes
82c2526
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
import re
import numpy as np
from sklearn.feature_extraction.text import TfidfVectorizer
from sklearn.metrics.pairwise import cosine_similarity
import nltk

# Ensure punkt is downloaded for sentence tokenization
try:
    nltk.data.find('tokenizers/punkt')
except LookupError:
    nltk.download('punkt')
    nltk.download('punkt_tab')

class FeatureExtractor:
    def __init__(self):
        pass

    def split_sentences(self, text):
        """
        Splits text into sentences.
        Using nltk's sent_tokenize which handles standard punctuation better.
        """
        # Balinese text might use standard punctuation.
        if not text:
            return []
        return nltk.sent_tokenize(text)

    def calculate_features(self, title, article_text):
        """
        Calculates the 6 features for each sentence in the article.

        Features:
        1. f1_lead: 1 if first sentence, 0 otherwise.
        2. f2_judul_sim: Cosine similarity between Sentence and Title.
        3. f3_freq_word: Sum of TF-IDF scores of words in sentence.
        4. f4_sim_sent: Average cosine similarity to all other sentences.
        5. f5_len_norm: Sentence length / Longest sentence length.
        6. f6_overlap: Number of common words between Sentence and Title.

        Returns:
            list of dicts (one per sentence) containing features and the sentence text.
        """
        sentences = self.split_sentences(article_text)
        if not sentences:
            return []

        # Precompute TF-IDF for the whole document (sentences + title)
        # We treat the Title as one document and Sentences as others for similarity calc.
        corpus = [title] + sentences
        vectorizer = TfidfVectorizer()

        try:
            tfidf_matrix = vectorizer.fit_transform(corpus)
        except ValueError:
            # Handle empty vocabulary or stop words issues
            return []

        # tfidf_matrix[0] is Title
        # tfidf_matrix[1:] are Sentences

        title_vector = tfidf_matrix[0]
        sentence_vectors = tfidf_matrix[1:]

        # 1. f1_lead
        f1_lead = [1 if i == 0 else 0 for i in range(len(sentences))]

        # 2. f2_judul_sim
        # Compute cosine similarity between each sentence and title
        # cosine_similarity returns shape (n_samples_X, n_samples_Y)
        f2_judul_sim = cosine_similarity(sentence_vectors, title_vector).flatten()

        # 3. f3_freq_word
        # Sum of TF-IDF scores for words in the sentence
        # sentence_vectors is a sparse matrix. sum(axis=1) gives sum of each row.
        f3_freq_word = np.array(sentence_vectors.sum(axis=1)).flatten()

        # 4. f4_sim_sent
        # Average cosine similarity to all other sentences
        # Compute similarity matrix between sentences
        if len(sentences) > 1:
            sim_matrix = cosine_similarity(sentence_vectors)
            # We want average similarity to OTHERS.
            # sum of similarities minus self-similarity (which is 1.0) / (n-1)
            f4_sim_sent = (sim_matrix.sum(axis=1) - 1) / (len(sentences) - 1)
        else:
            f4_sim_sent = [0.0] * len(sentences)

        # 5. f5_len_norm
        # Number of words in sentence / Length of longest sentence in doc
        # We use simple whitespace splitting for word count
        sent_word_counts = [len(re.findall(r'\w+', s)) for s in sentences]
        max_len = max(sent_word_counts) if sent_word_counts else 1
        f5_len_norm = [count / max_len for count in sent_word_counts]

        # 6. f6_overlap
        # Number of common words between sentence and Title
        title_words = set(re.findall(r'\w+', title.lower()))
        f6_overlap = []
        for s in sentences:
            s_words = set(re.findall(r'\w+', s.lower()))
            overlap_count = len(title_words.intersection(s_words))
            f6_overlap.append(overlap_count)

        # Combine all features
        features_list = []
        for i in range(len(sentences)):
            features_list.append({
                "teks_kalimat": sentences[i],
                "f1_lead": f1_lead[i],
                "f2_judul_sim": f2_judul_sim[i],
                "f3_freq_word": f3_freq_word[i],
                "f4_sim_sent": f4_sim_sent[i],
                "f5_len_norm": f5_len_norm[i],
                "f6_overlap": f6_overlap[i]
            })

        return features_list