File size: 1,526 Bytes
9a9467f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
# https://www.kaggle.com/datasets/programmerrdai/sinhala-english-singlish-translation-dataset?resource=download

import dataload

df = dataload.load_data('.\\','converted_data.csv')
text = df['Sinhala'].dropna().to_string()
print(f"{type(text)},{len(text)},{text[:10]},{text[-10:]}")

#with open('input.txt', 'r', encoding='utf-8') as f:
#    text = f.read()

total_chars = len(text)
unique_chars = sorted(list(set(text)))
vocab_size = len(unique_chars)

word_list = text.split()
word_lengths = [len(i) for i in word_list]
sentence_list = text.split('\n')
sente_lengths = [len(i) for i in sentence_list]

#print(len(sentence_list))

print(f"Total characters: {total_chars:,}")
print(f"Unique characters (Vocab Size): {vocab_size}")
print(f"Max word length: {max(word_lengths)},({word_list[word_lengths.index(max(word_lengths))]})")
print(f"Average word length: {sum(len(w) for w in text.split()) / len(text.split()):.2f}")
print(f"Max sentence length: {max(sente_lengths)},({sentence_list[sente_lengths.index(max(sente_lengths))]})")
print(f"Average sentence length: {sum(len(s) for s in sentence_list) / len(sentence_list):.2f}")

from collections import Counter
import matplotlib.pyplot as plt

# Count every character
char_counts = Counter(text)

# Sort by frequency
most_common = char_counts.most_common(20)
labels, values = zip(*most_common)

# Visualize
plt.figure(figsize=(10, 5))
plt.bar(labels, values)
plt.title("Top 20 Most Frequent Characters in This Dataset")
plt.show()