gpt2sin / stats.py
dastharak's picture
Upload 4 files
9a9467f verified
Raw History Blame Contribute Delete
1.53 kB
# https://www.kaggle.com/datasets/programmerrdai/sinhala-english-singlish-translation-dataset?resource=download
import dataload
df = dataload.load_data('.\\','converted_data.csv')
text = df['Sinhala'].dropna().to_string()
print(f"{type(text)},{len(text)},{text[:10]},{text[-10:]}")
#with open('input.txt', 'r', encoding='utf-8') as f:
# text = f.read()
total_chars = len(text)
unique_chars = sorted(list(set(text)))
vocab_size = len(unique_chars)
word_list = text.split()
word_lengths = [len(i) for i in word_list]
sentence_list = text.split('\n')
sente_lengths = [len(i) for i in sentence_list]
#print(len(sentence_list))
print(f"Total characters: {total_chars:,}")
print(f"Unique characters (Vocab Size): {vocab_size}")
print(f"Max word length: {max(word_lengths)},({word_list[word_lengths.index(max(word_lengths))]})")
print(f"Average word length: {sum(len(w) for w in text.split()) / len(text.split()):.2f}")
print(f"Max sentence length: {max(sente_lengths)},({sentence_list[sente_lengths.index(max(sente_lengths))]})")
print(f"Average sentence length: {sum(len(s) for s in sentence_list) / len(sentence_list):.2f}")
from collections import Counter
import matplotlib.pyplot as plt
# Count every character
char_counts = Counter(text)
# Sort by frequency
most_common = char_counts.most_common(20)
labels, values = zip(*most_common)
# Visualize
plt.figure(figsize=(10, 5))
plt.bar(labels, values)
plt.title("Top 20 Most Frequent Characters in This Dataset")
plt.show()