# https://www.kaggle.com/datasets/programmerrdai/sinhala-english-singlish-translation-dataset?resource=download import dataload df = dataload.load_data('.\\','converted_data.csv') text = df['Sinhala'].dropna().to_string() print(f"{type(text)},{len(text)},{text[:10]},{text[-10:]}") #with open('input.txt', 'r', encoding='utf-8') as f: # text = f.read() total_chars = len(text) unique_chars = sorted(list(set(text))) vocab_size = len(unique_chars) word_list = text.split() word_lengths = [len(i) for i in word_list] sentence_list = text.split('\n') sente_lengths = [len(i) for i in sentence_list] #print(len(sentence_list)) print(f"Total characters: {total_chars:,}") print(f"Unique characters (Vocab Size): {vocab_size}") print(f"Max word length: {max(word_lengths)},({word_list[word_lengths.index(max(word_lengths))]})") print(f"Average word length: {sum(len(w) for w in text.split()) / len(text.split()):.2f}") print(f"Max sentence length: {max(sente_lengths)},({sentence_list[sente_lengths.index(max(sente_lengths))]})") print(f"Average sentence length: {sum(len(s) for s in sentence_list) / len(sentence_list):.2f}") from collections import Counter import matplotlib.pyplot as plt # Count every character char_counts = Counter(text) # Sort by frequency most_common = char_counts.most_common(20) labels, values = zip(*most_common) # Visualize plt.figure(figsize=(10, 5)) plt.bar(labels, values) plt.title("Top 20 Most Frequent Characters in This Dataset") plt.show()