Download stats.py from dastharak/gpt2sin: direct link, hf CLI and curl.
- Browser
- Download file 1.53 kB
-
https://huggingface.co/dastharak/gpt2sin/resolve/main/stats.py
- Command line
-
hf download hf://dastharak/gpt2sin/stats.py
-
curl -L -o stats.py https://huggingface.co/dastharak/gpt2sin/resolve/main/stats.py
1.53 kB
| # https://www.kaggle.com/datasets/programmerrdai/sinhala-english-singlish-translation-dataset?resource=download | |
| import dataload | |
| df = dataload.load_data('.\\','converted_data.csv') | |
| text = df['Sinhala'].dropna().to_string() | |
| print(f"{type(text)},{len(text)},{text[:10]},{text[-10:]}") | |
| #with open('input.txt', 'r', encoding='utf-8') as f: | |
| # text = f.read() | |
| total_chars = len(text) | |
| unique_chars = sorted(list(set(text))) | |
| vocab_size = len(unique_chars) | |
| word_list = text.split() | |
| word_lengths = [len(i) for i in word_list] | |
| sentence_list = text.split('\n') | |
| sente_lengths = [len(i) for i in sentence_list] | |
| #print(len(sentence_list)) | |
| print(f"Total characters: {total_chars:,}") | |
| print(f"Unique characters (Vocab Size): {vocab_size}") | |
| print(f"Max word length: {max(word_lengths)},({word_list[word_lengths.index(max(word_lengths))]})") | |
| print(f"Average word length: {sum(len(w) for w in text.split()) / len(text.split()):.2f}") | |
| print(f"Max sentence length: {max(sente_lengths)},({sentence_list[sente_lengths.index(max(sente_lengths))]})") | |
| print(f"Average sentence length: {sum(len(s) for s in sentence_list) / len(sentence_list):.2f}") | |
| from collections import Counter | |
| import matplotlib.pyplot as plt | |
| # Count every character | |
| char_counts = Counter(text) | |
| # Sort by frequency | |
| most_common = char_counts.most_common(20) | |
| labels, values = zip(*most_common) | |
| # Visualize | |
| plt.figure(figsize=(10, 5)) | |
| plt.bar(labels, values) | |
| plt.title("Top 20 Most Frequent Characters in This Dataset") | |
| plt.show() |