File size: 1,662 Bytes
fb5d414
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
import pandas as pd
from datasets import load_dataset

print("=== 1. Jane Austen Corpus ===")
try:
    ds_ja = load_dataset('alexatran/jane-austen-corpus', split='train')
    print("Novels:", len(ds_ja), "Cols:", ds_ja.column_names)
    for i in range(len(ds_ja)):
        print(f"Novel {i}: {ds_ja[i]['title']} ({len(ds_ja[i]['text'])} chars)")
except Exception as e:
    print("Jane Austen Error:", e)

print("\n=== 2. Classic Novels ===")
try:
    ds_cn = load_dataset('hugfaceguy0001/ClassicNovels', split='train')
    print("Novels:", len(ds_cn), "Cols:", ds_cn.column_names)
    for i in range(min(5, len(ds_cn))):
        print(f"Novel {i}: {ds_cn[i]['title']} by {ds_cn[i].get('author', '')} ({len(ds_cn[i]['text'])} chars)")
except Exception as e:
    print("Classic Novels Error:", e)

print("\n=== 3. GoEmotions Hindi ===")
try:
    url = "https://huggingface.co/datasets/utkarsharora100/google_go_emotions_hindi_translated/resolve/main/hindi_translated_google_go_emotions_dataset.csv"
    df_emo = pd.read_csv(url)
    print("GoEmotions Hindi Shape:", df_emo.shape, "Cols:", df_emo.columns.tolist()[:10])
    # check emotion columns
    print("Sample text:", df_emo.iloc[0]['text'] if 'text' in df_emo.columns else df_emo.iloc[0])
except Exception as e:
    print("GoEmotions Hindi Error:", e)

print("\n=== 4. Shayari ===")
try:
    url = "https://huggingface.co/datasets/Shubham231/hindi_shayari_dataset/resolve/main/shayari_dataset.csv"
    df_sh = pd.read_csv(url)
    print("Shayari Shape:", df_sh.shape, "Cols:", df_sh.columns.tolist())
    print("Sample output:", df_sh.iloc[0]['output'])
except Exception as e:
    print("Shayari Error:", e)