ONNX
Thai
File size: 1,035 Bytes
c0a2429
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
import pandas as pd
import random
import pickle
import re
from collections import Counter, defaultdict
from sklearn.model_selection import train_test_split
import pandas as pd
import random
random.seed(42)
import datasets

# Call this at the start of your script
datasets.disable_caching()
datasets.builder.has_sufficient_disk_space = lambda needed_bytes, directory=".": True

from datasets import load_dataset

ds = load_dataset("pythainlp/thai-g2p-v4-dataset", download_mode="force_redownload")

train = [(i["word"],i["ipa"]) for i in ds["train"]]
# train += [(i["word"],i["ipa"]) for i in ds["validation"]]
random.shuffle(train)
test=[(i["word"],i["ipa"]) for i in ds["validation"]]

# ==========================================
# 7. Export Outputs
# ==========================================
with open(f'data-ipa.pkl', 'wb') as f:
    pickle.dump(train+test, f)

with open(f'data-train.pkl', 'wb') as f:
    pickle.dump(train, f)
    
with open(f'data-test.pkl', 'wb') as f:
    pickle.dump(test, f)
    
print("Export complete!")