File size: 1,035 Bytes
c0a2429 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 | import pandas as pd
import random
import pickle
import re
from collections import Counter, defaultdict
from sklearn.model_selection import train_test_split
import pandas as pd
import random
random.seed(42)
import datasets
# Call this at the start of your script
datasets.disable_caching()
datasets.builder.has_sufficient_disk_space = lambda needed_bytes, directory=".": True
from datasets import load_dataset
ds = load_dataset("pythainlp/thai-g2p-v4-dataset", download_mode="force_redownload")
train = [(i["word"],i["ipa"]) for i in ds["train"]]
# train += [(i["word"],i["ipa"]) for i in ds["validation"]]
random.shuffle(train)
test=[(i["word"],i["ipa"]) for i in ds["validation"]]
# ==========================================
# 7. Export Outputs
# ==========================================
with open(f'data-ipa.pkl', 'wb') as f:
pickle.dump(train+test, f)
with open(f'data-train.pkl', 'wb') as f:
pickle.dump(train, f)
with open(f'data-test.pkl', 'wb') as f:
pickle.dump(test, f)
print("Export complete!") |