Download make-data.py from pythainlp/thaig2p-v4: direct link, hf CLI and curl.
- Browser
- Download file 1.04 kB
-
https://huggingface.co/pythainlp/thaig2p-v4/resolve/main/make-data.py
- Command line
-
hf download hf://pythainlp/thaig2p-v4/make-data.py
-
curl -L -o make-data.py https://huggingface.co/pythainlp/thaig2p-v4/resolve/main/make-data.py
1.04 kB
| import pandas as pd | |
| import random | |
| import pickle | |
| import re | |
| from collections import Counter, defaultdict | |
| from sklearn.model_selection import train_test_split | |
| import pandas as pd | |
| import random | |
| random.seed(42) | |
| import datasets | |
| # Call this at the start of your script | |
| datasets.disable_caching() | |
| datasets.builder.has_sufficient_disk_space = lambda needed_bytes, directory=".": True | |
| from datasets import load_dataset | |
| ds = load_dataset("pythainlp/thai-g2p-v4-dataset", download_mode="force_redownload") | |
| train = [(i["word"],i["ipa"]) for i in ds["train"]] | |
| # train += [(i["word"],i["ipa"]) for i in ds["validation"]] | |
| random.shuffle(train) | |
| test=[(i["word"],i["ipa"]) for i in ds["validation"]] | |
| # ========================================== | |
| # 7. Export Outputs | |
| # ========================================== | |
| with open(f'data-ipa.pkl', 'wb') as f: | |
| pickle.dump(train+test, f) | |
| with open(f'data-train.pkl', 'wb') as f: | |
| pickle.dump(train, f) | |
| with open(f'data-test.pkl', 'wb') as f: | |
| pickle.dump(test, f) | |
| print("Export complete!") |