Download scripts/train_tokenizer.py from kings1/Nova: direct link, hf CLI and curl.
- Browser
- Download file 2.23 kB
-
https://huggingface.co/kings1/Nova/resolve/main/scripts/train_tokenizer.py
- Command line
-
hf download hf://kings1/Nova/scripts/train_tokenizer.py
-
curl -L -o train_tokenizer.py https://huggingface.co/kings1/Nova/resolve/main/scripts/train_tokenizer.py
2.23 kB
| """ | |
| Script 1: Train Custom 16K BPE Tokenizer for Scaled Nova 1.0 | |
| Trains across multiple instruction & conversational datasets (Alpaca & Dolly). | |
| """ | |
| import os | |
| import argparse | |
| from datasets import load_dataset | |
| from src.tokenizer.bpe_tokenizer import Nova1Tokenizer | |
| from src.dataset.hf_dataset import format_item_to_chat | |
| def main(): | |
| parser = argparse.ArgumentParser(description="Train Scaled Nova 1.0 Tokenizer") | |
| parser.add_argument("--vocab_size", type=int, default=16384, help="Vocabulary size") | |
| parser.add_argument("--output_path", type=str, default="checkpoints/nova1_tokenizer.json", help="Path to save tokenizer JSON") | |
| args = parser.parse_args() | |
| token = os.environ.get("HF_TOKEN") | |
| print("Loading datasets for high-capacity tokenizer training...") | |
| ds_alpaca = load_dataset("tatsu-lab/alpaca", split="train", token=token) | |
| ds_dolly = load_dataset("philschmid/dolly-15k-oai-style", split="train", token=token) | |
| def text_iterator(): | |
| # Include core conversational phrases explicitly in vocabulary training | |
| yield "User: Hello! How are you?\nAssistant: Hello! I am Nova 1.0, an AI model built from scratch. I am doing great, how can I help you today?" | |
| yield "User: What is quantum computing?\nAssistant: Quantum computing is a rapidly-emerging technology that harnesses the laws of quantum mechanics to solve complex problems." | |
| yield "User: Who created you?\nAssistant: I was built from scratch as Nova 1.0 using the Hierarchical Reasoning Model (HRM) architecture." | |
| for item in ds_alpaca: | |
| txt = format_item_to_chat(item) | |
| if txt and len(txt.strip()) > 0: | |
| yield txt | |
| for item in ds_dolly: | |
| txt = format_item_to_chat(item) | |
| if txt and len(txt.strip()) > 0: | |
| yield txt | |
| print(f"Training 16K subword BPE Tokenizer with vocab_size={args.vocab_size}...") | |
| tokenizer = Nova1Tokenizer() | |
| tokenizer.train_from_iterator(text_iterator(), vocab_size=args.vocab_size) | |
| tokenizer.save(args.output_path) | |
| print(f"Tokenizer saved successfully to '{args.output_path}'!") | |
| print(f"Actual Vocab Size: {tokenizer.vocab_size}") | |
| if __name__ == "__main__": | |
| main() | |