File size: 2,233 Bytes
23ea6bd
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
"""
Script 1: Train Custom 16K BPE Tokenizer for Scaled Nova 1.0
Trains across multiple instruction & conversational datasets (Alpaca & Dolly).
"""

import os
import argparse
from datasets import load_dataset
from src.tokenizer.bpe_tokenizer import Nova1Tokenizer
from src.dataset.hf_dataset import format_item_to_chat


def main():
    parser = argparse.ArgumentParser(description="Train Scaled Nova 1.0 Tokenizer")
    parser.add_argument("--vocab_size", type=int, default=16384, help="Vocabulary size")
    parser.add_argument("--output_path", type=str, default="checkpoints/nova1_tokenizer.json", help="Path to save tokenizer JSON")
    args = parser.parse_args()

    token = os.environ.get("HF_TOKEN")
    
    print("Loading datasets for high-capacity tokenizer training...")
    ds_alpaca = load_dataset("tatsu-lab/alpaca", split="train", token=token)
    ds_dolly = load_dataset("philschmid/dolly-15k-oai-style", split="train", token=token)

    def text_iterator():
        # Include core conversational phrases explicitly in vocabulary training
        yield "User: Hello! How are you?\nAssistant: Hello! I am Nova 1.0, an AI model built from scratch. I am doing great, how can I help you today?"
        yield "User: What is quantum computing?\nAssistant: Quantum computing is a rapidly-emerging technology that harnesses the laws of quantum mechanics to solve complex problems."
        yield "User: Who created you?\nAssistant: I was built from scratch as Nova 1.0 using the Hierarchical Reasoning Model (HRM) architecture."

        for item in ds_alpaca:
            txt = format_item_to_chat(item)
            if txt and len(txt.strip()) > 0:
                yield txt

        for item in ds_dolly:
            txt = format_item_to_chat(item)
            if txt and len(txt.strip()) > 0:
                yield txt

    print(f"Training 16K subword BPE Tokenizer with vocab_size={args.vocab_size}...")
    tokenizer = Nova1Tokenizer()
    tokenizer.train_from_iterator(text_iterator(), vocab_size=args.vocab_size)

    tokenizer.save(args.output_path)
    print(f"Tokenizer saved successfully to '{args.output_path}'!")
    print(f"Actual Vocab Size: {tokenizer.vocab_size}")


if __name__ == "__main__":
    main()