from pathlib import Path from datasets import interleave_datasets, load_dataset from tokenizers import ByteLevelBPETokenizer from tqdm import tqdm NUM_DOCUMENTS = 500_000 VOCAB_SIZE = 32_000 SEED = 42 OUTPUT_FILE = Path(__file__).with_name("hanse_tokenizer.json") SOURCES = ( ("HuggingFaceFW/fineweb-edu", "sample-10BT", 0.40), ("HuggingFaceFW/fineweb-2", "deu_Latn", 0.35), ("HuggingFaceFW/finewiki", "de", 0.20), ("HuggingFaceFW/finewiki", "en", 0.05), ) SPECIAL_TOKENS = ( "<|pad|>", "<|bos|>", "<|eos|>", "<|unk|>", "<|system|>", "<|user|>", "<|assistant|>", "<|tool|>", "<|tool_result|>", "<|end_of_turn|>", ) def training_corpus(dataset): accepted = 0 with tqdm(total=NUM_DOCUMENTS, desc="Feeding documents", unit="docs", dynamic_ncols=True) as progress: for example in dataset: text = example.get("text") if not isinstance(text, str) or not (text := text.strip()): continue yield text accepted += 1 progress.update() if accepted == NUM_DOCUMENTS: return raise RuntimeError(f"Dataset exhausted after {accepted:,} usable documents") def main() -> None: streams = [ load_dataset(name, config, split="train", streaming=True) for name, config, _ in SOURCES ] dataset = interleave_datasets( streams, probabilities=[probability for _, _, probability in SOURCES], seed=SEED, stopping_strategy="all_exhausted", ) tokenizer = ByteLevelBPETokenizer() tokenizer.train_from_iterator( training_corpus(dataset), vocab_size=VOCAB_SIZE, min_frequency=2, special_tokens=list(SPECIAL_TOKENS), length=NUM_DOCUMENTS, show_progress=True, ) assert tokenizer.get_vocab_size() == VOCAB_SIZE assert [tokenizer.token_to_id(token) for token in SPECIAL_TOKENS] == list(range(len(SPECIAL_TOKENS))) tokenizer.save(str(OUTPUT_FILE)) print(f"[+] Saved {tokenizer.get_vocab_size():,}-token tokenizer to {OUTPUT_FILE}") if __name__ == "__main__": main()