File size: 2,768 Bytes
ce3c8df
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
"""
Build a SentencePiece BPE tokenizer from LibriSpeech transcripts.

Usage (from project root, venv active):
    python -m scripts.prepare_tokenizer --librispeech-root data \
        --splits train-clean-100 --vocab-size 5000 --out configs/tokenizer

Expects the standard layout:
    <root>/LibriSpeech/<split>/<spk>/<chap>/<spk>-<chap>.trans.txt
"""

import argparse
import pathlib

from src.tokenizer import train_tokenizer


def build_corpus(librispeech_root: str, splits: list, corpus_out: str) -> int:
    root = pathlib.Path(librispeech_root) / "LibriSpeech"
    num_lines = 0
    with open(corpus_out, "w", encoding="utf-8") as out_f:
        for split in splits:
            split_dir = root / split
            if not split_dir.is_dir():
                raise FileNotFoundError(
                    f"Expected split directory at {split_dir}, but it doesn't exist. "
                    f"Download/extract LibriSpeech's {split}.tar.gz there first "
                    f"(see README.md for the exact layout)."
                )
            trans_files = sorted(split_dir.glob("*/*/*.trans.txt"))
            if not trans_files:
                raise FileNotFoundError(f"No *.trans.txt files found under {split_dir}")
            for trans_file in trans_files:
                with open(trans_file, encoding="utf-8") as f:
                    for line in f:
                        line = line.strip()
                        if not line:
                            continue
                        # format: "<utt-id> TRANSCRIPT TEXT..."
                        _, _, text = line.partition(" ")
                        out_f.write(text.lower() + "\n")
                        num_lines += 1
    return num_lines


def main():
    parser = argparse.ArgumentParser()
    parser.add_argument("--librispeech-root", default="data", help="Directory containing LibriSpeech/")
    parser.add_argument("--splits", nargs="+", default=["train-clean-100"])
    parser.add_argument("--vocab-size", type=int, default=5000)
    parser.add_argument("--out", default="configs/tokenizer", help="Output model prefix (no extension)")
    args = parser.parse_args()

    corpus_path = f"{args.out}_corpus.txt"
    pathlib.Path(args.out).parent.mkdir(parents=True, exist_ok=True)

    print(f"Building corpus from splits {args.splits} under {args.librispeech_root}/LibriSpeech ...")
    num_lines = build_corpus(args.librispeech_root, args.splits, corpus_path)
    print(f"Wrote {num_lines} transcript lines to {corpus_path}")

    print(f"Training SentencePiece BPE (vocab_size={args.vocab_size}) ...")
    train_tokenizer(corpus_path, args.out, vocab_size=args.vocab_size)
    print(f"Done. Wrote {args.out}.model and {args.out}.vocab")


if __name__ == "__main__":
    main()