Download scripts/prepare_tokenizer.py from shubhexists/asr: direct link, hf CLI and curl.
- Browser
- Download file 2.77 kB
-
https://huggingface.co/shubhexists/asr/resolve/main/scripts/prepare_tokenizer.py
- Command line
-
hf download hf://shubhexists/asr/scripts/prepare_tokenizer.py
-
curl -L -o prepare_tokenizer.py https://huggingface.co/shubhexists/asr/resolve/main/scripts/prepare_tokenizer.py
2.77 kB
| """ | |
| Build a SentencePiece BPE tokenizer from LibriSpeech transcripts. | |
| Usage (from project root, venv active): | |
| python -m scripts.prepare_tokenizer --librispeech-root data \ | |
| --splits train-clean-100 --vocab-size 5000 --out configs/tokenizer | |
| Expects the standard layout: | |
| <root>/LibriSpeech/<split>/<spk>/<chap>/<spk>-<chap>.trans.txt | |
| """ | |
| import argparse | |
| import pathlib | |
| from src.tokenizer import train_tokenizer | |
| def build_corpus(librispeech_root: str, splits: list, corpus_out: str) -> int: | |
| root = pathlib.Path(librispeech_root) / "LibriSpeech" | |
| num_lines = 0 | |
| with open(corpus_out, "w", encoding="utf-8") as out_f: | |
| for split in splits: | |
| split_dir = root / split | |
| if not split_dir.is_dir(): | |
| raise FileNotFoundError( | |
| f"Expected split directory at {split_dir}, but it doesn't exist. " | |
| f"Download/extract LibriSpeech's {split}.tar.gz there first " | |
| f"(see README.md for the exact layout)." | |
| ) | |
| trans_files = sorted(split_dir.glob("*/*/*.trans.txt")) | |
| if not trans_files: | |
| raise FileNotFoundError(f"No *.trans.txt files found under {split_dir}") | |
| for trans_file in trans_files: | |
| with open(trans_file, encoding="utf-8") as f: | |
| for line in f: | |
| line = line.strip() | |
| if not line: | |
| continue | |
| # format: "<utt-id> TRANSCRIPT TEXT..." | |
| _, _, text = line.partition(" ") | |
| out_f.write(text.lower() + "\n") | |
| num_lines += 1 | |
| return num_lines | |
| def main(): | |
| parser = argparse.ArgumentParser() | |
| parser.add_argument("--librispeech-root", default="data", help="Directory containing LibriSpeech/") | |
| parser.add_argument("--splits", nargs="+", default=["train-clean-100"]) | |
| parser.add_argument("--vocab-size", type=int, default=5000) | |
| parser.add_argument("--out", default="configs/tokenizer", help="Output model prefix (no extension)") | |
| args = parser.parse_args() | |
| corpus_path = f"{args.out}_corpus.txt" | |
| pathlib.Path(args.out).parent.mkdir(parents=True, exist_ok=True) | |
| print(f"Building corpus from splits {args.splits} under {args.librispeech_root}/LibriSpeech ...") | |
| num_lines = build_corpus(args.librispeech_root, args.splits, corpus_path) | |
| print(f"Wrote {num_lines} transcript lines to {corpus_path}") | |
| print(f"Training SentencePiece BPE (vocab_size={args.vocab_size}) ...") | |
| train_tokenizer(corpus_path, args.out, vocab_size=args.vocab_size) | |
| print(f"Done. Wrote {args.out}.model and {args.out}.vocab") | |
| if __name__ == "__main__": | |
| main() | |