#!/usr/bin/env python3 """Convert the tokenizer config from the transformers>=5 format to the 4.x format. `transformers` 5 stores `extra_special_tokens` as a list of token strings; 4.x expects a mapping. Run once if you use transformers 4.x: python patch_tokenizer_for_transformers4.py tokenizer/tokenizer_config.json """ from __future__ import annotations import json import sys from pathlib import Path def main() -> None: path = Path(sys.argv[1] if len(sys.argv) > 1 else "tokenizer/tokenizer_config.json") cfg = json.loads(path.read_text(encoding="utf-8")) extra = cfg.get("extra_special_tokens") if isinstance(extra, list): cfg["extra_special_tokens"] = {token: token for token in extra} path.write_text(json.dumps(cfg, ensure_ascii=False, indent=2), encoding="utf-8") print(f"patched {path}: extra_special_tokens list -> dict ({len(cfg['extra_special_tokens'])} entries)") elif isinstance(extra, dict): print(f"{path}: already in 4.x format") else: print(f"{path}: no extra_special_tokens field, nothing to do") if __name__ == "__main__": main()