Download python/example.py from AXERA-TECH/inflect_micro_v2: direct link, hf CLI and curl.
- Browser
- Download file 2.81 kB
-
https://huggingface.co/AXERA-TECH/inflect_micro_v2/resolve/main/python/example.py
- Command line
-
hf download hf://AXERA-TECH/inflect_micro_v2/python/example.py
-
curl -L -o example.py https://huggingface.co/AXERA-TECH/inflect_micro_v2/resolve/main/python/example.py
2.81 kB
| #!/usr/bin/env python3 | |
| """Inflect AX TTS SDK example. | |
| Real text synthesis (requires phonemizer + espeak-ng, see README): | |
| python example.py \ | |
| --encoder ../models/ax620e/encoder.axmodel \ | |
| --decoder ../models/ax620e/decoder.axmodel \ | |
| --text "The quick brown fox jumps over the lazy dog." \ | |
| --output out.wav | |
| Host dry-run without NPU and without eSpeak (onnxruntime stand-in on the | |
| fp32 ONNX graphs, deterministic dummy tokens from | |
| ../models/model_meta.json numeric_baseline): | |
| python example.py --demo-tokens \ | |
| --encoder ../model_convert/export/encoder.onnx \ | |
| --decoder ../model_convert/export/decoder.onnx \ | |
| --output demo.wav | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import sys | |
| from pathlib import Path | |
| sys.path.insert(0, str(Path(__file__).resolve().parent)) | |
| from inflect_ax_tts import DUMMY_PHONEME_IDS, InflectTTS, write_wav | |
| def main() -> None: | |
| parser = argparse.ArgumentParser(description="Inflect AX TTS SDK example") | |
| parser.add_argument("--encoder", required=True, help="encoder .axmodel (or .onnx)") | |
| parser.add_argument("--decoder", required=True, help="decoder .axmodel (or .onnx)") | |
| parser.add_argument("--text", help="text to synthesize (needs phonemizer + eSpeak-ng)") | |
| parser.add_argument( | |
| "--demo-tokens", | |
| action="store_true", | |
| help="skip the text frontend and use the deterministic EXPORT dummy " | |
| "phoneme ids (no eSpeak dependency)", | |
| ) | |
| parser.add_argument("--output", required=True, help="output wav path") | |
| parser.add_argument("--speed", type=float, default=1.0) | |
| parser.add_argument("--variation", type=float, default=0.667) | |
| parser.add_argument("--seed", type=int, default=0) | |
| parser.add_argument( | |
| "--backend", | |
| default="auto", | |
| choices=["auto", "axengine", "onnxruntime"], | |
| help="inference backend (auto: .axmodel -> pyaxengine, .onnx -> onnxruntime)", | |
| ) | |
| args = parser.parse_args() | |
| if bool(args.text) == bool(args.demo_tokens): | |
| parser.error("provide exactly one of --text or --demo-tokens") | |
| tts = InflectTTS(args.encoder, args.decoder, backend=args.backend) | |
| if args.demo_tokens: | |
| sample_rate, wav = tts.synthesize_tokens( | |
| DUMMY_PHONEME_IDS, | |
| speed=args.speed, | |
| variation=args.variation, | |
| seed=args.seed, | |
| ) | |
| else: | |
| sample_rate, wav = tts.synthesize( | |
| args.text, speed=args.speed, variation=args.variation, seed=args.seed | |
| ) | |
| write_wav(args.output, wav, sample_rate) | |
| print( | |
| f"wrote {args.output}: {wav.size} samples " | |
| f"({wav.size / sample_rate:.2f} s @ {sample_rate} Hz), " | |
| f"amp [{wav.min():.3f}, {wav.max():.3f}]" | |
| ) | |
| if __name__ == "__main__": | |
| main() | |