Download src/speech_text_data_aligners/cli/annotate.py from vertox-ai/speech-text-data-aligners: direct link, hf CLI and curl.
- Browser
- Download file 3.01 kB
-
https://huggingface.co/vertox-ai/speech-text-data-aligners/resolve/main/src/speech_text_data_aligners/cli/annotate.py
- Command line
-
hf download hf://vertox-ai/speech-text-data-aligners/src/speech_text_data_aligners/cli/annotate.py
-
curl -L -o annotate.py https://huggingface.co/vertox-ai/speech-text-data-aligners/resolve/main/src/speech_text_data_aligners/cli/annotate.py
3.01 kB
| # License: Apache-2.0 License | |
| # Created by: Patrick Lumbantobing, VertoX-AI | |
| # Copyright (c) 2026 VertoX-AI. All rights reserved. | |
| # | |
| # This work is licensed under the Apache-2.0 License. | |
| # To view a copy of this license, visit | |
| # https://www.apache.org/licenses/LICENSE-2.0 | |
| """Adapt annotation CLI arguments to the reusable application service.""" | |
| from __future__ import annotations | |
| import argparse | |
| import json | |
| from pathlib import Path | |
| from speech_text_data_aligners.application.annotate import ( | |
| AnnotationOptions, | |
| annotate_metadata, | |
| ) | |
| from speech_text_data_aligners.backends.registry import ( | |
| BackendRegistry, | |
| default_backend_registry, | |
| ) | |
| from speech_text_data_aligners.core.errors import ConfigurationError | |
| def configure_parser(parser: argparse.ArgumentParser) -> None: | |
| """Add annotation composition/presentation arguments to a subcommand parser. | |
| Parameters | |
| ---------- | |
| parser: Parser owned by the CLI composition root. | |
| Side Effects: | |
| Registers arguments and the command handler on ``parser``. | |
| """ | |
| parser.add_argument("metadata_csv", type=Path) | |
| parser.add_argument("audio_dir", type=Path) | |
| parser.add_argument("output_csv", type=Path) | |
| parser.add_argument("--backend", required=True) | |
| parser.add_argument("--backend-config-json", default="{}") | |
| parser.add_argument("--strict", action="store_true") | |
| parser.add_argument("--overwrite", action="store_true") | |
| parser.add_argument("--diagnostics", type=Path) | |
| parser.add_argument("--receipt", type=Path) | |
| parser.set_defaults(handler=run) | |
| def run( | |
| args: argparse.Namespace, | |
| *, | |
| registry: BackendRegistry | None = None, | |
| ) -> int: | |
| """Construct the selected backend and delegate annotation to the library service. | |
| Parameters | |
| ---------- | |
| args: Parsed annotation arguments. | |
| registry: Optional injected lazy registry for tests/custom composition. | |
| Returns | |
| ------- | |
| Zero after successful service completion. | |
| Raises | |
| ------ | |
| ConfigurationError: If backend configuration JSON is invalid/not an object. | |
| SpeechTextAlignerError: Propagated typed construction or service failures. | |
| """ | |
| try: | |
| config = json.loads(args.backend_config_json) | |
| except json.JSONDecodeError as error: | |
| raise ConfigurationError("backend configuration is not valid JSON") from error | |
| if not isinstance(config, dict): | |
| raise ConfigurationError("backend configuration JSON must be an object") | |
| selected_registry = registry or default_backend_registry() | |
| backend = selected_registry.create(args.backend, config=config) | |
| annotate_metadata( | |
| metadata_csv=args.metadata_csv, | |
| audio_dir=args.audio_dir, | |
| output_csv=args.output_csv, | |
| backend=backend, | |
| options=AnnotationOptions( | |
| strict=args.strict, | |
| overwrite=args.overwrite, | |
| diagnostics_path=args.diagnostics, | |
| receipt_path=args.receipt, | |
| ), | |
| ) | |
| return 0 | |