Automatic Speech Recognition
MLX
ONNX
GGUF
Rust
English
Chinese
audio8
streaming-asr
quantized
experimental
Instructions to use Reza2kn/Audio8-ASR-Infinite-Compressed with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- MLX
How to use Reza2kn/Audio8-ASR-Infinite-Compressed with MLX:
# Download the model from the Hub pip install huggingface_hub[hf_xet] hf download Reza2kn/Audio8-ASR-Infinite-Compressed --local-dir Audio8-ASR-Infinite-Compressed
- Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- LM Studio
- Atomic Chat
Download evaluation/published-consumer/linux/cpu-en.json from Reza2kn/Audio8-ASR-Infinite-Compressed: direct link, hf CLI and curl.
- Browser
- Download file 4.91 kB
-
https://huggingface.co/Reza2kn/Audio8-ASR-Infinite-Compressed/resolve/main/evaluation/published-consumer/linux/cpu-en.json
- Command line
-
hf download hf://Reza2kn/Audio8-ASR-Infinite-Compressed/evaluation/published-consumer/linux/cpu-en.json
-
curl -L -o cpu-en.json https://huggingface.co/Reza2kn/Audio8-ASR-Infinite-Compressed/resolve/main/evaluation/published-consumer/linux/cpu-en.json
4.91 kB
| { | |
| "status": "complete", | |
| "summary": { | |
| "kind": "caption_stream_complete", | |
| "status": "complete", | |
| "audio_samples": 93680, | |
| "audio_seconds": 5.855, | |
| "segments": 1, | |
| "selected_samples": 90864, | |
| "unconsumed_selected_samples": 0, | |
| "emissions": 85, | |
| "text_bytes": 88, | |
| "model_seconds": 4.573718884000001, | |
| "reset_seconds": 8.12e-07, | |
| "vad_seconds": 0.053173863999999973, | |
| "read_seconds": 1.9281761100000006, | |
| "sleep_seconds": 0, | |
| "elapsed_seconds": 6.558700714, | |
| "work_rtf": 0.7902465516652434, | |
| "peak_cache_bytes": 210169856, | |
| "max_retained_input_samples": 6912, | |
| "input_sha256": "22d472b40f913206b6917115d136306be88df3b278e77f68be5910423250fa4d", | |
| "generated_ids_sha256": "4cab5d89ace01105314a3ff9ea386dd217c2cb319c74d0295a34ead46d3c44f7", | |
| "all_source_audio_observed": true, | |
| "measured_word_aligned_latency": false, | |
| "production_accuracy_approved": false | |
| }, | |
| "score": { | |
| "case_id": "1272-128104-0000", | |
| "language": "en", | |
| "input_sha256": "22d472b40f913206b6917115d136306be88df3b278e77f68be5910423250fa4d", | |
| "raw_hypothesis": "Mr. Quilter is the apostle of the middle classes, and we are glad to welcome his gospel.", | |
| "normalized_reference": "mister quilter is the apostle of the middle classes and we are glad to welcome his gospel", | |
| "normalized_hypothesis": "mr quilter is the apostle of the middle classes and we are glad to welcome his gospel", | |
| "metric": "WER", | |
| "counts": { | |
| "hits": 16, | |
| "substitutions": 1, | |
| "deletions": 0, | |
| "insertions": 0, | |
| "reference_units": 17, | |
| "hypothesis_units": 17, | |
| "errors": 1, | |
| "error_rate": 0.058823529411764705 | |
| }, | |
| "longest_contiguous_deleted_reference_units": 0, | |
| "aligned_word_timing": { | |
| "measured": false, | |
| "actual_arrival_paced": true, | |
| "timing_basis": "Independent host receipt of flushed JSONL since paced20ms producer start; all source words retained", | |
| "correctly_matched_units": 0, | |
| "ambiguous_matched_units": 0, | |
| "ambiguity_method": "all minimum unit-cost Levenshtein paths, without timing-based selection", | |
| "unambiguous_matched_units": 0, | |
| "unambiguous_maximum_seconds": null, | |
| "unmatched_reference_units": 1, | |
| "matched_reference_fraction": null, | |
| "min_seconds": null, | |
| "p50_seconds": null, | |
| "p95_seconds": null, | |
| "maximum_seconds": null, | |
| "matched_units_above_one_second": null, | |
| "all_matched_units_within_one_second": null, | |
| "whole_release_pass": false, | |
| "limitation": "Lag summaries use one deterministic text alignment and are descriptive when alternative optimal alignments exist. Only unambiguous matches can support per-word timing claims. Omissions/substitutions remain errors, never zero-latency successes. Published reference timing uncertainty and overlapping-speaker word order are not independently corrected." | |
| }, | |
| "alignment": [], | |
| "error_spans": [ | |
| { | |
| "type": "substitute", | |
| "reference_range": [ | |
| 0, | |
| 1 | |
| ], | |
| "hypothesis_range": [ | |
| 0, | |
| 1 | |
| ] | |
| } | |
| ], | |
| "first_nonempty_delta_seconds": 1.3183811120688915, | |
| "production_accuracy_approved": false | |
| }, | |
| "segments": [ | |
| { | |
| "id": 0, | |
| "start_sample": 2816, | |
| "decision_sample": 9216, | |
| "end_sample": 93680, | |
| "final_eof": true, | |
| "windows": 85, | |
| "unconsumed_samples": 0 | |
| } | |
| ], | |
| "all_ids_exact_expected": null, | |
| "independent_tokenizer_oracle": true, | |
| "max_queue_seconds": 0.654069003, | |
| "no_synthetic_nonfinal_eof": true, | |
| "retention_bounded": true, | |
| "release_qualification": false, | |
| "independent_delta_token_clock_oracle": true, | |
| "feeder": { | |
| "chunks": 293, | |
| "samples": 93680, | |
| "duration_seconds": 5.8550937010440975, | |
| "max_write_seconds": 5.0443923100829124e-05, | |
| "max_delivery_lateness_seconds": 0.0013744130358099849, | |
| "process_started_monotonic": 1189174.972518553, | |
| "feeder_started_monotonic": 1189177.764669274, | |
| "ready_received_monotonic": 1189177.764503835, | |
| "clock_scope": "Feeder starts after host receipt of native readiness. Native and observer clocks are separately recorded; no microphone capture clock." | |
| }, | |
| "command": [ | |
| "/bin/sh", | |
| "/dev/shm/audio8-q3-public-linux-consumer-v1/run-cpu-linux-x86_64.sh", | |
| "-", | |
| "en", | |
| "/dev/shm/audio8-q3-public-linux-consumer-v1/verification/en.a8pfx" | |
| ], | |
| "invoked_shipped_launcher": true, | |
| "memory": { | |
| "os_peak_process_rss_bytes": 2025762816, | |
| "os_peak_memory_footprint_bytes": null, | |
| "footprint_available": false, | |
| "rss_source_unit": "KiB (1024 bytes)", | |
| "elapsed_seconds": 9.4, | |
| "user_seconds": 26.12, | |
| "system_seconds": 2.63, | |
| "scope": "Fresh native child process including loading and inference; excludes Python harness. RSS and footprint are separate measurements and must not be summed." | |
| } | |
| } | |