Automatic Speech Recognition
MLX
ONNX
GGUF
Rust
English
Chinese
audio8
streaming-asr
quantized
experimental
Instructions to use Reza2kn/Audio8-ASR-Infinite-Compressed with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- MLX
How to use Reza2kn/Audio8-ASR-Infinite-Compressed with MLX:
# Download the model from the Hub pip install huggingface_hub[hf_xet] hf download Reza2kn/Audio8-ASR-Infinite-Compressed --local-dir Audio8-ASR-Infinite-Compressed
- Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- LM Studio
- Atomic Chat
Download evaluation/published-consumer/linux/cpu-zh.json from Reza2kn/Audio8-ASR-Infinite-Compressed: direct link, hf CLI and curl.
- Browser
- Download file 4.69 kB
-
https://huggingface.co/Reza2kn/Audio8-ASR-Infinite-Compressed/resolve/main/evaluation/published-consumer/linux/cpu-zh.json
- Command line
-
hf download hf://Reza2kn/Audio8-ASR-Infinite-Compressed/evaluation/published-consumer/linux/cpu-zh.json
-
curl -L -o cpu-zh.json https://huggingface.co/Reza2kn/Audio8-ASR-Infinite-Compressed/resolve/main/evaluation/published-consumer/linux/cpu-zh.json
4.69 kB
| { | |
| "status": "complete", | |
| "summary": { | |
| "kind": "caption_stream_complete", | |
| "status": "complete", | |
| "audio_samples": 67263, | |
| "audio_seconds": 4.2039375, | |
| "segments": 1, | |
| "selected_samples": 64447, | |
| "unconsumed_selected_samples": 0, | |
| "emissions": 65, | |
| "text_bytes": 42, | |
| "model_seconds": 3.2648139269999996, | |
| "reset_seconds": 6.91e-07, | |
| "vad_seconds": 0.03091347100000001, | |
| "read_seconds": 1.662342578, | |
| "sleep_seconds": 0, | |
| "elapsed_seconds": 4.960410639, | |
| "work_rtf": 0.7839621994856011, | |
| "peak_cache_bytes": 210169856, | |
| "max_retained_input_samples": 6912, | |
| "input_sha256": "54d3a34be5c52ad1cf48c7f68b47299c7ce88bc893fefa4268f3e376631dee3a", | |
| "generated_ids_sha256": "1ceee1bbd9e9884b8497a9c8a04f5566c80e93bdb19a6164d7f4c86789f253d6", | |
| "all_source_audio_observed": true, | |
| "measured_word_aligned_latency": false, | |
| "production_accuracy_approved": false | |
| }, | |
| "score": { | |
| "case_id": "BAC009S0764W0121", | |
| "language": "zh", | |
| "input_sha256": "54d3a34be5c52ad1cf48c7f68b47299c7ce88bc893fefa4268f3e376631dee3a", | |
| "raw_hypothesis": "\u751a\u81f3\u51fa\u73b0\u4ea4\u6613\u51e0\u4e4e\u505c\u6ede\u7684\u60c5\u51b5\u3002", | |
| "normalized_reference": "\u751a\u81f3\u51fa\u73b0\u4ea4\u6613\u51e0\u4e4e\u505c\u6ede\u7684\u60c5\u51b5", | |
| "normalized_hypothesis": "\u751a\u81f3\u51fa\u73b0\u4ea4\u6613\u51e0\u4e4e\u505c\u6ede\u7684\u60c5\u51b5", | |
| "metric": "CER", | |
| "counts": { | |
| "hits": 13, | |
| "substitutions": 0, | |
| "deletions": 0, | |
| "insertions": 0, | |
| "reference_units": 13, | |
| "hypothesis_units": 13, | |
| "errors": 0, | |
| "error_rate": 0.0 | |
| }, | |
| "longest_contiguous_deleted_reference_units": 0, | |
| "aligned_word_timing": { | |
| "measured": false, | |
| "actual_arrival_paced": true, | |
| "timing_basis": "Independent host receipt of flushed JSONL since paced20ms producer start; all source words retained", | |
| "correctly_matched_units": 0, | |
| "ambiguous_matched_units": 0, | |
| "ambiguity_method": "all minimum unit-cost Levenshtein paths, without timing-based selection", | |
| "unambiguous_matched_units": 0, | |
| "unambiguous_maximum_seconds": null, | |
| "unmatched_reference_units": 0, | |
| "matched_reference_fraction": null, | |
| "min_seconds": null, | |
| "p50_seconds": null, | |
| "p95_seconds": null, | |
| "maximum_seconds": null, | |
| "matched_units_above_one_second": null, | |
| "all_matched_units_within_one_second": null, | |
| "whole_release_pass": false, | |
| "limitation": "Lag summaries use one deterministic text alignment and are descriptive when alternative optimal alignments exist. Only unambiguous matches can support per-word timing claims. Omissions/substitutions remain errors, never zero-latency successes. Published reference timing uncertainty and overlapping-speaker word order are not independently corrected." | |
| }, | |
| "alignment": [], | |
| "error_spans": [], | |
| "first_nonempty_delta_seconds": 1.212862549815327, | |
| "production_accuracy_approved": false | |
| }, | |
| "segments": [ | |
| { | |
| "id": 0, | |
| "start_sample": 2816, | |
| "decision_sample": 9216, | |
| "end_sample": 67263, | |
| "final_eof": true, | |
| "windows": 65, | |
| "unconsumed_samples": 0 | |
| } | |
| ], | |
| "all_ids_exact_expected": null, | |
| "independent_tokenizer_oracle": true, | |
| "max_queue_seconds": 0.7081696489999993, | |
| "no_synthetic_nonfinal_eof": true, | |
| "retention_bounded": true, | |
| "release_qualification": false, | |
| "independent_delta_token_clock_oracle": true, | |
| "feeder": { | |
| "chunks": 211, | |
| "samples": 67263, | |
| "duration_seconds": 4.204009119886905, | |
| "max_write_seconds": 8.585001341998577e-05, | |
| "max_delivery_lateness_seconds": 0.0007969938497991436, | |
| "process_started_monotonic": 1189187.799993712, | |
| "feeder_started_monotonic": 1189190.596342693, | |
| "ready_received_monotonic": 1189190.59608933, | |
| "clock_scope": "Feeder starts after host receipt of native readiness. Native and observer clocks are separately recorded; no microphone capture clock." | |
| }, | |
| "command": [ | |
| "/bin/sh", | |
| "/dev/shm/audio8-q3-public-linux-consumer-v1/run-cpu-linux-x86_64.sh", | |
| "-", | |
| "zh", | |
| "/dev/shm/audio8-q3-public-linux-consumer-v1/verification/zh.a8pfx" | |
| ], | |
| "invoked_shipped_launcher": true, | |
| "memory": { | |
| "os_peak_process_rss_bytes": 2004418560, | |
| "os_peak_memory_footprint_bytes": null, | |
| "footprint_available": false, | |
| "rss_source_unit": "KiB (1024 bytes)", | |
| "elapsed_seconds": 7.81, | |
| "user_seconds": 20.23, | |
| "system_seconds": 2.14, | |
| "scope": "Fresh native child process including loading and inference; excludes Python harness. RSS and footprint are separate measurements and must not be summed." | |
| } | |
| } | |