Download models.json from OpenVoiceOS/wakehubert-wakewords: direct link, hf CLI and curl.
- Browser
- Download file 36.7 kB
-
https://huggingface.co/OpenVoiceOS/wakehubert-wakewords/resolve/main/models.json
- Command line
-
hf download hf://OpenVoiceOS/wakehubert-wakewords/models.json
-
curl -L -o models.json https://huggingface.co/OpenVoiceOS/wakehubert-wakewords/resolve/main/models.json
36.7 kB
| { | |
| "models": [ | |
| { | |
| "name": "wakehubert_jarvis", | |
| "file": "models/wakehubert_jarvis.onnx", | |
| "word": "jarvis", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.57, | |
| "sha256": "b3db39a777aeb66f69523a7964dd4cae5545770494d24c0d12b1a46669e3de8c", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, edge-tts and Google Translate TTS voices, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.984, | |
| "false_activations_per_hour": 0.69, | |
| "recall_test_set": "Picovoice wake-word benchmark recordings of real speakers", | |
| "recall_test_clips": 384, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.938, | |
| "false_activations_per_hour": 0.15, | |
| "recall_test_set": "Picovoice wake-word benchmark recordings of real speakers", | |
| "recall_test_clips": 384, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_alexa", | |
| "file": "models/wakehubert_alexa.onnx", | |
| "word": "alexa", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.44, | |
| "sha256": "76380fedd9b5302449e45872bcf24c3a42096949b50235583dfacd6e09b2a274", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: text-to-speech clips from TigreGotico/synthetic-wakeword-alexa (multi-engine, edge-tts multi-voice and Piper with LibriTTS-R speakers, CC BY 4.0) and OmniVoice clips; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.984, | |
| "false_activations_per_hour": 0.6, | |
| "recall_test_set": "Picovoice wake-word benchmark recordings of real speakers", | |
| "recall_test_clips": 315, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.914, | |
| "false_activations_per_hour": 0.28, | |
| "recall_test_set": "Picovoice wake-word benchmark recordings of real speakers", | |
| "recall_test_clips": 315, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_hey_jarvis", | |
| "file": "models/wakehubert_hey_jarvis.onnx", | |
| "word": "hey jarvis", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.16, | |
| "sha256": "02851e30e8c5f81ce8dfcc8ddc8c506ad4e21e2eebb993bd3553ae0017445202", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: an edge-tts voice grid, edge-tts and Google Translate TTS voices, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.945, | |
| "false_activations_per_hour": 0.09, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 384, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.862, | |
| "false_activations_per_hour": 0.02, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 384, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_hey_marvin", | |
| "file": "models/wakehubert_hey_marvin.onnx", | |
| "word": "hey marvin", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.34, | |
| "sha256": "1d0929a37d107032c0b0c410bba01edc01da406acb1760cc0574e0c1e9112294", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.974, | |
| "false_activations_per_hour": 0.95, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 386, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.907, | |
| "false_activations_per_hour": 0.24, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 386, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_home_assistant", | |
| "file": "models/wakehubert_home_assistant.onnx", | |
| "word": "home assistant", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.19, | |
| "sha256": "50a57ca721607b80d7fc198d7875248ef6b93db609026370348ab84192f95d4f", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: an edge-tts voice grid, edge-tts and Google Translate TTS voices, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.911, | |
| "false_activations_per_hour": 0.3, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 380, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.842, | |
| "false_activations_per_hour": 0.15, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 380, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_okay_nabu", | |
| "file": "models/wakehubert_okay_nabu.onnx", | |
| "word": "okay nabu", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.45, | |
| "sha256": "41fbf1641ccd83a60c128aea27a79e7cdd8fa3612eb1bb8097b076120a8144bd", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: an edge-tts voice grid, edge-tts and Google Translate TTS voices, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.92, | |
| "false_activations_per_hour": 0.26, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 386, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.754, | |
| "false_activations_per_hour": 0.02, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 386, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_hello_nabu", | |
| "file": "models/wakehubert_hello_nabu.onnx", | |
| "word": "hello nabu", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.49, | |
| "sha256": "ca247672badf4684de5d11b0282fbf129ab5338b17bed0c2f18fc69004de0ece", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: an edge-tts voice grid, further edge-tts voices, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.809, | |
| "false_activations_per_hour": 0.39, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 382, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.652, | |
| "false_activations_per_hour": 0.02, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 382, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_hey_chatterbox", | |
| "file": "models/wakehubert_hey_chatterbox.onnx", | |
| "word": "hey chatterbox", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.13, | |
| "sha256": "aa3ba38acd01c2c6c001e67f2517bfcefb014842ab4ea84ac71c8b466418794f", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.828, | |
| "false_activations_per_hour": 0.09, | |
| "recall_test_set": "OVOS community recordings of real speakers", | |
| "recall_test_clips": 116, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.612, | |
| "false_activations_per_hour": 0.0, | |
| "recall_test_set": "OVOS community recordings of real speakers", | |
| "recall_test_clips": 116, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_hey_floyd", | |
| "file": "models/wakehubert_hey_floyd.onnx", | |
| "word": "hey floyd", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.43, | |
| "sha256": "e670d5543446c263e488301ffe337d58954edcc1fe57142c08bc4c783736722b", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.917, | |
| "false_activations_per_hour": 0.47, | |
| "recall_test_set": "OVOS community recordings of real speakers", | |
| "recall_test_clips": 96, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.823, | |
| "false_activations_per_hour": 0.02, | |
| "recall_test_set": "OVOS community recordings of real speakers", | |
| "recall_test_clips": 96, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_hey_rhasspy", | |
| "file": "models/wakehubert_hey_rhasspy.onnx", | |
| "word": "hey rhasspy", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.4, | |
| "sha256": "6daeeb497f5aac45f769a0c67e816e6d816399df0b704d949a76f752a7d54c89", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 1.0, | |
| "false_activations_per_hour": 0.6, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 374, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.976, | |
| "false_activations_per_hour": 0.11, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 374, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_hey_robin", | |
| "file": "models/wakehubert_hey_robin.onnx", | |
| "word": "hey robin", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.05, | |
| "sha256": "3e8a24885153ad920f688679c0c5ac292951af1f258cf3971fa6f527cd61dce3", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.995, | |
| "false_activations_per_hour": 0.39, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 380, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.979, | |
| "false_activations_per_hour": 0.06, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 380, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_marvin", | |
| "file": "models/wakehubert_marvin.onnx", | |
| "word": "marvin", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.06, | |
| "sha256": "e1cafc139880f074a8c2545518eb3531ca520df3648b873adf31c0a2a8dd5fa5", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.733, | |
| "false_activations_per_hour": 0.77, | |
| "recall_test_set": "Speech Commands test split, real speakers", | |
| "recall_test_clips": 195, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.718, | |
| "false_activations_per_hour": 0.67, | |
| "recall_test_set": "Speech Commands test split, real speakers", | |
| "recall_test_clips": 195, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_sheila", | |
| "file": "models/wakehubert_sheila.onnx", | |
| "word": "sheila", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.47, | |
| "sha256": "bdcfc2519fdc4e87dd13d3793b347d425c6ce8aa1f04ee5cb29d4e0d00ede234", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.887, | |
| "false_activations_per_hour": 2.08, | |
| "recall_test_set": "Speech Commands test split, real speakers", | |
| "recall_test_clips": 212, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.844, | |
| "false_activations_per_hour": 0.54, | |
| "recall_test_set": "Speech Commands test split, real speakers", | |
| "recall_test_clips": 212, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_stop", | |
| "file": "models/wakehubert_stop.onnx", | |
| "word": "stop", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.14, | |
| "sha256": "c071186bc0c2bdbb51dfaff10a1a99de5d2cb90c2d5f7976bb2ae445f089b4c2", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.866, | |
| "false_activations_per_hour": 2.06, | |
| "recall_test_set": "Speech Commands test split, real speakers", | |
| "recall_test_clips": 411, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.769, | |
| "false_activations_per_hour": 0.45, | |
| "recall_test_set": "Speech Commands test split, real speakers", | |
| "recall_test_clips": 411, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_android", | |
| "file": "models/wakehubert_android.onnx", | |
| "word": "android", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.42, | |
| "sha256": "753ad4800946a552c134e8c330d9eab1a042d4b6fb0b0500d8e9395d3761a2cf", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.987, | |
| "false_activations_per_hour": 0.95, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 390, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.964, | |
| "false_activations_per_hour": 0.11, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 390, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_hey_computer", | |
| "file": "models/wakehubert_hey_computer.onnx", | |
| "word": "hey computer", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.31, | |
| "sha256": "6ea782477908e8512a59a7c23541d42913ca39c95295a20b788693cc4a1b07f3", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.964, | |
| "false_activations_per_hour": 0.47, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 390, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.933, | |
| "false_activations_per_hour": 0.04, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 390, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_hey_k9", | |
| "file": "models/wakehubert_hey_k9.onnx", | |
| "word": "hey k9", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.06, | |
| "sha256": "3705ef79d9c7cce3c7e89669b4e1a55f0495ea2719287c9225fed5b6a09d63bf", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.994, | |
| "false_activations_per_hour": 0.19, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 338, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.935, | |
| "false_activations_per_hour": 0.04, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 338, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_hey_scout", | |
| "file": "models/wakehubert_hey_scout.onnx", | |
| "word": "hey scout", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.36, | |
| "sha256": "c72fba4f93cf69b4bd9510682a3a80db67f663a0686f4d331a34d44ff031fdf3", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.959, | |
| "false_activations_per_hour": 0.09, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 390, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.938, | |
| "false_activations_per_hour": 0.04, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 390, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_wake_up", | |
| "file": "models/wakehubert_wake_up.onnx", | |
| "word": "wake up", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.36, | |
| "sha256": "524e54bc24cf790339928bf8656d0c23d533108328e7b316afbd65b8cdfe28aa", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.981, | |
| "false_activations_per_hour": 1.1, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 378, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.968, | |
| "false_activations_per_hour": 0.19, | |
| "recall_test_set": "held-out synthetic voices", | |
| "recall_test_clips": 378, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_computer", | |
| "file": "models/wakehubert_computer.onnx", | |
| "word": "computer", | |
| "language": "en", | |
| "featurizer": "wakehubert", | |
| "calibrated": false, | |
| "default_threshold": 0.99, | |
| "sha256": "4ecda088bcce9951730de9fc8b1c4f801ea208add0cbf8a1b9397ed7d8329496", | |
| "license": "Apache-2.0", | |
| "training_data": "TigreGotico/synthetic-wakeword-computer (CC BY 4.0); negatives: TigreGotico/not-wake-words-speech-en (CC BY 4.0) and AudioSet-derived clips", | |
| "measured": null | |
| }, | |
| { | |
| "name": "wakehubert_hey_mycroft", | |
| "file": "models/wakehubert_hey_mycroft.onnx", | |
| "word": "hey mycroft", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.18, | |
| "sha256": "5eb388364e489785fd34edf668dd5f780b0481d0c1de3d286720ed3cf545a193", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: text-to-speech clips from TigreGotico/synthetic-wakeword-hey_mycroft (edge-tts, Piper with LibriTTS-R speakers, Amazon Polly, OVOS TTS plugins, OmniVoice, and Chatterbox and OpenVoice conversions onto Common Voice speakers, CC BY 4.0) and further edge-tts and Google Translate TTS clips, with eight held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.974, | |
| "false_activations_per_hour": 1.57, | |
| "recall_test_set": "held-out synthetic voices: held-out edge-tts voices (285 clips) and Kokoro voices converted to unseen speakers (379 clips)", | |
| "recall_test_clips": 664, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.848, | |
| "false_activations_per_hour": 0.24, | |
| "recall_test_set": "held-out synthetic voices: held-out edge-tts voices (285 clips) and Kokoro voices converted to unseen speakers (379 clips)", | |
| "recall_test_clips": 664, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_hey_ziggy", | |
| "file": "models/wakehubert_hey_ziggy.onnx", | |
| "word": "hey ziggy", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.27, | |
| "sha256": "55e849cb7a20ce037cbd9c6d6d4508f182731e35b1a3be159673c571d0890b02", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, each checked against a speech recogniser transcript, with eight held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.936, | |
| "false_activations_per_hour": 0.64, | |
| "recall_test_set": "held-out synthetic voices: OmniVoice clips and held-out edge-tts voices", | |
| "recall_test_clips": 374, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.893, | |
| "false_activations_per_hour": 0.24, | |
| "recall_test_set": "held-out synthetic voices: OmniVoice clips and held-out edge-tts voices", | |
| "recall_test_clips": 374, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_hey_potato", | |
| "file": "models/wakehubert_hey_potato.onnx", | |
| "word": "hey potato", | |
| "language": "en", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.15, | |
| "sha256": "3951a1f258bbe8d2685bb8c4c3921de0b8b26c89783af37d9ce999fb8219170e", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, each checked against a speech recogniser transcript, with eight held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.889, | |
| "false_activations_per_hour": 0.34, | |
| "recall_test_set": "held-out synthetic voices: 360 clips in held-out edge-tts voices only, without the OmniVoice test clips used for the other words", | |
| "recall_test_clips": 360, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.806, | |
| "false_activations_per_hour": 0.02, | |
| "recall_test_set": "held-out synthetic voices: 360 clips in held-out edge-tts voices only, without the OmniVoice test clips used for the other words", | |
| "recall_test_clips": 360, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_hey_stemcom", | |
| "file": "models/wakehubert_hey_stemcom.onnx", | |
| "word": "hey stemcom", | |
| "language": "nl", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.28, | |
| "sha256": "0610626100881dc8da8b416a0bad7bd03f4ea52537359ac9ebbe18ff49731b7a", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: a Dutch edge-tts voice grid (nl-NL and nl-BE), voice-converted copies of edge-tts clips, OmniVoice clips, each checked against a speech recogniser transcript, with one held-out edge-tts test voice excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", | |
| "measured": { | |
| "default": { | |
| "recall": 0.89, | |
| "false_activations_per_hour": 0.21, | |
| "recall_test_set": "held-out synthetic voices: OmniVoice clips and held-out edge-tts voices", | |
| "recall_test_clips": 337, | |
| "negative_hours": 46.5 | |
| }, | |
| "0.8": { | |
| "recall": 0.831, | |
| "false_activations_per_hour": 0.06, | |
| "recall_test_set": "held-out synthetic voices: OmniVoice clips and held-out edge-tts voices", | |
| "recall_test_clips": 337, | |
| "negative_hours": 46.5 | |
| } | |
| } | |
| }, | |
| { | |
| "name": "wakehubert_despierta", | |
| "file": "models/wakehubert_despierta.onnx", | |
| "word": "despierta", | |
| "language": "es", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.51, | |
| "sha256": "8e993142a4ee30572f1338ce205a6466e61645b3f47465658d0bf7161c4e3ba2", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: OmniVoice clips with no reference speaker, one seed per clip, each checked against a speech recogniser transcript, and the training clips of TigreGotico/synthetic-wakeword-despierta (CC BY 4.0); LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv, the even half of an AudioSet noise sample, and Spanish speech from Common Voice 17 train (CC0) and FLEURS train (CC BY 4.0) cut into 1.5 s windows", | |
| "measured": { | |
| "default": { | |
| "recall": 0.984, | |
| "false_activations_per_hour": 0.58, | |
| "recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses", | |
| "recall_test_clips": 370, | |
| "negative_hours": 31.1, | |
| "false_activations_per_hour_in_language": 0.6, | |
| "in_language_negative_hours": 3.34 | |
| }, | |
| "0.8": { | |
| "recall": 0.941, | |
| "false_activations_per_hour": 0.03, | |
| "recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses", | |
| "recall_test_clips": 370, | |
| "negative_hours": 31.1, | |
| "false_activations_per_hour_in_language": 0.3, | |
| "in_language_negative_hours": 3.34 | |
| } | |
| }, | |
| "notes": "In-language false activations: 0.60 per hour over 3.3 h of Spanish Common Voice 17 test and FLEURS dev speech at the default threshold. At the default threshold it fires on 19 of 20 edge-tts near-word probe clips (\"despiertas\", \"despierto\", \"depierta\", \"desperta\" and \"despertar\")." | |
| }, | |
| { | |
| "name": "wakehubert_aufwachen", | |
| "file": "models/wakehubert_aufwachen.onnx", | |
| "word": "aufwachen", | |
| "language": "de", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.57, | |
| "sha256": "4df0063c8087a39677ee708da05570447081571401e3142a8515abbd95c7fafb", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: OmniVoice clips with no reference speaker, one seed per clip, each checked against a speech recogniser transcript, and the training clips of TigreGotico/synthetic-wakeword-aufwachen (CC BY 4.0); LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv, the even half of an AudioSet noise sample, and German speech from Common Voice 17 train (CC0) and FLEURS train (CC BY 4.0) cut into 1.5 s windows", | |
| "measured": { | |
| "default": { | |
| "recall": 0.995, | |
| "false_activations_per_hour": 0.22, | |
| "recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses", | |
| "recall_test_clips": 390, | |
| "negative_hours": 31.1, | |
| "false_activations_per_hour_in_language": 1.85, | |
| "in_language_negative_hours": 3.25 | |
| }, | |
| "0.8": { | |
| "recall": 0.972, | |
| "false_activations_per_hour": 0.0, | |
| "recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses", | |
| "recall_test_clips": 390, | |
| "negative_hours": 31.1, | |
| "false_activations_per_hour_in_language": 0.31, | |
| "in_language_negative_hours": 3.25 | |
| } | |
| }, | |
| "notes": "In-language false activations: 1.85 per hour over 3.2 h of German Common Voice 17 test and FLEURS dev speech at the default threshold. At the default threshold it fires on 8 of 8 edge-tts near-word probe clips (\"aufmachen\" and \"aufwachten\")." | |
| }, | |
| { | |
| "name": "wakehubert_wakker_worden", | |
| "file": "models/wakehubert_wakker_worden.onnx", | |
| "word": "wakker worden", | |
| "language": "nl", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.56, | |
| "sha256": "5c77d9de56c923ce0aa61506d6faaf6378949fd6df1a764110eae3323427d4ad", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: OmniVoice clips with no reference speaker, one seed per clip, each checked against a speech recogniser transcript, and the training clips of TigreGotico/synthetic-wakeword-wakker_worden (CC BY 4.0); LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv, the even half of an AudioSet noise sample, and Dutch speech from Common Voice 17 train (CC0) and FLEURS train (CC BY 4.0) cut into 1.5 s windows", | |
| "measured": { | |
| "default": { | |
| "recall": 0.997, | |
| "false_activations_per_hour": 0.29, | |
| "recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses", | |
| "recall_test_clips": 390, | |
| "negative_hours": 31.1, | |
| "false_activations_per_hour_in_language": 0.0, | |
| "in_language_negative_hours": 2.43 | |
| }, | |
| "0.8": { | |
| "recall": 0.985, | |
| "false_activations_per_hour": 0.03, | |
| "recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses", | |
| "recall_test_clips": 390, | |
| "negative_hours": 31.1, | |
| "false_activations_per_hour_in_language": 0.0, | |
| "in_language_negative_hours": 2.43 | |
| } | |
| }, | |
| "notes": "In-language false activations: 0.00 per hour over 2.4 h of Dutch Common Voice 17 test and FLEURS dev speech at the default threshold. At the default threshold it fires on none of 8 edge-tts near-word probe clips (\"wakker\", \"worden\")." | |
| }, | |
| { | |
| "name": "wakehubert_sveglia", | |
| "file": "models/wakehubert_sveglia.onnx", | |
| "word": "sveglia", | |
| "language": "it", | |
| "featurizer": "wakehubert-int8", | |
| "calibrated": true, | |
| "default_threshold": 0.47, | |
| "sha256": "523544cc3629f4bd995b9acd4ec98ef217c1240b1c93ddd2f594f71399fdc473", | |
| "license": "Apache-2.0", | |
| "training_data": "synthetic only: OmniVoice clips with no reference speaker, one seed per clip, each checked against a speech recogniser transcript, and the training clips of TigreGotico/synthetic-wakeword-sveglia (CC BY 4.0); LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv, the even half of an AudioSet noise sample, and Italian speech from Common Voice 17 train (CC0) and FLEURS train (CC BY 4.0) cut into 1.5 s windows", | |
| "measured": { | |
| "default": { | |
| "recall": 0.995, | |
| "false_activations_per_hour": 1.16, | |
| "recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses", | |
| "recall_test_clips": 222, | |
| "negative_hours": 31.1, | |
| "false_activations_per_hour_in_language": 1.7, | |
| "in_language_negative_hours": 3.53 | |
| }, | |
| "0.8": { | |
| "recall": 0.977, | |
| "false_activations_per_hour": 0.1, | |
| "recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses", | |
| "recall_test_clips": 222, | |
| "negative_hours": 31.1, | |
| "false_activations_per_hour_in_language": 0.28, | |
| "in_language_negative_hours": 3.53 | |
| } | |
| }, | |
| "notes": "In-language false activations: 1.70 per hour over 3.5 h of Italian Common Voice 17 test and FLEURS dev speech at the default threshold. At the default threshold it fires on 18 of 24 edge-tts near-word probe clips (\"sveglio\", \"sveglie\", \"svegliati\", and the rhymes \"meraviglia\", \"bottiglia\" and \"voglia\")." | |
| } | |
| ] | |
| } | |