{ "models": [ { "name": "wakehubert_jarvis", "file": "models/wakehubert_jarvis.onnx", "word": "jarvis", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.57, "sha256": "b3db39a777aeb66f69523a7964dd4cae5545770494d24c0d12b1a46669e3de8c", "license": "Apache-2.0", "training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, edge-tts and Google Translate TTS voices, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.984, "false_activations_per_hour": 0.69, "recall_test_set": "Picovoice wake-word benchmark recordings of real speakers", "recall_test_clips": 384, "negative_hours": 46.5 }, "0.8": { "recall": 0.938, "false_activations_per_hour": 0.15, "recall_test_set": "Picovoice wake-word benchmark recordings of real speakers", "recall_test_clips": 384, "negative_hours": 46.5 } } }, { "name": "wakehubert_alexa", "file": "models/wakehubert_alexa.onnx", "word": "alexa", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.44, "sha256": "76380fedd9b5302449e45872bcf24c3a42096949b50235583dfacd6e09b2a274", "license": "Apache-2.0", "training_data": "synthetic only: text-to-speech clips from TigreGotico/synthetic-wakeword-alexa (multi-engine, edge-tts multi-voice and Piper with LibriTTS-R speakers, CC BY 4.0) and OmniVoice clips; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.984, "false_activations_per_hour": 0.6, "recall_test_set": "Picovoice wake-word benchmark recordings of real speakers", "recall_test_clips": 315, "negative_hours": 46.5 }, "0.8": { "recall": 0.914, "false_activations_per_hour": 0.28, "recall_test_set": "Picovoice wake-word benchmark recordings of real speakers", "recall_test_clips": 315, "negative_hours": 46.5 } } }, { "name": "wakehubert_hey_jarvis", "file": "models/wakehubert_hey_jarvis.onnx", "word": "hey jarvis", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.16, "sha256": "02851e30e8c5f81ce8dfcc8ddc8c506ad4e21e2eebb993bd3553ae0017445202", "license": "Apache-2.0", "training_data": "synthetic only: an edge-tts voice grid, edge-tts and Google Translate TTS voices, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.945, "false_activations_per_hour": 0.09, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 384, "negative_hours": 46.5 }, "0.8": { "recall": 0.862, "false_activations_per_hour": 0.02, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 384, "negative_hours": 46.5 } } }, { "name": "wakehubert_hey_marvin", "file": "models/wakehubert_hey_marvin.onnx", "word": "hey marvin", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.34, "sha256": "1d0929a37d107032c0b0c410bba01edc01da406acb1760cc0574e0c1e9112294", "license": "Apache-2.0", "training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.974, "false_activations_per_hour": 0.95, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 386, "negative_hours": 46.5 }, "0.8": { "recall": 0.907, "false_activations_per_hour": 0.24, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 386, "negative_hours": 46.5 } } }, { "name": "wakehubert_home_assistant", "file": "models/wakehubert_home_assistant.onnx", "word": "home assistant", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.19, "sha256": "50a57ca721607b80d7fc198d7875248ef6b93db609026370348ab84192f95d4f", "license": "Apache-2.0", "training_data": "synthetic only: an edge-tts voice grid, edge-tts and Google Translate TTS voices, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.911, "false_activations_per_hour": 0.3, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 380, "negative_hours": 46.5 }, "0.8": { "recall": 0.842, "false_activations_per_hour": 0.15, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 380, "negative_hours": 46.5 } } }, { "name": "wakehubert_okay_nabu", "file": "models/wakehubert_okay_nabu.onnx", "word": "okay nabu", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.45, "sha256": "41fbf1641ccd83a60c128aea27a79e7cdd8fa3612eb1bb8097b076120a8144bd", "license": "Apache-2.0", "training_data": "synthetic only: an edge-tts voice grid, edge-tts and Google Translate TTS voices, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.92, "false_activations_per_hour": 0.26, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 386, "negative_hours": 46.5 }, "0.8": { "recall": 0.754, "false_activations_per_hour": 0.02, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 386, "negative_hours": 46.5 } } }, { "name": "wakehubert_hello_nabu", "file": "models/wakehubert_hello_nabu.onnx", "word": "hello nabu", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.49, "sha256": "ca247672badf4684de5d11b0282fbf129ab5338b17bed0c2f18fc69004de0ece", "license": "Apache-2.0", "training_data": "synthetic only: an edge-tts voice grid, further edge-tts voices, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.809, "false_activations_per_hour": 0.39, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 382, "negative_hours": 46.5 }, "0.8": { "recall": 0.652, "false_activations_per_hour": 0.02, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 382, "negative_hours": 46.5 } } }, { "name": "wakehubert_hey_chatterbox", "file": "models/wakehubert_hey_chatterbox.onnx", "word": "hey chatterbox", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.13, "sha256": "aa3ba38acd01c2c6c001e67f2517bfcefb014842ab4ea84ac71c8b466418794f", "license": "Apache-2.0", "training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.828, "false_activations_per_hour": 0.09, "recall_test_set": "OVOS community recordings of real speakers", "recall_test_clips": 116, "negative_hours": 46.5 }, "0.8": { "recall": 0.612, "false_activations_per_hour": 0.0, "recall_test_set": "OVOS community recordings of real speakers", "recall_test_clips": 116, "negative_hours": 46.5 } } }, { "name": "wakehubert_hey_floyd", "file": "models/wakehubert_hey_floyd.onnx", "word": "hey floyd", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.43, "sha256": "e670d5543446c263e488301ffe337d58954edcc1fe57142c08bc4c783736722b", "license": "Apache-2.0", "training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.917, "false_activations_per_hour": 0.47, "recall_test_set": "OVOS community recordings of real speakers", "recall_test_clips": 96, "negative_hours": 46.5 }, "0.8": { "recall": 0.823, "false_activations_per_hour": 0.02, "recall_test_set": "OVOS community recordings of real speakers", "recall_test_clips": 96, "negative_hours": 46.5 } } }, { "name": "wakehubert_hey_rhasspy", "file": "models/wakehubert_hey_rhasspy.onnx", "word": "hey rhasspy", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.4, "sha256": "6daeeb497f5aac45f769a0c67e816e6d816399df0b704d949a76f752a7d54c89", "license": "Apache-2.0", "training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 1.0, "false_activations_per_hour": 0.6, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 374, "negative_hours": 46.5 }, "0.8": { "recall": 0.976, "false_activations_per_hour": 0.11, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 374, "negative_hours": 46.5 } } }, { "name": "wakehubert_hey_robin", "file": "models/wakehubert_hey_robin.onnx", "word": "hey robin", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.05, "sha256": "3e8a24885153ad920f688679c0c5ac292951af1f258cf3971fa6f527cd61dce3", "license": "Apache-2.0", "training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.995, "false_activations_per_hour": 0.39, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 380, "negative_hours": 46.5 }, "0.8": { "recall": 0.979, "false_activations_per_hour": 0.06, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 380, "negative_hours": 46.5 } } }, { "name": "wakehubert_marvin", "file": "models/wakehubert_marvin.onnx", "word": "marvin", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.06, "sha256": "e1cafc139880f074a8c2545518eb3531ca520df3648b873adf31c0a2a8dd5fa5", "license": "Apache-2.0", "training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.733, "false_activations_per_hour": 0.77, "recall_test_set": "Speech Commands test split, real speakers", "recall_test_clips": 195, "negative_hours": 46.5 }, "0.8": { "recall": 0.718, "false_activations_per_hour": 0.67, "recall_test_set": "Speech Commands test split, real speakers", "recall_test_clips": 195, "negative_hours": 46.5 } } }, { "name": "wakehubert_sheila", "file": "models/wakehubert_sheila.onnx", "word": "sheila", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.47, "sha256": "bdcfc2519fdc4e87dd13d3793b347d425c6ce8aa1f04ee5cb29d4e0d00ede234", "license": "Apache-2.0", "training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.887, "false_activations_per_hour": 2.08, "recall_test_set": "Speech Commands test split, real speakers", "recall_test_clips": 212, "negative_hours": 46.5 }, "0.8": { "recall": 0.844, "false_activations_per_hour": 0.54, "recall_test_set": "Speech Commands test split, real speakers", "recall_test_clips": 212, "negative_hours": 46.5 } } }, { "name": "wakehubert_stop", "file": "models/wakehubert_stop.onnx", "word": "stop", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.14, "sha256": "c071186bc0c2bdbb51dfaff10a1a99de5d2cb90c2d5f7976bb2ae445f089b4c2", "license": "Apache-2.0", "training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.866, "false_activations_per_hour": 2.06, "recall_test_set": "Speech Commands test split, real speakers", "recall_test_clips": 411, "negative_hours": 46.5 }, "0.8": { "recall": 0.769, "false_activations_per_hour": 0.45, "recall_test_set": "Speech Commands test split, real speakers", "recall_test_clips": 411, "negative_hours": 46.5 } } }, { "name": "wakehubert_android", "file": "models/wakehubert_android.onnx", "word": "android", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.42, "sha256": "753ad4800946a552c134e8c330d9eab1a042d4b6fb0b0500d8e9395d3761a2cf", "license": "Apache-2.0", "training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.987, "false_activations_per_hour": 0.95, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 390, "negative_hours": 46.5 }, "0.8": { "recall": 0.964, "false_activations_per_hour": 0.11, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 390, "negative_hours": 46.5 } } }, { "name": "wakehubert_hey_computer", "file": "models/wakehubert_hey_computer.onnx", "word": "hey computer", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.31, "sha256": "6ea782477908e8512a59a7c23541d42913ca39c95295a20b788693cc4a1b07f3", "license": "Apache-2.0", "training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.964, "false_activations_per_hour": 0.47, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 390, "negative_hours": 46.5 }, "0.8": { "recall": 0.933, "false_activations_per_hour": 0.04, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 390, "negative_hours": 46.5 } } }, { "name": "wakehubert_hey_k9", "file": "models/wakehubert_hey_k9.onnx", "word": "hey k9", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.06, "sha256": "3705ef79d9c7cce3c7e89669b4e1a55f0495ea2719287c9225fed5b6a09d63bf", "license": "Apache-2.0", "training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.994, "false_activations_per_hour": 0.19, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 338, "negative_hours": 46.5 }, "0.8": { "recall": 0.935, "false_activations_per_hour": 0.04, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 338, "negative_hours": 46.5 } } }, { "name": "wakehubert_hey_scout", "file": "models/wakehubert_hey_scout.onnx", "word": "hey scout", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.36, "sha256": "c72fba4f93cf69b4bd9510682a3a80db67f663a0686f4d331a34d44ff031fdf3", "license": "Apache-2.0", "training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.959, "false_activations_per_hour": 0.09, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 390, "negative_hours": 46.5 }, "0.8": { "recall": 0.938, "false_activations_per_hour": 0.04, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 390, "negative_hours": 46.5 } } }, { "name": "wakehubert_wake_up", "file": "models/wakehubert_wake_up.onnx", "word": "wake up", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.36, "sha256": "524e54bc24cf790339928bf8656d0c23d533108328e7b316afbd65b8cdfe28aa", "license": "Apache-2.0", "training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.981, "false_activations_per_hour": 1.1, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 378, "negative_hours": 46.5 }, "0.8": { "recall": 0.968, "false_activations_per_hour": 0.19, "recall_test_set": "held-out synthetic voices", "recall_test_clips": 378, "negative_hours": 46.5 } } }, { "name": "wakehubert_computer", "file": "models/wakehubert_computer.onnx", "word": "computer", "language": "en", "featurizer": "wakehubert", "calibrated": false, "default_threshold": 0.99, "sha256": "4ecda088bcce9951730de9fc8b1c4f801ea208add0cbf8a1b9397ed7d8329496", "license": "Apache-2.0", "training_data": "TigreGotico/synthetic-wakeword-computer (CC BY 4.0); negatives: TigreGotico/not-wake-words-speech-en (CC BY 4.0) and AudioSet-derived clips", "measured": null }, { "name": "wakehubert_hey_mycroft", "file": "models/wakehubert_hey_mycroft.onnx", "word": "hey mycroft", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.18, "sha256": "5eb388364e489785fd34edf668dd5f780b0481d0c1de3d286720ed3cf545a193", "license": "Apache-2.0", "training_data": "synthetic only: text-to-speech clips from TigreGotico/synthetic-wakeword-hey_mycroft (edge-tts, Piper with LibriTTS-R speakers, Amazon Polly, OVOS TTS plugins, OmniVoice, and Chatterbox and OpenVoice conversions onto Common Voice speakers, CC BY 4.0) and further edge-tts and Google Translate TTS clips, with eight held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.974, "false_activations_per_hour": 1.57, "recall_test_set": "held-out synthetic voices: held-out edge-tts voices (285 clips) and Kokoro voices converted to unseen speakers (379 clips)", "recall_test_clips": 664, "negative_hours": 46.5 }, "0.8": { "recall": 0.848, "false_activations_per_hour": 0.24, "recall_test_set": "held-out synthetic voices: held-out edge-tts voices (285 clips) and Kokoro voices converted to unseen speakers (379 clips)", "recall_test_clips": 664, "negative_hours": 46.5 } } }, { "name": "wakehubert_hey_ziggy", "file": "models/wakehubert_hey_ziggy.onnx", "word": "hey ziggy", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.27, "sha256": "55e849cb7a20ce037cbd9c6d6d4508f182731e35b1a3be159673c571d0890b02", "license": "Apache-2.0", "training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, each checked against a speech recogniser transcript, with eight held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.936, "false_activations_per_hour": 0.64, "recall_test_set": "held-out synthetic voices: OmniVoice clips and held-out edge-tts voices", "recall_test_clips": 374, "negative_hours": 46.5 }, "0.8": { "recall": 0.893, "false_activations_per_hour": 0.24, "recall_test_set": "held-out synthetic voices: OmniVoice clips and held-out edge-tts voices", "recall_test_clips": 374, "negative_hours": 46.5 } } }, { "name": "wakehubert_hey_potato", "file": "models/wakehubert_hey_potato.onnx", "word": "hey potato", "language": "en", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.15, "sha256": "3951a1f258bbe8d2685bb8c4c3921de0b8b26c89783af37d9ce999fb8219170e", "license": "Apache-2.0", "training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, each checked against a speech recogniser transcript, with eight held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.889, "false_activations_per_hour": 0.34, "recall_test_set": "held-out synthetic voices: 360 clips in held-out edge-tts voices only, without the OmniVoice test clips used for the other words", "recall_test_clips": 360, "negative_hours": 46.5 }, "0.8": { "recall": 0.806, "false_activations_per_hour": 0.02, "recall_test_set": "held-out synthetic voices: 360 clips in held-out edge-tts voices only, without the OmniVoice test clips used for the other words", "recall_test_clips": 360, "negative_hours": 46.5 } } }, { "name": "wakehubert_hey_stemcom", "file": "models/wakehubert_hey_stemcom.onnx", "word": "hey stemcom", "language": "nl", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.28, "sha256": "0610626100881dc8da8b416a0bad7bd03f4ea52537359ac9ebbe18ff49731b7a", "license": "Apache-2.0", "training_data": "synthetic only: a Dutch edge-tts voice grid (nl-NL and nl-BE), voice-converted copies of edge-tts clips, OmniVoice clips, each checked against a speech recogniser transcript, with one held-out edge-tts test voice excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample", "measured": { "default": { "recall": 0.89, "false_activations_per_hour": 0.21, "recall_test_set": "held-out synthetic voices: OmniVoice clips and held-out edge-tts voices", "recall_test_clips": 337, "negative_hours": 46.5 }, "0.8": { "recall": 0.831, "false_activations_per_hour": 0.06, "recall_test_set": "held-out synthetic voices: OmniVoice clips and held-out edge-tts voices", "recall_test_clips": 337, "negative_hours": 46.5 } } }, { "name": "wakehubert_despierta", "file": "models/wakehubert_despierta.onnx", "word": "despierta", "language": "es", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.51, "sha256": "8e993142a4ee30572f1338ce205a6466e61645b3f47465658d0bf7161c4e3ba2", "license": "Apache-2.0", "training_data": "synthetic only: OmniVoice clips with no reference speaker, one seed per clip, each checked against a speech recogniser transcript, and the training clips of TigreGotico/synthetic-wakeword-despierta (CC BY 4.0); LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv, the even half of an AudioSet noise sample, and Spanish speech from Common Voice 17 train (CC0) and FLEURS train (CC BY 4.0) cut into 1.5 s windows", "measured": { "default": { "recall": 0.984, "false_activations_per_hour": 0.58, "recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses", "recall_test_clips": 370, "negative_hours": 31.1, "false_activations_per_hour_in_language": 0.6, "in_language_negative_hours": 3.34 }, "0.8": { "recall": 0.941, "false_activations_per_hour": 0.03, "recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses", "recall_test_clips": 370, "negative_hours": 31.1, "false_activations_per_hour_in_language": 0.3, "in_language_negative_hours": 3.34 } }, "notes": "In-language false activations: 0.60 per hour over 3.3 h of Spanish Common Voice 17 test and FLEURS dev speech at the default threshold. At the default threshold it fires on 19 of 20 edge-tts near-word probe clips (\"despiertas\", \"despierto\", \"depierta\", \"desperta\" and \"despertar\")." }, { "name": "wakehubert_aufwachen", "file": "models/wakehubert_aufwachen.onnx", "word": "aufwachen", "language": "de", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.57, "sha256": "4df0063c8087a39677ee708da05570447081571401e3142a8515abbd95c7fafb", "license": "Apache-2.0", "training_data": "synthetic only: OmniVoice clips with no reference speaker, one seed per clip, each checked against a speech recogniser transcript, and the training clips of TigreGotico/synthetic-wakeword-aufwachen (CC BY 4.0); LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv, the even half of an AudioSet noise sample, and German speech from Common Voice 17 train (CC0) and FLEURS train (CC BY 4.0) cut into 1.5 s windows", "measured": { "default": { "recall": 0.995, "false_activations_per_hour": 0.22, "recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses", "recall_test_clips": 390, "negative_hours": 31.1, "false_activations_per_hour_in_language": 1.85, "in_language_negative_hours": 3.25 }, "0.8": { "recall": 0.972, "false_activations_per_hour": 0.0, "recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses", "recall_test_clips": 390, "negative_hours": 31.1, "false_activations_per_hour_in_language": 0.31, "in_language_negative_hours": 3.25 } }, "notes": "In-language false activations: 1.85 per hour over 3.2 h of German Common Voice 17 test and FLEURS dev speech at the default threshold. At the default threshold it fires on 8 of 8 edge-tts near-word probe clips (\"aufmachen\" and \"aufwachten\")." }, { "name": "wakehubert_wakker_worden", "file": "models/wakehubert_wakker_worden.onnx", "word": "wakker worden", "language": "nl", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.56, "sha256": "5c77d9de56c923ce0aa61506d6faaf6378949fd6df1a764110eae3323427d4ad", "license": "Apache-2.0", "training_data": "synthetic only: OmniVoice clips with no reference speaker, one seed per clip, each checked against a speech recogniser transcript, and the training clips of TigreGotico/synthetic-wakeword-wakker_worden (CC BY 4.0); LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv, the even half of an AudioSet noise sample, and Dutch speech from Common Voice 17 train (CC0) and FLEURS train (CC BY 4.0) cut into 1.5 s windows", "measured": { "default": { "recall": 0.997, "false_activations_per_hour": 0.29, "recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses", "recall_test_clips": 390, "negative_hours": 31.1, "false_activations_per_hour_in_language": 0.0, "in_language_negative_hours": 2.43 }, "0.8": { "recall": 0.985, "false_activations_per_hour": 0.03, "recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses", "recall_test_clips": 390, "negative_hours": 31.1, "false_activations_per_hour_in_language": 0.0, "in_language_negative_hours": 2.43 } }, "notes": "In-language false activations: 0.00 per hour over 2.4 h of Dutch Common Voice 17 test and FLEURS dev speech at the default threshold. At the default threshold it fires on none of 8 edge-tts near-word probe clips (\"wakker\", \"worden\")." }, { "name": "wakehubert_sveglia", "file": "models/wakehubert_sveglia.onnx", "word": "sveglia", "language": "it", "featurizer": "wakehubert-int8", "calibrated": true, "default_threshold": 0.47, "sha256": "523544cc3629f4bd995b9acd4ec98ef217c1240b1c93ddd2f594f71399fdc473", "license": "Apache-2.0", "training_data": "synthetic only: OmniVoice clips with no reference speaker, one seed per clip, each checked against a speech recogniser transcript, and the training clips of TigreGotico/synthetic-wakeword-sveglia (CC BY 4.0); LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv, the even half of an AudioSet noise sample, and Italian speech from Common Voice 17 train (CC0) and FLEURS train (CC BY 4.0) cut into 1.5 s windows", "measured": { "default": { "recall": 0.995, "false_activations_per_hour": 1.16, "recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses", "recall_test_clips": 222, "negative_hours": 31.1, "false_activations_per_hour_in_language": 1.7, "in_language_negative_hours": 3.53 }, "0.8": { "recall": 0.977, "false_activations_per_hour": 0.1, "recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses", "recall_test_clips": 222, "negative_hours": 31.1, "false_activations_per_hour_in_language": 0.28, "in_language_negative_hours": 3.53 } }, "notes": "In-language false activations: 1.70 per hour over 3.5 h of Italian Common Voice 17 test and FLEURS dev speech at the default threshold. At the default threshold it fires on 18 of 24 edge-tts near-word probe clips (\"sveglio\", \"sveglie\", \"svegliati\", and the rhymes \"meraviglia\", \"bottiglia\" and \"voglia\")." } ] }