wakehubert-wakewords / models.json
Jarbas's picture
Plugin format for the Spanish, German, Dutch and Italian wake-up models
4919c6c verified
Raw History Blame Contribute Delete
36.7 kB
{
"models": [
{
"name": "wakehubert_jarvis",
"file": "models/wakehubert_jarvis.onnx",
"word": "jarvis",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.57,
"sha256": "b3db39a777aeb66f69523a7964dd4cae5545770494d24c0d12b1a46669e3de8c",
"license": "Apache-2.0",
"training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, edge-tts and Google Translate TTS voices, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.984,
"false_activations_per_hour": 0.69,
"recall_test_set": "Picovoice wake-word benchmark recordings of real speakers",
"recall_test_clips": 384,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.938,
"false_activations_per_hour": 0.15,
"recall_test_set": "Picovoice wake-word benchmark recordings of real speakers",
"recall_test_clips": 384,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_alexa",
"file": "models/wakehubert_alexa.onnx",
"word": "alexa",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.44,
"sha256": "76380fedd9b5302449e45872bcf24c3a42096949b50235583dfacd6e09b2a274",
"license": "Apache-2.0",
"training_data": "synthetic only: text-to-speech clips from TigreGotico/synthetic-wakeword-alexa (multi-engine, edge-tts multi-voice and Piper with LibriTTS-R speakers, CC BY 4.0) and OmniVoice clips; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.984,
"false_activations_per_hour": 0.6,
"recall_test_set": "Picovoice wake-word benchmark recordings of real speakers",
"recall_test_clips": 315,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.914,
"false_activations_per_hour": 0.28,
"recall_test_set": "Picovoice wake-word benchmark recordings of real speakers",
"recall_test_clips": 315,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_hey_jarvis",
"file": "models/wakehubert_hey_jarvis.onnx",
"word": "hey jarvis",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.16,
"sha256": "02851e30e8c5f81ce8dfcc8ddc8c506ad4e21e2eebb993bd3553ae0017445202",
"license": "Apache-2.0",
"training_data": "synthetic only: an edge-tts voice grid, edge-tts and Google Translate TTS voices, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.945,
"false_activations_per_hour": 0.09,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 384,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.862,
"false_activations_per_hour": 0.02,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 384,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_hey_marvin",
"file": "models/wakehubert_hey_marvin.onnx",
"word": "hey marvin",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.34,
"sha256": "1d0929a37d107032c0b0c410bba01edc01da406acb1760cc0574e0c1e9112294",
"license": "Apache-2.0",
"training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.974,
"false_activations_per_hour": 0.95,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 386,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.907,
"false_activations_per_hour": 0.24,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 386,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_home_assistant",
"file": "models/wakehubert_home_assistant.onnx",
"word": "home assistant",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.19,
"sha256": "50a57ca721607b80d7fc198d7875248ef6b93db609026370348ab84192f95d4f",
"license": "Apache-2.0",
"training_data": "synthetic only: an edge-tts voice grid, edge-tts and Google Translate TTS voices, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.911,
"false_activations_per_hour": 0.3,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 380,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.842,
"false_activations_per_hour": 0.15,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 380,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_okay_nabu",
"file": "models/wakehubert_okay_nabu.onnx",
"word": "okay nabu",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.45,
"sha256": "41fbf1641ccd83a60c128aea27a79e7cdd8fa3612eb1bb8097b076120a8144bd",
"license": "Apache-2.0",
"training_data": "synthetic only: an edge-tts voice grid, edge-tts and Google Translate TTS voices, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.92,
"false_activations_per_hour": 0.26,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 386,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.754,
"false_activations_per_hour": 0.02,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 386,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_hello_nabu",
"file": "models/wakehubert_hello_nabu.onnx",
"word": "hello nabu",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.49,
"sha256": "ca247672badf4684de5d11b0282fbf129ab5338b17bed0c2f18fc69004de0ece",
"license": "Apache-2.0",
"training_data": "synthetic only: an edge-tts voice grid, further edge-tts voices, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.809,
"false_activations_per_hour": 0.39,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 382,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.652,
"false_activations_per_hour": 0.02,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 382,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_hey_chatterbox",
"file": "models/wakehubert_hey_chatterbox.onnx",
"word": "hey chatterbox",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.13,
"sha256": "aa3ba38acd01c2c6c001e67f2517bfcefb014842ab4ea84ac71c8b466418794f",
"license": "Apache-2.0",
"training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.828,
"false_activations_per_hour": 0.09,
"recall_test_set": "OVOS community recordings of real speakers",
"recall_test_clips": 116,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.612,
"false_activations_per_hour": 0.0,
"recall_test_set": "OVOS community recordings of real speakers",
"recall_test_clips": 116,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_hey_floyd",
"file": "models/wakehubert_hey_floyd.onnx",
"word": "hey floyd",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.43,
"sha256": "e670d5543446c263e488301ffe337d58954edcc1fe57142c08bc4c783736722b",
"license": "Apache-2.0",
"training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.917,
"false_activations_per_hour": 0.47,
"recall_test_set": "OVOS community recordings of real speakers",
"recall_test_clips": 96,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.823,
"false_activations_per_hour": 0.02,
"recall_test_set": "OVOS community recordings of real speakers",
"recall_test_clips": 96,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_hey_rhasspy",
"file": "models/wakehubert_hey_rhasspy.onnx",
"word": "hey rhasspy",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.4,
"sha256": "6daeeb497f5aac45f769a0c67e816e6d816399df0b704d949a76f752a7d54c89",
"license": "Apache-2.0",
"training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 1.0,
"false_activations_per_hour": 0.6,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 374,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.976,
"false_activations_per_hour": 0.11,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 374,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_hey_robin",
"file": "models/wakehubert_hey_robin.onnx",
"word": "hey robin",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.05,
"sha256": "3e8a24885153ad920f688679c0c5ac292951af1f258cf3971fa6f527cd61dce3",
"license": "Apache-2.0",
"training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.995,
"false_activations_per_hour": 0.39,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 380,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.979,
"false_activations_per_hour": 0.06,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 380,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_marvin",
"file": "models/wakehubert_marvin.onnx",
"word": "marvin",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.06,
"sha256": "e1cafc139880f074a8c2545518eb3531ca520df3648b873adf31c0a2a8dd5fa5",
"license": "Apache-2.0",
"training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.733,
"false_activations_per_hour": 0.77,
"recall_test_set": "Speech Commands test split, real speakers",
"recall_test_clips": 195,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.718,
"false_activations_per_hour": 0.67,
"recall_test_set": "Speech Commands test split, real speakers",
"recall_test_clips": 195,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_sheila",
"file": "models/wakehubert_sheila.onnx",
"word": "sheila",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.47,
"sha256": "bdcfc2519fdc4e87dd13d3793b347d425c6ce8aa1f04ee5cb29d4e0d00ede234",
"license": "Apache-2.0",
"training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.887,
"false_activations_per_hour": 2.08,
"recall_test_set": "Speech Commands test split, real speakers",
"recall_test_clips": 212,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.844,
"false_activations_per_hour": 0.54,
"recall_test_set": "Speech Commands test split, real speakers",
"recall_test_clips": 212,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_stop",
"file": "models/wakehubert_stop.onnx",
"word": "stop",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.14,
"sha256": "c071186bc0c2bdbb51dfaff10a1a99de5d2cb90c2d5f7976bb2ae445f089b4c2",
"license": "Apache-2.0",
"training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.866,
"false_activations_per_hour": 2.06,
"recall_test_set": "Speech Commands test split, real speakers",
"recall_test_clips": 411,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.769,
"false_activations_per_hour": 0.45,
"recall_test_set": "Speech Commands test split, real speakers",
"recall_test_clips": 411,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_android",
"file": "models/wakehubert_android.onnx",
"word": "android",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.42,
"sha256": "753ad4800946a552c134e8c330d9eab1a042d4b6fb0b0500d8e9395d3761a2cf",
"license": "Apache-2.0",
"training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.987,
"false_activations_per_hour": 0.95,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 390,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.964,
"false_activations_per_hour": 0.11,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 390,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_hey_computer",
"file": "models/wakehubert_hey_computer.onnx",
"word": "hey computer",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.31,
"sha256": "6ea782477908e8512a59a7c23541d42913ca39c95295a20b788693cc4a1b07f3",
"license": "Apache-2.0",
"training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.964,
"false_activations_per_hour": 0.47,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 390,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.933,
"false_activations_per_hour": 0.04,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 390,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_hey_k9",
"file": "models/wakehubert_hey_k9.onnx",
"word": "hey k9",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.06,
"sha256": "3705ef79d9c7cce3c7e89669b4e1a55f0495ea2719287c9225fed5b6a09d63bf",
"license": "Apache-2.0",
"training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.994,
"false_activations_per_hour": 0.19,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 338,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.935,
"false_activations_per_hour": 0.04,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 338,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_hey_scout",
"file": "models/wakehubert_hey_scout.onnx",
"word": "hey scout",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.36,
"sha256": "c72fba4f93cf69b4bd9510682a3a80db67f663a0686f4d331a34d44ff031fdf3",
"license": "Apache-2.0",
"training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.959,
"false_activations_per_hour": 0.09,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 390,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.938,
"false_activations_per_hour": 0.04,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 390,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_wake_up",
"file": "models/wakehubert_wake_up.onnx",
"word": "wake up",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.36,
"sha256": "524e54bc24cf790339928bf8656d0c23d533108328e7b316afbd65b8cdfe28aa",
"license": "Apache-2.0",
"training_data": "synthetic only: an edge-tts voice grid, OmniVoice clips, with six held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.981,
"false_activations_per_hour": 1.1,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 378,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.968,
"false_activations_per_hour": 0.19,
"recall_test_set": "held-out synthetic voices",
"recall_test_clips": 378,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_computer",
"file": "models/wakehubert_computer.onnx",
"word": "computer",
"language": "en",
"featurizer": "wakehubert",
"calibrated": false,
"default_threshold": 0.99,
"sha256": "4ecda088bcce9951730de9fc8b1c4f801ea208add0cbf8a1b9397ed7d8329496",
"license": "Apache-2.0",
"training_data": "TigreGotico/synthetic-wakeword-computer (CC BY 4.0); negatives: TigreGotico/not-wake-words-speech-en (CC BY 4.0) and AudioSet-derived clips",
"measured": null
},
{
"name": "wakehubert_hey_mycroft",
"file": "models/wakehubert_hey_mycroft.onnx",
"word": "hey mycroft",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.18,
"sha256": "5eb388364e489785fd34edf668dd5f780b0481d0c1de3d286720ed3cf545a193",
"license": "Apache-2.0",
"training_data": "synthetic only: text-to-speech clips from TigreGotico/synthetic-wakeword-hey_mycroft (edge-tts, Piper with LibriTTS-R speakers, Amazon Polly, OVOS TTS plugins, OmniVoice, and Chatterbox and OpenVoice conversions onto Common Voice speakers, CC BY 4.0) and further edge-tts and Google Translate TTS clips, with eight held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.974,
"false_activations_per_hour": 1.57,
"recall_test_set": "held-out synthetic voices: held-out edge-tts voices (285 clips) and Kokoro voices converted to unseen speakers (379 clips)",
"recall_test_clips": 664,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.848,
"false_activations_per_hour": 0.24,
"recall_test_set": "held-out synthetic voices: held-out edge-tts voices (285 clips) and Kokoro voices converted to unseen speakers (379 clips)",
"recall_test_clips": 664,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_hey_ziggy",
"file": "models/wakehubert_hey_ziggy.onnx",
"word": "hey ziggy",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.27,
"sha256": "55e849cb7a20ce037cbd9c6d6d4508f182731e35b1a3be159673c571d0890b02",
"license": "Apache-2.0",
"training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, each checked against a speech recogniser transcript, with eight held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.936,
"false_activations_per_hour": 0.64,
"recall_test_set": "held-out synthetic voices: OmniVoice clips and held-out edge-tts voices",
"recall_test_clips": 374,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.893,
"false_activations_per_hour": 0.24,
"recall_test_set": "held-out synthetic voices: OmniVoice clips and held-out edge-tts voices",
"recall_test_clips": 374,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_hey_potato",
"file": "models/wakehubert_hey_potato.onnx",
"word": "hey potato",
"language": "en",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.15,
"sha256": "3951a1f258bbe8d2685bb8c4c3921de0b8b26c89783af37d9ce999fb8219170e",
"license": "Apache-2.0",
"training_data": "synthetic only: an edge-tts voice grid, voice-converted copies of edge-tts clips, OmniVoice clips, each checked against a speech recogniser transcript, with eight held-out edge-tts test voices excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.889,
"false_activations_per_hour": 0.34,
"recall_test_set": "held-out synthetic voices: 360 clips in held-out edge-tts voices only, without the OmniVoice test clips used for the other words",
"recall_test_clips": 360,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.806,
"false_activations_per_hour": 0.02,
"recall_test_set": "held-out synthetic voices: 360 clips in held-out edge-tts voices only, without the OmniVoice test clips used for the other words",
"recall_test_clips": 360,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_hey_stemcom",
"file": "models/wakehubert_hey_stemcom.onnx",
"word": "hey stemcom",
"language": "nl",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.28,
"sha256": "0610626100881dc8da8b416a0bad7bd03f4ea52537359ac9ebbe18ff49731b7a",
"license": "Apache-2.0",
"training_data": "synthetic only: a Dutch edge-tts voice grid (nl-NL and nl-BE), voice-converted copies of edge-tts clips, OmniVoice clips, each checked against a speech recogniser transcript, with one held-out edge-tts test voice excluded; LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv and the even half of an AudioSet noise sample",
"measured": {
"default": {
"recall": 0.89,
"false_activations_per_hour": 0.21,
"recall_test_set": "held-out synthetic voices: OmniVoice clips and held-out edge-tts voices",
"recall_test_clips": 337,
"negative_hours": 46.5
},
"0.8": {
"recall": 0.831,
"false_activations_per_hour": 0.06,
"recall_test_set": "held-out synthetic voices: OmniVoice clips and held-out edge-tts voices",
"recall_test_clips": 337,
"negative_hours": 46.5
}
}
},
{
"name": "wakehubert_despierta",
"file": "models/wakehubert_despierta.onnx",
"word": "despierta",
"language": "es",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.51,
"sha256": "8e993142a4ee30572f1338ce205a6466e61645b3f47465658d0bf7161c4e3ba2",
"license": "Apache-2.0",
"training_data": "synthetic only: OmniVoice clips with no reference speaker, one seed per clip, each checked against a speech recogniser transcript, and the training clips of TigreGotico/synthetic-wakeword-despierta (CC BY 4.0); LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv, the even half of an AudioSet noise sample, and Spanish speech from Common Voice 17 train (CC0) and FLEURS train (CC BY 4.0) cut into 1.5 s windows",
"measured": {
"default": {
"recall": 0.984,
"false_activations_per_hour": 0.58,
"recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses",
"recall_test_clips": 370,
"negative_hours": 31.1,
"false_activations_per_hour_in_language": 0.6,
"in_language_negative_hours": 3.34
},
"0.8": {
"recall": 0.941,
"false_activations_per_hour": 0.03,
"recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses",
"recall_test_clips": 370,
"negative_hours": 31.1,
"false_activations_per_hour_in_language": 0.3,
"in_language_negative_hours": 3.34
}
},
"notes": "In-language false activations: 0.60 per hour over 3.3 h of Spanish Common Voice 17 test and FLEURS dev speech at the default threshold. At the default threshold it fires on 19 of 20 edge-tts near-word probe clips (\"despiertas\", \"despierto\", \"depierta\", \"desperta\" and \"despertar\")."
},
{
"name": "wakehubert_aufwachen",
"file": "models/wakehubert_aufwachen.onnx",
"word": "aufwachen",
"language": "de",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.57,
"sha256": "4df0063c8087a39677ee708da05570447081571401e3142a8515abbd95c7fafb",
"license": "Apache-2.0",
"training_data": "synthetic only: OmniVoice clips with no reference speaker, one seed per clip, each checked against a speech recogniser transcript, and the training clips of TigreGotico/synthetic-wakeword-aufwachen (CC BY 4.0); LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv, the even half of an AudioSet noise sample, and German speech from Common Voice 17 train (CC0) and FLEURS train (CC BY 4.0) cut into 1.5 s windows",
"measured": {
"default": {
"recall": 0.995,
"false_activations_per_hour": 0.22,
"recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses",
"recall_test_clips": 390,
"negative_hours": 31.1,
"false_activations_per_hour_in_language": 1.85,
"in_language_negative_hours": 3.25
},
"0.8": {
"recall": 0.972,
"false_activations_per_hour": 0.0,
"recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses",
"recall_test_clips": 390,
"negative_hours": 31.1,
"false_activations_per_hour_in_language": 0.31,
"in_language_negative_hours": 3.25
}
},
"notes": "In-language false activations: 1.85 per hour over 3.2 h of German Common Voice 17 test and FLEURS dev speech at the default threshold. At the default threshold it fires on 8 of 8 edge-tts near-word probe clips (\"aufmachen\" and \"aufwachten\")."
},
{
"name": "wakehubert_wakker_worden",
"file": "models/wakehubert_wakker_worden.onnx",
"word": "wakker worden",
"language": "nl",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.56,
"sha256": "5c77d9de56c923ce0aa61506d6faaf6378949fd6df1a764110eae3323427d4ad",
"license": "Apache-2.0",
"training_data": "synthetic only: OmniVoice clips with no reference speaker, one seed per clip, each checked against a speech recogniser transcript, and the training clips of TigreGotico/synthetic-wakeword-wakker_worden (CC BY 4.0); LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv, the even half of an AudioSet noise sample, and Dutch speech from Common Voice 17 train (CC0) and FLEURS train (CC BY 4.0) cut into 1.5 s windows",
"measured": {
"default": {
"recall": 0.997,
"false_activations_per_hour": 0.29,
"recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses",
"recall_test_clips": 390,
"negative_hours": 31.1,
"false_activations_per_hour_in_language": 0.0,
"in_language_negative_hours": 2.43
},
"0.8": {
"recall": 0.985,
"false_activations_per_hour": 0.03,
"recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses",
"recall_test_clips": 390,
"negative_hours": 31.1,
"false_activations_per_hour_in_language": 0.0,
"in_language_negative_hours": 2.43
}
},
"notes": "In-language false activations: 0.00 per hour over 2.4 h of Dutch Common Voice 17 test and FLEURS dev speech at the default threshold. At the default threshold it fires on none of 8 edge-tts near-word probe clips (\"wakker\", \"worden\")."
},
{
"name": "wakehubert_sveglia",
"file": "models/wakehubert_sveglia.onnx",
"word": "sveglia",
"language": "it",
"featurizer": "wakehubert-int8",
"calibrated": true,
"default_threshold": 0.47,
"sha256": "523544cc3629f4bd995b9acd4ec98ef217c1240b1c93ddd2f594f71399fdc473",
"license": "Apache-2.0",
"training_data": "synthetic only: OmniVoice clips with no reference speaker, one seed per clip, each checked against a speech recogniser transcript, and the training clips of TigreGotico/synthetic-wakeword-sveglia (CC BY 4.0); LibriSpeech train-clean-100 (CC BY 4.0) mixed in as background babble; negatives from wakeforge datasets/train.csv, the even half of an AudioSet noise sample, and Italian speech from Common Voice 17 train (CC0) and FLEURS train (CC BY 4.0) cut into 1.5 s windows",
"measured": {
"default": {
"recall": 0.995,
"false_activations_per_hour": 1.16,
"recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses",
"recall_test_clips": 222,
"negative_hours": 31.1,
"false_activations_per_hour_in_language": 1.7,
"in_language_negative_hours": 3.53
},
"0.8": {
"recall": 0.977,
"false_activations_per_hour": 0.1,
"recall_test_set": "held-out synthetic voices: OmniVoice clips whose seeds no training clip uses",
"recall_test_clips": 222,
"negative_hours": 31.1,
"false_activations_per_hour_in_language": 0.28,
"in_language_negative_hours": 3.53
}
},
"notes": "In-language false activations: 1.70 per hour over 3.5 h of Italian Common Voice 17 test and FLEURS dev speech at the default threshold. At the default threshold it fires on 18 of 24 edge-tts near-word probe clips (\"sveglio\", \"sveglie\", \"svegliati\", and the rhymes \"meraviglia\", \"bottiglia\" and \"voglia\")."
}
]
}