Publish Visar wake-word models: ONNX, TFLite, and fine-tuning state
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- .gitattributes +34 -0
- README.md +55 -0
- schema.fbs +1715 -0
- schema_generated.py +0 -0
- training_config.yaml +32 -0
- training_state/negative_features_test.npy +3 -0
- training_state/negative_features_train.npy +3 -0
- training_state/negative_test/0.wav +0 -0
- training_state/negative_test/1.wav +0 -0
- training_state/negative_test/10.wav +0 -0
- training_state/negative_test/100.wav +0 -0
- training_state/negative_test/101.wav +0 -0
- training_state/negative_test/102.wav +0 -0
- training_state/negative_test/103.wav +0 -0
- training_state/negative_test/104.wav +0 -0
- training_state/negative_test/105.wav +0 -0
- training_state/negative_test/106.wav +0 -0
- training_state/negative_test/107.wav +0 -0
- training_state/negative_test/108.wav +0 -0
- training_state/negative_test/109.wav +0 -0
- training_state/negative_test/11.wav +0 -0
- training_state/negative_test/110.wav +0 -0
- training_state/negative_test/111.wav +0 -0
- training_state/negative_test/112.wav +0 -0
- training_state/negative_test/113.wav +0 -0
- training_state/negative_test/114.wav +0 -0
- training_state/negative_test/115.wav +0 -0
- training_state/negative_test/116.wav +0 -0
- training_state/negative_test/117.wav +0 -0
- training_state/negative_test/118.wav +0 -0
- training_state/negative_test/119.wav +0 -0
- training_state/negative_test/12.wav +0 -0
- training_state/negative_test/120.wav +0 -0
- training_state/negative_test/121.wav +0 -0
- training_state/negative_test/122.wav +0 -0
- training_state/negative_test/123.wav +0 -0
- training_state/negative_test/124.wav +0 -0
- training_state/negative_test/125.wav +0 -0
- training_state/negative_test/126.wav +0 -0
- training_state/negative_test/127.wav +0 -0
- training_state/negative_test/128.wav +0 -0
- training_state/negative_test/129.wav +0 -0
- training_state/negative_test/13.wav +0 -0
- training_state/negative_test/130.wav +0 -0
- training_state/negative_test/131.wav +0 -0
- training_state/negative_test/132.wav +0 -0
- training_state/negative_test/133.wav +0 -0
- training_state/negative_test/134.wav +0 -0
- training_state/negative_test/135.wav +0 -0
- training_state/negative_test/136.wav +0 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,37 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
user_calibration_audio/20260223_165137066.wav filter=lfs diff=lfs merge=lfs -text
|
| 37 |
+
user_calibration_audio/20260223_165151990.wav filter=lfs diff=lfs merge=lfs -text
|
| 38 |
+
user_calibration_audio/20260223_165156182.wav filter=lfs diff=lfs merge=lfs -text
|
| 39 |
+
user_calibration_audio/20260223_165200415.wav filter=lfs diff=lfs merge=lfs -text
|
| 40 |
+
user_calibration_audio/20260223_165205430.wav filter=lfs diff=lfs merge=lfs -text
|
| 41 |
+
user_calibration_audio/20260223_165209065.wav filter=lfs diff=lfs merge=lfs -text
|
| 42 |
+
user_calibration_audio/20260223_165214153.wav filter=lfs diff=lfs merge=lfs -text
|
| 43 |
+
user_calibration_audio/20260223_165220006.wav filter=lfs diff=lfs merge=lfs -text
|
| 44 |
+
user_calibration_audio/20260223_165228805.wav filter=lfs diff=lfs merge=lfs -text
|
| 45 |
+
user_calibration_audio/20260223_165233919.wav filter=lfs diff=lfs merge=lfs -text
|
| 46 |
+
user_calibration_audio/20260223_165240393.wav filter=lfs diff=lfs merge=lfs -text
|
| 47 |
+
user_calibration_audio/20260223_165340217.wav filter=lfs diff=lfs merge=lfs -text
|
| 48 |
+
user_calibration_audio/20260223_165344343.wav filter=lfs diff=lfs merge=lfs -text
|
| 49 |
+
user_calibration_audio/20260223_165351382.wav filter=lfs diff=lfs merge=lfs -text
|
| 50 |
+
user_calibration_audio/20260223_165357948.wav filter=lfs diff=lfs merge=lfs -text
|
| 51 |
+
user_calibration_audio/20260223_165409386.wav filter=lfs diff=lfs merge=lfs -text
|
| 52 |
+
user_calibration_audio/20260223_165419082.wav filter=lfs diff=lfs merge=lfs -text
|
| 53 |
+
user_calibration_audio/20260223_165424062.wav filter=lfs diff=lfs merge=lfs -text
|
| 54 |
+
user_calibration_audio/20260223_165429632.wav filter=lfs diff=lfs merge=lfs -text
|
| 55 |
+
user_calibration_audio/20260223_165434870.wav filter=lfs diff=lfs merge=lfs -text
|
| 56 |
+
user_calibration_audio/20260223_165440008.wav filter=lfs diff=lfs merge=lfs -text
|
| 57 |
+
user_calibration_audio/20260223_165505293.wav filter=lfs diff=lfs merge=lfs -text
|
| 58 |
+
user_calibration_audio/20260223_165523705.wav filter=lfs diff=lfs merge=lfs -text
|
| 59 |
+
user_calibration_audio/20260223_165622234.wav filter=lfs diff=lfs merge=lfs -text
|
| 60 |
+
user_calibration_audio/20260223_165627330.wav filter=lfs diff=lfs merge=lfs -text
|
| 61 |
+
user_calibration_audio/20260223_165633167.wav filter=lfs diff=lfs merge=lfs -text
|
| 62 |
+
user_calibration_audio/20260223_165645302.wav filter=lfs diff=lfs merge=lfs -text
|
| 63 |
+
user_calibration_audio/20260223_165651345.wav filter=lfs diff=lfs merge=lfs -text
|
| 64 |
+
user_calibration_audio/20260223_165655777.wav filter=lfs diff=lfs merge=lfs -text
|
| 65 |
+
user_calibration_audio/20260223_165701273.wav filter=lfs diff=lfs merge=lfs -text
|
| 66 |
+
user_calibration_audio/20260223_165705473.wav filter=lfs diff=lfs merge=lfs -text
|
| 67 |
+
user_calibration_audio/20260223_165717600.wav filter=lfs diff=lfs merge=lfs -text
|
| 68 |
+
user_calibration_audio/normalized_voice_26.wav filter=lfs diff=lfs merge=lfs -text
|
| 69 |
+
visar_edge.onnx.data filter=lfs diff=lfs merge=lfs -text
|
README.md
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
language:
|
| 3 |
+
- hi
|
| 4 |
+
- en
|
| 5 |
+
tags:
|
| 6 |
+
- openwakeword
|
| 7 |
+
- wake-word-detection
|
| 8 |
+
- voice-assistant
|
| 9 |
+
- edge-ai
|
| 10 |
+
- onnx
|
| 11 |
+
- tflite
|
| 12 |
+
license: apache-2.0
|
| 13 |
+
---
|
| 14 |
+
|
| 15 |
+
# Visar (वी-सार) — Custom openWakeWord Edge Model
|
| 16 |
+
|
| 17 |
+
A lightweight, high-accuracy wake-word detection model custom-trained for low-power edge SBCs, PCs, and offline home automation systems.
|
| 18 |
+
|
| 19 |
+
## Model Highlights
|
| 20 |
+
- **Target Phrase:** "Visar" / "वी-सार" / "Vee-saar"
|
| 21 |
+
- **Acoustic Tuning:** Native Hindi cadence and Indian English phonetic variations (`hi-IN-Madhur`, `hi-IN-Swara`, `en-IN-Prabhat`, `en-IN-Neerja`).
|
| 22 |
+
- **Ground Truth:** Real-world microphone samples convolved with MIT environmental impulse responses (RIRs).
|
| 23 |
+
- **Hard Negatives:** Penalized against acoustically close Indian words (*vichar*, *vishal*, *vikas*, *bazaar*).
|
| 24 |
+
- **Target Hardware:** Raspberry Pi, Linux Edge SBCs, Intel x86, Android, ARM64 microcontrollers.
|
| 25 |
+
|
| 26 |
+
## Audio Input Requirements
|
| 27 |
+
- **Sample Rate:** 16,000 Hz
|
| 28 |
+
- **Channels:** 1 (Mono)
|
| 29 |
+
- **Format:** 16-bit Signed Linear PCM
|
| 30 |
+
|
| 31 |
+
## Repository Contents
|
| 32 |
+
| File / Directory | Description |
|
| 33 |
+
| :--- | :--- |
|
| 34 |
+
| `visar_edge.onnx` | Production ONNX inference graph |
|
| 35 |
+
| `visar_edge.onnx.data` | External weight buffer for ONNX runtime |
|
| 36 |
+
| `visar_edge.tflite` | Quantized FlatBuffer model for mobile/embedded devices |
|
| 37 |
+
| `training_config.yaml` | Exact hyperparameter configuration used during training |
|
| 38 |
+
| `training_state/` | Step checkpoints and optimizer states for fine-tuning |
|
| 39 |
+
| `user_calibration_audio/` | Microphone ground-truth calibration recordings |
|
| 40 |
+
|
| 41 |
+
## Quickstart (Python Inference)
|
| 42 |
+
```python
|
| 43 |
+
import numpy as np
|
| 44 |
+
import openwakeword
|
| 45 |
+
from openwakeword.model import Model
|
| 46 |
+
|
| 47 |
+
# Initialize wake word engine with custom Visar model
|
| 48 |
+
oww_model = Model(wakeword_models=["visar_edge.onnx"])
|
| 49 |
+
|
| 50 |
+
# Pass raw 16kHz 16-bit PCM audio frames (1280 samples / 80ms chunk)
|
| 51 |
+
# audio_frame = np.frombuffer(mic_stream.read(1280), dtype=np.int16)
|
| 52 |
+
# prediction = oww_model.predict(audio_frame)
|
| 53 |
+
# if prediction["visar_edge"] > 0.5:
|
| 54 |
+
# print("Wake-word detected: Visar!")
|
| 55 |
+
|
schema.fbs
ADDED
|
|
schema_generated.py
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
training_config.yaml
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
augmentation_batch_size: 16
|
| 2 |
+
augmentation_rounds: 1
|
| 3 |
+
background_paths:
|
| 4 |
+
- ./audioset_16k
|
| 5 |
+
- ./fma
|
| 6 |
+
background_paths_duplication_rate:
|
| 7 |
+
- 1
|
| 8 |
+
batch_n_per_class:
|
| 9 |
+
ACAV100M_sample: 1024
|
| 10 |
+
adversarial_negative: 50
|
| 11 |
+
positive: 50
|
| 12 |
+
custom_negative_phrases: []
|
| 13 |
+
false_positive_validation_data_path: validation_set_features.npy
|
| 14 |
+
feature_data_files:
|
| 15 |
+
ACAV100M_sample: openwakeword_features_ACAV100M_2000_hrs_16bit.npy
|
| 16 |
+
layer_size: 32
|
| 17 |
+
max_negative_weight: 3000
|
| 18 |
+
model_name: visar_edge
|
| 19 |
+
model_type: dnn
|
| 20 |
+
n_samples: 5000
|
| 21 |
+
n_samples_val: 500
|
| 22 |
+
output_dir: ./my_custom_model
|
| 23 |
+
piper_sample_generator_path: ./piper-sample-generator
|
| 24 |
+
rir_paths:
|
| 25 |
+
- ./mit_rirs
|
| 26 |
+
steps: 35000
|
| 27 |
+
target_accuracy: 0.5
|
| 28 |
+
target_false_positives_per_hour: 0.2
|
| 29 |
+
target_phrase:
|
| 30 |
+
- vee_saar
|
| 31 |
+
target_recall: 0.25
|
| 32 |
+
tts_batch_size: 50
|
training_state/negative_features_test.npy
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7af4ff28a9da26c2dd16c67b0050c4ff90814c298654c44013acdce1765a5411
|
| 3 |
+
size 3072128
|
training_state/negative_features_train.npy
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0134f18643998b0e0460605db43f4427ae8cb087b44c97b353ff892974d1c050
|
| 3 |
+
size 30720128
|
training_state/negative_test/0.wav
ADDED
|
Binary file (30.9 kB). View file
|
|
|
training_state/negative_test/1.wav
ADDED
|
Binary file (29 kB). View file
|
|
|
training_state/negative_test/10.wav
ADDED
|
Binary file (24.2 kB). View file
|
|
|
training_state/negative_test/100.wav
ADDED
|
Binary file (28 kB). View file
|
|
|
training_state/negative_test/101.wav
ADDED
|
Binary file (28 kB). View file
|
|
|
training_state/negative_test/102.wav
ADDED
|
Binary file (41.5 kB). View file
|
|
|
training_state/negative_test/103.wav
ADDED
|
Binary file (27.1 kB). View file
|
|
|
training_state/negative_test/104.wav
ADDED
|
Binary file (25.2 kB). View file
|
|
|
training_state/negative_test/105.wav
ADDED
|
Binary file (18.4 kB). View file
|
|
|
training_state/negative_test/106.wav
ADDED
|
Binary file (27.1 kB). View file
|
|
|
training_state/negative_test/107.wav
ADDED
|
Binary file (22.3 kB). View file
|
|
|
training_state/negative_test/108.wav
ADDED
|
Binary file (25.2 kB). View file
|
|
|
training_state/negative_test/109.wav
ADDED
|
Binary file (22.3 kB). View file
|
|
|
training_state/negative_test/11.wav
ADDED
|
Binary file (30.9 kB). View file
|
|
|
training_state/negative_test/110.wav
ADDED
|
Binary file (26.1 kB). View file
|
|
|
training_state/negative_test/111.wav
ADDED
|
Binary file (33.8 kB). View file
|
|
|
training_state/negative_test/112.wav
ADDED
|
Binary file (22.3 kB). View file
|
|
|
training_state/negative_test/113.wav
ADDED
|
Binary file (42.4 kB). View file
|
|
|
training_state/negative_test/114.wav
ADDED
|
Binary file (19.4 kB). View file
|
|
|
training_state/negative_test/115.wav
ADDED
|
Binary file (30.9 kB). View file
|
|
|
training_state/negative_test/116.wav
ADDED
|
Binary file (42.4 kB). View file
|
|
|
training_state/negative_test/117.wav
ADDED
|
Binary file (19.4 kB). View file
|
|
|
training_state/negative_test/118.wav
ADDED
|
Binary file (24.2 kB). View file
|
|
|
training_state/negative_test/119.wav
ADDED
|
Binary file (30 kB). View file
|
|
|
training_state/negative_test/12.wav
ADDED
|
Binary file (22.3 kB). View file
|
|
|
training_state/negative_test/120.wav
ADDED
|
Binary file (29 kB). View file
|
|
|
training_state/negative_test/121.wav
ADDED
|
Binary file (38.6 kB). View file
|
|
|
training_state/negative_test/122.wav
ADDED
|
Binary file (44.4 kB). View file
|
|
|
training_state/negative_test/123.wav
ADDED
|
Binary file (37.6 kB). View file
|
|
|
training_state/negative_test/124.wav
ADDED
|
Binary file (33.8 kB). View file
|
|
|
training_state/negative_test/125.wav
ADDED
|
Binary file (19.4 kB). View file
|
|
|
training_state/negative_test/126.wav
ADDED
|
Binary file (12.7 kB). View file
|
|
|
training_state/negative_test/127.wav
ADDED
|
Binary file (24.2 kB). View file
|
|
|
training_state/negative_test/128.wav
ADDED
|
Binary file (22.3 kB). View file
|
|
|
training_state/negative_test/129.wav
ADDED
|
Binary file (24.2 kB). View file
|
|
|
training_state/negative_test/13.wav
ADDED
|
Binary file (25.2 kB). View file
|
|
|
training_state/negative_test/130.wav
ADDED
|
Binary file (22.3 kB). View file
|
|
|
training_state/negative_test/131.wav
ADDED
|
Binary file (18.4 kB). View file
|
|
|
training_state/negative_test/132.wav
ADDED
|
Binary file (23.2 kB). View file
|
|
|
training_state/negative_test/133.wav
ADDED
|
Binary file (31.9 kB). View file
|
|
|
training_state/negative_test/134.wav
ADDED
|
Binary file (18.4 kB). View file
|
|
|
training_state/negative_test/135.wav
ADDED
|
Binary file (21.3 kB). View file
|
|
|
training_state/negative_test/136.wav
ADDED
|
Binary file (34.8 kB). View file
|
|
|