laughter-detection / export.json
alexcannan's picture
v0.1
849623f
Raw History Blame Contribute Delete
2.66 kB
{
"variants": {
"model.onnx": {
"dtype": "float32",
"opset": 17,
"exporter": "torchscript",
"clip_prob_max_delta": 6.556510925292969e-07,
"frame_prob_max_delta": 2.9578804969787598e-06,
"min_onnxruntime": "1.14",
"intended_for": "CPU or any GPU"
},
"model_fp16.onnx": {
"dtype": "float16",
"opset": 17,
"exporter": "torchscript",
"clip_prob_max_delta": 0.00027447938919067383,
"frame_prob_max_delta": 0.0024967193603515625,
"parity_reference": "model.onnx (fp32 graph, itself verified against the torch package)",
"parity_provider": "CUDAExecutionProvider",
"parity_battery": "32 real 4 s windows",
"min_onnxruntime": "1.14",
"intended_for": "GPU (CPU runtimes have few fp16 kernels)"
},
"model_fp16_opset23.onnx": {
"dtype": "float16",
"opset": 23,
"exporter": "dynamo",
"clip_prob_max_delta": 0.00022864341735839844,
"frame_prob_max_delta": 0.0015059113502502441,
"parity_reference": "model.onnx (fp32 graph, itself verified against the torch package)",
"parity_provider": "CUDAExecutionProvider",
"parity_battery": "32 real 4 s windows",
"min_onnxruntime": "1.22",
"intended_for": "GPU (CPU runtimes have few fp16 kernels)"
}
},
"default": "model.onnx",
"selection": "CPU: model.onnx. CUDA: model_fp16_opset23.onnx if the runtime loads it (>= 1.22), else model_fp16.onnx. events_onnx.py does this.",
"inputs": {
"waveform": "(batch, 64000) float32 PCM, 16 kHz mono, one 4 s window per row, zero-padded",
"n_valid": "(batch,) int64 real samples per row (64000 for a full window)"
},
"outputs": {
"clip_prob": "(batch,) float32 laughter probability of the window",
"frame_probs": "(batch, 24) float32 laughter probability per 0.16 s frame",
"frame_mask": "(batch, 24) bool, False on padding"
},
"front_end": "kaldi log-mel fbank inside the graph, float32 in every variant: 25 ms Hann, 10 ms shift, pre-emphasis 0.97, DC removal, 128 mel bins (20 Hz-8 kHz), log, normalised (x - -4.268) / (2 * 4.569). Resampling to 16 kHz is the caller's job.",
"windowing": {
"window_s": 4.0,
"hop_s": 2.0,
"stitch": "max over overlapping windows",
"decode": "see config.json operating_points"
},
"parity_battery": {
"fp32_graph": {
"windows": 32,
"fbank_max_delta": 3.9696693420410156e-05,
"against": "the torch package (modeling_laughter.py, fp32) on real 4 s windows incl. zero-padded short ones"
},
"fp16_graphs": "32 real 4 s windows from a 16 kHz conversation, compared against the fp32 graph on CUDAExecutionProvider (onnxruntime-gpu 1.29, cuDNN 9, NVIDIA GB10); tolerance 5e-3"
},
"dependencies": [
"onnxruntime",
"numpy"
]
}