{ "variants": { "model.onnx": { "dtype": "float32", "opset": 17, "exporter": "torchscript", "clip_prob_max_delta": 6.556510925292969e-07, "frame_prob_max_delta": 2.9578804969787598e-06, "min_onnxruntime": "1.14", "intended_for": "CPU or any GPU" }, "model_fp16.onnx": { "dtype": "float16", "opset": 17, "exporter": "torchscript", "clip_prob_max_delta": 0.00027447938919067383, "frame_prob_max_delta": 0.0024967193603515625, "parity_reference": "model.onnx (fp32 graph, itself verified against the torch package)", "parity_provider": "CUDAExecutionProvider", "parity_battery": "32 real 4 s windows", "min_onnxruntime": "1.14", "intended_for": "GPU (CPU runtimes have few fp16 kernels)" }, "model_fp16_opset23.onnx": { "dtype": "float16", "opset": 23, "exporter": "dynamo", "clip_prob_max_delta": 0.00022864341735839844, "frame_prob_max_delta": 0.0015059113502502441, "parity_reference": "model.onnx (fp32 graph, itself verified against the torch package)", "parity_provider": "CUDAExecutionProvider", "parity_battery": "32 real 4 s windows", "min_onnxruntime": "1.22", "intended_for": "GPU (CPU runtimes have few fp16 kernels)" } }, "default": "model.onnx", "selection": "CPU: model.onnx. CUDA: model_fp16_opset23.onnx if the runtime loads it (>= 1.22), else model_fp16.onnx. events_onnx.py does this.", "inputs": { "waveform": "(batch, 64000) float32 PCM, 16 kHz mono, one 4 s window per row, zero-padded", "n_valid": "(batch,) int64 real samples per row (64000 for a full window)" }, "outputs": { "clip_prob": "(batch,) float32 laughter probability of the window", "frame_probs": "(batch, 24) float32 laughter probability per 0.16 s frame", "frame_mask": "(batch, 24) bool, False on padding" }, "front_end": "kaldi log-mel fbank inside the graph, float32 in every variant: 25 ms Hann, 10 ms shift, pre-emphasis 0.97, DC removal, 128 mel bins (20 Hz-8 kHz), log, normalised (x - -4.268) / (2 * 4.569). Resampling to 16 kHz is the caller's job.", "windowing": { "window_s": 4.0, "hop_s": 2.0, "stitch": "max over overlapping windows", "decode": "see config.json operating_points" }, "parity_battery": { "fp32_graph": { "windows": 32, "fbank_max_delta": 3.9696693420410156e-05, "against": "the torch package (modeling_laughter.py, fp32) on real 4 s windows incl. zero-padded short ones" }, "fp16_graphs": "32 real 4 s windows from a 16 kHz conversation, compared against the fp32 graph on CUDAExecutionProvider (onnxruntime-gpu 1.29, cuDNN 9, NVIDIA GB10); tolerance 5e-3" }, "dependencies": [ "onnxruntime", "numpy" ] }