{ "architecture": "whisper encoder + attention pooling + MLP head (Pipecat Smart Turn v3 layout)", "input": {"name": "input_features", "shape": [-1, 80, 800], "description": "log-mel of the last 8 s of 16 kHz audio, WhisperFeatureExtractor(chunk_length=8), do_normalize=True"}, "output": {"name": "logits", "shape": [-1, 1], "description": "probability that the user's turn is complete"}, "threshold": 0.5, "languages": ["eng", "hin", "mar", "ben", "tam", "tel", "kan", "mal", "guj", "pan", "ori", "asm"], "files": { "indic-smart-turn-base-int8.onnx": {"encoder": "whisper-base", "precision": "int8 dynamic (weights)", "size_mb": 24, "recommended": true}, "indic-smart-turn-base-fp32.onnx": {"encoder": "whisper-base", "precision": "fp32", "size_mb": 81}, "indic-smart-turn-tiny-int8.onnx": {"encoder": "whisper-tiny", "precision": "int8 static", "size_mb": 8.8}, "indic-smart-turn-tiny-fp32.onnx": {"encoder": "whisper-tiny", "precision": "fp32", "size_mb": 32} } }