[ "rms_energy", "rms_std", "spectral_centroid_mean", "spectral_centroid_std", "spectral_flatness_mean", "spectral_flatness_std", "spectral_bandwidth_mean", "spectral_bandwidth_std", "spectral_rolloff_mean", "spectral_rolloff_std", "spectral_contrast_mean", "spectral_contrast_std", "mfcc_variance", "mfcc_delta_var", "mfcc_delta2_var", "mel_flatness", "tempo_bpm", "tempo_stability", "tempo_cv", "zero_crossing_rate", "zero_crossing_std", "onset_strength_mean", "onset_strength_std", "rms_dynamic_range", "beat_count", "chroma_entropy", "chroma_std", "chroma_transition_rate", "harmonic_ratio", "tonnetz_std", "spectral_regularity", "temporal_patterns", "harmonic_structure", "has_vocals", "vocal_confidence", "vocal_ai_score", "pitch_stability_score", "vibrato_regularity_score", "formant_consistency_score", "breath_pattern_score", "vocal_texture_score", "pitch_mean_hz", "pitch_std_cents", "vibrato_rate_hz", "vibrato_extent_cents", "vocal_harmonic_ratio", "vocal_energy_ratio" ]