File size: 6,149 Bytes
81882f8
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
import gradio as gr
import numpy as np
import scipy.signal as signal
import librosa
import plotly.graph_objects as go
from plotly.subplots import make_subplots

def run_forensic_analysis(vocal_file, master_file):
    if not vocal_file or not master_file:
        return None, None, None, "โŒ Error: Please upload both assets to run diagnostics."
    
    # Load audio clips (Optimized downsampling & 60-second execution cap)
    y_v, sr_v = librosa.load(vocal_file, sr=22050, duration=60.0)
    y_m, sr_m = librosa.load(master_file, sr=22050, duration=60.0)
    
    # =================================================================
    # MODULE 1: STRUCTURAL RECURRENCE MAPPER (SSM)
    # =================================================================
    # Extract chroma features to map harmonic patterns
    chroma = librosa.feature.chroma_stft(y=y_m, sr=sr_m, hop_length=1024)
    chroma = librosa.util.normalize(chroma, axis=0)
    ssm = np.dot(chroma.T, chroma)
    
    fig_ssm = go.Figure(data=gr.Plot(go.Heatmap(
        z=ssm, colorscale='Viridis', showscale=False,
        x0=0, dx=1024/sr_m, y0=0, dy=1024/sr_m
    )))
    fig_ssm.update_layout(
        title="Structural Recurrence Matrix (Pattern Matches Over Time)",
        xaxis_title="Timeline (Seconds)",
        yaxis_title="Timeline (Seconds)",
        height=400, margin=dict(l=10, r=10, t=40, b=10)
    )
    
    # =================================================================
    # MODULE 2: MIX FORENSICS & FREQUENCY HEALTH
    # =================================================================
    centroid = librosa.feature.spectral_centroid(y=y_m, sr=sr_m, hop_length=512)[0]
    flatness = librosa.feature.spectral_flatness(y=y_m, hop_length=512)[0]
    t_m = librosa.times_like(centroid, sr=sr_m, hop_length=512)
    
    rms_m = librosa.feature.rms(y=y_m)[0]
    crest_factor = np.max(np.abs(y_m)) / (np.mean(rms_m) + 1e-6)
    
    fig_mix = make_subplots(rows=2, cols=1, shared_xaxes=True, vertical_spacing=0.1)
    fig_mix.add_trace(go.Scatter(x=t_m, y=centroid, mode='lines', name='Centroid (Brightness)', line=dict(color='#FF4B4B')), row=1, col=1)
    fig_mix.add_trace(go.Scatter(x=t_m, y=flatness, mode='lines', name='Flatness (Noise/Harsh)', line=dict(color='#00E5FF')), row=2, col=1)
    
    fig_mix.update_layout(
        title="Spectral Diagnostic Vectors",
        xaxis2_title="Timeline (Seconds)",
        height=400, showlegend=False, margin=dict(l=10, r=10, t=40, b=10)
    )
    
    # =================================================================
    # MODULE 3: SYLLABLE VELOCITY & PHRASING DENSITY
    # =================================================================
    onset_v = librosa.onset.onset_strength(y=y_v, sr=sr_v, hop_length=512)
    t_v = librosa.times_like(onset_v, sr=sr_v, hop_length=512)
    
    peaks, _ = signal.find_peaks(onset_v, height=0.2, distance=10)
    peak_times = t_v[peaks]
    
    # Calculate local density (syllables per second window)
    window_sz = 2.0
    density_y = []
    for t in t_v:
        count = np.sum((peak_times >= t - window_sz/2) & (peak_times <= t + window_sz/2))
        density_y.append(count / window_sz)
        
    fig_vocal = go.Figure()
    fig_vocal.add_trace(go.Scatter(x=t_v, y=density_y, mode='lines', fill='tozeroy', line=dict(color='#00FF66', width=2)))
    fig_vocal.update_layout(
        title="Lyrical Delivery Velocity (Syllables Per Second)",
        xaxis_title="Timeline (Seconds)",
        yaxis_title="Syllables/Sec",
        height=350, margin=dict(l=10, r=10, t=40, b=10)
    )
    
    # =================================================================
    # DIAGNOSTIC SUMMARY GENERATOR
    # =================================================================
    avg_flatness = np.mean(flatness)
    mix_critique = "Balanced clean spectrum." if avg_flatness < 0.01 else "High harshness/noise ratios detected."
    if crest_factor > 5.0:
        dynamics_status = "Dynamic, punchy transient preservation."
    else:
        dynamics_status = "Highly compressed / squashed brickwall mix."
        
    report = f"""### ๐Ÿ“Š Forensic Lab Insights (First 60s)
* **Dynamic Crest Factor:** {crest_factor:.2f} $\\rightarrow$ *{dynamics_status}*
* **Total Vocal Artifact Onsets:** {len(peaks)} individual syllables mapped.
* **Peak Delivery Speed:** {max(density_y):.1f} syllables per second.
* **Acoustic Blueprint Profile:** {mix_critique}

> **How to interpret charts:** > * **Recurrence Map:** Look for solid diagonal block shapes. These explicitly flag identical choruses or looping sample patterns.
> * **Spectral Vectors:** Sharp dips or valleys reveal filtering tricks, transitions, or frequency drops.
> * **Delivery Velocity:** Waves illustrate the emotional phrasing of a performance, showing where the artist rushes or drags words."""
    
    return fig_ssm, fig_mix, fig_vocal, report

# =====================================================================
# INTERFACE LAYOUT
# =====================================================================
with gr.Blocks(title="Audio Forensic Analytics Lab") as demo:
    gr.Markdown("# ๐Ÿ”ฌ Audio Forensic & Structural Intelligence Lab")
    gr.Markdown("Drop clean stems extracted from Space 1 to extract advanced frequency blueprints, structural maps, and vocal data matrices.")
    
    with gr.Row():
        vocal_input = gr.Audio(type="filepath", label="Upload Isolated Vocal Stem (vocals.wav)")
        master_input = gr.Audio(type="filepath", label="Upload Instrumental or Full Mix Master")
        
    analyze_btn = gr.Button("Deploy Forensic Diagnosis Engines", variant="primary")
    
    with gr.Tab("1. Structural Recurrence Map"):
        plot_ssm = gr.Plot()
    with gr.Tab("2. Mix Health & Spectral Vectors"):
        plot_mix = gr.Plot()
    with gr.Tab("3. Lyrical Velocity Matrix"):
        plot_vocal = gr.Plot()
        
    text_report = gr.Markdown()
    
    analyze_btn.click(
        fn=run_forensic_analysis,
        inputs=[vocal_input, master_input],
        outputs=[plot_ssm, plot_mix, plot_vocal, text_report]
    )

if __name__ == "__main__":
    demo.launch()