File size: 4,300 Bytes
22eb159
e3a3f05
17b1467
b28b067
 
 
e3a3f05
817c485
 
e3a3f05
 
 
817c485
e3a3f05
 
817c485
e3a3f05
 
 
 
 
817c485
e3a3f05
 
 
 
 
43e9d5b
817c485
e3a3f05
 
 
817c485
e3a3f05
 
 
 
 
43e9d5b
c9c1031
b28b067
 
 
c9c1031
b28b067
 
 
 
 
 
 
c9c1031
b28b067
 
 
 
 
 
 
 
 
c9c1031
 
 
 
 
b28b067
c9c1031
 
43e9d5b
817c485
e3a3f05
 
43e9d5b
2910796
 
 
 
 
 
 
 
 
 
 
 
43e9d5b
e3a3f05
 
 
 
43e9d5b
c9c1031
 
 
 
 
 
b28b067
c9c1031
e3a3f05
43e9d5b
 
817c485
 
 
 
 
 
 
 
e3a3f05
817c485
e3a3f05
 
 
b28b067
 
e3a3f05
817c485
43e9d5b
ece9a8f
2910796
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
import os
import glob
import gradio as gr
import numpy as np
import soundfile as sf
import noisereduce as nr
from audio_separator.separator import Separator

# -----------------------------
# Locate the local ONNX model already committed to this repo
# (avoids any network download and avoids hardcoding a filename
# that may have changed in past rename commits)
# -----------------------------
MODELS_DIR = "models"
onnx_candidates = sorted(glob.glob(os.path.join(MODELS_DIR, "*.onnx")))

if not onnx_candidates:
    raise FileNotFoundError(
        f"No .onnx model found in '{MODELS_DIR}/'. "
        f"Make sure your UVR-MDX-Net model file is committed to that folder."
    )

MODEL_PATH = onnx_candidates[0]
MODEL_DIR = os.path.dirname(MODEL_PATH) or "."
MODEL_FILENAME = os.path.basename(MODEL_PATH)

print(f"Using local model: {MODEL_PATH}")

# -----------------------------
# Load the separator once at startup
# (model_file_dir points at the folder that ALREADY has the file,
# so audio-separator uses it directly instead of trying to download it)
# -----------------------------
separator = Separator(
    output_dir="/tmp/audio_separator_outputs",
    model_file_dir=MODEL_DIR,
)
separator.load_model(model_filename=MODEL_FILENAME)


def denoise_vocals(vocals_path: str) -> str:
    """
    Run spectral-gating noise reduction on the isolated vocals track.

    Uses `noisereduce` instead of a neural denoiser (e.g. DeepFilterNet2)
    because that pulled in a conflicting, unmaintained torch/torchaudio/onnx
    dependency chain that could no longer be resolved alongside
    audio-separator. noisereduce has no heavy ML dependencies, so it can't
    collide with audio-separator's stack the same way.
    """
    audio, sr = sf.read(vocals_path)

    # soundfile returns shape (frames, channels) for stereo; noisereduce
    # expects channels-first for multi-channel input.
    is_stereo = audio.ndim == 2
    y = audio.T if is_stereo else audio

    reduced = nr.reduce_noise(y=y, sr=sr)

    if is_stereo:
        reduced = reduced.T

    denoised_path = os.path.join(
        os.path.dirname(vocals_path),
        f"denoised_{os.path.basename(vocals_path)}",
    )
    sf.write(denoised_path, reduced, sr)
    return denoised_path


def separate_audio(audio_file):
    if audio_file is None:
        return None, None

    raw_outputs = separator.separate(audio_file)

    # separator.separate() returns bare filenames, not full paths — they
    # actually get written under separator.output_dir. Without joining them
    # back to that directory, Gradio resolves the relative filename against
    # the app's cwd instead of where the files really are, which is what
    # caused the empty players / 403 on download.
    output_paths = [
        p if os.path.isabs(p) else os.path.join(separator.output_dir, p)
        for p in raw_outputs
    ]
    output_paths = [os.path.abspath(p) for p in output_paths]

    vocals_path = next((p for p in output_paths if "vocal" in p.lower()), None)
    instrumental_path = next(
        (p for p in output_paths if p != vocals_path), None
    )

    if vocals_path is not None:
        try:
            vocals_path = denoise_vocals(vocals_path)
        except Exception as e:
            # If denoising fails for any reason, fall back to the raw
            # separated vocals rather than losing the whole result.
            print(f"Denoising failed, returning raw vocals: {e}")

    return vocals_path, instrumental_path


# -----------------------------
# Gradio Interface
# -----------------------------
ui = gr.Interface(
    fn=separate_audio,
    inputs=gr.Audio(type="filepath", label="Upload Audio"),
    outputs=[
        gr.Audio(type="filepath", label="Vocals"),
        gr.Audio(type="filepath", label="Instrumental"),
    ],
    title="Vocal / Instrumental Separator",
    description=(
        f"CPU-friendly vocal & instrumental separation using the local "
        f"UVR-MDX-Net ONNX model ({MODEL_FILENAME}), with vocals cleaned "
        f"up via spectral-gating noise reduction."
    ),
)

if __name__ == "__main__":
    # Explicitly whitelist the output directory for Gradio's file server —
    # avoids 403s when serving files from a directory outside the app's cwd.
    ui.launch(allowed_paths=[separator.output_dir])