AudioSplitter / app.py
Rezuwan's picture
Update app.py
b28b067 verified
Raw History Blame Contribute Delete
4.3 kB
import os
import glob
import gradio as gr
import numpy as np
import soundfile as sf
import noisereduce as nr
from audio_separator.separator import Separator
# -----------------------------
# Locate the local ONNX model already committed to this repo
# (avoids any network download and avoids hardcoding a filename
# that may have changed in past rename commits)
# -----------------------------
MODELS_DIR = "models"
onnx_candidates = sorted(glob.glob(os.path.join(MODELS_DIR, "*.onnx")))
if not onnx_candidates:
raise FileNotFoundError(
f"No .onnx model found in '{MODELS_DIR}/'. "
f"Make sure your UVR-MDX-Net model file is committed to that folder."
)
MODEL_PATH = onnx_candidates[0]
MODEL_DIR = os.path.dirname(MODEL_PATH) or "."
MODEL_FILENAME = os.path.basename(MODEL_PATH)
print(f"Using local model: {MODEL_PATH}")
# -----------------------------
# Load the separator once at startup
# (model_file_dir points at the folder that ALREADY has the file,
# so audio-separator uses it directly instead of trying to download it)
# -----------------------------
separator = Separator(
output_dir="/tmp/audio_separator_outputs",
model_file_dir=MODEL_DIR,
)
separator.load_model(model_filename=MODEL_FILENAME)
def denoise_vocals(vocals_path: str) -> str:
"""
Run spectral-gating noise reduction on the isolated vocals track.
Uses `noisereduce` instead of a neural denoiser (e.g. DeepFilterNet2)
because that pulled in a conflicting, unmaintained torch/torchaudio/onnx
dependency chain that could no longer be resolved alongside
audio-separator. noisereduce has no heavy ML dependencies, so it can't
collide with audio-separator's stack the same way.
"""
audio, sr = sf.read(vocals_path)
# soundfile returns shape (frames, channels) for stereo; noisereduce
# expects channels-first for multi-channel input.
is_stereo = audio.ndim == 2
y = audio.T if is_stereo else audio
reduced = nr.reduce_noise(y=y, sr=sr)
if is_stereo:
reduced = reduced.T
denoised_path = os.path.join(
os.path.dirname(vocals_path),
f"denoised_{os.path.basename(vocals_path)}",
)
sf.write(denoised_path, reduced, sr)
return denoised_path
def separate_audio(audio_file):
if audio_file is None:
return None, None
raw_outputs = separator.separate(audio_file)
# separator.separate() returns bare filenames, not full paths — they
# actually get written under separator.output_dir. Without joining them
# back to that directory, Gradio resolves the relative filename against
# the app's cwd instead of where the files really are, which is what
# caused the empty players / 403 on download.
output_paths = [
p if os.path.isabs(p) else os.path.join(separator.output_dir, p)
for p in raw_outputs
]
output_paths = [os.path.abspath(p) for p in output_paths]
vocals_path = next((p for p in output_paths if "vocal" in p.lower()), None)
instrumental_path = next(
(p for p in output_paths if p != vocals_path), None
)
if vocals_path is not None:
try:
vocals_path = denoise_vocals(vocals_path)
except Exception as e:
# If denoising fails for any reason, fall back to the raw
# separated vocals rather than losing the whole result.
print(f"Denoising failed, returning raw vocals: {e}")
return vocals_path, instrumental_path
# -----------------------------
# Gradio Interface
# -----------------------------
ui = gr.Interface(
fn=separate_audio,
inputs=gr.Audio(type="filepath", label="Upload Audio"),
outputs=[
gr.Audio(type="filepath", label="Vocals"),
gr.Audio(type="filepath", label="Instrumental"),
],
title="Vocal / Instrumental Separator",
description=(
f"CPU-friendly vocal & instrumental separation using the local "
f"UVR-MDX-Net ONNX model ({MODEL_FILENAME}), with vocals cleaned "
f"up via spectral-gating noise reduction."
),
)
if __name__ == "__main__":
# Explicitly whitelist the output directory for Gradio's file server —
# avoids 403s when serving files from a directory outside the app's cwd.
ui.launch(allowed_paths=[separator.output_dir])