AudioEditor / app.py
jacky3102's picture
Update app.py
7b1d9bb verified
Raw History Blame Contribute Delete
14.8 kB
"""
Vocal Studio Space
===================
一条龙处理:
1) 人声/伴奏分离 (audio-separator, MDX23C / UVR-MDX-NET)
2) AI 降噪与音质增强 (DeepFilterNet)
3) 室内混响 (pedalboard)
4) 响度归一化 (ffmpeg loudnorm, 支持 auto 模式)
5) 可选:处理后人声与原伴奏混音
同时提供:
- Gradio 网页界面 (手动试听调参)
- REST API POST /api/process (给 OpenClaw 之类的自动化流水线调用)
- GET /health
"""
import json
import os
import subprocess
import tempfile
import uuid
from pathlib import Path
import gradio as gr
import numpy as np
import soundfile as sf
from fastapi import FastAPI, File, Form, UploadFile
from fastapi.responses import FileResponse, JSONResponse
from pydub import AudioSegment
TMP_DIR = Path(os.environ.get("TMP_DIR", "/app/tmp"))
TMP_DIR.mkdir(parents=True, exist_ok=True)
MODEL_DIR = Path(os.environ.get("MODEL_DIR", "/app/models"))
MODEL_DIR.mkdir(parents=True, exist_ok=True)
# 可选的分离模型。UVR-MDX-NET-Inst-HQ_4 速度快、人声干净;
# MDX23C 质量更高但更慢,CPU上建议先用前者测试。
# 注意:文件名必须与 audio-separator 库内部模型清单完全一致
# (已用 sep.list_supported_model_files() 实际核对过)。
SEPARATION_MODELS = {
"UVR-MDX-NET-Inst-HQ_4 (推荐, 速度快)": "UVR-MDX-NET-Inst_HQ_4.onnx",
"MDX23C-InstVoc-HQ (质量更高, 更慢)": "MDX23C-8KFFT-InstVoc_HQ.ckpt",
}
# ---------------------------------------------------------------------------
# 全局懒加载的模型对象,避免每次请求重新加载权重
# ---------------------------------------------------------------------------
_separator_cache = {}
_df_model = None
_df_state = None
def _new_path(suffix=".wav"):
return str(TMP_DIR / f"{uuid.uuid4().hex}{suffix}")
def get_separator(model_filename: str):
"""按模型名缓存 Separator 实例,避免重复加载权重。"""
from audio_separator.separator import Separator
if model_filename not in _separator_cache:
sep = Separator(
output_dir=str(TMP_DIR),
model_file_dir=str(MODEL_DIR),
)
sep.load_model(model_filename=model_filename)
_separator_cache[model_filename] = sep
return _separator_cache[model_filename]
def get_df_model():
global _df_model, _df_state
if _df_model is None:
from df.enhance import init_df
_df_model, _df_state, _ = init_df()
return _df_model, _df_state
# ---------------------------------------------------------------------------
# 步骤 1: 人声 / 伴奏分离
# ---------------------------------------------------------------------------
def separate_vocals(input_path: str, model_label: str):
model_filename = SEPARATION_MODELS[model_label]
sep = get_separator(model_filename)
output_files = sep.separate(input_path)
# audio-separator 会返回若干输出文件路径,按文件名里的 (Vocals)/(Instrumental) 区分
vocals_path, instrumental_path = None, None
for f in output_files:
full = f if os.path.isabs(f) else str(TMP_DIR / f)
lower = full.lower()
if "vocal" in lower and "inst" not in lower:
vocals_path = full
elif "instrument" in lower or "accomp" in lower:
instrumental_path = full
if vocals_path is None and output_files:
vocals_path = output_files[0] if os.path.isabs(output_files[0]) else str(TMP_DIR / output_files[0])
if instrumental_path is None and len(output_files) > 1:
instrumental_path = output_files[1] if os.path.isabs(output_files[1]) else str(TMP_DIR / output_files[1])
return vocals_path, instrumental_path
# ---------------------------------------------------------------------------
# 步骤 2: AI 降噪 / 音质增强 (DeepFilterNet)
# ---------------------------------------------------------------------------
def enhance_audio(input_path: str) -> str:
from df.enhance import enhance, load_audio, save_audio
model, df_state = get_df_model()
audio, _ = load_audio(input_path, sr=df_state.sr())
enhanced = enhance(model, df_state, audio)
out_path = _new_path("_enhanced.wav")
save_audio(out_path, enhanced, df_state.sr())
return out_path
# ---------------------------------------------------------------------------
# 步骤 3: 室内混响 (pedalboard)
# ---------------------------------------------------------------------------
def add_reverb(
input_path: str,
room_size: float = 0.5,
damping: float = 0.5,
wet_level: float = 0.3,
dry_level: float = 0.7,
) -> str:
from pedalboard import Pedalboard, Reverb
from pedalboard.io import AudioFile
out_path = _new_path("_reverb.wav")
board = Pedalboard(
[
Reverb(
room_size=room_size,
damping=damping,
wet_level=wet_level,
dry_level=dry_level,
)
]
)
with AudioFile(input_path) as f:
audio = f.read(f.frames)
sr = f.samplerate
effected = board(audio, sr)
with AudioFile(out_path, "w", sr, effected.shape[0]) as f:
f.write(effected)
return out_path
# ---------------------------------------------------------------------------
# 步骤 4: 响度归一化 (ffmpeg loudnorm 两遍分析, 支持 auto 模式)
# ---------------------------------------------------------------------------
def _measure_loudness(path: str) -> dict:
cmd = [
"ffmpeg", "-i", path,
"-af", "loudnorm=print_format=json",
"-f", "null", "-",
]
result = subprocess.run(cmd, capture_output=True, text=True)
stderr = result.stderr
start = stderr.rfind("{")
end = stderr.rfind("}") + 1
stats = json.loads(stderr[start:end])
return stats
def normalize_loudness(input_path: str, mode: str = "auto", target_lufs: float = -16.0) -> str:
stats = _measure_loudness(input_path)
if mode == "auto":
# auto 模式:保留原始响度感受,只修正真峰值超标/响度范围过大的问题,
# 不强行拉到某个固定 LUFS。
i = float(stats["input_i"])
else:
i = target_lufs
out_path = _new_path("_norm.wav")
af = (
f"loudnorm=I={i}:TP=-1.5:LRA=11:"
f"measured_I={stats['input_i']}:measured_TP={stats['input_tp']}:"
f"measured_LRA={stats['input_lra']}:measured_thresh={stats['input_thresh']}:"
f"offset={stats['target_offset']}:linear=true"
)
cmd = ["ffmpeg", "-y", "-i", input_path, "-af", af, "-ar", "44100", out_path]
subprocess.run(cmd, capture_output=True, text=True, check=True)
return out_path
# ---------------------------------------------------------------------------
# 步骤 5: 处理后人声与伴奏重新混音
# ---------------------------------------------------------------------------
def remix_with_instrumental(vocal_path: str, instrumental_path: str, vocal_gain_db: float = 0.0) -> str:
vocal = AudioSegment.from_file(vocal_path) + vocal_gain_db
instrumental = AudioSegment.from_file(instrumental_path)
length = max(len(vocal), len(instrumental))
vocal = vocal + AudioSegment.silent(duration=max(0, length - len(vocal)))
instrumental = instrumental + AudioSegment.silent(duration=max(0, length - len(instrumental)))
mixed = instrumental.overlay(vocal)
out_path = _new_path("_mixed.wav")
mixed.export(out_path, format="wav")
return out_path
# ---------------------------------------------------------------------------
# 整合流水线
# ---------------------------------------------------------------------------
def run_pipeline(
input_path: str,
do_separate: bool = True,
separation_model: str = list(SEPARATION_MODELS.keys())[0],
do_enhance: bool = True,
do_reverb: bool = True,
room_size: float = 0.5,
damping: float = 0.5,
wet_level: float = 0.3,
dry_level: float = 0.7,
do_remix: bool = False,
vocal_gain_db: float = 0.0,
do_normalize: bool = True,
normalize_mode: str = "auto",
target_lufs: float = -16.0,
):
vocals_path = None
instrumental_path = None
working = input_path
if do_separate:
vocals_path, instrumental_path = separate_vocals(input_path, separation_model)
working = vocals_path
if do_enhance:
working = enhance_audio(working)
if do_reverb:
working = add_reverb(working, room_size, damping, wet_level, dry_level)
if do_remix and instrumental_path:
working = remix_with_instrumental(working, instrumental_path, vocal_gain_db)
if do_normalize:
working = normalize_loudness(working, normalize_mode, target_lufs)
return working, vocals_path, instrumental_path
# ---------------------------------------------------------------------------
# Gradio 界面
# ---------------------------------------------------------------------------
def gradio_handler(
audio_file,
do_separate,
separation_model,
do_enhance,
do_reverb,
room_size,
damping,
wet_level,
dry_level,
do_remix,
vocal_gain_db,
do_normalize,
normalize_mode,
target_lufs,
progress=gr.Progress(),
):
if audio_file is None:
raise gr.Error("请先上传一个音频文件")
progress(0.1, desc="处理中...")
final_path, vocals_path, instrumental_path = run_pipeline(
audio_file,
do_separate=do_separate,
separation_model=separation_model,
do_enhance=do_enhance,
do_reverb=do_reverb,
room_size=room_size,
damping=damping,
wet_level=wet_level,
dry_level=dry_level,
do_remix=do_remix,
vocal_gain_db=vocal_gain_db,
do_normalize=do_normalize,
normalize_mode=normalize_mode,
target_lufs=target_lufs,
)
progress(1.0, desc="完成")
return final_path, vocals_path, instrumental_path
with gr.Blocks(title="Vocal Studio") as demo:
gr.Markdown("## 🎙️ Vocal Studio — 人声分离 · 降噪增强 · 室内混响 · 响度归一化")
with gr.Row():
with gr.Column():
audio_input = gr.Audio(label="上传原始歌曲/音频", type="filepath")
with gr.Accordion("① 人声/伴奏分离", open=True):
do_separate = gr.Checkbox(value=True, label="启用分离")
separation_model = gr.Dropdown(
choices=list(SEPARATION_MODELS.keys()),
value=list(SEPARATION_MODELS.keys())[0],
label="分离模型",
)
with gr.Accordion("② AI 降噪 / 音质增强", open=True):
do_enhance = gr.Checkbox(value=True, label="启用增强 (DeepFilterNet)")
with gr.Accordion("③ 室内混响", open=True):
do_reverb = gr.Checkbox(value=True, label="启用混响")
room_size = gr.Slider(0.0, 1.0, value=0.5, label="房间大小 room_size")
damping = gr.Slider(0.0, 1.0, value=0.5, label="阻尼 damping (越大高频衰减越快)")
wet_level = gr.Slider(0.0, 1.0, value=0.3, label="湿声比例 wet_level")
dry_level = gr.Slider(0.0, 1.0, value=0.7, label="干声比例 dry_level")
with gr.Accordion("④ 混回伴奏 (可选)", open=False):
do_remix = gr.Checkbox(value=False, label="处理后人声与原伴奏混音")
vocal_gain_db = gr.Slider(-12, 12, value=0.0, label="人声增益 (dB)")
with gr.Accordion("⑤ 响度归一化", open=True):
do_normalize = gr.Checkbox(value=True, label="启用归一化")
normalize_mode = gr.Radio(
choices=["auto", "manual"], value="auto",
label="模式 (auto=保留原响度感受, manual=拉到固定LUFS)",
)
target_lufs = gr.Slider(-30, -6, value=-16.0, label="目标 LUFS (仅 manual 模式生效)")
run_btn = gr.Button("🚀 开始处理", variant="primary")
with gr.Column():
final_output = gr.Audio(label="最终输出", type="filepath")
vocals_output = gr.Audio(label="分离出的人声(处理前)", type="filepath")
instrumental_output = gr.Audio(label="分离出的伴奏", type="filepath")
run_btn.click(
fn=gradio_handler,
inputs=[
audio_input, do_separate, separation_model,
do_enhance,
do_reverb, room_size, damping, wet_level, dry_level,
do_remix, vocal_gain_db,
do_normalize, normalize_mode, target_lufs,
],
outputs=[final_output, vocals_output, instrumental_output],
)
# ---------------------------------------------------------------------------
# REST API (给 OpenClaw 等自动化流水线用)
# ---------------------------------------------------------------------------
app = FastAPI(title="Vocal Studio API")
@app.get("/health")
def health():
return {"status": "ok"}
@app.post("/api/process")
async def api_process(
file: UploadFile = File(...),
do_separate: bool = Form(True),
separation_model: str = Form(list(SEPARATION_MODELS.keys())[0]),
do_enhance: bool = Form(True),
do_reverb: bool = Form(True),
room_size: float = Form(0.5),
damping: float = Form(0.5),
wet_level: float = Form(0.3),
dry_level: float = Form(0.7),
do_remix: bool = Form(False),
vocal_gain_db: float = Form(0.0),
do_normalize: bool = Form(True),
normalize_mode: str = Form("auto"),
target_lufs: float = Form(-16.0),
):
input_path = _new_path(Path(file.filename).suffix or ".wav")
with open(input_path, "wb") as f:
f.write(await file.read())
try:
final_path, vocals_path, instrumental_path = run_pipeline(
input_path,
do_separate=do_separate,
separation_model=separation_model,
do_enhance=do_enhance,
do_reverb=do_reverb,
room_size=room_size,
damping=damping,
wet_level=wet_level,
dry_level=dry_level,
do_remix=do_remix,
vocal_gain_db=vocal_gain_db,
do_normalize=do_normalize,
normalize_mode=normalize_mode,
target_lufs=target_lufs,
)
except Exception as e:
return JSONResponse(status_code=500, content={"error": str(e)})
return FileResponse(final_path, media_type="audio/wav", filename="processed.wav")
app = gr.mount_gradio_app(app, demo, path="/")
if __name__ == "__main__":
import uvicorn
uvicorn.run(app, host="0.0.0.0", port=int(os.environ.get("GRADIO_SERVER_PORT", 7860)))