crowncode-backend / app /services /multitrack_mixer.py
Rthur2003's picture
feat: remix ve Multitrack API uç noktaları ile rate limiting ve Pydantic schema doğrulama eklendi
26c0314
Raw History Blame Contribute Delete
4.98 kB
"""
Creator Studio Multitrack Mixer — real DSP, not a mock.
Takes 2-6 uploaded tracks, applies per-track gain (dB) and stereo pan,
aligns them to a common length (shorter tracks are zero-padded — silence,
not looped, since these are independent parts of one arrangement rather
than a loopable texture like remix_engine's crossfade blend), sums them
into a single stereo mix, and peak-normalizes to prevent clipping.
Equal-power panning (constant-power law) is used so a track panned to
either side doesn't lose perceived loudness relative to center.
"""
import io
import logging
import librosa
import numpy as np
import soundfile as sf
from app.schemas import MultitrackChannelSettings, MultitrackOptions
logger = logging.getLogger(__name__)
MAX_CHANNELS = 6
MIN_CHANNELS = 2
class MultitrackResult:
"""Plain result container, mirroring RemixResult's pattern."""
def __init__(self, audio: io.BytesIO, analysis: dict) -> None:
self.audio = audio
self.analysis = analysis
def _db_to_linear(db: float) -> float:
return float(10.0 ** (db / 20.0))
def _equal_power_pan(mono: np.ndarray, pan: float) -> np.ndarray:
"""Pan a mono signal to stereo using the constant-power law.
pan=-1 -> full left, 0 -> center (both channels at -3dB), 1 -> full right."""
angle = (pan + 1.0) * (np.pi / 4.0) # maps [-1, 1] -> [0, pi/2]
left_gain = float(np.cos(angle))
right_gain = float(np.sin(angle))
stereo = np.stack([mono * left_gain, mono * right_gain], axis=1)
return stereo
def mix_tracks(
track_bytes_list: list[bytes],
options: MultitrackOptions,
) -> MultitrackResult:
if len(track_bytes_list) < MIN_CHANNELS:
raise ValueError(f"At least {MIN_CHANNELS} tracks are required for a multitrack mix")
if len(track_bytes_list) > MAX_CHANNELS:
raise ValueError(f"At most {MAX_CHANNELS} tracks are supported per mix")
channels = options.channels
if len(channels) != len(track_bytes_list):
# Missing settings default to unity gain, centered, unmuted —
# a caller can send fewer channel configs than files and get sane
# defaults for the rest rather than a validation error.
channels = list(channels) + [
MultitrackChannelSettings() for _ in range(len(track_bytes_list) - len(channels))
]
sr = 22050
loaded: list[np.ndarray] = []
for i, raw in enumerate(track_bytes_list):
try:
y, _ = librosa.load(io.BytesIO(raw), sr=sr, mono=True)
except Exception as e:
logger.error(f"Failed to load track {i} for multitrack mix: {e}", exc_info=True)
raise ValueError(f"Could not read audio for track {i + 1}: {e}")
if y.size == 0:
raise ValueError(f"Track {i + 1} is empty")
loaded.append(y)
max_len = max(len(y) for y in loaded)
channel_reports = []
mix = np.zeros((max_len, 2), dtype=np.float64)
for i, (y, settings) in enumerate(zip(loaded, channels)):
duration_sec = round(float(len(y) / sr), 2)
if settings.muted:
channel_reports.append({
"index": i,
"duration_sec": duration_sec,
"applied_gain_db": settings.gain_db,
"pan": settings.pan,
"muted": True,
"peak_level": 0.0,
})
continue
# Zero-pad to common length — independent arrangement parts should
# not loop into each other the way a two-track crossfade blend would.
padded = np.pad(y, (0, max_len - len(y)))
gained = padded * _db_to_linear(settings.gain_db)
stereo = _equal_power_pan(gained, settings.pan)
mix += stereo
peak = float(np.max(np.abs(gained))) if gained.size else 0.0
channel_reports.append({
"index": i,
"duration_sec": duration_sec,
"applied_gain_db": settings.gain_db,
"pan": settings.pan,
"muted": False,
"peak_level": round(peak, 4),
})
output_peak = float(np.max(np.abs(mix))) if mix.size else 0.0
clipping_prevented = False
if options.normalize_output and output_peak > 1.0:
mix = mix / output_peak * 0.98
clipping_prevented = True
output_peak = 0.98
elif output_peak > 1.0:
# Caller explicitly disabled normalization — hard-clip rather than
# let soundfile silently wrap/distort on write.
mix = np.clip(mix, -1.0, 1.0)
clipping_prevented = True
output_peak = 1.0
out_buffer = io.BytesIO()
sf.write(out_buffer, mix.astype(np.float32), sr, format="WAV")
out_buffer.seek(0)
analysis = {
"channels": channel_reports,
"output_duration_sec": round(float(max_len / sr), 2),
"output_peak_level": round(output_peak, 4),
"clipping_prevented": clipping_prevented,
}
return MultitrackResult(audio=out_buffer, analysis=analysis)