""" Creator Studio Multitrack Mixer — real DSP, not a mock. Takes 2-6 uploaded tracks, applies per-track gain (dB) and stereo pan, aligns them to a common length (shorter tracks are zero-padded — silence, not looped, since these are independent parts of one arrangement rather than a loopable texture like remix_engine's crossfade blend), sums them into a single stereo mix, and peak-normalizes to prevent clipping. Equal-power panning (constant-power law) is used so a track panned to either side doesn't lose perceived loudness relative to center. """ import io import logging import librosa import numpy as np import soundfile as sf from app.schemas import MultitrackChannelSettings, MultitrackOptions logger = logging.getLogger(__name__) MAX_CHANNELS = 6 MIN_CHANNELS = 2 class MultitrackResult: """Plain result container, mirroring RemixResult's pattern.""" def __init__(self, audio: io.BytesIO, analysis: dict) -> None: self.audio = audio self.analysis = analysis def _db_to_linear(db: float) -> float: return float(10.0 ** (db / 20.0)) def _equal_power_pan(mono: np.ndarray, pan: float) -> np.ndarray: """Pan a mono signal to stereo using the constant-power law. pan=-1 -> full left, 0 -> center (both channels at -3dB), 1 -> full right.""" angle = (pan + 1.0) * (np.pi / 4.0) # maps [-1, 1] -> [0, pi/2] left_gain = float(np.cos(angle)) right_gain = float(np.sin(angle)) stereo = np.stack([mono * left_gain, mono * right_gain], axis=1) return stereo def mix_tracks( track_bytes_list: list[bytes], options: MultitrackOptions, ) -> MultitrackResult: if len(track_bytes_list) < MIN_CHANNELS: raise ValueError(f"At least {MIN_CHANNELS} tracks are required for a multitrack mix") if len(track_bytes_list) > MAX_CHANNELS: raise ValueError(f"At most {MAX_CHANNELS} tracks are supported per mix") channels = options.channels if len(channels) != len(track_bytes_list): # Missing settings default to unity gain, centered, unmuted — # a caller can send fewer channel configs than files and get sane # defaults for the rest rather than a validation error. channels = list(channels) + [ MultitrackChannelSettings() for _ in range(len(track_bytes_list) - len(channels)) ] sr = 22050 loaded: list[np.ndarray] = [] for i, raw in enumerate(track_bytes_list): try: y, _ = librosa.load(io.BytesIO(raw), sr=sr, mono=True) except Exception as e: logger.error(f"Failed to load track {i} for multitrack mix: {e}", exc_info=True) raise ValueError(f"Could not read audio for track {i + 1}: {e}") if y.size == 0: raise ValueError(f"Track {i + 1} is empty") loaded.append(y) max_len = max(len(y) for y in loaded) channel_reports = [] mix = np.zeros((max_len, 2), dtype=np.float64) for i, (y, settings) in enumerate(zip(loaded, channels)): duration_sec = round(float(len(y) / sr), 2) if settings.muted: channel_reports.append({ "index": i, "duration_sec": duration_sec, "applied_gain_db": settings.gain_db, "pan": settings.pan, "muted": True, "peak_level": 0.0, }) continue # Zero-pad to common length — independent arrangement parts should # not loop into each other the way a two-track crossfade blend would. padded = np.pad(y, (0, max_len - len(y))) gained = padded * _db_to_linear(settings.gain_db) stereo = _equal_power_pan(gained, settings.pan) mix += stereo peak = float(np.max(np.abs(gained))) if gained.size else 0.0 channel_reports.append({ "index": i, "duration_sec": duration_sec, "applied_gain_db": settings.gain_db, "pan": settings.pan, "muted": False, "peak_level": round(peak, 4), }) output_peak = float(np.max(np.abs(mix))) if mix.size else 0.0 clipping_prevented = False if options.normalize_output and output_peak > 1.0: mix = mix / output_peak * 0.98 clipping_prevented = True output_peak = 0.98 elif output_peak > 1.0: # Caller explicitly disabled normalization — hard-clip rather than # let soundfile silently wrap/distort on write. mix = np.clip(mix, -1.0, 1.0) clipping_prevented = True output_peak = 1.0 out_buffer = io.BytesIO() sf.write(out_buffer, mix.astype(np.float32), sr, format="WAV") out_buffer.seek(0) analysis = { "channels": channel_reports, "output_duration_sec": round(float(max_len / sr), 2), "output_peak_level": round(output_peak, 4), "clipping_prevented": clipping_prevented, } return MultitrackResult(audio=out_buffer, analysis=analysis)