crowncode-backend / app /services /remix_engine.py
Rthur2003's picture
feat: remix engine routes dream analysis servisleri ve dataset organization utilities eklendi
e49c006
Raw History Blame Contribute Delete
5.13 kB
"""
Creator Studio AI Remix engine — real DSP, not a mock.
Takes two uploaded tracks, detects each one's tempo (via librosa beat
tracking) and musical key (via the same Krumhansl-Schmuckler chroma
correlation used by dataset_organizer.py), time-stretches/pitch-shifts the
second track to match the first, and blends them with a crossfade so the
end of track A overlaps into the start of track B.
This is deliberately a "remix as a smooth two-track blend" rather than a
full beat-synced stem remix (that needs source separation, which this
codebase doesn't ship) — but every number it reports (BPM, key, applied
stretch rate) comes from real analysis of the real uploaded audio.
"""
import io
import logging
import librosa
import numpy as np
import soundfile as sf
from app.schemas import RemixOptions
from app.services.dataset_organizer import PITCH_CLASSES, detect_key
logger = logging.getLogger(__name__)
_MAX_STRETCH_RATE = 1.6
_MIN_STRETCH_RATE = 0.625 # 1 / _MAX_STRETCH_RATE, kept symmetric in log-space
class RemixResult:
"""Plain result container, mirroring DatasetEntryMetadata's pattern —
keeps this module free of a pydantic dependency."""
def __init__(self, audio: io.BytesIO, analysis: dict) -> None:
self.audio = audio
self.analysis = analysis
def _analyze_track(y: np.ndarray, sr: int) -> tuple[float, str]:
tempo, _ = librosa.beat.beat_track(y=y, sr=sr)
bpm = round(float(np.atleast_1d(tempo)[0]), 1)
chroma = librosa.feature.chroma_cqt(y=y, sr=sr)
key = detect_key(np.mean(chroma, axis=1))
return bpm, key
def _key_root_index(key: str) -> int:
root = key.split(" ")[0]
return PITCH_CLASSES.index(root) if root in PITCH_CLASSES else 0
def _semitone_distance(from_key: str, to_key: str) -> float:
"""Shortest signed semitone distance between two detected keys' roots,
e.g. "C major" -> "A major" is -3 (down 3), not +9 (up 9)."""
diff = (_key_root_index(to_key) - _key_root_index(from_key)) % 12
if diff > 6:
diff -= 12
return float(diff)
def build_remix(
track_a_bytes: bytes,
track_b_bytes: bytes,
options: RemixOptions,
) -> RemixResult:
try:
y_a, sr = librosa.load(io.BytesIO(track_a_bytes), sr=22050, mono=True)
y_b, _ = librosa.load(io.BytesIO(track_b_bytes), sr=sr, mono=True)
except Exception as e:
logger.error(f"Failed to load audio for remix: {e}", exc_info=True)
raise ValueError(f"Could not read audio file: {e}")
if y_a.size == 0 or float(np.max(np.abs(y_a))) < 1e-6:
raise ValueError("Track A is empty or silent")
if y_b.size == 0 or float(np.max(np.abs(y_b))) < 1e-6:
raise ValueError("Track B is empty or silent")
bpm_a, key_a = _analyze_track(y_a, sr)
bpm_b, key_b = _analyze_track(y_b, sr)
applied_rate = 1.0
if options.match_tempo and bpm_a >= 1 and bpm_b >= 1:
raw_rate = bpm_b / bpm_a
# Clamp to a musically sane range — librosa's phase vocoder degrades
# badly outside roughly 0.6x-1.6x, and a tempo detector octave error
# (BPM doubled/halved) would otherwise send this to an extreme.
applied_rate = float(np.clip(raw_rate, _MIN_STRETCH_RATE, _MAX_STRETCH_RATE))
y_b = librosa.effects.time_stretch(y_b, rate=applied_rate)
logger.info(f"Remix: stretched track B by {applied_rate:.3f}x ({bpm_b}->{bpm_a} bpm)")
applied_semitones = 0.0
if options.match_key:
applied_semitones = _semitone_distance(key_b, key_a)
if abs(applied_semitones) > 0.01:
y_b = librosa.effects.pitch_shift(y_b, sr=sr, n_steps=applied_semitones)
logger.info(f"Remix: pitch-shifted track B by {applied_semitones:+.1f} semitones ({key_b}->{key_a})")
crossfade_samples = int(options.crossfade_seconds * sr)
crossfade_samples = min(crossfade_samples, len(y_a), len(y_b))
if crossfade_samples <= 0:
# No usable overlap — just concatenate.
mixed = np.concatenate([y_a, y_b])
else:
t = np.linspace(0.0, 1.0, crossfade_samples, dtype=np.float64)
if options.crossfade_curve == "equal_power":
fade_out = np.cos(t * np.pi / 2)
fade_in = np.sin(t * np.pi / 2)
else:
fade_out = 1.0 - t
fade_in = t
head = y_a[:-crossfade_samples]
tail_a = y_a[-crossfade_samples:]
overlap = tail_a * fade_out.astype(np.float32) + y_b[:crossfade_samples] * fade_in.astype(np.float32)
rest_b = y_b[crossfade_samples:]
mixed = np.concatenate([head, overlap, rest_b])
mixed = librosa.util.normalize(mixed)
out_buffer = io.BytesIO()
sf.write(out_buffer, mixed, sr, format="WAV")
out_buffer.seek(0)
analysis = {
"track_a_bpm": bpm_a,
"track_b_bpm": bpm_b,
"track_a_key": key_a,
"track_b_key": key_b,
"applied_stretch_rate": round(applied_rate, 3),
"applied_pitch_shift_semitones": applied_semitones,
"output_duration_sec": round(float(len(mixed) / sr), 2),
}
return RemixResult(audio=out_buffer, analysis=analysis)