Spaces:
Sleeping
Sleeping
Download app/services/remix_engine.py from Rthur2003/crowncode-backend: direct link, hf CLI and curl.
- Browser
- Download file 5.13 kB
-
https://huggingface.co/spaces/Rthur2003/crowncode-backend/resolve/main/app/services/remix_engine.py
- Command line
-
hf download hf://spaces/Rthur2003/crowncode-backend/app/services/remix_engine.py
-
curl -L -o remix_engine.py https://huggingface.co/spaces/Rthur2003/crowncode-backend/resolve/main/app/services/remix_engine.py
5.13 kB
| """ | |
| Creator Studio AI Remix engine — real DSP, not a mock. | |
| Takes two uploaded tracks, detects each one's tempo (via librosa beat | |
| tracking) and musical key (via the same Krumhansl-Schmuckler chroma | |
| correlation used by dataset_organizer.py), time-stretches/pitch-shifts the | |
| second track to match the first, and blends them with a crossfade so the | |
| end of track A overlaps into the start of track B. | |
| This is deliberately a "remix as a smooth two-track blend" rather than a | |
| full beat-synced stem remix (that needs source separation, which this | |
| codebase doesn't ship) — but every number it reports (BPM, key, applied | |
| stretch rate) comes from real analysis of the real uploaded audio. | |
| """ | |
| import io | |
| import logging | |
| import librosa | |
| import numpy as np | |
| import soundfile as sf | |
| from app.schemas import RemixOptions | |
| from app.services.dataset_organizer import PITCH_CLASSES, detect_key | |
| logger = logging.getLogger(__name__) | |
| _MAX_STRETCH_RATE = 1.6 | |
| _MIN_STRETCH_RATE = 0.625 # 1 / _MAX_STRETCH_RATE, kept symmetric in log-space | |
| class RemixResult: | |
| """Plain result container, mirroring DatasetEntryMetadata's pattern — | |
| keeps this module free of a pydantic dependency.""" | |
| def __init__(self, audio: io.BytesIO, analysis: dict) -> None: | |
| self.audio = audio | |
| self.analysis = analysis | |
| def _analyze_track(y: np.ndarray, sr: int) -> tuple[float, str]: | |
| tempo, _ = librosa.beat.beat_track(y=y, sr=sr) | |
| bpm = round(float(np.atleast_1d(tempo)[0]), 1) | |
| chroma = librosa.feature.chroma_cqt(y=y, sr=sr) | |
| key = detect_key(np.mean(chroma, axis=1)) | |
| return bpm, key | |
| def _key_root_index(key: str) -> int: | |
| root = key.split(" ")[0] | |
| return PITCH_CLASSES.index(root) if root in PITCH_CLASSES else 0 | |
| def _semitone_distance(from_key: str, to_key: str) -> float: | |
| """Shortest signed semitone distance between two detected keys' roots, | |
| e.g. "C major" -> "A major" is -3 (down 3), not +9 (up 9).""" | |
| diff = (_key_root_index(to_key) - _key_root_index(from_key)) % 12 | |
| if diff > 6: | |
| diff -= 12 | |
| return float(diff) | |
| def build_remix( | |
| track_a_bytes: bytes, | |
| track_b_bytes: bytes, | |
| options: RemixOptions, | |
| ) -> RemixResult: | |
| try: | |
| y_a, sr = librosa.load(io.BytesIO(track_a_bytes), sr=22050, mono=True) | |
| y_b, _ = librosa.load(io.BytesIO(track_b_bytes), sr=sr, mono=True) | |
| except Exception as e: | |
| logger.error(f"Failed to load audio for remix: {e}", exc_info=True) | |
| raise ValueError(f"Could not read audio file: {e}") | |
| if y_a.size == 0 or float(np.max(np.abs(y_a))) < 1e-6: | |
| raise ValueError("Track A is empty or silent") | |
| if y_b.size == 0 or float(np.max(np.abs(y_b))) < 1e-6: | |
| raise ValueError("Track B is empty or silent") | |
| bpm_a, key_a = _analyze_track(y_a, sr) | |
| bpm_b, key_b = _analyze_track(y_b, sr) | |
| applied_rate = 1.0 | |
| if options.match_tempo and bpm_a >= 1 and bpm_b >= 1: | |
| raw_rate = bpm_b / bpm_a | |
| # Clamp to a musically sane range — librosa's phase vocoder degrades | |
| # badly outside roughly 0.6x-1.6x, and a tempo detector octave error | |
| # (BPM doubled/halved) would otherwise send this to an extreme. | |
| applied_rate = float(np.clip(raw_rate, _MIN_STRETCH_RATE, _MAX_STRETCH_RATE)) | |
| y_b = librosa.effects.time_stretch(y_b, rate=applied_rate) | |
| logger.info(f"Remix: stretched track B by {applied_rate:.3f}x ({bpm_b}->{bpm_a} bpm)") | |
| applied_semitones = 0.0 | |
| if options.match_key: | |
| applied_semitones = _semitone_distance(key_b, key_a) | |
| if abs(applied_semitones) > 0.01: | |
| y_b = librosa.effects.pitch_shift(y_b, sr=sr, n_steps=applied_semitones) | |
| logger.info(f"Remix: pitch-shifted track B by {applied_semitones:+.1f} semitones ({key_b}->{key_a})") | |
| crossfade_samples = int(options.crossfade_seconds * sr) | |
| crossfade_samples = min(crossfade_samples, len(y_a), len(y_b)) | |
| if crossfade_samples <= 0: | |
| # No usable overlap — just concatenate. | |
| mixed = np.concatenate([y_a, y_b]) | |
| else: | |
| t = np.linspace(0.0, 1.0, crossfade_samples, dtype=np.float64) | |
| if options.crossfade_curve == "equal_power": | |
| fade_out = np.cos(t * np.pi / 2) | |
| fade_in = np.sin(t * np.pi / 2) | |
| else: | |
| fade_out = 1.0 - t | |
| fade_in = t | |
| head = y_a[:-crossfade_samples] | |
| tail_a = y_a[-crossfade_samples:] | |
| overlap = tail_a * fade_out.astype(np.float32) + y_b[:crossfade_samples] * fade_in.astype(np.float32) | |
| rest_b = y_b[crossfade_samples:] | |
| mixed = np.concatenate([head, overlap, rest_b]) | |
| mixed = librosa.util.normalize(mixed) | |
| out_buffer = io.BytesIO() | |
| sf.write(out_buffer, mixed, sr, format="WAV") | |
| out_buffer.seek(0) | |
| analysis = { | |
| "track_a_bpm": bpm_a, | |
| "track_b_bpm": bpm_b, | |
| "track_a_key": key_a, | |
| "track_b_key": key_b, | |
| "applied_stretch_rate": round(applied_rate, 3), | |
| "applied_pitch_shift_semitones": applied_semitones, | |
| "output_duration_sec": round(float(len(mixed) / sr), 2), | |
| } | |
| return RemixResult(audio=out_buffer, analysis=analysis) | |