Rthur2003 commited on
Commit
2eb5dd7
·
1 Parent(s): 8de7693

feat: audio processing ve organization API uç noktaları ile unit testing eklendi

Browse files
app/routes/data_processing.py CHANGED
@@ -11,8 +11,9 @@ from pydantic import ValidationError
11
  import json
12
  import logging
13
 
14
- from app.schemas import AudioAugmentationOptions
15
- from app.services.audio_processor import process_audio
 
16
 
17
  router = APIRouter(prefix="/api/process", tags=["Data Processing"])
18
  logger = logging.getLogger(__name__)
@@ -44,15 +45,34 @@ async def _process_rate_limit(request: Request) -> None:
44
  _process_rate_store[client_ip] = hits
45
 
46
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
47
  @router.post("/audio", dependencies=[Depends(_process_rate_limit)])
48
  async def process_audio_endpoint(
49
  file: UploadFile = File(...),
50
- options: str = Form(default="{}")
 
51
  ):
52
  """
53
  Process an audio file with the given augmentation options.
54
  Returns the processed WAV file.
55
  options is a JSON string; if missing or empty, defaults to all-off.
 
 
56
  """
57
  MAX_PAYLOAD_BYTES = 30 * 1024 * 1024 # 30 MB
58
  logger.info(f"Received audio processing request for file: {file.filename}")
@@ -72,13 +92,86 @@ async def process_audio_endpoint(
72
  if not file.content_type or not file.content_type.startswith("audio/"):
73
  raise HTTPException(status_code=400, detail={"code": "invalid_file_type", "message": "Invalid file type. Must be audio."})
74
 
 
 
 
 
 
 
 
 
75
  try:
76
  # Sanitize filename to prevent injection attacks
77
  safe_filename = re.sub(r'[^a-zA-Z0-9._-]', '_', file.filename or 'audio')
78
  if safe_filename.endswith('.wav'):
79
  safe_filename = safe_filename[:-4]
80
 
81
- # Read file in chunks to enforce size limit before full allocation
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
82
  chunks = []
83
  total_read = 0
84
  chunk_size = 1024 * 1024 # 1 MB chunks
@@ -92,19 +185,61 @@ async def process_audio_endpoint(
92
  chunks.append(chunk)
93
  content = b"".join(chunks)
94
 
95
- # Process audio
96
- processed_audio = process_audio(content, parsed_options)
97
 
98
- # Return as downloadable file
99
- output_filename = f"processed_{safe_filename}.wav"
100
  return StreamingResponse(
101
- processed_audio,
102
- media_type="audio/wav",
103
  headers={"Content-Disposition": f"attachment; filename={output_filename}"}
104
  )
105
 
106
  except ValueError as e:
107
  raise HTTPException(status_code=400, detail={"code": "validation_error", "message": str(e)})
 
 
 
 
108
  except Exception as e:
109
- logger.error(f"Unexpected error in audio processing: {e}", exc_info=True)
110
- raise HTTPException(status_code=500, detail={"code": "internal_error", "message": "Internal server error during audio processing"})
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
11
  import json
12
  import logging
13
 
14
+ from app.schemas import AudioAugmentationOptions, AudioConvertOptions, DatasetEntryMetadata
15
+ from app.services.audio_processor import process_audio, convert_audio_format
16
+ from app.services.dataset_organizer import analyze_for_organization
17
 
18
  router = APIRouter(prefix="/api/process", tags=["Data Processing"])
19
  logger = logging.getLogger(__name__)
 
45
  _process_rate_store[client_ip] = hits
46
 
47
 
48
+ async def _read_capped(file: UploadFile, max_bytes: int) -> bytes:
49
+ """Read an UploadFile in chunks, raising 413 if it exceeds max_bytes."""
50
+ chunks = []
51
+ total_read = 0
52
+ chunk_size = 1024 * 1024 # 1 MB chunks
53
+ while True:
54
+ chunk = await file.read(chunk_size)
55
+ if not chunk:
56
+ break
57
+ total_read += len(chunk)
58
+ if total_read > max_bytes:
59
+ raise HTTPException(status_code=413, detail={"code": "file_too_large", "message": f"File too large. Maximum size is {max_bytes // (1024*1024)} MB."})
60
+ chunks.append(chunk)
61
+ return b"".join(chunks)
62
+
63
+
64
  @router.post("/audio", dependencies=[Depends(_process_rate_limit)])
65
  async def process_audio_endpoint(
66
  file: UploadFile = File(...),
67
+ options: str = Form(default="{}"),
68
+ mix_file: UploadFile | None = File(default=None),
69
  ):
70
  """
71
  Process an audio file with the given augmentation options.
72
  Returns the processed WAV file.
73
  options is a JSON string; if missing or empty, defaults to all-off.
74
+ mix_file: required when options.mixAudio is true — a second audio file
75
+ to blend into the primary track.
76
  """
77
  MAX_PAYLOAD_BYTES = 30 * 1024 * 1024 # 30 MB
78
  logger.info(f"Received audio processing request for file: {file.filename}")
 
92
  if not file.content_type or not file.content_type.startswith("audio/"):
93
  raise HTTPException(status_code=400, detail={"code": "invalid_file_type", "message": "Invalid file type. Must be audio."})
94
 
95
+ if parsed_options.mix_audio and mix_file is None:
96
+ raise HTTPException(
97
+ status_code=400,
98
+ detail={"code": "missing_mix_file", "message": "mixAudio is enabled but no second file was provided."}
99
+ )
100
+ if mix_file is not None and (not mix_file.content_type or not mix_file.content_type.startswith("audio/")):
101
+ raise HTTPException(status_code=400, detail={"code": "invalid_file_type", "message": "Second file for mixing must be audio."})
102
+
103
  try:
104
  # Sanitize filename to prevent injection attacks
105
  safe_filename = re.sub(r'[^a-zA-Z0-9._-]', '_', file.filename or 'audio')
106
  if safe_filename.endswith('.wav'):
107
  safe_filename = safe_filename[:-4]
108
 
109
+ content = await _read_capped(file, MAX_PAYLOAD_BYTES)
110
+ mix_content = await _read_capped(mix_file, MAX_PAYLOAD_BYTES) if mix_file is not None else None
111
+
112
+ # Process audio
113
+ processed_audio = process_audio(content, parsed_options, mix_with_bytes=mix_content)
114
+
115
+ # Return as downloadable file
116
+ output_filename = f"processed_{safe_filename}.wav"
117
+ return StreamingResponse(
118
+ processed_audio,
119
+ media_type="audio/wav",
120
+ headers={"Content-Disposition": f"attachment; filename={output_filename}"}
121
+ )
122
+
123
+ except ValueError as e:
124
+ raise HTTPException(status_code=400, detail={"code": "validation_error", "message": str(e)})
125
+ except HTTPException:
126
+ # Re-raise as-is (e.g. the 413 file_too_large above) — without this,
127
+ # the bare `except Exception` below catches it too (HTTPException IS
128
+ # an Exception) and replaces a correct 413 with a misleading 500.
129
+ raise
130
+ except Exception as e:
131
+ logger.error(f"Unexpected error in audio processing: {e}", exc_info=True)
132
+ raise HTTPException(status_code=500, detail={"code": "internal_error", "message": "Internal server error during audio processing"})
133
+
134
+
135
+ _CONVERT_MEDIA_TYPES = {
136
+ "wav": "audio/wav",
137
+ "mp3": "audio/mpeg",
138
+ "flac": "audio/flac",
139
+ "ogg": "audio/ogg",
140
+ }
141
+
142
+
143
+ @router.post("/audio/convert", dependencies=[Depends(_process_rate_limit)])
144
+ async def convert_audio_endpoint(
145
+ file: UploadFile = File(...),
146
+ options: str = Form(default="{}")
147
+ ):
148
+ """
149
+ Convert an audio file to another format (wav/mp3/flac/ogg).
150
+ Returns the converted file for download.
151
+ options is a JSON string; if missing or empty, defaults to wav @ 192kbps.
152
+ """
153
+ MAX_PAYLOAD_BYTES = 30 * 1024 * 1024 # 30 MB
154
+ logger.info(f"Received audio conversion request for file: {file.filename}")
155
+
156
+ raw_options = options.strip() if options else "{}"
157
+ if not raw_options:
158
+ raw_options = "{}"
159
+ try:
160
+ parsed_options = AudioConvertOptions.model_validate_json(raw_options)
161
+ except (ValidationError, json.JSONDecodeError) as e:
162
+ raise HTTPException(
163
+ status_code=422,
164
+ detail={"code": "invalid_options", "message": f"Invalid options format: {e}"}
165
+ )
166
+
167
+ if not file.content_type or not file.content_type.startswith("audio/"):
168
+ raise HTTPException(status_code=400, detail={"code": "invalid_file_type", "message": "Invalid file type. Must be audio."})
169
+
170
+ try:
171
+ safe_filename = re.sub(r'[^a-zA-Z0-9._-]', '_', file.filename or 'audio')
172
+ # Strip any existing extension so we don't end up with e.g. "song.mp3.flac"
173
+ safe_filename = re.sub(r'\.[a-zA-Z0-9]{1,5}$', '', safe_filename)
174
+
175
  chunks = []
176
  total_read = 0
177
  chunk_size = 1024 * 1024 # 1 MB chunks
 
185
  chunks.append(chunk)
186
  content = b"".join(chunks)
187
 
188
+ converted = convert_audio_format(content, parsed_options)
 
189
 
190
+ target = parsed_options.target_format
191
+ output_filename = f"{safe_filename}.{target}"
192
  return StreamingResponse(
193
+ converted,
194
+ media_type=_CONVERT_MEDIA_TYPES[target],
195
  headers={"Content-Disposition": f"attachment; filename={output_filename}"}
196
  )
197
 
198
  except ValueError as e:
199
  raise HTTPException(status_code=400, detail={"code": "validation_error", "message": str(e)})
200
+ except HTTPException:
201
+ # Re-raise as-is — see the matching comment in the /audio endpoint
202
+ # above for why this is needed before the bare `except Exception`.
203
+ raise
204
  except Exception as e:
205
+ logger.error(f"Unexpected error in audio conversion: {e}", exc_info=True)
206
+ raise HTTPException(status_code=500, detail={"code": "internal_error", "message": "Internal server error during audio conversion"})
207
+
208
+
209
+ @router.post("/audio/organize", response_model=DatasetEntryMetadata, dependencies=[Depends(_process_rate_limit)])
210
+ async def organize_audio_endpoint(file: UploadFile = File(...)):
211
+ """
212
+ Analyze an audio file and return real acoustic metadata (duration,
213
+ tempo, key, loudness) plus auto-generated tags for dataset tagging
214
+ and categorization. No transformation — read-only analysis.
215
+ """
216
+ MAX_PAYLOAD_BYTES = 30 * 1024 * 1024 # 30 MB
217
+ logger.info(f"Received audio organization request for file: {file.filename}")
218
+
219
+ if not file.content_type or not file.content_type.startswith("audio/"):
220
+ raise HTTPException(status_code=400, detail={"code": "invalid_file_type", "message": "Invalid file type. Must be audio."})
221
+
222
+ try:
223
+ chunks = []
224
+ total_read = 0
225
+ chunk_size = 1024 * 1024
226
+ while True:
227
+ chunk = await file.read(chunk_size)
228
+ if not chunk:
229
+ break
230
+ total_read += len(chunk)
231
+ if total_read > MAX_PAYLOAD_BYTES:
232
+ raise HTTPException(status_code=413, detail={"code": "file_too_large", "message": f"File too large. Maximum size is {MAX_PAYLOAD_BYTES // (1024*1024)} MB."})
233
+ chunks.append(chunk)
234
+ content = b"".join(chunks)
235
+
236
+ metadata = analyze_for_organization(content)
237
+ return DatasetEntryMetadata(**metadata.to_dict())
238
+
239
+ except ValueError as e:
240
+ raise HTTPException(status_code=400, detail={"code": "validation_error", "message": str(e)})
241
+ except HTTPException:
242
+ raise
243
+ except Exception as e:
244
+ logger.error(f"Unexpected error in audio organization: {e}", exc_info=True)
245
+ raise HTTPException(status_code=500, detail={"code": "internal_error", "message": "Internal server error during audio organization"})
app/schemas.py CHANGED
@@ -62,5 +62,30 @@ class AudioAugmentationOptions(BaseModel):
62
  speed_change: bool = Field(default=False, alias="speedChange", description="Apply random speed change")
63
  bass_boost: bool = Field(default=False, alias="bassBoost", description="Apply bass boost equalization")
64
  trim_silence: bool = Field(default=False, alias="trimSilence", description="Trim leading and trailing silence")
65
- mix_audio: bool = Field(default=False, alias="mixAudio", description="Mix with another audio track (placeholder)")
66
  add_noise: bool = Field(default=False, alias="addNoise", description="Add Gaussian noise")
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
62
  speed_change: bool = Field(default=False, alias="speedChange", description="Apply random speed change")
63
  bass_boost: bool = Field(default=False, alias="bassBoost", description="Apply bass boost equalization")
64
  trim_silence: bool = Field(default=False, alias="trimSilence", description="Trim leading and trailing silence")
65
+ mix_audio: bool = Field(default=False, alias="mixAudio", description="Mix with a second uploaded audio track")
66
  add_noise: bool = Field(default=False, alias="addNoise", description="Add Gaussian noise")
67
+
68
+
69
+ AudioTargetFormat = Literal["wav", "mp3", "flac", "ogg"]
70
+
71
+
72
+ class AudioConvertOptions(BaseModel):
73
+ model_config = {"populate_by_name": True}
74
+
75
+ target_format: AudioTargetFormat = Field(
76
+ default="wav", alias="targetFormat", description="Output container/codec to convert to"
77
+ )
78
+ bitrate_kbps: int = Field(
79
+ default=192, alias="bitrateKbps", ge=64, le=320,
80
+ description="Target bitrate for lossy formats (mp3/ogg); ignored for wav/flac"
81
+ )
82
+
83
+
84
+ class DatasetEntryMetadata(BaseModel):
85
+ model_config = {"populate_by_name": True}
86
+
87
+ duration_sec: float = Field(alias="durationSec")
88
+ tempo_bpm: float = Field(alias="tempoBpm")
89
+ key: str
90
+ loudness_db: float = Field(alias="loudnessDb")
91
+ tags: List[str]
app/services/audio_processor.py CHANGED
@@ -4,26 +4,57 @@ Audio processing service for data augmentation and manipulation.
4
 
5
  import io
6
  import logging
 
 
 
 
7
  import numpy as np
8
  import librosa
9
  import soundfile as sf
10
  import scipy.signal
11
  from fastapi import UploadFile
12
 
13
- from app.schemas import AudioAugmentationOptions
14
 
15
  logger = logging.getLogger(__name__)
16
 
17
- def process_audio(file_bytes: bytes, options: AudioAugmentationOptions) -> io.BytesIO:
 
 
 
 
 
 
 
 
 
 
 
18
  """
19
  Process audio file with requested augmentation options.
20
  Returns processed audio as BytesIO (WAV format).
 
 
 
 
21
  """
22
  try:
23
  # Load audio from bytes
24
  # librosa.load expects a file path or file-like object
25
  y, sr = librosa.load(io.BytesIO(file_bytes), sr=None)
26
-
 
 
 
 
 
 
 
 
 
 
 
 
27
  # 1. Trim Silence
28
  if options.trim_silence:
29
  y, _ = librosa.effects.trim(y, top_db=20)
@@ -69,3 +100,47 @@ def process_audio(file_bytes: bytes, options: AudioAugmentationOptions) -> io.By
69
  except Exception as e:
70
  logger.error(f"Error processing audio: {str(e)}", exc_info=True)
71
  raise ValueError(f"Audio processing failed: {str(e)}")
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
4
 
5
  import io
6
  import logging
7
+ import subprocess
8
+ import tempfile
9
+ from pathlib import Path
10
+
11
  import numpy as np
12
  import librosa
13
  import soundfile as sf
14
  import scipy.signal
15
  from fastapi import UploadFile
16
 
17
+ from app.schemas import AudioAugmentationOptions, AudioConvertOptions
18
 
19
  logger = logging.getLogger(__name__)
20
 
21
+ # soundfile/libsndfile writes these natively — no subprocess needed.
22
+ _NATIVE_FORMATS = {"wav": "WAV", "flac": "FLAC"}
23
+ # Everything else (mp3, ogg) goes through ffmpeg, same subprocess pattern
24
+ # already used elsewhere in this codebase for audio decode (see
25
+ # feature_extractor.py / vocal_analyzer.py's _ffmpeg_decode).
26
+ _FFMPEG_CODEC = {"mp3": "libmp3lame", "ogg": "libvorbis"}
27
+
28
+ def process_audio(
29
+ file_bytes: bytes,
30
+ options: AudioAugmentationOptions,
31
+ mix_with_bytes: bytes | None = None,
32
+ ) -> io.BytesIO:
33
  """
34
  Process audio file with requested augmentation options.
35
  Returns processed audio as BytesIO (WAV format).
36
+
37
+ mix_with_bytes: a second audio file to mix in when options.mix_audio is
38
+ set. Silently ignored (mix skipped) if mix_audio is off or no second
39
+ file was provided — the route layer decides whether that's an error.
40
  """
41
  try:
42
  # Load audio from bytes
43
  # librosa.load expects a file path or file-like object
44
  y, sr = librosa.load(io.BytesIO(file_bytes), sr=None)
45
+
46
+ # 0. Mix Audio — blend in a second track, resampled to match and
47
+ # looped/trimmed to the primary track's length so levels stay sane
48
+ # regardless of which clip is longer.
49
+ if options.mix_audio and mix_with_bytes:
50
+ y2, sr2 = librosa.load(io.BytesIO(mix_with_bytes), sr=sr)
51
+ if len(y2) < len(y):
52
+ repeats = int(np.ceil(len(y) / max(len(y2), 1)))
53
+ y2 = np.tile(y2, repeats)
54
+ y2 = y2[: len(y)]
55
+ y = librosa.util.normalize(y * 0.6 + y2 * 0.6)
56
+ logger.info("Applied mix_audio")
57
+
58
  # 1. Trim Silence
59
  if options.trim_silence:
60
  y, _ = librosa.effects.trim(y, top_db=20)
 
100
  except Exception as e:
101
  logger.error(f"Error processing audio: {str(e)}", exc_info=True)
102
  raise ValueError(f"Audio processing failed: {str(e)}")
103
+
104
+
105
+ def convert_audio_format(file_bytes: bytes, options: AudioConvertOptions) -> io.BytesIO:
106
+ """
107
+ Convert an audio file to the requested target format.
108
+ Returns the converted audio as BytesIO.
109
+ """
110
+ target = options.target_format
111
+
112
+ if target in _NATIVE_FORMATS:
113
+ try:
114
+ y, sr = librosa.load(io.BytesIO(file_bytes), sr=None)
115
+ out_buffer = io.BytesIO()
116
+ sf.write(out_buffer, y, sr, format=_NATIVE_FORMATS[target])
117
+ out_buffer.seek(0)
118
+ return out_buffer
119
+ except Exception as e:
120
+ logger.error(f"Error converting audio to {target}: {str(e)}", exc_info=True)
121
+ raise ValueError(f"Audio conversion failed: {str(e)}")
122
+
123
+ if target in _FFMPEG_CODEC:
124
+ with tempfile.NamedTemporaryFile(suffix=f".{target}", delete=False) as tmp:
125
+ tmp_path = tmp.name
126
+ try:
127
+ result = subprocess.run(
128
+ [
129
+ "ffmpeg", "-y", "-i", "pipe:0",
130
+ "-c:a", _FFMPEG_CODEC[target],
131
+ "-b:a", f"{options.bitrate_kbps}k",
132
+ tmp_path,
133
+ ],
134
+ input=file_bytes,
135
+ capture_output=True,
136
+ timeout=60,
137
+ )
138
+ if result.returncode != 0:
139
+ logger.error(f"ffmpeg conversion to {target} failed: {result.stderr.decode(errors='replace')[:300]}")
140
+ raise ValueError(f"Audio conversion to {target} failed")
141
+ with open(tmp_path, "rb") as f:
142
+ return io.BytesIO(f.read())
143
+ finally:
144
+ Path(tmp_path).unlink(missing_ok=True)
145
+
146
+ raise ValueError(f"Unsupported target format: {target}")
app/services/dataset_organizer.py ADDED
@@ -0,0 +1,161 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Dataset organization service — analyzes an audio file and derives real
3
+ acoustic metadata (duration, tempo, key, loudness, energy tier) used to
4
+ auto-tag and categorize entries in a dataset, per the "Veri Seti
5
+ Organizasyonu" tool advertised on the data-manipulation page.
6
+
7
+ Deliberately independent from feature_extractor.py (AURIS's 49-feature
8
+ AI-detection pipeline) — that module is tuned for AI-vs-human classification
9
+ and pulls in a much heavier feature set than dataset tagging needs. Coupling
10
+ this to it would make a simple tagging tool fragile to AI-detection changes.
11
+ """
12
+
13
+ import io
14
+ import logging
15
+
16
+ import librosa
17
+ import numpy as np
18
+
19
+ logger = logging.getLogger(__name__)
20
+
21
+ _PITCH_CLASSES = [
22
+ "C", "C#", "D", "D#", "E", "F",
23
+ "F#", "G", "G#", "A", "A#", "B",
24
+ ]
25
+
26
+ # Krumhansl-Schmuckler key profiles — standard weights for major/minor
27
+ # key detection via correlation against the chroma vector.
28
+ _MAJOR_PROFILE = np.array(
29
+ [6.35, 2.23, 3.48, 2.33, 4.38, 4.09, 2.52, 5.19, 2.39, 3.66, 2.29, 2.88]
30
+ )
31
+ _MINOR_PROFILE = np.array(
32
+ [6.33, 2.68, 3.52, 5.38, 2.60, 3.53, 2.54, 4.75, 3.98, 2.69, 3.34, 3.17]
33
+ )
34
+
35
+
36
+ def _detect_key(chroma_mean: np.ndarray) -> str:
37
+ """Correlate the mean chroma vector against all 24 rotated key profiles."""
38
+ best_score = -np.inf
39
+ best_key = "C major"
40
+ for shift in range(12):
41
+ major_rot = np.roll(_MAJOR_PROFILE, shift)
42
+ minor_rot = np.roll(_MINOR_PROFILE, shift)
43
+ major_score = float(np.corrcoef(chroma_mean, major_rot)[0, 1])
44
+ minor_score = float(np.corrcoef(chroma_mean, minor_rot)[0, 1])
45
+ if major_score > best_score:
46
+ best_score = major_score
47
+ best_key = f"{_PITCH_CLASSES[shift]} major"
48
+ if minor_score > best_score:
49
+ best_score = minor_score
50
+ best_key = f"{_PITCH_CLASSES[shift]} minor"
51
+ return best_key
52
+
53
+
54
+ def _tempo_tag(bpm: float) -> str:
55
+ # librosa.beat.beat_track returns ~0 when it can't lock onto a beat at
56
+ # all (e.g. a sustained tone or ambient texture with no rhythm) — that's
57
+ # "no beat detected", not literally "slow", so tag it separately rather
58
+ # than lump it into the slow bucket.
59
+ if bpm < 1:
60
+ return "beatless"
61
+ if bpm < 76:
62
+ return "slow"
63
+ if bpm < 120:
64
+ return "moderate"
65
+ if bpm < 150:
66
+ return "upbeat"
67
+ return "fast"
68
+
69
+
70
+ def _energy_tag(rms_mean: float) -> str:
71
+ # RMS on a float32 waveform normalized to [-1, 1]; these thresholds are
72
+ # calibrated against typical mixed/mastered music, not raw voice memos.
73
+ if rms_mean < 0.03:
74
+ return "ambient"
75
+ if rms_mean < 0.08:
76
+ return "calm"
77
+ if rms_mean < 0.15:
78
+ return "energetic"
79
+ return "intense"
80
+
81
+
82
+ def _loudness_lufs_approx(y: np.ndarray) -> float:
83
+ """
84
+ Rough integrated-loudness estimate in LUFS-like dB, using RMS as a
85
+ stand-in for full ITU-R BS.1770 K-weighting (that needs a dedicated
86
+ filter chain this tool doesn't need for a dataset-tagging heuristic).
87
+ """
88
+ rms = float(np.sqrt(np.mean(np.square(y)))) if len(y) else 0.0
89
+ if rms <= 0:
90
+ return -70.0
91
+ return round(float(20 * np.log10(rms)), 1)
92
+
93
+
94
+ class DatasetEntryMetadata:
95
+ """Plain result container — kept dependency-free so routes can shape
96
+ the JSON response without importing pydantic into this module."""
97
+
98
+ def __init__(
99
+ self,
100
+ duration_sec: float,
101
+ tempo_bpm: float,
102
+ key: str,
103
+ loudness_db: float,
104
+ tags: list[str],
105
+ ) -> None:
106
+ self.duration_sec = duration_sec
107
+ self.tempo_bpm = tempo_bpm
108
+ self.key = key
109
+ self.loudness_db = loudness_db
110
+ self.tags = tags
111
+
112
+ def to_dict(self) -> dict:
113
+ return {
114
+ "durationSec": self.duration_sec,
115
+ "tempoBpm": self.tempo_bpm,
116
+ "key": self.key,
117
+ "loudnessDb": self.loudness_db,
118
+ "tags": self.tags,
119
+ }
120
+
121
+
122
+ def analyze_for_organization(file_bytes: bytes) -> DatasetEntryMetadata:
123
+ """
124
+ Extract real acoustic metadata from an audio file for dataset tagging:
125
+ duration, tempo (BPM), musical key, approximate loudness, and a set of
126
+ auto-generated descriptive tags (tempo feel, energy level, duration
127
+ bucket) suitable for filtering/categorizing a growing dataset.
128
+ """
129
+ try:
130
+ y, sr = librosa.load(io.BytesIO(file_bytes), sr=22050, mono=True)
131
+ except Exception as e:
132
+ logger.error(f"Failed to load audio for organization: {e}", exc_info=True)
133
+ raise ValueError(f"Could not read audio file: {e}")
134
+
135
+ if y.size == 0 or float(np.max(np.abs(y))) < 1e-6:
136
+ raise ValueError("Audio file is empty or silent")
137
+
138
+ duration_sec = round(float(librosa.get_duration(y=y, sr=sr)), 2)
139
+
140
+ tempo, _ = librosa.beat.beat_track(y=y, sr=sr)
141
+ tempo_bpm = round(float(np.atleast_1d(tempo)[0]), 1)
142
+
143
+ chroma = librosa.feature.chroma_cqt(y=y, sr=sr)
144
+ key = _detect_key(np.mean(chroma, axis=1))
145
+
146
+ loudness_db = _loudness_lufs_approx(y)
147
+ rms_mean = float(np.mean(librosa.feature.rms(y=y)[0]))
148
+
149
+ tags = [_tempo_tag(tempo_bpm), _energy_tag(rms_mean)]
150
+ if duration_sec < 30:
151
+ tags.append("short-clip")
152
+ elif duration_sec > 240:
153
+ tags.append("long-form")
154
+
155
+ return DatasetEntryMetadata(
156
+ duration_sec=duration_sec,
157
+ tempo_bpm=tempo_bpm,
158
+ key=key,
159
+ loudness_db=loudness_db,
160
+ tags=tags,
161
+ )
tests/test_data_processing.py CHANGED
@@ -5,8 +5,23 @@ from __future__ import annotations
5
  import io
6
  import json
7
 
 
8
  from fastapi.testclient import TestClient
9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
10
 
11
  def _valid_options(**overrides: object) -> str:
12
  """Return a valid JSON options string with optional overrides."""
@@ -100,3 +115,157 @@ def test_audio_defaults_when_options_missing(client: TestClient) -> None:
100
  )
101
  # Should proceed to processing (200) or processing error — never 422
102
  assert response.status_code != 422
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5
  import io
6
  import json
7
 
8
+ import pytest
9
  from fastapi.testclient import TestClient
10
 
11
+ from app.routes.data_processing import _process_rate_store
12
+
13
+
14
+ @pytest.fixture(autouse=True)
15
+ def _reset_rate_limiter():
16
+ """The rate limiter's store is module-level and shared across every
17
+ test in this file (10 requests/60s per IP, and TestClient always uses
18
+ the same fake client IP) — without resetting it, tests that pass in
19
+ isolation start failing with 429 once enough tests run before them in
20
+ the same process."""
21
+ _process_rate_store.clear()
22
+ yield
23
+ _process_rate_store.clear()
24
+
25
 
26
  def _valid_options(**overrides: object) -> str:
27
  """Return a valid JSON options string with optional overrides."""
 
115
  )
116
  # Should proceed to processing (200) or processing error — never 422
117
  assert response.status_code != 422
118
+
119
+
120
+ def test_audio_rejects_oversized_file_with_413_not_500(client: TestClient) -> None:
121
+ """Regression test: the 413 raised inside the read loop's try block was
122
+ being caught by the bare `except Exception` below it (HTTPException IS
123
+ an Exception) and replaced with a misleading 500. A file over the 30MB
124
+ cap must surface as 413 file_too_large, not 500 internal_error."""
125
+ oversized = _fake_audio(b"\x00" * (31 * 1024 * 1024))
126
+ response = client.post(
127
+ "/api/process/audio",
128
+ data={"options": _valid_options()},
129
+ files={"file": ("big.wav", oversized, "audio/wav")},
130
+ )
131
+ assert response.status_code == 413
132
+ detail = response.json()["detail"]
133
+ assert detail["code"] == "file_too_large"
134
+
135
+
136
+ # ─────────────────────────── /api/process/audio/convert ───────────────────────────
137
+
138
+ def _valid_convert_options(**overrides: object) -> str:
139
+ defaults = {"targetFormat": "wav", "bitrateKbps": 192}
140
+ defaults.update(overrides)
141
+ return json.dumps(defaults)
142
+
143
+
144
+ def test_convert_valid_request_accepted(client: TestClient) -> None:
145
+ """Valid audio file + valid convert options should not return 422."""
146
+ response = client.post(
147
+ "/api/process/audio/convert",
148
+ data={"options": _valid_convert_options()},
149
+ files={"file": ("test.wav", _fake_audio(), "audio/wav")},
150
+ )
151
+ assert response.status_code != 422
152
+
153
+
154
+ def test_convert_rejects_non_audio(client: TestClient) -> None:
155
+ response = client.post(
156
+ "/api/process/audio/convert",
157
+ data={"options": _valid_convert_options()},
158
+ files={"file": ("test.txt", _fake_audio(), "text/plain")},
159
+ )
160
+ assert response.status_code == 400
161
+ detail = response.json()["detail"]
162
+ assert detail["code"] == "invalid_file_type"
163
+
164
+
165
+ def test_convert_rejects_invalid_target_format(client: TestClient) -> None:
166
+ """targetFormat outside the wav/mp3/flac/ogg enum should 422."""
167
+ response = client.post(
168
+ "/api/process/audio/convert",
169
+ data={"options": _valid_convert_options(targetFormat="exe")},
170
+ files={"file": ("test.wav", _fake_audio(), "audio/wav")},
171
+ )
172
+ assert response.status_code == 422
173
+ detail = response.json()["detail"]
174
+ assert detail["code"] == "invalid_options"
175
+
176
+
177
+ def test_convert_rejects_bitrate_out_of_range(client: TestClient) -> None:
178
+ response = client.post(
179
+ "/api/process/audio/convert",
180
+ data={"options": _valid_convert_options(bitrateKbps=999)},
181
+ files={"file": ("test.wav", _fake_audio(), "audio/wav")},
182
+ )
183
+ assert response.status_code == 422
184
+ detail = response.json()["detail"]
185
+ assert detail["code"] == "invalid_options"
186
+
187
+
188
+ def test_convert_defaults_when_options_missing(client: TestClient) -> None:
189
+ """Missing options should default to wav @ 192kbps, not 422."""
190
+ response = client.post(
191
+ "/api/process/audio/convert",
192
+ files={"file": ("test.wav", _fake_audio(), "audio/wav")},
193
+ )
194
+ assert response.status_code != 422
195
+
196
+
197
+ def test_convert_rejects_oversized_file_with_413_not_500(client: TestClient) -> None:
198
+ """Same regression as the /audio endpoint: 413 must not become 500."""
199
+ oversized = _fake_audio(b"\x00" * (31 * 1024 * 1024))
200
+ response = client.post(
201
+ "/api/process/audio/convert",
202
+ data={"options": _valid_convert_options()},
203
+ files={"file": ("big.wav", oversized, "audio/wav")},
204
+ )
205
+ assert response.status_code == 413
206
+ detail = response.json()["detail"]
207
+ assert detail["code"] == "file_too_large"
208
+
209
+
210
+ # ─────────────────────────── /api/process/audio/organize ───────────────────────────
211
+
212
+ def test_organize_valid_request_accepted(client: TestClient) -> None:
213
+ response = client.post(
214
+ "/api/process/audio/organize",
215
+ files={"file": ("test.wav", _fake_audio(), "audio/wav")},
216
+ )
217
+ # A silent all-zero WAV is a legitimate 400 (analyzer rejects silence) —
218
+ # never 422 (that's the "malformed request" status, not "bad content").
219
+ assert response.status_code != 422
220
+
221
+
222
+ def test_organize_rejects_non_audio(client: TestClient) -> None:
223
+ response = client.post(
224
+ "/api/process/audio/organize",
225
+ files={"file": ("test.txt", _fake_audio(), "text/plain")},
226
+ )
227
+ assert response.status_code == 400
228
+ detail = response.json()["detail"]
229
+ assert detail["code"] == "invalid_file_type"
230
+
231
+
232
+ def test_organize_rejects_missing_file(client: TestClient) -> None:
233
+ response = client.post("/api/process/audio/organize")
234
+ assert response.status_code == 422
235
+
236
+
237
+ def test_organize_rejects_oversized_file_with_413(client: TestClient) -> None:
238
+ oversized = _fake_audio(b"\x00" * (31 * 1024 * 1024))
239
+ response = client.post(
240
+ "/api/process/audio/organize",
241
+ files={"file": ("big.wav", oversized, "audio/wav")},
242
+ )
243
+ assert response.status_code == 413
244
+ detail = response.json()["detail"]
245
+ assert detail["code"] == "file_too_large"
246
+
247
+
248
+ def test_organize_returns_real_metadata_for_real_audio(client: TestClient) -> None:
249
+ """End-to-end with an actual sine wave — not a silent/zero fixture —
250
+ to prove the analyzer runs and returns genuine acoustic metadata."""
251
+ import io as _io
252
+ import numpy as _np
253
+ import soundfile as _sf
254
+
255
+ sr = 22050
256
+ t = _np.linspace(0, 2, sr * 2)
257
+ y = (0.4 * _np.sin(2 * _np.pi * 440 * t)).astype(_np.float32)
258
+ buf = _io.BytesIO()
259
+ _sf.write(buf, y, sr, format="WAV")
260
+ buf.seek(0)
261
+
262
+ response = client.post(
263
+ "/api/process/audio/organize",
264
+ files={"file": ("tone.wav", buf, "audio/wav")},
265
+ )
266
+ assert response.status_code == 200
267
+ body = response.json()
268
+ assert body["durationSec"] == pytest.approx(2.0, abs=0.1)
269
+ assert isinstance(body["tempoBpm"], (int, float))
270
+ assert "major" in body["key"] or "minor" in body["key"]
271
+ assert isinstance(body["tags"], list) and len(body["tags"]) > 0