Spaces:
Running
Running
Download app/routes/data_processing.py from Rthur2003/crowncode-backend: direct link, hf CLI and curl.
- Browser
- Download file 10.4 kB
-
https://huggingface.co/spaces/Rthur2003/crowncode-backend/resolve/main/app/routes/data_processing.py
- Command line
-
hf download hf://spaces/Rthur2003/crowncode-backend/app/routes/data_processing.py
-
curl -L -o data_processing.py https://huggingface.co/spaces/Rthur2003/crowncode-backend/resolve/main/app/routes/data_processing.py
10.4 kB
| """ | |
| Routes for data processing and manipulation (Audio/Image). | |
| """ | |
| import time | |
| import re | |
| from collections import defaultdict | |
| from fastapi import APIRouter, File, UploadFile, Form, HTTPException, Request, Depends, status | |
| from fastapi.responses import StreamingResponse | |
| from pydantic import ValidationError | |
| import json | |
| import logging | |
| from app.schemas import AudioAugmentationOptions, AudioConvertOptions, DatasetEntryMetadata | |
| from app.services.audio_processor import process_audio, convert_audio_format | |
| from app.services.dataset_organizer import analyze_for_organization | |
| from app.services.client_ip import client_ip as _client_ip | |
| router = APIRouter(prefix="/api/process", tags=["Data Processing"]) | |
| logger = logging.getLogger(__name__) | |
| # Rate limiter for heavy processing endpoints | |
| _process_rate_store: dict[str, list[float]] = defaultdict(list) | |
| _PROCESS_RATE_WINDOW = 60 | |
| _PROCESS_RATE_MAX = 10 | |
| _PROCESS_MAX_IPS = 10_000 | |
| async def _process_rate_limit(request: Request) -> None: | |
| client_ip = _client_ip(request) | |
| now = time.time() | |
| cutoff = now - _PROCESS_RATE_WINDOW | |
| if len(_process_rate_store) > _PROCESS_MAX_IPS: | |
| stale = [ip for ip, ts in _process_rate_store.items() if not ts or all(t <= cutoff for t in ts)] | |
| for ip in stale: | |
| del _process_rate_store[ip] | |
| hits = [t for t in _process_rate_store[client_ip] if t > cutoff] | |
| if len(hits) >= _PROCESS_RATE_MAX: | |
| raise HTTPException( | |
| status_code=status.HTTP_429_TOO_MANY_REQUESTS, | |
| detail={"code": "rate_limit_exceeded", "message": "Too many processing requests. Please wait."} | |
| ) | |
| hits.append(now) | |
| _process_rate_store[client_ip] = hits | |
| async def _read_capped(file: UploadFile, max_bytes: int) -> bytes: | |
| """Read an UploadFile in chunks, raising 413 if it exceeds max_bytes.""" | |
| chunks = [] | |
| total_read = 0 | |
| chunk_size = 1024 * 1024 # 1 MB chunks | |
| while True: | |
| chunk = await file.read(chunk_size) | |
| if not chunk: | |
| break | |
| total_read += len(chunk) | |
| if total_read > max_bytes: | |
| raise HTTPException(status_code=413, detail={"code": "file_too_large", "message": f"File too large. Maximum size is {max_bytes // (1024*1024)} MB."}) | |
| chunks.append(chunk) | |
| return b"".join(chunks) | |
| async def process_audio_endpoint( | |
| file: UploadFile = File(...), | |
| options: str = Form(default="{}"), | |
| mix_file: UploadFile | None = File(default=None), | |
| ): | |
| """ | |
| Process an audio file with the given augmentation options. | |
| Returns the processed WAV file. | |
| options is a JSON string; if missing or empty, defaults to all-off. | |
| mix_file: required when options.mixAudio is true — a second audio file | |
| to blend into the primary track. | |
| """ | |
| MAX_PAYLOAD_BYTES = 30 * 1024 * 1024 # 30 MB | |
| logger.info(f"Received audio processing request for file: {file.filename}") | |
| # Parse options JSON string into validated Pydantic model | |
| raw_options = options.strip() if options else "{}" | |
| if not raw_options: | |
| raw_options = "{}" | |
| try: | |
| parsed_options = AudioAugmentationOptions.model_validate_json(raw_options) | |
| except (ValidationError, json.JSONDecodeError) as e: | |
| raise HTTPException( | |
| status_code=422, | |
| detail={"code": "invalid_options", "message": f"Invalid options format: {e}"} | |
| ) | |
| if not file.content_type or not file.content_type.startswith("audio/"): | |
| raise HTTPException(status_code=400, detail={"code": "invalid_file_type", "message": "Invalid file type. Must be audio."}) | |
| if parsed_options.mix_audio and mix_file is None: | |
| raise HTTPException( | |
| status_code=400, | |
| detail={"code": "missing_mix_file", "message": "mixAudio is enabled but no second file was provided."} | |
| ) | |
| if mix_file is not None and (not mix_file.content_type or not mix_file.content_type.startswith("audio/")): | |
| raise HTTPException(status_code=400, detail={"code": "invalid_file_type", "message": "Second file for mixing must be audio."}) | |
| try: | |
| # Sanitize filename to prevent injection attacks | |
| safe_filename = re.sub(r'[^a-zA-Z0-9._-]', '_', file.filename or 'audio') | |
| if safe_filename.endswith('.wav'): | |
| safe_filename = safe_filename[:-4] | |
| content = await _read_capped(file, MAX_PAYLOAD_BYTES) | |
| mix_content = await _read_capped(mix_file, MAX_PAYLOAD_BYTES) if mix_file is not None else None | |
| # Process audio | |
| processed_audio = process_audio(content, parsed_options, mix_with_bytes=mix_content) | |
| # Return as downloadable file | |
| output_filename = f"processed_{safe_filename}.wav" | |
| return StreamingResponse( | |
| processed_audio, | |
| media_type="audio/wav", | |
| headers={"Content-Disposition": f"attachment; filename={output_filename}"} | |
| ) | |
| except ValueError as e: | |
| raise HTTPException(status_code=400, detail={"code": "validation_error", "message": str(e)}) | |
| except HTTPException: | |
| # Re-raise as-is (e.g. the 413 file_too_large above) — without this, | |
| # the bare `except Exception` below catches it too (HTTPException IS | |
| # an Exception) and replaces a correct 413 with a misleading 500. | |
| raise | |
| except Exception as e: | |
| logger.error(f"Unexpected error in audio processing: {e}", exc_info=True) | |
| raise HTTPException(status_code=500, detail={"code": "internal_error", "message": "Internal server error during audio processing"}) | |
| _CONVERT_MEDIA_TYPES = { | |
| "wav": "audio/wav", | |
| "mp3": "audio/mpeg", | |
| "flac": "audio/flac", | |
| "ogg": "audio/ogg", | |
| } | |
| async def convert_audio_endpoint( | |
| file: UploadFile = File(...), | |
| options: str = Form(default="{}") | |
| ): | |
| """ | |
| Convert an audio file to another format (wav/mp3/flac/ogg). | |
| Returns the converted file for download. | |
| options is a JSON string; if missing or empty, defaults to wav @ 192kbps. | |
| """ | |
| MAX_PAYLOAD_BYTES = 30 * 1024 * 1024 # 30 MB | |
| logger.info(f"Received audio conversion request for file: {file.filename}") | |
| raw_options = options.strip() if options else "{}" | |
| if not raw_options: | |
| raw_options = "{}" | |
| try: | |
| parsed_options = AudioConvertOptions.model_validate_json(raw_options) | |
| except (ValidationError, json.JSONDecodeError) as e: | |
| raise HTTPException( | |
| status_code=422, | |
| detail={"code": "invalid_options", "message": f"Invalid options format: {e}"} | |
| ) | |
| if not file.content_type or not file.content_type.startswith("audio/"): | |
| raise HTTPException(status_code=400, detail={"code": "invalid_file_type", "message": "Invalid file type. Must be audio."}) | |
| try: | |
| safe_filename = re.sub(r'[^a-zA-Z0-9._-]', '_', file.filename or 'audio') | |
| # Strip any existing extension so we don't end up with e.g. "song.mp3.flac" | |
| safe_filename = re.sub(r'\.[a-zA-Z0-9]{1,5}$', '', safe_filename) | |
| chunks = [] | |
| total_read = 0 | |
| chunk_size = 1024 * 1024 # 1 MB chunks | |
| while True: | |
| chunk = await file.read(chunk_size) | |
| if not chunk: | |
| break | |
| total_read += len(chunk) | |
| if total_read > MAX_PAYLOAD_BYTES: | |
| raise HTTPException(status_code=413, detail={"code": "file_too_large", "message": f"File too large. Maximum size is {MAX_PAYLOAD_BYTES // (1024*1024)} MB."}) | |
| chunks.append(chunk) | |
| content = b"".join(chunks) | |
| converted = convert_audio_format(content, parsed_options) | |
| target = parsed_options.target_format | |
| output_filename = f"{safe_filename}.{target}" | |
| return StreamingResponse( | |
| converted, | |
| media_type=_CONVERT_MEDIA_TYPES[target], | |
| headers={"Content-Disposition": f"attachment; filename={output_filename}"} | |
| ) | |
| except ValueError as e: | |
| raise HTTPException(status_code=400, detail={"code": "validation_error", "message": str(e)}) | |
| except HTTPException: | |
| # Re-raise as-is — see the matching comment in the /audio endpoint | |
| # above for why this is needed before the bare `except Exception`. | |
| raise | |
| except Exception as e: | |
| logger.error(f"Unexpected error in audio conversion: {e}", exc_info=True) | |
| raise HTTPException(status_code=500, detail={"code": "internal_error", "message": "Internal server error during audio conversion"}) | |
| async def organize_audio_endpoint(file: UploadFile = File(...)): | |
| """ | |
| Analyze an audio file and return real acoustic metadata (duration, | |
| tempo, key, loudness) plus auto-generated tags for dataset tagging | |
| and categorization. No transformation — read-only analysis. | |
| """ | |
| MAX_PAYLOAD_BYTES = 30 * 1024 * 1024 # 30 MB | |
| logger.info(f"Received audio organization request for file: {file.filename}") | |
| if not file.content_type or not file.content_type.startswith("audio/"): | |
| raise HTTPException(status_code=400, detail={"code": "invalid_file_type", "message": "Invalid file type. Must be audio."}) | |
| try: | |
| chunks = [] | |
| total_read = 0 | |
| chunk_size = 1024 * 1024 | |
| while True: | |
| chunk = await file.read(chunk_size) | |
| if not chunk: | |
| break | |
| total_read += len(chunk) | |
| if total_read > MAX_PAYLOAD_BYTES: | |
| raise HTTPException(status_code=413, detail={"code": "file_too_large", "message": f"File too large. Maximum size is {MAX_PAYLOAD_BYTES // (1024*1024)} MB."}) | |
| chunks.append(chunk) | |
| content = b"".join(chunks) | |
| metadata = analyze_for_organization(content) | |
| return DatasetEntryMetadata(**metadata.to_dict()) | |
| except ValueError as e: | |
| raise HTTPException(status_code=400, detail={"code": "validation_error", "message": str(e)}) | |
| except HTTPException: | |
| raise | |
| except Exception as e: | |
| logger.error(f"Unexpected error in audio organization: {e}", exc_info=True) | |
| raise HTTPException(status_code=500, detail={"code": "internal_error", "message": "Internal server error during audio organization"}) | |