FROM nvidia/cuda:12.1.1-cudnn8-runtime-ubuntu22.04 ENV DEBIAN_FRONTEND=noninteractive ENV PYTHONUNBUFFERED=1 RUN apt-get update && apt-get install -y --no-install-recommends \ python3 python3-pip python3-dev \ git ffmpeg libsndfile1 wget \ build-essential \ && rm -rf /var/lib/apt/lists/* RUN ln -sf /usr/bin/python3 /usr/bin/python WORKDIR /app # Install PyTorch with CUDA 12.1 RUN pip install --no-cache-dir \ torch==2.5.1 torchaudio==2.5.1 \ --index-url https://download.pytorch.org/whl/cu121 COPY requirements.txt . RUN pip install --no-cache-dir -r requirements.txt # Clone SafeEar repository (for model code imports) RUN git clone --depth 1 https://github.com/LetterLiGo/SafeEar.git /app/safeear_repo # Install the fairseq fork with C extensions PATCHED OUT # (same proven patch used by ShiftySpeech and Nes2Net -- # C extensions are not needed for checkpoint loading) WORKDIR /app/safeear_repo/fairseq_ours RUN sed -i 's/ext_modules=extensions/ext_modules=[]/' setup.py && \ pip install --no-cache-dir --no-deps -e . WORKDIR /app # Download model weights from HuggingFace RUN mkdir -p /app/weights && \ wget -q -O /app/weights/SpeechTokenizer.pt \ "https://huggingface.co/TEC2004/SafeEar-ASV19-spoof-detection/resolve/main/SpeechTokenizer.pt" && \ wget -q -O /app/weights/model.ckpt \ "https://huggingface.co/TEC2004/SafeEar-ASV19-spoof-detection/resolve/main/model.ckpt" COPY api.py . EXPOSE 8002 RUN adduser --disabled-password --gecos '' appuser USER appuser CMD ["python", "api.py"]