# Application source and immutable model-lock.json are copied together. # Model download happens at runtime; no token or model weights in build layers. FROM python:3.10.18-slim-bookworm ENV PYTHONUNBUFFERED=1 PYTHONDONTWRITEBYTECODE=1 HF_HUB_DISABLE_TELEMETRY=1 \ MODEL_CACHE_DIR=/model-cache TOKENIZERS_PARALLELISM=false RUN apt-get update && apt-get install -y --no-install-recommends ffmpeg libsndfile1 ca-certificates libgomp1 \ && rm -rf /var/lib/apt/lists/* WORKDIR /app # Bootstrap a modern pinned installer from PyPI. The CUDA index is not a PyPI mirror. RUN python -m pip install --no-cache-dir --only-binary=:all: --index-url https://pypi.org/simple pip==25.3 # Only CUDA wheels belong in this step; resolve their dependencies from PyPI below. RUN python -m pip install --no-cache-dir --no-deps --only-binary=:all: \ --index-url https://download.pytorch.org/whl/cu128 torch==2.8.0+cu128 torchaudio==2.8.0+cu128 COPY requirements-gpu.txt /app/requirements-gpu.txt RUN python -m pip install --no-cache-dir --index-url https://pypi.org/simple \ -r requirements-gpu.txt # Decord's official py3 wheel incorrectly embeds a CPython 3.6 tag (upstream #356). # Correct only that exact metadata, retain full pip check, and decode a CPU test clip. COPY scripts/verify_gpu_install.py /app/scripts/verify_gpu_install.py RUN python /app/scripts/verify_gpu_install.py COPY gpu_service /app/gpu_service COPY model-lock.json /app/model-lock.json RUN useradd --uid 1000 --create-home app && mkdir -p /model-cache && chown app:app /model-cache USER app EXPOSE 8090 CMD ["uvicorn", "gpu_service.app:app", "--host", "0.0.0.0", "--port", "8090", "--no-access-log", "--workers", "1"]