omniAI / deploy /gpu.Dockerfile
hasimjaneef's picture
Sync release with verified GPU dependency build fix
cd7e317 verified
Raw History Blame Contribute Delete
1.69 kB
# Application source and immutable model-lock.json are copied together.
# Model download happens at runtime; no token or model weights in build layers.
FROM python:3.10.18-slim-bookworm
ENV PYTHONUNBUFFERED=1 PYTHONDONTWRITEBYTECODE=1 HF_HUB_DISABLE_TELEMETRY=1 \
MODEL_CACHE_DIR=/model-cache TOKENIZERS_PARALLELISM=false
RUN apt-get update && apt-get install -y --no-install-recommends ffmpeg libsndfile1 ca-certificates libgomp1 \
&& rm -rf /var/lib/apt/lists/*
WORKDIR /app
# Bootstrap a modern pinned installer from PyPI. The CUDA index is not a PyPI mirror.
RUN python -m pip install --no-cache-dir --only-binary=:all: --index-url https://pypi.org/simple pip==25.3
# Only CUDA wheels belong in this step; resolve their dependencies from PyPI below.
RUN python -m pip install --no-cache-dir --no-deps --only-binary=:all: \
--index-url https://download.pytorch.org/whl/cu128 torch==2.8.0+cu128 torchaudio==2.8.0+cu128
COPY requirements-gpu.txt /app/requirements-gpu.txt
RUN python -m pip install --no-cache-dir --index-url https://pypi.org/simple \
-r requirements-gpu.txt
# Decord's official py3 wheel incorrectly embeds a CPython 3.6 tag (upstream #356).
# Correct only that exact metadata, retain full pip check, and decode a CPU test clip.
COPY scripts/verify_gpu_install.py /app/scripts/verify_gpu_install.py
RUN python /app/scripts/verify_gpu_install.py
COPY gpu_service /app/gpu_service
COPY model-lock.json /app/model-lock.json
RUN useradd --uid 1000 --create-home app && mkdir -p /model-cache && chown app:app /model-cache
USER app
EXPOSE 8090
CMD ["uvicorn", "gpu_service.app:app", "--host", "0.0.0.0", "--port", "8090", "--no-access-log", "--workers", "1"]