precis / docker-compose.yml
compendious's picture
GPU speedup
4a6590d unverified
Raw History Blame Contribute Delete
1.9 kB
services:
api:
build:
context: .
dockerfile: Dockerfile
args:
# Baked into frontend at `npm run build` time (Vite inlines env).
# api still reads PRECIS_API_KEY at runtime via env_file / environment.
VITE_API_BASE_URL: "" # same-origin in Docker (frontend served by FastAPI)
PRECIS_API_KEY: ${PRECIS_API_KEY:-}
VITE_API_KEY: ${VITE_API_KEY:-${PRECIS_API_KEY:-}}
container_name: precis-api
ports:
- "8000:8000"
env_file:
- .env
environment:
# Host Ollama (Option A) — blazing fast GPU, no 2.5GB re-download.
# api inside Docker talks to host via host.docker.internal:11434.
# Host must have `ollama serve` running (see README). On Windows Docker Desktop this resolves automatically.
# If host Ollama was started with default 127.0.0.1 binding, set OLLAMA_HOST=0.0.0.0 on host before `ollama serve`.
PORT: ${PORT:-8000}
OLLAMA_BASE_URL: http://host.docker.internal:11434
DEFAULT_MODEL: ${DEFAULT_MODEL:-phi4-mini:latest}
AVAILABLE_MODELS: ${AVAILABLE_MODELS:-phi4-mini:latest}
PRECIS_ALLOWED_ORIGINS: ${PRECIS_ALLOWED_ORIGINS:-http://localhost:5173,http://localhost:8000,http://localhost:7860,https://*.hf.space,https://*.huggingface.co}
PRECIS_API_KEY: ${PRECIS_API_KEY:?PRECIS_API_KEY must be set in .env}
OLLAMA_KEEP_ALIVE: ${OLLAMA_KEEP_ALIVE:-30m}
extra_hosts:
- "host.docker.internal:host-gateway"
volumes:
# Mount .env read-only so config.py hot-reload isn't needed — container restarts pick up changes
- type: bind
source: ./.env
target: /app/.env
read_only: true
restart: unless-stopped
healthcheck:
test: ["CMD", "curl", "-fsS", "http://localhost:${PORT:-8000}/health"]
interval: 15s
timeout: 5s
retries: 5
start_period: 20s
volumes: {}