services: api: build: context: . dockerfile: Dockerfile args: # Baked into frontend at `npm run build` time (Vite inlines env). # api still reads PRECIS_API_KEY at runtime via env_file / environment. VITE_API_BASE_URL: "" # same-origin in Docker (frontend served by FastAPI) PRECIS_API_KEY: ${PRECIS_API_KEY:-} VITE_API_KEY: ${VITE_API_KEY:-${PRECIS_API_KEY:-}} container_name: precis-api ports: - "8000:8000" env_file: - .env environment: # Host Ollama (Option A) — blazing fast GPU, no 2.5GB re-download. # api inside Docker talks to host via host.docker.internal:11434. # Host must have `ollama serve` running (see README). On Windows Docker Desktop this resolves automatically. # If host Ollama was started with default 127.0.0.1 binding, set OLLAMA_HOST=0.0.0.0 on host before `ollama serve`. PORT: ${PORT:-8000} OLLAMA_BASE_URL: http://host.docker.internal:11434 DEFAULT_MODEL: ${DEFAULT_MODEL:-phi4-mini:latest} AVAILABLE_MODELS: ${AVAILABLE_MODELS:-phi4-mini:latest} PRECIS_ALLOWED_ORIGINS: ${PRECIS_ALLOWED_ORIGINS:-http://localhost:5173,http://localhost:8000,http://localhost:7860,https://*.hf.space,https://*.huggingface.co} PRECIS_API_KEY: ${PRECIS_API_KEY:?PRECIS_API_KEY must be set in .env} OLLAMA_KEEP_ALIVE: ${OLLAMA_KEEP_ALIVE:-30m} extra_hosts: - "host.docker.internal:host-gateway" volumes: # Mount .env read-only so config.py hot-reload isn't needed — container restarts pick up changes - type: bind source: ./.env target: /app/.env read_only: true restart: unless-stopped healthcheck: test: ["CMD", "curl", "-fsS", "http://localhost:${PORT:-8000}/health"] interval: 15s timeout: 5s retries: 5 start_period: 20s volumes: {}