instance-2 / Dockerfile
validops-east-3's picture
Deploy 0cf602f4c2fecaf8c206bad7e3f4cc7b7642e444
6ac032f verified
Raw History Blame Contribute Delete
3.8 kB
# ============================================================================
# ACTIVE: llama-server (Qwen3-0.6B Q8_0), OpenAI-compatible, port 7860
# ============================================================================
FROM ghcr.io/ggml-org/llama.cpp:server
# pre-build: bake the official Q8_0 GGUF into the image (610 MB)
RUN mkdir -p /models && \
curl -fL --retry 3 -o /models/Qwen3-0.6B-Q8_0.gguf \
"https://huggingface.co/Qwen/Qwen3-0.6B-GGUF/resolve/main/Qwen3-0.6B-Q8_0.gguf" && \
ls -lh /models
# key lives in .env (llama-server reads the LLAMA_API_KEY env var natively)
COPY .env /app/.env
EXPOSE 7860
HEALTHCHECK --interval=30s --timeout=5s --start-period=120s --retries=6 \
CMD curl -fsS http://127.0.0.1:7860/health || exit 1
# source .env unless the key is already set (Space secret / --env-file win)
ENTRYPOINT ["/bin/sh", "-c", "if [ -z \"$LLAMA_API_KEY\" ]; then set -a; . /app/.env; set +a; fi; exec /app/llama-server \"$@\"", "--"]
# -c is TOTAL across slots, so 32768 / 2 slots = 16384 ctx per request.
# Raise -c to 65536 if you want a full 32k per concurrent request.
CMD ["-m", "/models/Qwen3-0.6B-Q8_0.gguf", \
"--host", "0.0.0.0", \
"--port", "7860", \
"-c", "32768", \
"-np", "2", \
"-t", "2", \
"-tb", "4", \
"-ctk", "q8_0", \
"-ctv", "q8_0", \
"--jinja", \
"--alias", "Qwen3-0.6B"]
# ============================================================================
# COMMENTED OUT: docqa — Document Data Extraction API (Python + FastAPI)
#
# Kept for later, not built. To restore it, delete the FROM/ENTRYPOINT/CMD above
# and uncomment the block below. Its own README is in docqa/.
# ============================================================================
#
# FROM python:3.11-slim-bookworm
#
# # poppler-utils -> pdftotext/pdftoppm for the PDF text layer and rasterising
# # scans; tesseract-ocr -> OCR for images and scanned pages.
# RUN apt-get update && apt-get install -y --no-install-recommends \
# poppler-utils \
# tesseract-ocr \
# tesseract-ocr-eng \
# curl \
# libgomp1 \
# && rm -rf /var/lib/apt/lists/*
#
# WORKDIR /app
#
# COPY requirements.txt ./
# RUN pip install --no-cache-dir -r requirements.txt
#
# COPY docqa ./docqa
# ENV PYTHONPATH=/app/docqa
#
# # The model is pulled from the Hub at first start and cached under
# # /root/.cache/huggingface. Pre-fetching during build means a cold Space does
# # not serve requests until the weights are on disk.
# ARG MODEL_ID=impira/layoutlm-invoices
# RUN python -c "\
# from huggingface_hub import snapshot_download; \
# print(snapshot_download('${MODEL_ID}', \
# allow_patterns=['*.json','*.txt','*.bin','*.safetensors','*.model'])); \
# " || echo 'prefetch failed; model will download at first start'
#
# ENV DOCX_MODEL_ID=impira/layoutlm-invoices \
# DOCX_DEVICE=cpu \
# DOCX_TORCH_THREADS=4 \
# DOCX_BATCH_SIZE=8 \
# DOCX_MAX_PAGES=3 \
# DOCX_MAX_UPLOAD_MB=25 \
# DOCX_CONFIDENCE_THRESHOLD=0.5 \
# DOCX_REQUEST_TIMEOUT_S=60 \
# DOCX_LOG_JSON=false \
# HF_HOME=/app/.cache/huggingface \
# PORT=7860
#
# EXPOSE 7860
#
# # Warm the model first, then serve. The model loads before uvicorn binds, so a
# # passing health check means inference will actually work rather than the
# # process merely being alive.
# HEALTHCHECK --interval=30s --timeout=10s --start-period=600s --retries=6 \
# CMD curl -fsS http://127.0.0.1:7860/health || exit 1
#
# CMD ["sh", "-c", "python -c \"from docxextract.engine import get_engine; from docxextract.config import get_settings; s=get_settings(); get_engine(s).warmup(); print('model warm')\" && exec uvicorn docxextract.api:app --host 0.0.0.0 --port ${PORT} --workers 1 --timeout-keep-alive 65"]