File size: 3,796 Bytes
6ac032f
 
 
 
 
 
 
 
 
 
 
 
 
 
16126fc
 
 
6ac032f
16126fc
 
6ac032f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
# ============================================================================
# ACTIVE: llama-server (Qwen3-0.6B Q8_0), OpenAI-compatible, port 7860
# ============================================================================

FROM ghcr.io/ggml-org/llama.cpp:server

# pre-build: bake the official Q8_0 GGUF into the image (610 MB)
RUN mkdir -p /models && \
    curl -fL --retry 3 -o /models/Qwen3-0.6B-Q8_0.gguf \
      "https://huggingface.co/Qwen/Qwen3-0.6B-GGUF/resolve/main/Qwen3-0.6B-Q8_0.gguf" && \
    ls -lh /models

# key lives in .env (llama-server reads the LLAMA_API_KEY env var natively)
COPY .env /app/.env

EXPOSE 7860

HEALTHCHECK --interval=30s --timeout=5s --start-period=120s --retries=6 \
  CMD curl -fsS http://127.0.0.1:7860/health || exit 1

# source .env unless the key is already set (Space secret / --env-file win)
ENTRYPOINT ["/bin/sh", "-c", "if [ -z \"$LLAMA_API_KEY\" ]; then set -a; . /app/.env; set +a; fi; exec /app/llama-server \"$@\"", "--"]

# -c is TOTAL across slots, so 32768 / 2 slots = 16384 ctx per request.
# Raise -c to 65536 if you want a full 32k per concurrent request.
CMD ["-m", "/models/Qwen3-0.6B-Q8_0.gguf", \
     "--host", "0.0.0.0", \
     "--port", "7860", \
     "-c", "32768", \
     "-np", "2", \
     "-t", "2", \
     "-tb", "4", \
     "-ctk", "q8_0", \
     "-ctv", "q8_0", \
     "--jinja", \
     "--alias", "Qwen3-0.6B"]

# ============================================================================
# COMMENTED OUT: docqa — Document Data Extraction API (Python + FastAPI)
#
# Kept for later, not built. To restore it, delete the FROM/ENTRYPOINT/CMD above
# and uncomment the block below. Its own README is in docqa/.
# ============================================================================
#
# FROM python:3.11-slim-bookworm
#
# # poppler-utils -> pdftotext/pdftoppm for the PDF text layer and rasterising
# # scans; tesseract-ocr -> OCR for images and scanned pages.
# RUN apt-get update && apt-get install -y --no-install-recommends \
#         poppler-utils \
#         tesseract-ocr \
#         tesseract-ocr-eng \
#         curl \
#         libgomp1 \
#     && rm -rf /var/lib/apt/lists/*
#
# WORKDIR /app
#
# COPY requirements.txt ./
# RUN pip install --no-cache-dir -r requirements.txt
#
# COPY docqa ./docqa
# ENV PYTHONPATH=/app/docqa
#
# # The model is pulled from the Hub at first start and cached under
# # /root/.cache/huggingface. Pre-fetching during build means a cold Space does
# # not serve requests until the weights are on disk.
# ARG MODEL_ID=impira/layoutlm-invoices
# RUN python -c "\
# from huggingface_hub import snapshot_download; \
# print(snapshot_download('${MODEL_ID}', \
#     allow_patterns=['*.json','*.txt','*.bin','*.safetensors','*.model'])); \
# " || echo 'prefetch failed; model will download at first start'
#
# ENV DOCX_MODEL_ID=impira/layoutlm-invoices \
#     DOCX_DEVICE=cpu \
#     DOCX_TORCH_THREADS=4 \
#     DOCX_BATCH_SIZE=8 \
#     DOCX_MAX_PAGES=3 \
#     DOCX_MAX_UPLOAD_MB=25 \
#     DOCX_CONFIDENCE_THRESHOLD=0.5 \
#     DOCX_REQUEST_TIMEOUT_S=60 \
#     DOCX_LOG_JSON=false \
#     HF_HOME=/app/.cache/huggingface \
#     PORT=7860
#
# EXPOSE 7860
#
# # Warm the model first, then serve. The model loads before uvicorn binds, so a
# # passing health check means inference will actually work rather than the
# # process merely being alive.
# HEALTHCHECK --interval=30s --timeout=10s --start-period=600s --retries=6 \
#   CMD curl -fsS http://127.0.0.1:7860/health || exit 1
#
# CMD ["sh", "-c", "python -c \"from docxextract.engine import get_engine; from docxextract.config import get_settings; s=get_settings(); get_engine(s).warmup(); print('model warm')\" && exec uvicorn docxextract.api:app --host 0.0.0.0 --port ${PORT} --workers 1 --timeout-keep-alive 65"]