File size: 4,799 Bytes
0f0e3d8
 
 
ecd6ba5
 
0f0e3d8
ecd6ba5
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
0f0e3d8
ecd6ba5
 
0f0e3d8
ecd6ba5
 
 
0f0e3d8
ecd6ba5
0f0e3d8
 
 
 
ecd6ba5
 
 
 
0f0e3d8
ecd6ba5
0f0e3d8
 
35bae08
7c6bd51
 
ecd6ba5
 
 
 
 
7c6bd51
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
ecd6ba5
 
 
 
 
0f0e3d8
ecd6ba5
 
 
6ed924b
ecd6ba5
6ed924b
0f0e3d8
 
ecd6ba5
 
 
 
0f0e3d8
 
 
 
 
ecd6ba5
 
 
 
 
 
0f0e3d8
 
ecd6ba5
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
# syntax=docker/dockerfile:1

# ---------------------------------------------------------------------------
# GPU image: the official prebuilt llama.cpp CUDA server plus this repo's
# FastAPI gateway. Nothing is compiled here.
#
# Earlier revisions built llama.cpp from source and cherry-picked the "STQ1_0"
# quant kernel from PR #22836, because the Hy-MT2 model card warns its GGUFs
# depend on that kernel. Reading the actual tensor tables shows that is not
# true for the quants this project uses: Hy-MT2-7B-Q4_K_M and
# Hy-MT2-1.8B-Q4_K_M are both 354 tensors of plain Q4_K / Q6_K / F32 on the
# `hunyuan-dense` architecture, all of which stock llama.cpp already supports.
# (The STQ warning applies to the separate 2-bit / 1.25-bit repos.) Dropping
# the source build removes 10-60 minutes of compiling per deploy and with it
# every build failure this project has hit - OOM-thrashed builders, missing
# git identity, a COPY of a file that wasn't in the context.
#
# The tag is pinned rather than floating on :server-cuda so a redeploy cannot
# silently land on a different llama.cpp. b10398 postdates `hunyuan-dense`
# support, and llama.cpp's default CUDA architecture list includes 75-virtual,
# which covers the T4 (sm_75) this targets - as PTX, so the driver JIT-compiles
# it once on first model load. Bump with:
#   --build-arg LLAMA_CPP_IMAGE=ghcr.io/ggml-org/llama.cpp:server-cuda-bNNNNN
# ---------------------------------------------------------------------------
ARG LLAMA_CPP_IMAGE=ghcr.io/ggml-org/llama.cpp:server-cuda-b10398
FROM ${LLAMA_CPP_IMAGE}

# The base image carries llama-server and its shared libraries in /app and
# little else - notably no Python. Its own layer so it stays cached whatever
# happens to the application below.
RUN apt-get update && apt-get install -y --no-install-recommends \
      python3 python3-pip \
    && rm -rf /var/lib/apt/lists/*

WORKDIR /app

# Dependencies before source, so editing app/ or test-ui/ rebuilds only the
# last few (instant) layers instead of reinstalling everything.
# --break-system-packages: the CUDA base is Ubuntu, whose system Python is
# PEP 668 "externally managed"; llama.cpp's own images install the same way.
COPY requirements.txt .
RUN pip install --no-cache-dir --break-system-packages -r requirements.txt

COPY app ./app
COPY test-ui ./test-ui
COPY entrypoint.sh ./

# Written inline rather than COPYed so the build cannot fail with "not found"
# just because this one extra file didn't make it into the build context -
# which is exactly what broke a previous deploy. Must stay identical to
# chat-template-rosetta.jinja in the repo; see the comment there for why the
# template exists at all.
RUN cat <<'CHAT_TEMPLATE' > /app/chat-template-rosetta.jinja
{{- bos_token -}}
{%- for m in messages -%}
{%- if m['role'] == 'system' -%}
{{- '<start_of_turn>instruction\n' -}}{{- m['content'] | trim -}}{{- '<end_of_turn>\n' -}}
{%- endif -%}
{%- endfor -%}
{%- for m in messages -%}
{%- if m['role'] == 'user' -%}
{{- '<start_of_turn>source\n' -}}{{- m['content'] | trim -}}{{- '<end_of_turn>\n' -}}
{%- endif -%}
{%- endfor -%}
{%- if add_generation_prompt -%}
{{- '<start_of_turn>translation\n' -}}
{%- endif -%}
CHAT_TEMPLATE

# Deliberately not `llama-server --version` as a smoke test any more: the
# binary now comes from an image that is already known-good, while the build
# worker has no GPU, so touching the CUDA backend at build time risks failing
# a build that would have run fine.
RUN chmod +x /app/entrypoint.sh && test -x /app/llama-server

ENV MODEL_REPO=tencent/Hy-MT2-7B-GGUF \
    MODEL_FILE=Hy-MT2-7B-Q4_K_M.gguf \
    MODEL_DIR=/app/models \
    GLOSSARY_FILE="" \
    DEFAULT_STYLE="" \
    GROUP_SIZE=10 \
    LLAMA_SERVER_HOST=127.0.0.1 \
    LLAMA_SERVER_PORT=8080 \
    GPU_LAYERS=all \
    THREADS=4 \
    PARALLEL_SLOTS=16 \
    CTX_SIZE=32768 \
    MAX_TOKENS=512 \
    MAX_BATCH_SIZE=200 \
    API_KEY="" \
    PORT=7860

# The base image sets LLAMA_ARG_HOST=0.0.0.0, which llama-server reads as a
# fallback for --host. entrypoint.sh passes --host explicitly, but overriding
# the variable too means the inference server stays loopback-only even if that
# flag is ever dropped: only the gateway on $PORT should be reachable.
ENV LLAMA_ARG_HOST=127.0.0.1

EXPOSE 7860

# The base image's ENTRYPOINT is /app/llama-server. Left in place, a CMD here
# would be appended to it and Docker would run `llama-server ./entrypoint.sh`.
ENTRYPOINT ["/app/entrypoint.sh"]

# Long start period: a cold start downloads several GB of GGUF before the
# gateway answers. /health returns 200 while the model is still loading, so
# this only fails if the process is genuinely dead.
HEALTHCHECK --start-period=15m --interval=1m --timeout=10s \
    CMD curl -fsS http://localhost:7860/health || exit 1