FROM pytorch/pytorch:2.6.0-cuda12.4-cudnn9-runtime ENV DEBIAN_FRONTEND=noninteractive \ PYTHONUNBUFFERED=1 \ PYTHONDONTWRITEBYTECODE=1 \ PERSIST_ROOT=/app/data-v7 \ HF_HOME=/app/data-v7/cache/huggingface \ TRANSFORMERS_CACHE=/app/data-v7/cache/huggingface \ HF_HUB_ENABLE_HF_TRANSFER=0 \ TOKENIZERS_PARALLELISM=false \ PORT=7860 RUN apt-get update && apt-get install -y --no-install-recommends \ build-essential ca-certificates cmake curl git wget && \ rm -rf /var/lib/apt/lists/* WORKDIR /app COPY requirements.txt . RUN python -m pip install --no-cache-dir --upgrade pip && \ python -m pip install --no-cache-dir -r requirements.txt # Cache the public base model while the Space image is building. Hugging Face # does not bill requested GPU hardware during the build stage. RUN python -c "from huggingface_hub import snapshot_download; snapshot_download('Qwen/Qwen3-1.7B')" # Pinned official llama.cpp revision for reproducible GGUF conversion. RUN git clone https://github.com/ggml-org/llama.cpp.git /opt/llama.cpp && \ cd /opt/llama.cpp && \ git checkout 4762ad7316dcdec20016ab5985fb46a27902204d && \ cmake -B build -DGGML_CUDA=OFF -DLLAMA_CURL=OFF -DGGML_NATIVE=OFF \ -DGGML_AVX512=OFF -DGGML_AVX512_VBMI=OFF -DGGML_AVX512_VNNI=OFF \ -DGGML_AVX512_BF16=OFF -DCMAKE_BUILD_TYPE=Release && \ cmake --build build --config Release -j2 --target llama-quantize llama-server COPY . /app EXPOSE 7860 CMD ["python", "app.py"]