Download entrypoint.sh from ramedde/gemma4cpu: direct link, hf CLI and curl.
- Browser
- Download file 1.17 kB
-
https://huggingface.co/spaces/ramedde/gemma4cpu/resolve/main/entrypoint.sh
- Command line
-
hf download hf://spaces/ramedde/gemma4cpu/entrypoint.sh
-
curl -L -o entrypoint.sh https://huggingface.co/spaces/ramedde/gemma4cpu/resolve/main/entrypoint.sh
1.17 kB
| set -eu | |
| MODEL_SPEC="${MODEL_SPEC:-unsloth/gemma-4-E2B-it-qat-GGUF:UD-Q4_K_XL}" | |
| HOST="${HOST:-0.0.0.0}" | |
| PORT="${PORT:-7860}" | |
| CTX_SIZE="${CTX_SIZE:-8192}" | |
| THREADS="${THREADS:-2}" | |
| PARALLEL="${PARALLEL:-1}" | |
| REASONING_MODE="${REASONING_MODE:-off}" | |
| CACHE_TYPE_K="${CACHE_TYPE_K:-q8_0}" | |
| CACHE_TYPE_V="${CACHE_TYPE_V:-q8_0}" | |
| FLASH_ATTN="${FLASH_ATTN:-on}" | |
| SPEC_TYPE="${SPEC_TYPE:-draft-mtp}" | |
| # NOTE: the b8840 CPU build does not support draft-mtp (needs a newer | |
| # llama.cpp build than this image ships). ngram-cache is a self-speculation | |
| # mode that needs no external draft model: it speculates from n-grams already | |
| # seen in context/output, verified by the target model, so it's zero-cost on | |
| # accuracy. Set SPEC_TYPE=none in Space secrets to disable entirely. | |
| SPEC_FLAGS="--spec-type $SPEC_TYPE" | |
| exec /app/llama-server \ | |
| -hf "$MODEL_SPEC" \ | |
| --host "$HOST" \ | |
| --port "$PORT" \ | |
| --ctx-size "$CTX_SIZE" \ | |
| --threads "$THREADS" \ | |
| --threads-batch "$THREADS" \ | |
| --parallel "$PARALLEL" \ | |
| --cache-type-k "$CACHE_TYPE_K" \ | |
| --cache-type-v "$CACHE_TYPE_V" \ | |
| --reasoning "$REASONING_MODE" \ | |
| -fa "$FLASH_ATTN" \ | |
| --spec-type draft-mtp \ | |
| --spec-draft-n-max 4 | |