ShoaibSSM's picture
Add pinned A100 Qwen API runtime and benchmark scripts
42b02cd verified
Raw History Blame Contribute Delete
3.3 kB
#!/usr/bin/env bash
set -euo pipefail
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
BUILD=b11177
RELEASE="https://github.com/ggml-org/llama.cpp/releases/download/${BUILD}"
mkdir -p "$ROOT/runtime" "$ROOT/models" "$ROOT/logs" "$ROOT/secrets" "$ROOT/results"
download() {
local url="$1" dest="$2" expected="$3"
if [[ -f "$dest" ]] && [[ "$(stat -c %s "$dest")" == "$expected" ]]; then
printf 'Already complete: %s\n' "$dest"
return
fi
curl --http1.1 -fL --retry 8 --retry-all-errors -C - "$url" -o "$dest"
[[ "$(stat -c %s "$dest")" == "$expected" ]] || { echo "Size mismatch: $dest" >&2; exit 1; }
}
download "$RELEASE/llama-${BUILD}-bin-ubuntu-cuda-12.8-x64.tar.gz" \
"$ROOT/runtime/llama-${BUILD}-bin-ubuntu-cuda-12.8-x64.tar.gz" 169512356
download "$RELEASE/cudart-llama-${BUILD}-bin-ubuntu-cuda-12.8-x64.tar.gz" \
"$ROOT/runtime/cudart-llama-${BUILD}-bin-ubuntu-cuda-12.8-x64.tar.gz" 594377693
printf '%s %s\n' \
7f584b4911be091788468412022dde72c8337af8db880d5f748e87563b25a82a "$ROOT/runtime/llama-${BUILD}-bin-ubuntu-cuda-12.8-x64.tar.gz" \
cea160366caea83923d76a676fc8591033d92138f35250cd3dc8b11f2edd57d8 "$ROOT/runtime/cudart-llama-${BUILD}-bin-ubuntu-cuda-12.8-x64.tar.gz" \
| sha256sum --check --status
mkdir -p "$ROOT/runtime/${BUILD}"
tar -xzf "$ROOT/runtime/llama-${BUILD}-bin-ubuntu-cuda-12.8-x64.tar.gz" -C "$ROOT/runtime/${BUILD}" --strip-components=1
tar -xzf "$ROOT/runtime/cudart-llama-${BUILD}-bin-ubuntu-cuda-12.8-x64.tar.gz" -C "$ROOT/runtime/${BUILD}" --strip-components=1
download_model() {
local repo="$1" file="$2" revision="$3" expected="$4"
if [[ -f "$ROOT/models/$file" ]] && [[ "$(stat -c %s "$ROOT/models/$file")" == "$expected" ]]; then
printf 'Already complete: %s\n' "$ROOT/models/$file"
return
fi
if [[ ! -x "$ROOT/.venv/bin/python" ]]; then
python -m venv "$ROOT/.venv"
fi
"$ROOT/.venv/bin/python" -m pip install -q 'huggingface_hub[hf_xet]==1.33.0'
HF_XET_HIGH_PERFORMANCE=1 "$ROOT/.venv/bin/python" - "$repo" "$file" "$revision" "$ROOT/models" <<'PY'
import sys
from huggingface_hub import hf_hub_download
repo, filename, revision, local_dir = sys.argv[1:]
print(hf_hub_download(repo_id=repo, filename=filename, revision=revision, local_dir=local_dir))
PY
[[ "$(stat -c %s "$ROOT/models/$file")" == "$expected" ]] || { echo "Size mismatch: $file" >&2; exit 1; }
}
download_model '6block/Qwen3.8-27B-GGUF' 'Qwen3.8-27B-Q6_K.gguf' \
5312b0f1c3705b45a48a18f051b4dc482627c17f 22563871392
download_model 'unsloth/Qwen3.6-35B-A3B-GGUF' 'Qwen3.6-35B-A3B-UD-Q5_K_M.gguf' \
a483e9e6cbd595906af30beda3187c2663a1118c 26456194016
if [[ "${WITH_MTP:-0}" == 1 ]]; then
download_model '6block/Qwen3.8-27B-GGUF' 'mtp-Qwen3.8-27B-Q8_0.gguf' \
5312b0f1c3705b45a48a18f051b4dc482627c17f 3164006816
printf '%s %s\n' e871427e49333b6eb05bef1745c811f4259455af9d67e5b577e0ef5fe9a8c1e5 \
"$ROOT/models/mtp-Qwen3.8-27B-Q8_0.gguf" | sha256sum --check --status
fi
printf '%s %s\n' \
87a379aa1ec5d06b740c73ab5610d3925de0e9d7bc99c933447f27eb6a8f2bcc "$ROOT/models/Qwen3.8-27B-Q6_K.gguf" \
c13ce26253ea334df472bd8fbd2d6da66d8a41195c17f6fcbf44c4d20ece0932 "$ROOT/models/Qwen3.6-35B-A3B-UD-Q5_K_M.gguf" \
| sha256sum --check --status
printf 'Installed llama.cpp %s and both models.\n' "$BUILD"