Download scripts/bench-throughput.sh from ShoaibSSM/a100-qwen-api-runtime: direct link, hf CLI and curl.
- Browser
- Download file 659 Bytes
-
https://huggingface.co/ShoaibSSM/a100-qwen-api-runtime/resolve/main/scripts/bench-throughput.sh
- Command line
-
hf download hf://ShoaibSSM/a100-qwen-api-runtime/scripts/bench-throughput.sh
-
curl -L -o bench-throughput.sh https://huggingface.co/ShoaibSSM/a100-qwen-api-runtime/resolve/main/scripts/bench-throughput.sh
659 Bytes
| set -euo pipefail | |
| ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" | |
| MODEL="${1:-qwen38}" | |
| case "$MODEL" in | |
| qwen38) MODEL_PATH="$ROOT/models/Qwen3.8-27B-Q6_K.gguf" ;; | |
| qwen36) MODEL_PATH="$ROOT/models/Qwen3.6-35B-A3B-UD-Q5_K_M.gguf" ;; | |
| *) echo "Usage: $0 [qwen38|qwen36]" >&2; exit 2 ;; | |
| esac | |
| mkdir -p "$ROOT/results" | |
| export LD_LIBRARY_PATH="$ROOT/runtime/b11177:/opt/conda/lib:${LD_LIBRARY_PATH:-}" | |
| exec "$ROOT/runtime/b11177/llama-bench" \ | |
| --model "$MODEL_PATH" \ | |
| --n-gpu-layers 999 \ | |
| --flash-attn on \ | |
| --cache-type-k q8_0 \ | |
| --cache-type-v q8_0 \ | |
| --n-prompt 2048 \ | |
| --n-gen 256 \ | |
| --repetitions 3 \ | |
| --output json | |