Download scripts/bench.py from ShoaibSSM/a100-qwen-api-runtime: direct link, hf CLI and curl.
- Browser
- Download file 3.07 kB
-
https://huggingface.co/ShoaibSSM/a100-qwen-api-runtime/resolve/main/scripts/bench.py
- Command line
-
hf download hf://ShoaibSSM/a100-qwen-api-runtime/scripts/bench.py
-
curl -L -o bench.py https://huggingface.co/ShoaibSSM/a100-qwen-api-runtime/resolve/main/scripts/bench.py
3.07 kB
| #!/usr/bin/env python3 | |
| """Small, repeatable API timing check; no benchmark accuracy claims.""" | |
| import argparse | |
| import json | |
| import os | |
| import time | |
| from pathlib import Path | |
| import requests | |
| ROOT = Path(__file__).resolve().parent.parent | |
| PROMPTS = { | |
| "reasoning": "A lab has three boxes labelled A, B, and C. Exactly one label is true. A says 'the sample is in B'. B says 'the sample is not in B'. C says 'the sample is not in A'. Which box holds the sample? Show a short check of all cases.", | |
| "coding": "Find the bug in this Python function and give a corrected implementation plus two tests:\n\ndef merge_intervals(xs):\n xs = sorted(xs)\n out = []\n for a, b in xs:\n if out and a < out[-1][1]:\n out[-1][1] = b\n else:\n out.append([a, b])\n return out\n", | |
| "analysis": "A trial reports 42/100 successes in treatment and 35/100 in control. Compute the absolute and relative differences, explain the uncertainty without claiming significance from these figures alone, and list the additional information needed for a sound conclusion.", | |
| } | |
| def main(): | |
| p = argparse.ArgumentParser() | |
| p.add_argument("model", choices=["qwen38", "qwen36"]) | |
| p.add_argument("--base-url", default="http://127.0.0.1:8080") | |
| p.add_argument("--max-tokens", type=int, default=4096) | |
| args = p.parse_args() | |
| key = os.environ.get("LLM_API_KEY") or (ROOT / "secrets/api_keys.txt").read_text().strip() | |
| records = [] | |
| for name, prompt in PROMPTS.items(): | |
| started = time.perf_counter() | |
| response = requests.post( | |
| args.base_url + "/v1/chat/completions", | |
| headers={"Authorization": f"Bearer {key}"}, | |
| json={ | |
| "model": args.model, | |
| "messages": [{"role": "user", "content": prompt}], | |
| "temperature": 1.0, | |
| "top_p": 0.95, | |
| "max_tokens": args.max_tokens, | |
| "reasoning_effort": "medium", | |
| "stream": False, | |
| }, | |
| timeout=900, | |
| ) | |
| elapsed = time.perf_counter() - started | |
| response.raise_for_status() | |
| data = response.json() | |
| usage = data.get("usage", {}) | |
| record = { | |
| "task": name, | |
| "elapsed_s": round(elapsed, 3), | |
| "prompt_tokens": usage.get("prompt_tokens"), | |
| "completion_tokens": usage.get("completion_tokens"), | |
| "output_tokens_per_s": round(usage.get("completion_tokens", 0) / elapsed, 2), | |
| "finish_reason": data["choices"][0].get("finish_reason"), | |
| "message": data["choices"][0]["message"], | |
| } | |
| records.append(record) | |
| print(f"{name}: {record['elapsed_s']}s, {record['completion_tokens']} output tokens, {record['output_tokens_per_s']} tok/s") | |
| outdir = ROOT / "results" | |
| outdir.mkdir(exist_ok=True) | |
| output = outdir / f"{args.model}-{time.strftime('%Y%m%d-%H%M%S', time.gmtime())}.json" | |
| output.write_text(json.dumps(records, indent=2)) | |
| print(output) | |
| if __name__ == "__main__": | |
| main() | |