Download tools/benchmark_prefix_cache.py from Sariel00/Ling-3.0-tiny-RKNN: direct link, hf CLI and curl.
- Browser
- Download file 2.17 kB
-
https://huggingface.co/Sariel00/Ling-3.0-tiny-RKNN/resolve/main/tools/benchmark_prefix_cache.py
- Command line
-
hf download hf://Sariel00/Ling-3.0-tiny-RKNN/tools/benchmark_prefix_cache.py
-
curl -L -o benchmark_prefix_cache.py https://huggingface.co/Sariel00/Ling-3.0-tiny-RKNN/resolve/main/tools/benchmark_prefix_cache.py
2.17 kB
| #!/usr/bin/env python3 | |
| """Measure an appended conversation against a cold evaluation of identical messages.""" | |
| import argparse | |
| import json | |
| from pathlib import Path | |
| import urllib.request | |
| p = argparse.ArgumentParser(description=__doc__) | |
| p.add_argument("--url", required=True) | |
| p.add_argument("--prompt-tokens", type=int, default=4096) | |
| p.add_argument("--output", type=Path, required=True) | |
| a = p.parse_args() | |
| client = urllib.request.build_opener(urllib.request.ProxyHandler({})) | |
| result = {} | |
| def call(path, body=None): | |
| req = urllib.request.Request(a.url.rstrip("/") + path, | |
| None if body is None else json.dumps(body).encode(), {"Content-Type": "application/json"}) | |
| with client.open(req, timeout=3600) as response: | |
| return json.load(response) | |
| def infer(name, messages, cache): | |
| r = call("/v1/chat/completions", {"model": "mindnano-ling3-tiny", "messages": messages, | |
| "max_tokens": 16, "temperature": 0, "cache_prompt": cache, "user": "prefix-benchmark"}) | |
| result[name] = r | |
| a.output.parent.mkdir(parents=True, exist_ok=True) | |
| a.output.write_text(json.dumps(result, ensure_ascii=False, indent=2) + "\n") | |
| print(json.dumps({"case": name, "metrics": r["mindnano_metrics"]}), flush=True) | |
| return r | |
| call("/v1/cache/clear", {}) | |
| result["health"] = call("/health") | |
| msgs = [{"role": "user", "content": "1" * (a.prompt_tokens - 21)}] | |
| seed = infer("seed", msgs, True) | |
| assert seed["usage"]["prompt_tokens"] == a.prompt_tokens | |
| msgs += [{"role": "assistant", "content": seed["choices"][0]["message"]["content"]}, | |
| {"role": "user", "content": "请简短总结。"}] | |
| hit = infer("prefix_hit", msgs, True) | |
| assert hit["mindnano_metrics"]["cached_tokens"] == a.prompt_tokens // 128 * 128 | |
| exact = infer("exact_hit", msgs, True) | |
| assert exact["mindnano_metrics"]["prompt_evaluated_tokens"] == 0 | |
| cold = infer("cold", msgs, False) | |
| assert hit["choices"] == exact["choices"] == cold["choices"] | |
| for key in ("prompt_tokens", "completion_tokens", "total_tokens"): | |
| assert hit["usage"][key] == exact["usage"][key] == cold["usage"][key] | |
| result["passed"] = True | |
| a.output.write_text(json.dumps(result, ensure_ascii=False, indent=2) + "\n") | |