Download tests/output_budget_api_test.py from Sariel00/Ling-3.0-tiny-RKNN: direct link, hf CLI and curl.
- Browser
- Download file 2.98 kB
-
https://huggingface.co/Sariel00/Ling-3.0-tiny-RKNN/resolve/main/tests/output_budget_api_test.py
- Command line
-
hf download hf://Sariel00/Ling-3.0-tiny-RKNN/tests/output_budget_api_test.py
-
curl -L -o output_budget_api_test.py https://huggingface.co/Sariel00/Ling-3.0-tiny-RKNN/resolve/main/tests/output_budget_api_test.py
2.98 kB
| #!/usr/bin/env python3 | |
| """Validate real HTTP parsing and generation at allocated KV boundaries.""" | |
| import argparse | |
| import json | |
| import urllib.error | |
| import urllib.request | |
| p = argparse.ArgumentParser(description=__doc__) | |
| p.add_argument("--url", default="http://127.0.0.1:19095") | |
| a = p.parse_args() | |
| client = urllib.request.build_opener(urllib.request.ProxyHandler({})) | |
| def call(path, body=None, padding=0): | |
| payload = None if body is None else json.dumps(body).encode() + b" " * padding | |
| req = urllib.request.Request(a.url + path, payload, {"Content-Type": "application/json"}) | |
| with client.open(req, timeout=600) as response: | |
| return json.load(response) | |
| def request(text, **extra): | |
| return {"model": "mindnano-ling3-tiny", "temperature": 0, | |
| "messages": [{"role": "user", "content": text}], **extra} | |
| capacity = call("/health")["context_length"] | |
| results = {} | |
| for budget in (4096, 2**64 - 1): | |
| r = call("/v1/chat/completions", request("1" * 107, max_tokens=budget, stop="1")) | |
| assert r["choices"][0]["finish_reason"] == "stop", r | |
| assert r["mindnano_metrics"]["effective_max_tokens"] == min(budget, capacity - 128) | |
| results[str(budget)] = r["mindnano_metrics"] | |
| r = call("/v1/chat/completions", request("你好", max_tokens=1), padding=1024*1024+1) | |
| assert r["usage"]["completion_tokens"] == 1 | |
| results["body_over_1mib"] = r["mindnano_metrics"] | |
| # Repeated digit input has one token per character; the chat template adds 21. | |
| # No requested max: verify actual output crosses the previous default of 128. | |
| prompt_length = capacity - 160 | |
| r = call("/v1/chat/completions", request("1" * (prompt_length - 21))) | |
| assert r["usage"] == {"prompt_tokens": prompt_length, "completion_tokens": 160, | |
| "total_tokens": capacity, "prompt_tokens_details": {"cached_tokens": r["mindnano_metrics"]["cached_tokens"]}}, r | |
| assert r["choices"][0]["finish_reason"] == "length" | |
| assert r["mindnano_metrics"]["context_capacity_reached"] | |
| results["default_over_128"] = r["mindnano_metrics"] | |
| for remaining in (1, 0): | |
| r = call("/v1/chat/completions", request("1" * (capacity - remaining - 21), max_tokens=2**64-1)) | |
| assert r["usage"]["prompt_tokens"] == capacity - remaining | |
| assert r["usage"]["completion_tokens"] == remaining, r | |
| assert r["choices"][0]["finish_reason"] == "length" | |
| assert r["mindnano_metrics"]["context_capacity_reached"] | |
| if remaining == 0: | |
| assert r["choices"][0]["message"]["content"] == "" | |
| assert r["mindnano_metrics"]["ttft_ms"] is None | |
| results[f"remaining_{remaining}"] = r["mindnano_metrics"] | |
| try: | |
| call("/v1/chat/completions", request("1" * (capacity + 1 - 21), max_tokens=1)) | |
| raise AssertionError("oversized input accepted") | |
| except urllib.error.HTTPError as e: | |
| assert e.code == 400 and json.load(e)["error"]["code"] == "context_length_exceeded" | |
| assert not call("/health")["active"] | |
| print(json.dumps({"passed": True, "capacity": capacity, "cases": results}, indent=2)) | |