Download tools/benchmark_long_context.py from Sariel00/Ling-3.0-tiny-RKNN: direct link, hf CLI and curl.
- Browser
- Download file 7.15 kB
-
https://huggingface.co/Sariel00/Ling-3.0-tiny-RKNN/resolve/main/tools/benchmark_long_context.py
- Command line
-
hf download hf://Sariel00/Ling-3.0-tiny-RKNN/tools/benchmark_long_context.py
-
curl -L -o benchmark_long_context.py https://huggingface.co/Sariel00/Ling-3.0-tiny-RKNN/resolve/main/tools/benchmark_long_context.py
7.15 kB
| #!/usr/bin/env python3 | |
| """Run one real long API request and collect process/system memory on the board.""" | |
| import argparse | |
| import concurrent.futures | |
| import json | |
| import os | |
| from pathlib import Path | |
| import signal | |
| import subprocess | |
| import time | |
| import urllib.request | |
| def read_kib(path): | |
| result = {} | |
| for line in path.read_text().splitlines(): | |
| key, _, value = line.partition(":") | |
| if value.strip().endswith("kB"): | |
| result[key] = int(value.split()[0]) / 1024 | |
| return result | |
| def main(): | |
| p = argparse.ArgumentParser(description=__doc__) | |
| p.add_argument("--binary", type=Path, required=True) | |
| p.add_argument("--model", type=Path, required=True) | |
| p.add_argument("--output-dir", type=Path, required=True) | |
| p.add_argument("--context", type=int, default=65536) | |
| p.add_argument("--input-tokens", type=int, default=65472) | |
| p.add_argument("--new-tokens", type=int, default=64) | |
| p.add_argument("--port", type=int, default=19095) | |
| a = p.parse_args() | |
| if a.input_tokens < 22 or a.input_tokens + a.new_tokens > a.context: | |
| p.error("input + output must fit the selected capacity") | |
| a.output_dir.mkdir(parents=True, exist_ok=False) | |
| client = urllib.request.build_opener(urllib.request.ProxyHandler({})) | |
| def call(path, body=None, timeout=10): | |
| request = urllib.request.Request( | |
| f"http://127.0.0.1:{a.port}" + path, | |
| data=None if body is None else json.dumps(body).encode(), | |
| headers={"Content-Type": "application/json"}) | |
| with client.open(request, timeout=timeout) as response: | |
| return json.load(response) | |
| def infer(prompt, count): | |
| return call("/v1/chat/completions", { | |
| "model": "mindnano-ling3-tiny", "temperature": 0, "cache_prompt": False, | |
| "messages": [{"role": "user", "content": prompt}], | |
| "max_tokens": count}, timeout=86400) | |
| result = {"context": a.context, "requested_input_tokens": a.input_tokens, | |
| "requested_output_tokens": a.new_tokens, "started_unix": time.time()} | |
| runtime = a.output_dir / "runtime.log" | |
| samples = a.output_dir / "memory.jsonl" | |
| child = None | |
| engine_pid = None | |
| started = time.monotonic() | |
| def sample(phase): | |
| nonlocal engine_pid | |
| mem = read_kib(Path("/proc/meminfo")) | |
| data = {"timestamp": time.time(), "elapsed_s": time.monotonic() - started, | |
| "phase": phase, "available_mib": mem.get("MemAvailable"), | |
| "system_swap_used_mib": mem.get("SwapTotal", 0) - mem.get("SwapFree", 0)} | |
| if child is not None and child.poll() is None: | |
| children = Path(f"/proc/{child.pid}/task/{child.pid}/children") | |
| if children.exists(): | |
| ids = children.read_text().split() | |
| if ids: | |
| engine_pid = int(ids[0]) | |
| status = Path(f"/proc/{engine_pid}/status") | |
| if status.exists(): | |
| proc = read_kib(status) | |
| data.update({"engine_pid": engine_pid, | |
| "rss_mib": proc.get("VmRSS"), "peak_rss_mib": proc.get("VmHWM"), | |
| "swap_mib": proc.get("VmSwap")}) | |
| with samples.open("a") as out: | |
| out.write(json.dumps(data) + "\n") | |
| return data | |
| try: | |
| sample("before_start") | |
| with runtime.open("w") as log: | |
| child = subprocess.Popen([ | |
| str(a.binary.resolve()), "--model", str(a.model.resolve()), | |
| "--context", str(a.context), "--host", "127.0.0.1", "--port", str(a.port), | |
| "--no-console", "--log", str((a.output_dir / "metrics.jsonl").resolve())], | |
| stdout=log, stderr=subprocess.STDOUT, start_new_session=True, | |
| env={"PATH": "/usr/bin:/bin"}) | |
| print(json.dumps({"launcher_pid": child.pid, "runtime_log": str(runtime)}), flush=True) | |
| deadline = time.monotonic() + 300 | |
| while True: | |
| sample("initialization") | |
| if child.poll() is not None: | |
| raise RuntimeError(f"service exited during startup: {child.returncode}") | |
| try: | |
| result["health_before"] = call("/health") | |
| break | |
| except (OSError, ValueError): | |
| if time.monotonic() >= deadline: | |
| raise TimeoutError("startup did not complete within 300 seconds") | |
| time.sleep(2) | |
| result["short_before"] = infer("1" * 107, 64) | |
| sample("ready") | |
| print(json.dumps({"ready": result["health_before"], | |
| "short_metrics": result["short_before"]["mindnano_metrics"]}), flush=True) | |
| (a.output_dir / "initialization.json").write_text(json.dumps(result, indent=2) + "\n") | |
| request_start = time.monotonic() | |
| with concurrent.futures.ThreadPoolExecutor(max_workers=1) as pool: | |
| future = pool.submit(infer, "1" * (a.input_tokens - 21), a.new_tokens) | |
| next_report = 0 | |
| while not future.done(): | |
| data = sample("long_request") | |
| elapsed = time.monotonic() - request_start | |
| if elapsed >= next_report: | |
| print(json.dumps({"request_elapsed_s": elapsed, **data}), flush=True) | |
| next_report = elapsed + 60 | |
| if child.poll() is not None: | |
| raise RuntimeError(f"service exited during request: {child.returncode}") | |
| try: | |
| future.result(timeout=10) | |
| except concurrent.futures.TimeoutError: | |
| pass | |
| result["long_request"] = future.result() | |
| result["request_wall_s"] = time.monotonic() - request_start | |
| long = result["long_request"] | |
| assert long["usage"]["prompt_tokens"] == a.input_tokens, long["usage"] | |
| assert long["mindnano_metrics"]["prompt_evaluated_tokens"] == a.input_tokens | |
| assert long["mindnano_metrics"]["cached_tokens"] == 0 | |
| result["short_after"] = infer("1" * 107, 64) | |
| assert result["short_before"]["choices"] == result["short_after"]["choices"] | |
| result["health_after"] = call("/health") | |
| sample("complete") | |
| result["passed"] = True | |
| print(json.dumps({"passed": True, "metrics": long["mindnano_metrics"]}), flush=True) | |
| except BaseException as exc: | |
| result["passed"] = False | |
| result["error"] = repr(exc) | |
| raise | |
| finally: | |
| result["finished_unix"] = time.time() | |
| (a.output_dir / "result.json").write_text(json.dumps(result, ensure_ascii=False, indent=2) + "\n") | |
| if child is not None and child.poll() is None: | |
| os.killpg(child.pid, signal.SIGTERM) | |
| try: | |
| child.wait(timeout=120) | |
| except subprocess.TimeoutExpired: | |
| os.killpg(child.pid, signal.SIGKILL) | |
| child.wait() | |
| sample("after_stop") | |
| if __name__ == "__main__": | |
| main() | |