Download scripts/stream_benchmark.py from devildasdf/NEXORA: direct link, hf CLI and curl.
- Browser
- Download file 1.11 kB
-
https://huggingface.co/devildasdf/NEXORA/resolve/main/scripts/stream_benchmark.py
- Command line
-
hf download hf://devildasdf/NEXORA/scripts/stream_benchmark.py
-
curl -L -o stream_benchmark.py https://huggingface.co/devildasdf/NEXORA/resolve/main/scripts/stream_benchmark.py
1.11 kB
| from pathlib import Path | |
| import argparse | |
| import json | |
| import sys | |
| sys.path.insert(0, str(Path(__file__).resolve().parents[1])) | |
| from nexora.inference import HFBackend | |
| def main(): | |
| p = argparse.ArgumentParser() | |
| p.add_argument("--model", default=".cache/Qwen3.5-0.8B") | |
| args = p.parse_args() | |
| backend = HFBackend(args.model, max_new_tokens=32) | |
| messages = [{"role": "user", "content": "Describe a queue in one short sentence."}] | |
| nonstream = backend.complete(messages) | |
| chunks = list(backend.stream(messages)) | |
| report = {"model": args.model, "prompt": messages[0]["content"], "nonstream": nonstream, "streamed": "".join(chunks), | |
| "stream_matches_nonstream": nonstream == "".join(chunks), "chunks": len(chunks), **backend.last_metrics, | |
| "limitations": "One greedy request, warm model, no concurrent load or audio"} | |
| if not report["stream_matches_nonstream"]: | |
| raise AssertionError("Streaming parity failed") | |
| Path("reports/streaming.json").write_text(json.dumps(report, indent=2)) | |
| print(json.dumps(report)) | |
| if __name__ == "__main__": | |
| main() | |