PEScn commited on
Commit
28164ee
·
verified ·
1 Parent(s): 227714c

Add knowline_engine.py (one-command Decision Index runs); --backend hf without accelerate

Browse files
Files changed (4) hide show
  1. INFERENCE.md +24 -0
  2. SHA256SUMS +4 -3
  3. knowline_engine.py +138 -0
  4. knowline_server.py +6 -1
INFERENCE.md CHANGED
@@ -95,9 +95,33 @@ decision-index score --suite-dir suite-0.2 --results results.jsonl --engine http
95
  - NVIDIA H20 (96 GB), driver 590.48.01, one server per GPU.
96
  - Latency was not measured on the board's reference hardware (1x RTX PRO 6000, latency-v1).
97
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
98
  ## Changes
99
 
100
  - 2026-10-08: `knowline_server.py` now accepts chats whose roles the chat template rejects (for example `customer` /
101
  `agent`): such a state is rendered as one user message, like any other structured state, instead of the request
102
  failing. Any other unexpected error now returns HTTP 500 instead of closing the connection. Requests that worked
103
  before give the same answers; our Decision Index runs had none that failed.
 
 
 
95
  - NVIDIA H20 (96 GB), driver 590.48.01, one server per GPU.
96
  - Latency was not measured on the board's reference hardware (1x RTX PRO 6000, latency-v1).
97
 
98
+ ## One-command Decision Index run
99
+
100
+ `knowline_engine.py` (in this repository) is a Decision Index engine. It runs the same front end in process, so there
101
+ is no server to start by hand:
102
+
103
+ ```bash
104
+ pip install "sglang==0.5.21" "transformers==5.12.1" requests # plus the decision-index kit
105
+ git clone https://huggingface.co/PelaAI/KnowLine-4B-Gen1 && cd KnowLine-4B-Gen1 # puts both files on the path
106
+ python -m decision_index pipeline --engine knowline_engine:KnowLine \
107
+ --option model=PelaAI/KnowLine-4B-Gen1 --option revision=<commit> --out runs/KnowLine-4B-Gen1
108
+ ```
109
+
110
+ - **Default backend (the setting of our runs):** the engine starts SGLang with the flags above (FP8 at load) on a free
111
+ local port and stops it when the run ends. Use `--option gpu=<n>` to pick a GPU, and
112
+ `--option sglang_python=<python>` if SGLang lives in another environment.
113
+ - **`--option backend=hf`:** transformers only, bf16, no server; slower, and not the setting of our runs.
114
+ - **Limits:** up to 64 questions per request and 255 options per question. Larger requests are reported as
115
+ unsupported; nothing is truncated.
116
+ - **Checked on Gen2:** on 300 random Decision Index 0.3 rows, the default backend gave the same answer as our published
117
+ Gen2 run on 706 of 709 questions. The 3 differences are near-ties (top two options within 0.06), from FP8 numerical
118
+ noise. Requests took 30.7 ms median, one at a time on one H20.
119
+
120
  ## Changes
121
 
122
  - 2026-10-08: `knowline_server.py` now accepts chats whose roles the chat template rejects (for example `customer` /
123
  `agent`): such a state is rendered as one user message, like any other structured state, instead of the request
124
  failing. Any other unexpected error now returns HTTP 500 instead of closing the connection. Requests that worked
125
  before give the same answers; our Decision Index runs had none that failed.
126
+ - 2026-10-08: added `knowline_engine.py` (one-command Decision Index runs). `--backend hf` of `knowline_server.py` no
127
+ longer needs `accelerate`: without it, the model is loaded and moved to one device.
SHA256SUMS CHANGED
@@ -1,8 +1,9 @@
1
  a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715 chat_template.jinja
2
  ddc63e1c717afa86c865bb5e01313d89d72bb53b97ad4a8a03ba8510c0621670 config.json
3
  f888421726665e8a84b738eed42a64875aed79de8be7daade851ac8bf4c0cef9 configuration.json
4
- 62be3575c09145de8c8a824831ede917d4138b6c585e4092278fa5888f7aa8cf INFERENCE.md
5
- 8f2f3b944722ec4a8c1de68e0da9f12eb1feb1d2b02d5d82a152f9652fc70046 knowline_server.py
 
6
  50cbab8a892c5f2993b8c7351a99182507472def3b1374558308605d99b86b32 LICENSE
7
  a9d356d7bdf1ef4949e3e748e95b8e10ad9d4e2e838eddc38a0a7b6b94d1db8d merges.txt
8
  77d5d691c4c13ae08eacbddf9d145b2ebdaa0f6e69f4917630eb5a87691f15e5 model-00001-of-00002.safetensors
@@ -11,7 +12,7 @@ e5fc485d419f2c554169c1fc441f57a1a45a50673bdb4e41d3dd0261b21abe7c model-extra-fr
11
  48b56300c2b19dcd9a43001c9098098e7eaf4c6bc9f5cdd108148611fdfdca56 model.safetensors.index.json
12
  27225450ac9c6529872ee1924fcb0962ff5634834f817040f444118116f4e516 preprocessor_config.json
13
  fd774ff020c02f7c5e032d94be9ae0ea9f139e895a077515a06b4ca509f9afac serve_knowline.sh
14
- 316230d6a809701f4db5ea8f8fc862bc3a6f3229c937c174e674ff3ca0a64ac8 tokenizer_config.json
15
  5f9e4d4901a92b997e463c1f46055088b6cca5ca61a6522d1b9f64c4bb81cb42 tokenizer.json
 
16
  7768af27c1fafa9cc9011c1dc20067e03f8915e03b63504550e11d5066986d13 video_preprocessor_config.json
17
  ce99b4cb2983d118806ce0a8b777a35b093e2000a503ebde25853284c9dfa003 vocab.json
 
1
  a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715 chat_template.jinja
2
  ddc63e1c717afa86c865bb5e01313d89d72bb53b97ad4a8a03ba8510c0621670 config.json
3
  f888421726665e8a84b738eed42a64875aed79de8be7daade851ac8bf4c0cef9 configuration.json
4
+ 176afd91aa41f4b053b3265e523cbdf4945d3601c03843cd8970ea7b527de05c INFERENCE.md
5
+ 6e4e32cfb44747202aa39030f7ac6d0893d2ad5de3e2413c2b8f4788c275132d knowline_engine.py
6
+ 4bf7c0392ff679948012a8db2876b86d7159c83e5c918bbee28c696c0cfb85b4 knowline_server.py
7
  50cbab8a892c5f2993b8c7351a99182507472def3b1374558308605d99b86b32 LICENSE
8
  a9d356d7bdf1ef4949e3e748e95b8e10ad9d4e2e838eddc38a0a7b6b94d1db8d merges.txt
9
  77d5d691c4c13ae08eacbddf9d145b2ebdaa0f6e69f4917630eb5a87691f15e5 model-00001-of-00002.safetensors
 
12
  48b56300c2b19dcd9a43001c9098098e7eaf4c6bc9f5cdd108148611fdfdca56 model.safetensors.index.json
13
  27225450ac9c6529872ee1924fcb0962ff5634834f817040f444118116f4e516 preprocessor_config.json
14
  fd774ff020c02f7c5e032d94be9ae0ea9f139e895a077515a06b4ca509f9afac serve_knowline.sh
 
15
  5f9e4d4901a92b997e463c1f46055088b6cca5ca61a6522d1b9f64c4bb81cb42 tokenizer.json
16
+ 316230d6a809701f4db5ea8f8fc862bc3a6f3229c937c174e674ff3ca0a64ac8 tokenizer_config.json
17
  7768af27c1fafa9cc9011c1dc20067e03f8915e03b63504550e11d5066986d13 video_preprocessor_config.json
18
  ce99b4cb2983d118806ce0a8b777a35b093e2000a503ebde25853284c9dfa003 vocab.json
knowline_engine.py ADDED
@@ -0,0 +1,138 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Decision Index engine for KnowLine (PelaAI): the model repo's `knowline_server.py` front end, in process.
2
+
3
+ One command, no server to start by hand:
4
+
5
+ # default: the setting of our submitted runs. Starts SGLang with the flags of serve_knowline.sh (FP8 at load) on a
6
+ # free local port, scores through it, and stops it when the run ends.
7
+ python -m decision_index pipeline --engine knowline_engine:KnowLine \\
8
+ --option model=PelaAI/KnowLine-4B-Gen2 --option revision=<commit> --out runs/KnowLine-4B-Gen2
9
+
10
+ # without SGLang: transformers only, bf16 (slower; not the setting of our runs)
11
+ ... --option backend=hf
12
+
13
+ Put this file and `knowline_server.py` (both in the model repo) on PYTHONPATH, e.g. run from a clone of the model repo.
14
+ Requirements: `transformers` and `requests`; `sglang==0.5.21` for the default backend, `torch` for backend=hf.
15
+ Options: model, revision, backend (sglang | hf), gpu (CUDA device for SGLang; default: as CUDA_VISIBLE_DEVICES), mem (0.72), port (free one),
16
+ temperature (1.0), workers (16), startup_timeout (s, 1800), sglang_python (interpreter with SGLang installed, if it is
17
+ not the one running the kit), device (backend=hf: torch device or device map, default auto / cuda).
18
+ Licence of this file: MIT.
19
+ """
20
+
21
+ import atexit
22
+ import os
23
+ import signal
24
+ import socket
25
+ import subprocess
26
+ import sys
27
+ import time
28
+ from pathlib import Path
29
+
30
+ from decision_index.engines.base import Engine, Unsupported
31
+
32
+ sys.path.insert(0, str(Path(__file__).resolve().parent))
33
+ import knowline_server as ks # noqa: E402
34
+
35
+ SGLANG_FLAGS = ["--served-model-name", "m", "--tp", "1", "--quantization", "fp8", "--mamba-radix-cache-strategy",
36
+ "extra_buffer", "--enable-fp32-lm-head"]
37
+
38
+
39
+ def _free_port():
40
+ with socket.socket() as s:
41
+ s.bind(("127.0.0.1", 0))
42
+ return s.getsockname()[1]
43
+
44
+
45
+ class KnowLine(Engine):
46
+ name = "knowline"
47
+ latency = ("In-process request wall time through knowline_server's KnowLine engine (rendering + one prefill per "
48
+ "question), against a local SGLang server for backend=sglang; excludes model loading and server startup.")
49
+
50
+ def __init__(self, model, revision=None, backend="sglang", gpu=None, mem=0.72, port=None, temperature=1.0,
51
+ workers=16, startup_timeout=1800, sglang_python=None, device=None, **options):
52
+ super().__init__(**options)
53
+ import requests
54
+ import transformers
55
+ from transformers import AutoTokenizer
56
+
57
+ path = model
58
+ if not Path(model).exists(): # a Hub repo id: pin the files once so SGLang and the tokenizer read the same ones
59
+ from huggingface_hub import snapshot_download
60
+ path = snapshot_download(model, revision=revision)
61
+ self.model_id, self.backend_name, self.server = model, backend, None
62
+ tok = AutoTokenizer.from_pretrained(path)
63
+ if backend == "sglang":
64
+ port = int(port or _free_port())
65
+ env = dict(os.environ)
66
+ if gpu is not None:
67
+ env["CUDA_VISIBLE_DEVICES"] = str(gpu)
68
+ cmd = [sglang_python or sys.executable, "-m", "sglang.launch_server", "--model-path", path, *SGLANG_FLAGS,
69
+ "--mem-fraction-static", str(mem), "--port", str(port)]
70
+ self.server = subprocess.Popen(cmd, env=env, start_new_session=True)
71
+ atexit.register(self.close)
72
+ url, deadline = f"http://127.0.0.1:{port}", time.time() + float(startup_timeout)
73
+ while True:
74
+ if self.server.poll() is not None:
75
+ raise RuntimeError(f"SGLang exited with code {self.server.returncode} before becoming healthy")
76
+ try:
77
+ if requests.get(f"{url}/health", timeout=3).status_code == 200:
78
+ break
79
+ except requests.RequestException:
80
+ pass
81
+ if time.time() > deadline:
82
+ raise RuntimeError("SGLang did not become healthy in time")
83
+ time.sleep(5)
84
+ back = ks.SGLang(url)
85
+ elif backend == "hf":
86
+ back = ks.HF(path, device=device)
87
+ else:
88
+ raise ValueError("backend must be 'sglang' or 'hf'")
89
+ self.engine = ks.KnowLine(tok, back, float(temperature), int(workers), {})
90
+ self.provenance = {
91
+ "kind": f"knowline_server.KnowLine in process, backend {backend}",
92
+ "repo": model, "revision": revision, "front_end": "knowline_server.py (chat style, label-token softmax, "
93
+ f"temperature {temperature}, {workers} scoring threads, no calibration file)",
94
+ "sglang": " ".join(SGLANG_FLAGS + ["--mem-fraction-static", str(mem)]) if backend == "sglang" else None,
95
+ "transformers": transformers.__version__,
96
+ "policy": f"Unmodified state and questions; up to {ks.MAX_QUESTIONS} questions per request and "
97
+ f"{ks.MAX_LABELS} options per question, larger requests are unsupported, nothing is truncated.",
98
+ }
99
+
100
+ def __call__(self, state, questions):
101
+ if len(questions) > ks.MAX_QUESTIONS:
102
+ raise Unsupported(f"{len(questions)} questions; at most {ks.MAX_QUESTIONS} per request")
103
+ try:
104
+ answers, usage = self.engine.run(state, questions)
105
+ except ValueError as exc:
106
+ if any(k in str(exc) for k in ("criteria", "levels", "options", "questions")):
107
+ raise Unsupported(str(exc)) from exc
108
+ raise
109
+ return {"model": self.model_id, "answers": answers, "usage": usage}, None
110
+
111
+ def runtime(self):
112
+ info = {"backend": self.backend_name}
113
+ try:
114
+ import torch
115
+ info.update(torch=torch.__version__, cuda=torch.version.cuda)
116
+ if torch.cuda.is_available():
117
+ info["gpu"] = torch.cuda.get_device_name()
118
+ except ImportError:
119
+ pass
120
+ if self.backend_name == "sglang":
121
+ try:
122
+ import sglang
123
+ info["sglang"] = sglang.__version__
124
+ except Exception: # noqa: BLE001 - version is informational
125
+ pass
126
+ return info
127
+
128
+ def close(self):
129
+ if self.server is not None and self.server.poll() is None:
130
+ try:
131
+ os.killpg(self.server.pid, signal.SIGTERM)
132
+ self.server.wait(timeout=60)
133
+ except Exception: # noqa: BLE001 - make sure the server does not outlive the run
134
+ try:
135
+ os.killpg(self.server.pid, signal.SIGKILL)
136
+ except ProcessLookupError:
137
+ pass
138
+ self.server = None
knowline_server.py CHANGED
@@ -223,7 +223,12 @@ class HF:
223
  multimodal = hasattr(cfg, "vision_config") or hasattr(cfg, "audio_config")
224
  cls = getattr(transformers, "AutoModelForMultimodalLM", transformers.AutoModelForImageTextToText) if multimodal \
225
  else AutoModelForCausalLM
226
- self.model = cls.from_pretrained(model, dtype=getattr(torch, dtype), device_map=device or "auto").eval()
 
 
 
 
 
227
  self.tok = AutoTokenizer.from_pretrained(model)
228
  self.lock = threading.Lock()
229
 
 
223
  multimodal = hasattr(cfg, "vision_config") or hasattr(cfg, "audio_config")
224
  cls = getattr(transformers, "AutoModelForMultimodalLM", transformers.AutoModelForImageTextToText) if multimodal \
225
  else AutoModelForCausalLM
226
+ try:
227
+ import accelerate # noqa: F401 (transformers needs it for device_map)
228
+ self.model = cls.from_pretrained(model, dtype=getattr(torch, dtype), device_map=device or "auto").eval()
229
+ except ImportError: # without accelerate: load, then move to one device
230
+ device = device or ("cuda" if torch.cuda.is_available() else "cpu")
231
+ self.model = cls.from_pretrained(model, dtype=getattr(torch, dtype)).to(device).eval()
232
  self.tok = AutoTokenizer.from_pretrained(model)
233
  self.lock = threading.Lock()
234