File size: 16,895 Bytes
28c70af | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 | #!/usr/bin/env python3
# SPDX-License-Identifier: MIT OR Apache-2.0
"""The model importer (sAGI/models.py) without the network: a tiny GGUF served from a loopback HTTP server stands in for
Hugging Face / the Ollama registry. Checks the pin (a download is kept only if its sha256 is the published one), the
licence gate (open source or refused), resume, the guard, and the URL and search parsers. With BANKML_TEST_CARRIER=1
and Bonsai-1.7B present it also starts, switches and rolls back a real carrier on spare ports.
run: python3 testing/test_models.py"""
import hashlib, http.server, json, os, shutil, struct, sys, tempfile, threading
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
tmp = Path(tempfile.mkdtemp(prefix="bankml-models-"))
os.environ.update(BANKML_MODELS=str(tmp / "models"), BANKML_FORKS=str(tmp / "forks"), BANKML_UI_STATE=str(tmp / "state"),
BANKML_SERVE_LISTEN="127.0.0.1:18293", BANKML_UPSTREAM="127.0.0.1:18292")
sys.dont_write_bytecode = True
sys.path.insert(0, str(ROOT / "sAGI"))
import models as M # noqa: E402
fails = 0
def check(name, ok):
global fails
fails += not ok
print(("ok " if ok else "FAIL ") + name)
def gguf(arch="qwen3", name="tiny", ty=0, n=64) -> bytes:
"""A minimal valid GGUF v3: two metadata strings, one tensor of `n` elements of type `ty` (F32), aligned to 32."""
s = lambda x: struct.pack("<Q", len(x.encode())) + x.encode()
h = b"GGUF" + struct.pack("<IQQ", 3, 1, 2)
for k, v in (("general.architecture", arch), ("general.name", name)):
h += s(k) + struct.pack("<I", 8) + s(v)
h += s("t") + struct.pack("<I", 1) + struct.pack("<Q", n) + struct.pack("<I", ty) + struct.pack("<Q", 0)
h += b"\0" * (-len(h) % 32)
return h + b"\0" * (n * 4)
BLOB = gguf()
SHA = hashlib.sha256(BLOB).hexdigest()
class H(http.server.BaseHTTPRequestHandler):
def do_GET(self):
body = BLOB
rng = self.headers.get("Range")
cr = None
if self.path == "/oversend":
body = BLOB + b"\0" * 4096
self.send_response(200)
elif rng and self.path != "/norange":
a = int(rng.split("=")[1].split("-")[0])
if a >= len(BLOB):
self.send_response(416)
self.end_headers()
return
if self.path == "/badrange": # a 206 that starts somewhere else than asked
a = 0
self.send_response(206)
body = BLOB[a:]
cr = f"bytes {a}-{len(BLOB) - 1}/{len(BLOB)}"
else:
self.send_response(200)
if cr:
self.send_header("Content-Range", cr)
self.send_header("Content-Length", str(len(body)))
self.end_headers()
self.wfile.write(body)
def log_message(self, *a):
pass
srv = http.server.ThreadingHTTPServer(("127.0.0.1", 0), H)
threading.Thread(target=srv.serve_forever, daemon=True).start()
url = f"http://127.0.0.1:{srv.server_address[1]}/tiny.gguf"
spec = lambda **k: {"file": "tiny.gguf", "bytes": len(BLOB), "sha256": SHA, "url": url, "repo": "test/tiny", "revision": "r1",
"licence": "apache-2.0", "pinned_from": "test", **k}
# the engine is found as install.sh finds it (BANKML_LLAMA_SERVER, install.env, the download, the dev checkout)
lt = tmp / "llama-resolve"
data, home = lt / "data", lt / "home"
dl = data / f"llama-{M.LLAMA_TAG}" / "llama-server"
dev = home / "sAGI" / "bonsai" / f"llama-{M.LLAMA_TAG}" / "llama-server"
check("llama_server(): nothing installed → the installer's download path (named in the refusal)", M.llama_server({}, data, home) == dl)
dev.parent.mkdir(parents=True); dev.write_text("#!/bin/sh\n"); dev.chmod(0o755)
check("llama_server(): the development checkout when it is the only build", M.llama_server({}, data, home) == dev)
dl.parent.mkdir(parents=True); dl.write_text("#!/bin/sh\n")
check("llama_server(): a download that is not executable is passed over", M.llama_server({}, data, home) == dev)
dl.chmod(0o755)
check("llama_server(): the installer's download before the development checkout", M.llama_server({}, data, home) == dl)
spaced = lt / "my engines" / "llama-server"
(data / "install.env").write_text(f"INSTALL_PYTHON=python3\nINSTALL_LLAMA_SERVER={spaced.as_posix().replace(' ', chr(92) + ' ')}\n")
check("llama_server(): install.env's INSTALL_LLAMA_SERVER, shell-quoted as printf %q writes it", M.llama_server({}, data, home) == spaced)
check("llama_server(): BANKML_LLAMA_SERVER over everything", M.llama_server({"BANKML_LLAMA_SERVER": "/opt/x/llama-server"}, data, home) == Path("/opt/x/llama-server"))
(data / "install.env").write_text("INSTALL_LLAMA_SERVER='unterminated\n$(touch pwned)=1\n")
check("llama_server(): a malformed install.env is ignored, never executed", M.llama_server({}, data, home) == dl and not (Path.cwd() / "pwned").exists())
try:
# parsers
check("hf_parse: repo, blob URL, resolve URL, hf:// id", M.hf_parse("https://huggingface.co/Qwen/Qwen3-4B-GGUF") == ("Qwen/Qwen3-4B-GGUF", None, None)
and M.hf_parse("https://huggingface.co/a/b/blob/main/x.gguf?download=true") == ("a/b", "main", "x.gguf")
and M.hf_parse("https://hf.co/a/b/resolve/abc123/sub/y%20z.gguf") == ("a/b", "abc123", "sub/y z.gguf")
and M.hf_parse("hf://a/b") == ("a/b", None, None))
try:
M.hf_parse("https://example.com/../../etc/passwd")
check("hf_parse refuses what is not a repo id", False)
except ValueError:
check("hf_parse refuses what is not a repo id", True)
bad = 0
for u in ("https://huggingface.co/../..", "hf://a/b/blob/../x.gguf", "https://huggingface.co/a/b/resolve/main/../../x", "https://huggingface.co/a/b/blob/ma%2Fin/x.gguf"):
try:
M.hf_parse(u)
except ValueError:
bad += 1
for n in ("qwen3:../../x", "a/../b", "qwen3:a/b"):
try:
M._ollama_name(n)
except ValueError:
bad += 1
check("'.' and '..' segments, slashes in a revision and odd Ollama tags are refused", bad == 7)
check("licence gate: OSI licences pass; gemma, llama and none are refused", M.licence_open("apache-2.0") and M.licence_open(["license:mit"])
and not M.licence_open("gemma") and not M.licence_open("llama3.2") and not M.licence_open(None) and not M.licence_open([]))
check("Ollama licence text: Apache and MIT recognised; Gemma's terms named and refused",
M._licence_of_text("Apache License\nVersion 2.0, January 2004") == "apache-2.0"
and M._licence_of_text("MIT License\n\nPermission is hereby granted, free of charge") == "mit"
and not M.licence_open(M._licence_of_text("Gemma Terms of Use\nLast modified")))
check("Ollama names: model[:tag], namespaces; anything else refused", M._ollama_name("qwen3:1.7b") == ("qwen3", "1.7b")
and M._ollama_name("library/granite3.3") == ("granite3.3", "latest") and M._ollama_name("user/model:q4") == ("user/model", "q4"))
fx = ('<ul><li class="x"><a href="/library/granite4" class="g"><div><h2><span>granite4</span></h2>'
'<p class="d">IBM Granite, released under Apache 2.0.</p></div><span class="a text-blue-600 b">1b</span>'
'<span class="a text-blue-600 b">3b</span><span >12.3K</span>\n<span class="hidden"> Pulls</span></a></li></ul>')
import unittest.mock as um
with um.patch.object(M, "_get", lambda *a, **k: fx.encode()):
r = M.ollama_search("granite")
check("ollama_search parses name, description, sizes, pulls", r == [{"name": "granite4", "description": "IBM Granite, released under Apache 2.0.",
"sizes": ["1b", "3b"], "pulls": "12.3K", "cloud": False}])
# the pin
out = M.import_spec(spec())
fk = json.loads(Path(out["fork"]).read_text())
check("an import is kept when its sha256 is the published one, guarded, and pinned in FORK.json", out["arch"] == "qwen3"
and (M.MODELS / "tiny.gguf").read_bytes() == BLOB and fk["files"] == [{"path": "tiny.gguf", "bytes": len(BLOB), "sha256": SHA}])
check("installed() lists it as pinned", [m["pinned"] for m in M.installed() if m["file"] == "tiny.gguf"] == [True])
try:
M.import_spec(spec(file="tampered.gguf", sha256="0" * 64))
check("a download whose sha256 differs is discarded", False)
except RuntimeError as e:
check("a download whose sha256 differs is discarded", "discarded" in str(e) and not list(M.MODELS.glob("tampered*")))
try:
M.import_spec(spec(file="gemma.gguf", licence="gemma"))
check("a non-open-source licence is refused before any byte is fetched", False)
except PermissionError:
check("a non-open-source licence is refused before any byte is fetched", not (M.MODELS / "gemma.gguf").exists())
try:
M.import_spec(spec(file="unpinned.gguf", sha256=None))
check("no published sha256, no import", False)
except ValueError:
check("no published sha256, no import", True)
(M.MODELS / "resume.gguf.part").write_bytes(BLOB[:100])
M.import_spec(spec(file="resume.gguf"))
check("an interrupted download resumes from its .part and still verifies", (M.MODELS / "resume.gguf").read_bytes() == BLOB)
(M.MODELS / "restart.gguf.part").write_bytes(BLOB[:100])
M.import_spec(spec(file="restart.gguf", url=url.rsplit("/", 1)[0] + "/norange"))
check("a server that ignores Range: the download starts over and verifies", (M.MODELS / "restart.gguf").read_bytes() == BLOB)
(M.MODELS / "whole.gguf.part").write_bytes(BLOB) # the process died between the last byte and the rename
M.import_spec(spec(file="whole.gguf", url="http://127.0.0.1:9/unreachable"))
check("a complete .part is verified and kept without asking the server again (no 416 loop)", (M.MODELS / "whole.gguf").read_bytes() == BLOB)
(M.MODELS / "badrange.gguf.part").write_bytes(BLOB[:100])
M.import_spec(spec(file="badrange.gguf", url=url.rsplit("/", 1)[0] + "/badrange"))
check("a 206 whose Content-Range starts elsewhere is not appended: the download starts over and verifies",
(M.MODELS / "badrange.gguf").read_bytes() == BLOB)
try:
M.import_spec(spec(file="oversend.gguf", url=url.rsplit("/", 1)[0] + "/oversend"))
check("a source that sends more than the published size is cut off and discarded", False)
except RuntimeError as e:
check("a source that sends more than the published size is cut off and discarded", "more than" in str(e) and not list(M.MODELS.glob("oversend*")))
M.JOB.update(state="idle")
ev = threading.Event()
first = M.start_job("hold", ev.wait, 5)
second = M.start_job("second", lambda: None)
ev.set()
import time
for _ in range(50):
if M.JOB["state"] != "running":
break
time.sleep(0.05)
check("one job at a time: a second import or switch is refused while one runs", first and not second and M.JOB["state"] == "done")
with um.patch.object(M, "fits_memory", lambda n: (False, "too big for memory")):
try:
M.import_spec(spec(file="adopted2.gguf", local=str(tmp / "ollama-blob"), url=""))
check("the memory check applies to adoption too (not only to downloads)", False)
except RuntimeError as e:
check("the memory check applies to adoption too (not only to downloads)", "memory" in str(e))
ss = ("LISTEN 0 128 0.0.0.0:18293 0.0.0.0:* users:((\"bankml\",pid=%d,fd=3))\n"
"LISTEN 0 128 127.0.0.1:18292 0.0.0.0:* users:((\"llama-server\",pid=%d,fd=3))\n"
"LISTEN 0 128 127.0.0.1:182930 0.0.0.0:* users:((\"x\",pid=1,fd=3))\n") % (os.getpid(), os.getpid())
with um.patch.object(M.subprocess, "run", lambda *a, **k: type("R", (), {"stdout": ss})()), \
um.patch.object(M, "LISTEN", "127.0.0.1:18293"):
l1 = M._listeners()
with um.patch.object(M.subprocess, "run", lambda *a, **k: type("R", (), {"stdout": ss})()), \
um.patch.object(M, "LISTEN", "0.0.0.0:18293"):
l2 = M._listeners()
check("_listeners matches the configured host exactly, and only this user's bankml / llama-server", l1 == {} and l2 == {})
fake = tmp / "fake-bankml"
fake.write_text("#!/bin/sh\nsleep 30\n")
fake.chmod(0o755)
with um.patch.object(M, "BANKML", fake), um.patch.object(M, "serve_status",
lambda timeout=3: {"verified": {"model_sha256": "a" * 64}}):
try:
M._start_carrier(M.MODELS / "tiny.gguf", M.FORKS / "tiny.gguf.FORK.json", want_sha=SHA, wait=5)
check("a different model answering is not success, and the new carrier is killed", False)
except RuntimeError as e:
check("a different model answering is not success, and the new carrier is killed", "another carrier answers" in str(e))
with um.patch.object(M, "BANKML", fake), um.patch.object(M, "serve_status", lambda timeout=3: {"error": "down"}):
import time as _t
t0 = _t.time()
try:
M._start_carrier(M.MODELS / "tiny.gguf", M.FORKS / "tiny.gguf.FORK.json", want_sha=SHA, wait=25)
check("a slow start is waited for (no 20 s port heuristic), then killed at the deadline", False)
except RuntimeError as e:
check("a slow start is waited for (no 20 s port heuristic), then killed at the deadline", "timed out" in str(e) and _t.time() - t0 >= 24)
bad = gguf(ty=99)
SHA_BAD = hashlib.sha256(bad).hexdigest()
BLOB, keep = bad, BLOB
try:
M.import_spec(spec(file="unplayable.gguf", sha256=SHA_BAD))
check("bankml's guard refuses an unplayable file after download, and it is removed", False)
except RuntimeError as e:
check("bankml's guard refuses an unplayable file after download, and it is removed", "guard refused" in str(e) and not (M.MODELS / "unplayable.gguf").exists())
BLOB = keep
local = tmp / "ollama-blob"
local.write_bytes(BLOB)
M.import_spec(spec(file="adopted.gguf", local=str(local), url="http://127.0.0.1:9/unreachable"))
check("a local Ollama blob is adopted by link after hashing, with no download", (M.MODELS / "adopted.gguf").is_symlink()
and (M.MODELS / "adopted.gguf").resolve() == local.resolve())
with um.patch.object(M, "free_bytes", lambda p=None: 10 ** 9):
ok, why = M.fits(2 * 10 ** 9)
check("fits(): a file that would leave less than the margin free is refused, with the numbers", not ok and "GB free" in why)
check("the catalogue: every entry open source, fully pinned (64-hex sha256, 40-hex revision), one default, no Gemma",
all(M.licence_open(c["licence"]) and len(c["sha256"]) == 64 and len(c["revision"]) == 40 for c in M.CATALOG)
and sum(bool(c.get("default")) for c in M.CATALOG) == 1 and not any("gemma" in c["id"] for c in M.CATALOG))
check("the recorded conversions (0.3.4): open source, the GGUF and its source both pinned by sha256, a 40-hex revision, the tools named",
all(M.licence_open(c["licence"]) and len(c["sha256"]) == 64 and len(c["source_sha256"]) == 64 and len(c["revision"]) == 40
and "convert_hf_to_gguf.py" in c["tools"] for c in M.CONVERTED))
try:
M.pin_converted("not-a-conversion.gguf")
check("pin_converted refuses a file it has no record of", False)
except RuntimeError:
check("pin_converted refuses a file it has no record of", True)
# a real carrier, on spare ports, when asked and when the small Bonsai is here
b17 = ROOT / ".models" / "Bonsai-1.7B-Q1_0.gguf"
if os.environ.get("BANKML_TEST_CARRIER") == "1" and b17.exists() and M.LLAMA.exists():
os.symlink(b17.resolve(), M.MODELS / b17.name)
c = next(x for x in M.CATALOG if x["file"] == b17.name)
M.write_fork(b17.name, c["bytes"], c["sha256"], c["repo"], "", c["revision"], c["licence"], "test: the catalogue's pin")
st = M.switch(b17.name)
check("switch(): the carrier comes up verified on the chosen model", st["verified"]["model_sha256"] == c["sha256"] and st["verified"]["arch"] == "qwen3")
try:
M.switch("tiny.gguf") # verifies, but llama-server cannot load a 64-float toy: it must roll back
check("a switch that fails rolls back to the previous model", False)
except RuntimeError:
check("a switch that fails rolls back to the previous model", (M.serve_status().get("verified") or {}).get("model_sha256") == c["sha256"])
M._stop_carrier()
else:
print("skip carrier switch (set BANKML_TEST_CARRIER=1 with .models/Bonsai-1.7B-Q1_0.gguf present)")
finally:
srv.shutdown()
shutil.rmtree(tmp, ignore_errors=True)
print("all ok" if not fails else f"{fails} FAILED")
sys.exit(1 if fails else 0)
|