File size: 6,431 Bytes
12496fc | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 | from dataclasses import asdict
from hashlib import sha256
import asyncio
import json
import os
import sys
import time
import pytest
from nexora.tools import Executor, Policy
from nexora.agent import Agent
from nexora.coding import index_repository, retrieve
from nexora.memory import Memory
from nexora.voice import VoiceSession
from nexora.inference import HTTPBackend
@pytest.fixture
def executor(tmp_path):
return Executor(Policy(str(tmp_path), permissions=["READ", "WRITE"]))
@pytest.mark.parametrize("path", ["../escape", "C:/Windows/a", ".git/config", ".env", ".env.local", "private/key", "x:y"])
def test_path_denied(executor, path):
assert not executor.execute("filesystem.read", {"path": path}).ok
def test_schema_permission(executor):
assert not executor.execute("filesystem.read", {"path": 4}).ok
assert not executor.execute("filesystem.read", {"path": "a", "extra": True}).ok
assert not executor.execute("unknown", {}).ok
assert not executor.execute("shell.exec", {"command": "test"}).ok
assert "shell.exec" not in executor.available_tools()
assert "filesystem.write" in executor.available_tools()
def test_agent_advertises_only_allowed_tools(executor):
class SchemaObserver:
def complete(self, messages, schema=None):
assert "shell.exec" not in schema["properties"]["tool"]["enum"]
assert "shell.exec" not in messages[0]["content"]
return '{"kind":"finish","summary":"Inspected policy"}'
assert Agent(SchemaObserver(), executor).run("Inspect")["status"] == "unverified"
def test_compare_write_idempotency(executor):
a = executor.execute("filesystem.write", {"path": "a.txt", "text": "first"}, call_id="1")
assert a.ok
assert executor.execute("filesystem.write", {"path": "a.txt", "text": "first"}, call_id="1") == a
with pytest.raises(ValueError):
executor.execute("filesystem.write", {"path": "a.txt", "text": "other"}, call_id="1")
assert not executor.execute("filesystem.write", {"path": "a.txt", "text": "overwrite"}).ok
assert executor.execute("filesystem.write", {"path": "a.txt", "text": "second", "expected_sha256": sha256(b"first").hexdigest()}).ok
assert executor.execute("filesystem.read", {"path": "a.txt"}).output == "second"
def test_symlink_escape(executor, tmp_path):
outside = tmp_path.parent / (tmp_path.name + "-outside")
outside.mkdir()
(outside / "secret").write_text("private")
try:
(tmp_path / "link").symlink_to(outside, target_is_directory=True)
except OSError:
pytest.skip("OS account cannot create symlinks")
assert not executor.execute("filesystem.read", {"path": "link/secret"}).ok
def test_command_failure_timeout_output(tmp_path):
commands = {"ok": [sys.executable, "-I", "-c", "print('verified')"],
"fail": [sys.executable, "-I", "-c", "raise SystemExit(3)"],
"flood": [sys.executable, "-I", "-c", "print('a'*10000)"],
"sleep": [sys.executable, "-I", "-c", "import time; time.sleep(10)"]}
e = Executor(Policy(str(tmp_path), ["EXECUTE"], commands, timeout_seconds=1, output_limit=100, allow_host_execution=True))
assert e.execute("shell.exec", {"command": "ok"}).ok
assert e.execute("shell.exec", {"command": "fail"}).exit_code == 3
assert e.execute("shell.exec", {"command": "flood"}).truncated
assert "timed out" in e.execute("shell.exec", {"command": "sleep"}).error
assert not e.execute("shell.exec", {"command": "arbitrary"}).ok
class ScriptedModel:
"""Test double for state-machine tests; never used for capability benchmarks."""
def __init__(self, actions):
self.actions = iter(actions)
def complete(self, messages, schema=None):
return json.dumps(next(self.actions))
def test_agent_verified_e2e(executor):
model = ScriptedModel([{"kind": "tool", "tool": "filesystem.write", "arguments": {"path": "result.txt", "text": "42"}},
{"kind": "finish", "summary": "Created result."}])
result = Agent(model, executor, lambda: (executor.root / "result.txt").read_text() == "42").run("Create result")
assert result["status"] == "verified"
def test_agent_cannot_self_certify(executor):
a = {"kind": "finish", "summary": "Everything passed."}
assert Agent(ScriptedModel([a]), executor).run("task")["status"] == "unverified"
assert Agent(ScriptedModel([a]*3), executor, lambda: False).run("task")["status"] == "failed"
def test_loop_guard(executor):
a = {"kind": "tool", "tool": "filesystem.list", "arguments": {"path": "."}}
assert Agent(ScriptedModel([a]*4), executor).run("task")["reason"] == "loop_detected"
def test_index(executor):
(executor.root / "main.py").write_text("import math\n\ndef compute():\n return math.sqrt(9)\n")
index = index_repository(executor)
assert index[0]["symbols"][0]["name"] == "compute"
assert retrieve(index, "sqrt")[0]["path"] == "main.py"
def test_memory_lifecycle(tmp_path):
m = Memory(tmp_path / "mem.sqlite")
rid = m.put("Project uses Python", "owner", .9, provenance={"conversation": "test"})
assert m.retrieve("Python")[0]["id"] == rid
m.put("Project uses Rust", "owner correction", 1, record_id=rid)
assert not m.retrieve("Python")
assert m.retrieve("Rust")
m.delete(rid)
assert not m.retrieve("Rust")
rid = m.put("Temporary note", "session", ttl_seconds=.01)
time.sleep(.02)
assert not m.retrieve("Temporary")
m.expire()
m.close()
def test_network_requires_opt_in():
with pytest.raises(PermissionError):
HTTPBackend("https://example.com/v1", "model")
with pytest.raises(ValueError):
HTTPBackend("file:///etc/passwd", "model")
def test_voice_barge_in():
async def scenario():
heard, stopped = [], []
async def reply(text):
yield text + " first"
await asyncio.sleep(.2)
yield text + " stale"
async def speak(text):
heard.append(text)
return time.perf_counter()
async def stop():
stopped.append(True)
s = VoiceSession(reply, speak, stop)
await s.transcript("old")
await asyncio.sleep(.02)
t = await s.transcript("new")
await t
assert "old stale" not in heard and "new stale" in heard
assert stopped and s.metrics
asyncio.run(scenario())
|