File size: 4,540 Bytes
12496fc | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 | """Measured baseline, tokenizer and agent experiments. All failures are retained."""
from pathlib import Path
from dataclasses import asdict
import json
import sys
import tempfile
import time
import argparse
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from nexora.inference import HFBackend
from nexora.tokenizer import ByteTokenizer
from nexora.agent import Agent
from nexora.tools import Executor, Policy
from nexora.evaluation import wilson, percentiles
SAMPLES = {
"english": "A reliable assistant inspects evidence before claiming that a task succeeded.",
"hindi": "सहायक को कार्य पूरा होने का दावा करने से पहले प्रमाण की जाँच करनी चाहिए।",
"hinglish": "Pehle code run karo aur tests ka output check karo, phir result batao.",
"python": "def total(values):\n return sum(x for x in values if x is not None)\n",
"json": '{"tool":"filesystem.read","arguments":{"path":"src/main.py"}}',
"math": "For x ∈ ℝ, x² + 2x + 1 = (x + 1)²; ∑ᵢ xᵢ / n.",
"shell": "git diff --stat && python -m pytest -q",
"xml_url": '<source href="https://example.org/api?q=a%20b">Unicode: λ</source>',
}
def main():
parser = argparse.ArgumentParser()
parser.add_argument("--model", default=".cache/Qwen3-0.6B")
parser.add_argument("--report", default="baseline")
args = parser.parse_args()
out = Path("reports")
out.mkdir(exist_ok=True)
backend = HFBackend(args.model, max_new_tokens=96)
byte_report = ByteTokenizer().benchmark(SAMPLES)
bpe = {}
for name, text in SAMPLES.items():
ids = backend.tokenizer.encode(text, add_special_tokens=False)
bpe[name] = {"tokens": len(ids), "characters_per_token": len(text)/len(ids), "roundtrip": backend.tokenizer.decode(ids) == text}
(out / f"tokenizer-{args.report}.json").write_text(json.dumps({"byte": byte_report, "qwen_bpe": bpe, "limitation": "Eight public examples, not a representative benchmark or vocabulary-size study"}, indent=2, ensure_ascii=False), encoding="utf-8")
cases = [
("arithmetic", "Return only the integer: 17 * 23.", lambda s: s.strip() == "391"),
("json", 'Return only a JSON object with key "ready" and boolean value true.', lambda s: json.loads(s) == {"ready": True}),
("ordering", "Sort these integers ascending. Return only a JSON array: 19, -3, 7, 0.", lambda s: json.loads(s) == [-3, 0, 7, 19]),
("reasoning", "Every raven is a bird. Some birds swim. Must every raven swim? Answer only yes or no.", lambda s: s.strip().lower().rstrip(".") == "no"),
]
results = []
for name, prompt, verify in cases:
answer = backend.complete([{"role": "user", "content": prompt}])
try:
passed = verify(answer) is True
except Exception:
passed = False
row = {"category": name, "prompt": prompt, "answer": answer, "passed": passed, **backend.last_metrics}
results.append(row)
print(json.dumps(row), flush=True)
with tempfile.TemporaryDirectory(prefix="nexora-agent-") as folder:
root = Path(folder)
(root / "note.txt").write_text("The test project uses a bounded queue.")
executor = Executor(Policy(folder, permissions=["READ", "WRITE"]))
task = "Read note.txt, then create summary.txt containing exactly: bounded queue. Finish after the file is written."
result = Agent(backend, executor, verifier=lambda: (root / "summary.txt").is_file() and (root / "summary.txt").read_text() == "bounded queue", max_steps=6).run(task)
n, successes = len(results), sum(r["passed"] for r in results)
report = {"model": args.model, "total_parameters": sum(p.numel() for p in backend.model.parameters()), "modified_weights": False, "device": "cpu", "cases": results,
"pass_at_1": successes/n, "wilson_95": wilson(successes, n),
"latency_seconds": percentiles([r["seconds"] for r in results]), "agent": result,
"limitations": ["Public hand-authored smoke tests; contamination unknown", "Four cases cannot rank models", "Agent file task is not repository-level coding evaluation", "Non-streaming latency only", "No private data used"]}
(out / f"{args.report}.json").write_text(json.dumps(report, indent=2, ensure_ascii=False), encoding="utf-8")
print(json.dumps({"pass_at_1": successes/n, "agent_status": result["status"]}))
if __name__ == "__main__":
main()
|