"""Measured baseline, tokenizer and agent experiments. All failures are retained.""" from pathlib import Path from dataclasses import asdict import json import sys import tempfile import time import argparse sys.path.insert(0, str(Path(__file__).resolve().parents[1])) from nexora.inference import HFBackend from nexora.tokenizer import ByteTokenizer from nexora.agent import Agent from nexora.tools import Executor, Policy from nexora.evaluation import wilson, percentiles SAMPLES = { "english": "A reliable assistant inspects evidence before claiming that a task succeeded.", "hindi": "सहायक को कार्य पूरा होने का दावा करने से पहले प्रमाण की जाँच करनी चाहिए।", "hinglish": "Pehle code run karo aur tests ka output check karo, phir result batao.", "python": "def total(values):\n return sum(x for x in values if x is not None)\n", "json": '{"tool":"filesystem.read","arguments":{"path":"src/main.py"}}', "math": "For x ∈ ℝ, x² + 2x + 1 = (x + 1)²; ∑ᵢ xᵢ / n.", "shell": "git diff --stat && python -m pytest -q", "xml_url": 'Unicode: λ', } def main(): parser = argparse.ArgumentParser() parser.add_argument("--model", default=".cache/Qwen3-0.6B") parser.add_argument("--report", default="baseline") args = parser.parse_args() out = Path("reports") out.mkdir(exist_ok=True) backend = HFBackend(args.model, max_new_tokens=96) byte_report = ByteTokenizer().benchmark(SAMPLES) bpe = {} for name, text in SAMPLES.items(): ids = backend.tokenizer.encode(text, add_special_tokens=False) bpe[name] = {"tokens": len(ids), "characters_per_token": len(text)/len(ids), "roundtrip": backend.tokenizer.decode(ids) == text} (out / f"tokenizer-{args.report}.json").write_text(json.dumps({"byte": byte_report, "qwen_bpe": bpe, "limitation": "Eight public examples, not a representative benchmark or vocabulary-size study"}, indent=2, ensure_ascii=False), encoding="utf-8") cases = [ ("arithmetic", "Return only the integer: 17 * 23.", lambda s: s.strip() == "391"), ("json", 'Return only a JSON object with key "ready" and boolean value true.', lambda s: json.loads(s) == {"ready": True}), ("ordering", "Sort these integers ascending. Return only a JSON array: 19, -3, 7, 0.", lambda s: json.loads(s) == [-3, 0, 7, 19]), ("reasoning", "Every raven is a bird. Some birds swim. Must every raven swim? Answer only yes or no.", lambda s: s.strip().lower().rstrip(".") == "no"), ] results = [] for name, prompt, verify in cases: answer = backend.complete([{"role": "user", "content": prompt}]) try: passed = verify(answer) is True except Exception: passed = False row = {"category": name, "prompt": prompt, "answer": answer, "passed": passed, **backend.last_metrics} results.append(row) print(json.dumps(row), flush=True) with tempfile.TemporaryDirectory(prefix="nexora-agent-") as folder: root = Path(folder) (root / "note.txt").write_text("The test project uses a bounded queue.") executor = Executor(Policy(folder, permissions=["READ", "WRITE"])) task = "Read note.txt, then create summary.txt containing exactly: bounded queue. Finish after the file is written." result = Agent(backend, executor, verifier=lambda: (root / "summary.txt").is_file() and (root / "summary.txt").read_text() == "bounded queue", max_steps=6).run(task) n, successes = len(results), sum(r["passed"] for r in results) report = {"model": args.model, "total_parameters": sum(p.numel() for p in backend.model.parameters()), "modified_weights": False, "device": "cpu", "cases": results, "pass_at_1": successes/n, "wilson_95": wilson(successes, n), "latency_seconds": percentiles([r["seconds"] for r in results]), "agent": result, "limitations": ["Public hand-authored smoke tests; contamination unknown", "Four cases cannot rank models", "Agent file task is not repository-level coding evaluation", "Non-streaming latency only", "No private data used"]} (out / f"{args.report}.json").write_text(json.dumps(report, indent=2, ensure_ascii=False), encoding="utf-8") print(json.dumps({"pass_at_1": successes/n, "agent_status": result["status"]})) if __name__ == "__main__": main()