File size: 4,540 Bytes
12496fc
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
"""Measured baseline, tokenizer and agent experiments. All failures are retained."""
from pathlib import Path
from dataclasses import asdict
import json
import sys
import tempfile
import time
import argparse
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from nexora.inference import HFBackend
from nexora.tokenizer import ByteTokenizer
from nexora.agent import Agent
from nexora.tools import Executor, Policy
from nexora.evaluation import wilson, percentiles

SAMPLES = {
    "english": "A reliable assistant inspects evidence before claiming that a task succeeded.",
    "hindi": "सहायक को कार्य पूरा होने का दावा करने से पहले प्रमाण की जाँच करनी चाहिए।",
    "hinglish": "Pehle code run karo aur tests ka output check karo, phir result batao.",
    "python": "def total(values):\n    return sum(x for x in values if x is not None)\n",
    "json": '{"tool":"filesystem.read","arguments":{"path":"src/main.py"}}',
    "math": "For x ∈ ℝ, x² + 2x + 1 = (x + 1)²; ∑ᵢ xᵢ / n.",
    "shell": "git diff --stat && python -m pytest -q",
    "xml_url": '<source href="https://example.org/api?q=a%20b">Unicode: λ</source>',
}


def main():
    parser = argparse.ArgumentParser()
    parser.add_argument("--model", default=".cache/Qwen3-0.6B")
    parser.add_argument("--report", default="baseline")
    args = parser.parse_args()
    out = Path("reports")
    out.mkdir(exist_ok=True)
    backend = HFBackend(args.model, max_new_tokens=96)
    byte_report = ByteTokenizer().benchmark(SAMPLES)
    bpe = {}
    for name, text in SAMPLES.items():
        ids = backend.tokenizer.encode(text, add_special_tokens=False)
        bpe[name] = {"tokens": len(ids), "characters_per_token": len(text)/len(ids), "roundtrip": backend.tokenizer.decode(ids) == text}
    (out / f"tokenizer-{args.report}.json").write_text(json.dumps({"byte": byte_report, "qwen_bpe": bpe, "limitation": "Eight public examples, not a representative benchmark or vocabulary-size study"}, indent=2, ensure_ascii=False), encoding="utf-8")
    cases = [
        ("arithmetic", "Return only the integer: 17 * 23.", lambda s: s.strip() == "391"),
        ("json", 'Return only a JSON object with key "ready" and boolean value true.', lambda s: json.loads(s) == {"ready": True}),
        ("ordering", "Sort these integers ascending. Return only a JSON array: 19, -3, 7, 0.", lambda s: json.loads(s) == [-3, 0, 7, 19]),
        ("reasoning", "Every raven is a bird. Some birds swim. Must every raven swim? Answer only yes or no.", lambda s: s.strip().lower().rstrip(".") == "no"),
    ]
    results = []
    for name, prompt, verify in cases:
        answer = backend.complete([{"role": "user", "content": prompt}])
        try:
            passed = verify(answer) is True
        except Exception:
            passed = False
        row = {"category": name, "prompt": prompt, "answer": answer, "passed": passed, **backend.last_metrics}
        results.append(row)
        print(json.dumps(row), flush=True)
    with tempfile.TemporaryDirectory(prefix="nexora-agent-") as folder:
        root = Path(folder)
        (root / "note.txt").write_text("The test project uses a bounded queue.")
        executor = Executor(Policy(folder, permissions=["READ", "WRITE"]))
        task = "Read note.txt, then create summary.txt containing exactly: bounded queue. Finish after the file is written."
        result = Agent(backend, executor, verifier=lambda: (root / "summary.txt").is_file() and (root / "summary.txt").read_text() == "bounded queue", max_steps=6).run(task)
    n, successes = len(results), sum(r["passed"] for r in results)
    report = {"model": args.model, "total_parameters": sum(p.numel() for p in backend.model.parameters()), "modified_weights": False, "device": "cpu", "cases": results,
              "pass_at_1": successes/n, "wilson_95": wilson(successes, n),
              "latency_seconds": percentiles([r["seconds"] for r in results]), "agent": result,
              "limitations": ["Public hand-authored smoke tests; contamination unknown", "Four cases cannot rank models", "Agent file task is not repository-level coding evaluation", "Non-streaming latency only", "No private data used"]}
    (out / f"{args.report}.json").write_text(json.dumps(report, indent=2, ensure_ascii=False), encoding="utf-8")
    print(json.dumps({"pass_at_1": successes/n, "agent_status": result["status"]}))


if __name__ == "__main__":
    main()