File size: 2,699 Bytes
4d97285
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
"""Drive the agent loop against the fakes and print every event.

    python tests/test_agent_loop.py

Exits non-zero if the loop does not reach a final answer, so it doubles as a
smoke test before a deploy.
"""

from __future__ import annotations

import asyncio
import os
import sys
import time
from pathlib import Path

sys.path.insert(0, str(Path(__file__).resolve().parent.parent))

from tests.fake_servers import LLM_PORT, MCP_PORT, start_background  # noqa: E402

os.environ.update(
    {
        "LLM_BASE_URL": f"http://127.0.0.1:{LLM_PORT}/v1",
        "LLM_MODEL": "fake-model",
        "LLM_API_KEY": "test-key",
        "VFRPLAN_MCP_URL": f"http://127.0.0.1:{MCP_PORT}/mcp",
        "VFRPLAN_MCP_TOKEN": "",
        "APP_ADMIN_TOKEN": "admin-secret",
    }
)

from app.agent import run_agent  # noqa: E402
from app.config import AppConfig  # noqa: E402


async def main() -> int:
    start_background()
    time.sleep(2.0)  # let both servers bind

    config = AppConfig()
    print(f"LLM  : {config.llm.chat_url}  model={config.llm.model}")
    print(f"MCP  : {config.mcp.url}\n")

    saw_final = False
    saw_tool_call = False
    saw_tool_result = False
    saw_reasoning = False

    async for event in run_agent(config, "What is the current METAR at LEBL?"):
        kind = event["type"]
        if kind == "status":
            print(f"  · {event['message']}")
        elif kind == "tools_loaded":
            print(f"  ✓ MCP tools ({event['count']}): {', '.join(event['names'])}")
        elif kind == "reasoning":
            saw_reasoning = True
            print(f"  🧠 reasoning[{event['step']}]: {event['content']}")
        elif kind == "thinking":
            print(f"  💭 thinking[{event['step']}]: {event['content']}")
        elif kind == "tool_call":
            saw_tool_call = True
            print(f"  🔧 call[{event['step']}]: {event['name']}({event['arguments']})")
        elif kind == "tool_result":
            saw_tool_result = True
            print(f"  📄 result: ok={event['ok']} {event['content'][:120]}")
        elif kind == "final":
            saw_final = True
            print(f"\n  ✅ FINAL:\n{event['content']}\n")
        elif kind == "error":
            print(f"\n  ❌ ERROR: {event['message']}\n")

    checks = {
        "reasoning surfaced": saw_reasoning,
        "tool called": saw_tool_call,
        "tool result fed back": saw_tool_result,
        "final answer produced": saw_final,
    }
    print("-" * 60)
    for label, passed in checks.items():
        print(f"  [{'PASS' if passed else 'FAIL'}] {label}")

    return 0 if all(checks.values()) else 1


if __name__ == "__main__":
    sys.exit(asyncio.run(main()))