cortex.6.sol / tests /test_agent.py
asdfasdfqrqwer's picture
feat(cortex-ai): agent engine, tool registry, OpenAI-compatible API, fine-tuning pipeline
c63bc31
Raw History Blame Contribute Delete
4.73 kB
"""Tests for the agent loop, run end to end with the mock adapter.
These are real tests of the real code path: the prompt is built by the actual
DeepSeek-V4 encoder, the completion is parsed by the actual parser, and the
tool is really executed. Only the model weights are replaced.
"""
import pytest
from cortex_ai.adapters import MockAdapter
from cortex_ai.config import EngineConfig
from cortex_ai.engine import CortexAgent, build_answer, build_tool_call
from cortex_ai.engine.agent import _normalize_call
from cortex_ai.tools import default_registry
@pytest.fixture
def agent():
return CortexAgent(
MockAdapter(),
default_registry(),
EngineConfig(max_tool_rounds=4, thinking_mode="chat"),
system_prompt="Tu es CORTEX AI.",
)
def test_plain_answer_needs_one_round(agent):
turn = agent.run([{"role": "user", "content": "Bonjour"}])
assert turn.rounds == 1
assert turn.used_tools is False
assert turn.content
def test_tool_is_called_and_result_reaches_the_answer(agent):
turn = agent.run([{"role": "user", "content": "Combien font 12 * 8 ?"}])
assert turn.rounds == 2
assert len(turn.tool_calls) == 1
call = turn.tool_calls[0]
assert call.name == "calculate"
assert call.arguments == {"expression": "12 * 8"}
assert call.result == "96"
assert call.ok is True
assert "outil" in turn.content
def test_tool_result_is_present_in_the_second_prompt(agent):
agent.run([{"role": "user", "content": "Combien font 9 * 9 ?"}])
# The adapter records every prompt it was given.
assert len(agent.adapter.calls) == 2
assert "81" in agent.adapter.calls[1]
def test_thinking_mode_produces_reasoning():
agent = CortexAgent(
MockAdapter(),
default_registry(),
EngineConfig(max_tool_rounds=4, thinking_mode="thinking"),
system_prompt="Tu es CORTEX AI.",
)
turn = agent.run([{"role": "user", "content": "Combien font 5 * 5 ?"}])
assert turn.reasoning
assert turn.tool_calls[0].result == "25"
def test_max_tool_rounds_is_respected():
"""A model that always calls a tool must not loop forever."""
def always_calls(prompt):
return build_tool_call("current_time", {}, reasoning="")
agent = CortexAgent(
MockAdapter(responder=always_calls),
default_registry(),
EngineConfig(max_tool_rounds=3, thinking_mode="chat"),
system_prompt="sys",
)
turn = agent.run([{"role": "user", "content": "boucle"}])
assert turn.rounds == 3
assert len(turn.tool_calls) == 3
assert "Limite" in turn.content
def test_unknown_tool_is_reported_without_crashing(agent):
turn = agent.run([{"role": "user", "content": "Combien font 1 * 1 ?"}])
assert turn.rounds >= 1 # the loop completed
def calls_missing_tool(prompt):
return build_tool_call("outil_inexistant", {"x": "1"}, reasoning="")
agent2 = CortexAgent(
MockAdapter(responder=calls_missing_tool),
default_registry(),
EngineConfig(max_tool_rounds=2, thinking_mode="chat"),
system_prompt="sys",
)
turn2 = agent2.run([{"role": "user", "content": "test"}])
assert turn2.tool_calls[0].ok is False
assert "unknown tool" in turn2.tool_calls[0].result
def test_normalize_call_accepts_both_shapes():
openai_shape = {
"type": "function",
"function": {"name": "calculate", "arguments": '{"expression": "1+1"}'},
}
flat_shape = {"name": "calculate", "arguments": {"expression": "1+1"}}
assert _normalize_call(openai_shape) == ("calculate", '{"expression": "1+1"}')
assert _normalize_call(flat_shape) == ("calculate", {"expression": "1+1"})
def test_build_answer_has_no_opening_thinking_token():
"""The opening thinking token belongs to the prompt, never to the completion."""
import encoding_dsv4 as enc
text = build_answer("ok", reasoning="raisonnement")
assert text.startswith("raisonnement")
assert enc.thinking_start_token not in text
assert enc.thinking_end_token in text
def test_build_tool_call_parses_back_to_the_same_arguments():
import encoding_dsv4 as enc
text = build_tool_call("calculate", {"expression": "2*21"})
parsed = enc.parse_message_from_completion_text(text, thinking_mode="chat")
assert parsed["tool_calls"][0]["function"]["name"] == "calculate"
assert "2*21" in parsed["tool_calls"][0]["function"]["arguments"]
def test_empty_tool_registry_means_no_tool_instructions():
agent = CortexAgent(
MockAdapter(), None, EngineConfig(thinking_mode="chat"), system_prompt="sys"
)
turn = agent.run([{"role": "user", "content": "Combien font 3 * 3 ?"}])
assert turn.used_tools is False