"""Tests for the agent loop, run end to end with the mock adapter. These are real tests of the real code path: the prompt is built by the actual DeepSeek-V4 encoder, the completion is parsed by the actual parser, and the tool is really executed. Only the model weights are replaced. """ import pytest from cortex_ai.adapters import MockAdapter from cortex_ai.config import EngineConfig from cortex_ai.engine import CortexAgent, build_answer, build_tool_call from cortex_ai.engine.agent import _normalize_call from cortex_ai.tools import default_registry @pytest.fixture def agent(): return CortexAgent( MockAdapter(), default_registry(), EngineConfig(max_tool_rounds=4, thinking_mode="chat"), system_prompt="Tu es CORTEX AI.", ) def test_plain_answer_needs_one_round(agent): turn = agent.run([{"role": "user", "content": "Bonjour"}]) assert turn.rounds == 1 assert turn.used_tools is False assert turn.content def test_tool_is_called_and_result_reaches_the_answer(agent): turn = agent.run([{"role": "user", "content": "Combien font 12 * 8 ?"}]) assert turn.rounds == 2 assert len(turn.tool_calls) == 1 call = turn.tool_calls[0] assert call.name == "calculate" assert call.arguments == {"expression": "12 * 8"} assert call.result == "96" assert call.ok is True assert "outil" in turn.content def test_tool_result_is_present_in_the_second_prompt(agent): agent.run([{"role": "user", "content": "Combien font 9 * 9 ?"}]) # The adapter records every prompt it was given. assert len(agent.adapter.calls) == 2 assert "81" in agent.adapter.calls[1] def test_thinking_mode_produces_reasoning(): agent = CortexAgent( MockAdapter(), default_registry(), EngineConfig(max_tool_rounds=4, thinking_mode="thinking"), system_prompt="Tu es CORTEX AI.", ) turn = agent.run([{"role": "user", "content": "Combien font 5 * 5 ?"}]) assert turn.reasoning assert turn.tool_calls[0].result == "25" def test_max_tool_rounds_is_respected(): """A model that always calls a tool must not loop forever.""" def always_calls(prompt): return build_tool_call("current_time", {}, reasoning="") agent = CortexAgent( MockAdapter(responder=always_calls), default_registry(), EngineConfig(max_tool_rounds=3, thinking_mode="chat"), system_prompt="sys", ) turn = agent.run([{"role": "user", "content": "boucle"}]) assert turn.rounds == 3 assert len(turn.tool_calls) == 3 assert "Limite" in turn.content def test_unknown_tool_is_reported_without_crashing(agent): turn = agent.run([{"role": "user", "content": "Combien font 1 * 1 ?"}]) assert turn.rounds >= 1 # the loop completed def calls_missing_tool(prompt): return build_tool_call("outil_inexistant", {"x": "1"}, reasoning="") agent2 = CortexAgent( MockAdapter(responder=calls_missing_tool), default_registry(), EngineConfig(max_tool_rounds=2, thinking_mode="chat"), system_prompt="sys", ) turn2 = agent2.run([{"role": "user", "content": "test"}]) assert turn2.tool_calls[0].ok is False assert "unknown tool" in turn2.tool_calls[0].result def test_normalize_call_accepts_both_shapes(): openai_shape = { "type": "function", "function": {"name": "calculate", "arguments": '{"expression": "1+1"}'}, } flat_shape = {"name": "calculate", "arguments": {"expression": "1+1"}} assert _normalize_call(openai_shape) == ("calculate", '{"expression": "1+1"}') assert _normalize_call(flat_shape) == ("calculate", {"expression": "1+1"}) def test_build_answer_has_no_opening_thinking_token(): """The opening thinking token belongs to the prompt, never to the completion.""" import encoding_dsv4 as enc text = build_answer("ok", reasoning="raisonnement") assert text.startswith("raisonnement") assert enc.thinking_start_token not in text assert enc.thinking_end_token in text def test_build_tool_call_parses_back_to_the_same_arguments(): import encoding_dsv4 as enc text = build_tool_call("calculate", {"expression": "2*21"}) parsed = enc.parse_message_from_completion_text(text, thinking_mode="chat") assert parsed["tool_calls"][0]["function"]["name"] == "calculate" assert "2*21" in parsed["tool_calls"][0]["function"]["arguments"] def test_empty_tool_registry_means_no_tool_instructions(): agent = CortexAgent( MockAdapter(), None, EngineConfig(thinking_mode="chat"), system_prompt="sys" ) turn = agent.run([{"role": "user", "content": "Combien font 3 * 3 ?"}]) assert turn.used_tools is False