diff --git a/lib/crewai/src/crewai/llms/providers/openai/completion.py b/lib/crewai/src/crewai/llms/providers/openai/completion.py index c6b75b142..4e7079266 100644 --- a/lib/crewai/src/crewai/llms/providers/openai/completion.py +++ b/lib/crewai/src/crewai/llms/providers/openai/completion.py @@ -747,7 +747,7 @@ class OpenAICompletion(BaseLLM): ) @staticmethod - def _to_responses_input(message: LLMMessage) -> list[Any]: + def _to_responses_input(message: LLMMessage) -> list[dict[str, Any] | LLMMessage]: """Translate a chat-format message into Responses ``input`` items. Tool calling is expressed differently by the two APIs. Chat Completions @@ -765,18 +765,23 @@ class OpenAICompletion(BaseLLM): role = message.get("role") if role == "assistant" and message.get("tool_calls"): - items: list[Any] = [] + items: list[dict[str, Any] | LLMMessage] = [] content = message.get("content") if content: items.append({"role": "assistant", "content": content}) for call in message["tool_calls"]: function = call.get("function", {}) + args = function.get("arguments") + if args is None or args == "": + args = "{}" + elif not isinstance(args, str): + args = json.dumps(args) items.append( { "type": "function_call", "call_id": call.get("id", ""), "name": function.get("name", ""), - "arguments": function.get("arguments", "{}"), + "arguments": args, } ) return items @@ -807,7 +812,7 @@ class OpenAICompletion(BaseLLM): - Internally-tagged tool format (flat structure) """ instructions: str | None = self.instructions - input_messages: list[LLMMessage] = [] + input_messages: list[dict[str, Any] | LLMMessage] = [] for message in messages: if message.get("role") == "system": @@ -821,7 +826,7 @@ class OpenAICompletion(BaseLLM): input_messages.extend(self._to_responses_input(message)) # Prepend reasoning items for ZDR (zero-data-retention) chaining when configured - final_input: list[Any] = [] + final_input: list[dict[str, Any] | LLMMessage] = [] if self.auto_chain_reasoning and self._last_reasoning_items: final_input.extend(self._last_reasoning_items) final_input.extend(input_messages if input_messages else messages) diff --git a/lib/crewai/src/crewai/utilities/agent_utils.py b/lib/crewai/src/crewai/utilities/agent_utils.py index a3fbe43ce..b2ab2fc33 100644 --- a/lib/crewai/src/crewai/utilities/agent_utils.py +++ b/lib/crewai/src/crewai/utilities/agent_utils.py @@ -1404,9 +1404,9 @@ def is_tool_call_list(response: list[Any]) -> bool: if isinstance(first_item, dict) and "name" in first_item and "input" in first_item: return True # OpenAI Responses API style: {"id", "name", "arguments"}, with no nested - # "function" object and no "input". Without this the list isn't recognized as - # tool calls, so the executor hands it back verbatim and the agent returns raw - # tool-call JSON instead of running the tool and producing a final answer. + # "function" object and no "input". This intentionally accepts the same broad + # shape as the Bedrock check above; only provider paths that return lists reach + # this classifier. if ( isinstance(first_item, dict) and "name" in first_item diff --git a/lib/crewai/tests/llms/openai/test_openai.py b/lib/crewai/tests/llms/openai/test_openai.py index d5bc797d8..925a941cb 100644 --- a/lib/crewai/tests/llms/openai/test_openai.py +++ b/lib/crewai/tests/llms/openai/test_openai.py @@ -970,6 +970,170 @@ def test_openai_responses_api_with_system_message_extraction(): assert result.isupper() or "HELLO" in result.upper() +def test_openai_responses_api_converts_assistant_tool_calls_message(): + """Regression: assistant messages carrying tool_calls (Chat-Completions + shape) must become standalone function_call input items, since the + Responses API has no message shape for an assistant tool-call turn. + """ + llm = OpenAICompletion(model="gpt-4o-mini", api="responses") + + messages = [ + {"role": "user", "content": "Fetch https://example.com"}, + { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "id": "call_abc123", + "type": "function", + "function": { + "name": "fetch_page", + "arguments": '{"url": "https://example.com"}', + }, + } + ], + }, + ] + + params = llm._prepare_responses_params(messages) + + assert params["input"][0] == {"role": "user", "content": "Fetch https://example.com"} + assert params["input"][1] == { + "type": "function_call", + "call_id": "call_abc123", + "name": "fetch_page", + "arguments": '{"url": "https://example.com"}', + } + + +def test_openai_responses_api_preserves_assistant_content_with_tool_calls(): + """Assistant text must be retained when it accompanies tool calls.""" + llm = OpenAICompletion(model="gpt-4o-mini", api="responses") + + messages = [ + { + "role": "assistant", + "content": "I'll fetch that page now.", + "tool_calls": [ + { + "type": "function", + "id": "call_fetch_page", + "function": { + "name": "fetch_page", + "arguments": {"url": "https://example.com"}, + }, + } + ], + } + ] + + params = llm._prepare_responses_params(messages) + + assert params["input"][0] == { + "role": "assistant", + "content": "I'll fetch that page now.", + } + assert params["input"][1]["type"] == "function_call" + assert params["input"][1]["call_id"] == "call_fetch_page" + assert params["input"][1]["arguments"] == '{"url": "https://example.com"}' + + +def test_openai_responses_api_defaults_missing_tool_call_arguments(): + """Missing or empty tool-call arguments must become a valid JSON object.""" + llm = OpenAICompletion(model="gpt-4o-mini", api="responses") + + messages = [ + { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "id": "call_no_args", + "type": "function", + "function": {"name": "ping"}, + }, + { + "id": "call_empty_args", + "type": "function", + "function": {"name": "ping", "arguments": ""}, + }, + ], + } + ] + + params = llm._prepare_responses_params(messages) + + assert params["input"][0]["arguments"] == "{}" + assert params["input"][1]["arguments"] == "{}" + + +def test_openai_responses_api_converts_tool_result_message(): + """Regression: tool-role messages (Chat-Completions shape) must become + function_call_output input items for the Responses API. + """ + llm = OpenAICompletion(model="gpt-4o-mini", api="responses") + + messages = [ + { + "role": "tool", + "tool_call_id": "call_abc123", + "name": "fetch_page", + "content": "page text", + }, + ] + + params = llm._prepare_responses_params(messages) + + assert params["input"] == [ + { + "type": "function_call_output", + "call_id": "call_abc123", + "output": "page text", + } + ] + + +def test_openai_responses_api_multi_turn_tool_conversation_shape(): + """Regression: a full multi-turn tool-calling conversation (user -> + assistant tool_calls -> tool result) must convert entirely into valid + Responses API input items, with no leftover Chat-Completions-only keys + ("tool_calls", "tool_call_id") that the Responses API would reject. + """ + llm = OpenAICompletion(model="gpt-4o-mini", api="responses") + + messages = [ + {"role": "user", "content": "Fetch https://example.com"}, + { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "id": "call_abc123", + "type": "function", + "function": { + "name": "fetch_page", + "arguments": '{"url": "https://example.com"}', + }, + } + ], + }, + { + "role": "tool", + "tool_call_id": "call_abc123", + "name": "fetch_page", + "content": "page text", + }, + ] + + params = llm._prepare_responses_params(messages) + + for item in params["input"]: + assert "tool_calls" not in item + assert "tool_call_id" not in item + assert params["input"][1]["type"] == "function_call" + assert params["input"][2]["type"] == "function_call_output" + + @pytest.mark.vcr() def test_openai_responses_api_streaming(): """Test Responses API with streaming enabled.""" diff --git a/lib/crewai/tests/utilities/test_agent_utils.py b/lib/crewai/tests/utilities/test_agent_utils.py index 755befdbe..c4a77006e 100644 --- a/lib/crewai/tests/utilities/test_agent_utils.py +++ b/lib/crewai/tests/utilities/test_agent_utils.py @@ -25,6 +25,8 @@ from crewai.utilities.agent_utils import ( _split_messages_into_chunks, convert_tools_to_openai_schema, execute_single_native_tool_call, + extract_tool_call_info, + is_tool_call_list, NativeToolCallResult, parse_tool_call_args, summarize_messages, @@ -981,6 +983,88 @@ class TestParallelSummarizationVCR: assert "report.pdf" in summary_msg["files"] +class TestIsToolCallListResponsesApiShape: + """Regression tests: OpenAI Responses API tool-call dicts must be recognized. + + Responses API function_call output items are flat dicts shaped + {"id", "name", "arguments"} - no nested "function" key, and "arguments" + instead of Anthropic/Bedrock-style "input". + """ + + def test_responses_api_dict_is_recognized_as_tool_call(self) -> None: + response = [ + { + "id": "call_abc123", + "name": "fetch_page", + "arguments": '{"url": "https://example.com"}', + } + ] + assert is_tool_call_list(response) is True + + def test_plain_text_answer_not_misclassified(self) -> None: + assert is_tool_call_list(["just a string, not a tool call"]) is False + + def test_empty_list_returns_false(self) -> None: + assert is_tool_call_list([]) is False + + def test_chat_completions_style_still_recognized(self) -> None: + response = [{"function": {"name": "fetch_page", "arguments": "{}"}}] + assert is_tool_call_list(response) is True + + def test_bedrock_anthropic_style_still_recognized(self) -> None: + response = [{"name": "fetch_page", "input": {"url": "https://example.com"}}] + assert is_tool_call_list(response) is True + + +class TestExtractToolCallInfoResponsesApiShape: + """Regression tests: extract_tool_call_info must parse Responses API dicts.""" + + def test_responses_api_dict_extracts_real_arguments(self) -> None: + tool_call = { + "id": "call_abc123", + "name": "fetch_page", + "arguments": '{"url": "https://example.com"}', + } + result = extract_tool_call_info(tool_call) + assert result is not None + call_id, func_name, func_args = result + assert call_id == "call_abc123" + assert func_name == "fetch_page" + assert func_args == '{"url": "https://example.com"}' + + def test_responses_api_dict_does_not_return_empty_args(self) -> None: + tool_call = { + "id": "call_xyz", + "name": "fetch_page", + "arguments": '{"url": "https://example.com"}', + } + _, _, func_args = extract_tool_call_info(tool_call) + assert func_args != {} + + def test_bedrock_anthropic_style_still_uses_input(self) -> None: + tool_call = {"name": "fetch_page", "input": {"url": "https://example.com"}} + _, func_name, func_args = extract_tool_call_info(tool_call) + assert func_name == "fetch_page" + assert func_args == {"url": "https://example.com"} + + def test_chat_completions_style_still_uses_nested_function(self) -> None: + tool_call = { + "id": "call_1", + "function": {"name": "fetch_page", "arguments": "{}"}, + } + _, func_name, func_args = extract_tool_call_info(tool_call) + assert func_name == "fetch_page" + assert func_args == "{}" + + def test_non_dict_unrecognized_shape_returns_none(self) -> None: + assert extract_tool_call_info("just a string") is None + + def test_unrecognized_dict_shape_returns_empty_name_and_args(self) -> None: + call_id, func_name, func_args = extract_tool_call_info({"unrelated": "data"}) + assert func_name == "" + assert func_args == {} + + class TestParseToolCallArgs: """Unit tests for parse_tool_call_args."""