From 754d7323beb2fd042e33444a115ea2d5a47193f0 Mon Sep 17 00:00:00 2001 From: Rip&Tear <84775494+theCyberTech@users.noreply.github.com> Date: Fri, 14 Aug 2026 09:27:35 +0800 Subject: [PATCH] fix: native tool calls broken over OpenAI Responses API (#6515) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(agent): recognize OpenAI Responses API tool-call shape in native tool loop is_tool_call_list() and extract_tool_call_info() only recognized Chat-Completions-style ({"function": {...}}), Anthropic-style ({"name", "input"}), and Gemini-style tool-call shapes. The Responses API's function_call output items are flat dicts shaped {"id", "name", "arguments"} with no nested "function" key and no "input" key, so they matched none of the checks. This caused is_tool_call_list() to misclassify a genuine tool call as a plain text answer, so the native tool loop returned the raw tool-call list as the agent's final output instead of executing the tool. Even after recognizing the shape, extract_tool_call_info() would have passed an empty arguments dict, since it only read "input" for the dict fallback. Verified against LLM(api="responses") with tools attached: the agent now correctly executes the tool with the parsed arguments instead of returning the unexecuted tool-call JSON as its answer. Co-Authored-By: Claude Sonnet 5 * test(agent_utils): cover OpenAI Responses API tool-call shape Regression tests for is_tool_call_list() and extract_tool_call_info() against the Responses API's flat {"id", "name", "arguments"} dict shape, alongside existing Chat-Completions and Bedrock/Anthropic shapes to confirm no regression there. Confirmed these tests fail against the pre-fix version of agent_utils.py (3 failures matching exactly the Responses API cases) and pass against the fix in 37087b7e1. Co-Authored-By: Claude Sonnet 5 * fix(llms/openai): convert Chat-Completions tool messages to Responses API input items _prepare_responses_params() passed non-system messages straight through as the Responses API "input" array without converting them. That's fine for plain user/assistant text (matches the API's lenient "easy input message" shape), but Chat-Completions-style assistant messages carrying "tool_calls" and "tool"-role messages have no equivalent shape in the Responses API - it expects standalone "function_call" and "function_call_output" input items instead. Sending the raw Chat-Completions shapes gets rejected with a 400 (union-type validation failure against every Responses API input item variant). This broke every multi-turn tool-calling conversation over api="responses" that doesn't rely on auto_chain/previous_response_id (i.e. the common case: resending full history each turn instead of referencing server-side state). Added _convert_message_to_responses_input_items() to translate: - assistant + tool_calls -> one function_call item per call - tool role -> function_call_output item - everything else -> passed through unchanged Verified against a real multi-turn tool-calling run: the agent now completes the full conversation and returns the actual extracted answer instead of erroring on the second turn. Co-Authored-By: Claude Sonnet 5 * fix(llms/openai): fix mypy list-item type error in message conversion _convert_message_to_responses_input_items() was annotated to return list[dict[str, Any]], but the passthrough branch returns the LLMMessage argument unchanged. Lists are invariant in mypy, so a bare LLMMessage (TypedDict) isn't assignable into a list[dict[str, Any]] return - this was flagged by CI's type-checker job across all Python versions. Widened the return type (and the local list built in the tool_calls branch) to list[dict[str, Any] | LLMMessage], matching what the function actually returns. Confirmed with a local mypy run and the full openai/agent_utils test suites (176 passed). Co-Authored-By: Claude Sonnet 5 * fix: harden Responses API tool call conversion * style: format Responses API conversion * fix: upgrade nltk to 3.10.0 to resolve path traversal vulnerabilities Upgrades nltk from 3.9.4 to 3.10.0 which fixes three path traversal vulnerabilities (GHSA-qvv7-cg9c-w4x3, GHSA-fg7f-2386-8897, GHSA-xh95-f55m-82fw) that were causing the pip-audit CI job to fail. Also removes the now-obsolete PYSEC-2026-597 ignore entry from the vulnerability-scan workflow since the vulnerability is fixed in 3.10.0. * fix: bump aiohttp to >=3.14.2 and cryptography to >=50.0.0 to fix pip-audit vulns Co-authored-by: theCyberTech <84775494+theCyberTech@users.noreply.github.com> * fix: default empty Responses tool-call arguments to {} Missing or empty Chat Completions tool-call arguments were forwarded as an empty string, which is invalid JSON for Responses API function_call items and breaks parse_tool_call_args. Normalize to "{}" and cover the case in regression tests. Also assert a supplied tool-call id is preserved as call_id. Co-authored-by: Rip&Tear * Removing fallback from call_id * Removing fallback from call_id --------- Co-authored-by: Claude Sonnet 5 Co-authored-by: João Moura Co-authored-by: copilot-swe-agent[bot] <198982749+Copilot@users.noreply.github.com> Co-authored-by: Cursor Agent Co-authored-by: Rip&Tear Co-authored-by: ViditOstwal --- .../llms/providers/openai/completion.py | 15 +- .../src/crewai/utilities/agent_utils.py | 6 +- lib/crewai/tests/llms/openai/test_openai.py | 164 ++++++++++++++++++ .../tests/utilities/test_agent_utils.py | 84 +++++++++ 4 files changed, 261 insertions(+), 8 deletions(-) diff --git a/lib/crewai/src/crewai/llms/providers/openai/completion.py b/lib/crewai/src/crewai/llms/providers/openai/completion.py index c6b75b142..4e7079266 100644 --- a/lib/crewai/src/crewai/llms/providers/openai/completion.py +++ b/lib/crewai/src/crewai/llms/providers/openai/completion.py @@ -747,7 +747,7 @@ class OpenAICompletion(BaseLLM): ) @staticmethod - def _to_responses_input(message: LLMMessage) -> list[Any]: + def _to_responses_input(message: LLMMessage) -> list[dict[str, Any] | LLMMessage]: """Translate a chat-format message into Responses ``input`` items. Tool calling is expressed differently by the two APIs. Chat Completions @@ -765,18 +765,23 @@ class OpenAICompletion(BaseLLM): role = message.get("role") if role == "assistant" and message.get("tool_calls"): - items: list[Any] = [] + items: list[dict[str, Any] | LLMMessage] = [] content = message.get("content") if content: items.append({"role": "assistant", "content": content}) for call in message["tool_calls"]: function = call.get("function", {}) + args = function.get("arguments") + if args is None or args == "": + args = "{}" + elif not isinstance(args, str): + args = json.dumps(args) items.append( { "type": "function_call", "call_id": call.get("id", ""), "name": function.get("name", ""), - "arguments": function.get("arguments", "{}"), + "arguments": args, } ) return items @@ -807,7 +812,7 @@ class OpenAICompletion(BaseLLM): - Internally-tagged tool format (flat structure) """ instructions: str | None = self.instructions - input_messages: list[LLMMessage] = [] + input_messages: list[dict[str, Any] | LLMMessage] = [] for message in messages: if message.get("role") == "system": @@ -821,7 +826,7 @@ class OpenAICompletion(BaseLLM): input_messages.extend(self._to_responses_input(message)) # Prepend reasoning items for ZDR (zero-data-retention) chaining when configured - final_input: list[Any] = [] + final_input: list[dict[str, Any] | LLMMessage] = [] if self.auto_chain_reasoning and self._last_reasoning_items: final_input.extend(self._last_reasoning_items) final_input.extend(input_messages if input_messages else messages) diff --git a/lib/crewai/src/crewai/utilities/agent_utils.py b/lib/crewai/src/crewai/utilities/agent_utils.py index a3fbe43ce..b2ab2fc33 100644 --- a/lib/crewai/src/crewai/utilities/agent_utils.py +++ b/lib/crewai/src/crewai/utilities/agent_utils.py @@ -1404,9 +1404,9 @@ def is_tool_call_list(response: list[Any]) -> bool: if isinstance(first_item, dict) and "name" in first_item and "input" in first_item: return True # OpenAI Responses API style: {"id", "name", "arguments"}, with no nested - # "function" object and no "input". Without this the list isn't recognized as - # tool calls, so the executor hands it back verbatim and the agent returns raw - # tool-call JSON instead of running the tool and producing a final answer. + # "function" object and no "input". This intentionally accepts the same broad + # shape as the Bedrock check above; only provider paths that return lists reach + # this classifier. if ( isinstance(first_item, dict) and "name" in first_item diff --git a/lib/crewai/tests/llms/openai/test_openai.py b/lib/crewai/tests/llms/openai/test_openai.py index d5bc797d8..925a941cb 100644 --- a/lib/crewai/tests/llms/openai/test_openai.py +++ b/lib/crewai/tests/llms/openai/test_openai.py @@ -970,6 +970,170 @@ def test_openai_responses_api_with_system_message_extraction(): assert result.isupper() or "HELLO" in result.upper() +def test_openai_responses_api_converts_assistant_tool_calls_message(): + """Regression: assistant messages carrying tool_calls (Chat-Completions + shape) must become standalone function_call input items, since the + Responses API has no message shape for an assistant tool-call turn. + """ + llm = OpenAICompletion(model="gpt-4o-mini", api="responses") + + messages = [ + {"role": "user", "content": "Fetch https://example.com"}, + { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "id": "call_abc123", + "type": "function", + "function": { + "name": "fetch_page", + "arguments": '{"url": "https://example.com"}', + }, + } + ], + }, + ] + + params = llm._prepare_responses_params(messages) + + assert params["input"][0] == {"role": "user", "content": "Fetch https://example.com"} + assert params["input"][1] == { + "type": "function_call", + "call_id": "call_abc123", + "name": "fetch_page", + "arguments": '{"url": "https://example.com"}', + } + + +def test_openai_responses_api_preserves_assistant_content_with_tool_calls(): + """Assistant text must be retained when it accompanies tool calls.""" + llm = OpenAICompletion(model="gpt-4o-mini", api="responses") + + messages = [ + { + "role": "assistant", + "content": "I'll fetch that page now.", + "tool_calls": [ + { + "type": "function", + "id": "call_fetch_page", + "function": { + "name": "fetch_page", + "arguments": {"url": "https://example.com"}, + }, + } + ], + } + ] + + params = llm._prepare_responses_params(messages) + + assert params["input"][0] == { + "role": "assistant", + "content": "I'll fetch that page now.", + } + assert params["input"][1]["type"] == "function_call" + assert params["input"][1]["call_id"] == "call_fetch_page" + assert params["input"][1]["arguments"] == '{"url": "https://example.com"}' + + +def test_openai_responses_api_defaults_missing_tool_call_arguments(): + """Missing or empty tool-call arguments must become a valid JSON object.""" + llm = OpenAICompletion(model="gpt-4o-mini", api="responses") + + messages = [ + { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "id": "call_no_args", + "type": "function", + "function": {"name": "ping"}, + }, + { + "id": "call_empty_args", + "type": "function", + "function": {"name": "ping", "arguments": ""}, + }, + ], + } + ] + + params = llm._prepare_responses_params(messages) + + assert params["input"][0]["arguments"] == "{}" + assert params["input"][1]["arguments"] == "{}" + + +def test_openai_responses_api_converts_tool_result_message(): + """Regression: tool-role messages (Chat-Completions shape) must become + function_call_output input items for the Responses API. + """ + llm = OpenAICompletion(model="gpt-4o-mini", api="responses") + + messages = [ + { + "role": "tool", + "tool_call_id": "call_abc123", + "name": "fetch_page", + "content": "page text", + }, + ] + + params = llm._prepare_responses_params(messages) + + assert params["input"] == [ + { + "type": "function_call_output", + "call_id": "call_abc123", + "output": "page text", + } + ] + + +def test_openai_responses_api_multi_turn_tool_conversation_shape(): + """Regression: a full multi-turn tool-calling conversation (user -> + assistant tool_calls -> tool result) must convert entirely into valid + Responses API input items, with no leftover Chat-Completions-only keys + ("tool_calls", "tool_call_id") that the Responses API would reject. + """ + llm = OpenAICompletion(model="gpt-4o-mini", api="responses") + + messages = [ + {"role": "user", "content": "Fetch https://example.com"}, + { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "id": "call_abc123", + "type": "function", + "function": { + "name": "fetch_page", + "arguments": '{"url": "https://example.com"}', + }, + } + ], + }, + { + "role": "tool", + "tool_call_id": "call_abc123", + "name": "fetch_page", + "content": "page text", + }, + ] + + params = llm._prepare_responses_params(messages) + + for item in params["input"]: + assert "tool_calls" not in item + assert "tool_call_id" not in item + assert params["input"][1]["type"] == "function_call" + assert params["input"][2]["type"] == "function_call_output" + + @pytest.mark.vcr() def test_openai_responses_api_streaming(): """Test Responses API with streaming enabled.""" diff --git a/lib/crewai/tests/utilities/test_agent_utils.py b/lib/crewai/tests/utilities/test_agent_utils.py index 755befdbe..c4a77006e 100644 --- a/lib/crewai/tests/utilities/test_agent_utils.py +++ b/lib/crewai/tests/utilities/test_agent_utils.py @@ -25,6 +25,8 @@ from crewai.utilities.agent_utils import ( _split_messages_into_chunks, convert_tools_to_openai_schema, execute_single_native_tool_call, + extract_tool_call_info, + is_tool_call_list, NativeToolCallResult, parse_tool_call_args, summarize_messages, @@ -981,6 +983,88 @@ class TestParallelSummarizationVCR: assert "report.pdf" in summary_msg["files"] +class TestIsToolCallListResponsesApiShape: + """Regression tests: OpenAI Responses API tool-call dicts must be recognized. + + Responses API function_call output items are flat dicts shaped + {"id", "name", "arguments"} - no nested "function" key, and "arguments" + instead of Anthropic/Bedrock-style "input". + """ + + def test_responses_api_dict_is_recognized_as_tool_call(self) -> None: + response = [ + { + "id": "call_abc123", + "name": "fetch_page", + "arguments": '{"url": "https://example.com"}', + } + ] + assert is_tool_call_list(response) is True + + def test_plain_text_answer_not_misclassified(self) -> None: + assert is_tool_call_list(["just a string, not a tool call"]) is False + + def test_empty_list_returns_false(self) -> None: + assert is_tool_call_list([]) is False + + def test_chat_completions_style_still_recognized(self) -> None: + response = [{"function": {"name": "fetch_page", "arguments": "{}"}}] + assert is_tool_call_list(response) is True + + def test_bedrock_anthropic_style_still_recognized(self) -> None: + response = [{"name": "fetch_page", "input": {"url": "https://example.com"}}] + assert is_tool_call_list(response) is True + + +class TestExtractToolCallInfoResponsesApiShape: + """Regression tests: extract_tool_call_info must parse Responses API dicts.""" + + def test_responses_api_dict_extracts_real_arguments(self) -> None: + tool_call = { + "id": "call_abc123", + "name": "fetch_page", + "arguments": '{"url": "https://example.com"}', + } + result = extract_tool_call_info(tool_call) + assert result is not None + call_id, func_name, func_args = result + assert call_id == "call_abc123" + assert func_name == "fetch_page" + assert func_args == '{"url": "https://example.com"}' + + def test_responses_api_dict_does_not_return_empty_args(self) -> None: + tool_call = { + "id": "call_xyz", + "name": "fetch_page", + "arguments": '{"url": "https://example.com"}', + } + _, _, func_args = extract_tool_call_info(tool_call) + assert func_args != {} + + def test_bedrock_anthropic_style_still_uses_input(self) -> None: + tool_call = {"name": "fetch_page", "input": {"url": "https://example.com"}} + _, func_name, func_args = extract_tool_call_info(tool_call) + assert func_name == "fetch_page" + assert func_args == {"url": "https://example.com"} + + def test_chat_completions_style_still_uses_nested_function(self) -> None: + tool_call = { + "id": "call_1", + "function": {"name": "fetch_page", "arguments": "{}"}, + } + _, func_name, func_args = extract_tool_call_info(tool_call) + assert func_name == "fetch_page" + assert func_args == "{}" + + def test_non_dict_unrecognized_shape_returns_none(self) -> None: + assert extract_tool_call_info("just a string") is None + + def test_unrecognized_dict_shape_returns_empty_name_and_args(self) -> None: + call_id, func_name, func_args = extract_tool_call_info({"unrelated": "data"}) + assert func_name == "" + assert func_args == {} + + class TestParseToolCallArgs: """Unit tests for parse_tool_call_args."""