From 675f43f9497ef2df632d801401fa5f9e58dd60b5 Mon Sep 17 00:00:00 2001 From: Alexander Zaikman Date: Wed, 9 Sep 2026 12:49:28 +0300 Subject: [PATCH 1/4] fix: retry CrewAI's first-call empty completion before failing (INT-1427) CrewAI's native OpenAI tool-calling path can return an empty completion (finish_reason='stop', no content, no tool_calls) on a turn's very first LLM call, before any tool has run. That's indistinguishable from a real provider outage, so the adapter always failed the delivery outright -- even though nothing ran yet this turn, so a single retry duplicates nothing and often recovers the turn. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01Fw8AZwnoa3Fen6HBhHEQ7H --- src/band/adapters/crewai.py | 46 ++++++++++++++++++++++++- tests/adapters/test_crewai_adapter.py | 48 +++++++++++++++++++++++++++ 2 files changed, 93 insertions(+), 1 deletion(-) diff --git a/src/band/adapters/crewai.py b/src/band/adapters/crewai.py index f49924fe1..74a69b32e 100644 --- a/src/band/adapters/crewai.py +++ b/src/band/adapters/crewai.py @@ -56,6 +56,11 @@ # is the only discriminator. One definition, matched here and faked in tests. EMPTY_LLM_RESPONSE_MARKER = "Invalid response from LLM call" +# Bound on the in-process retry for a turn's very first LLM call coming back +# empty (see _kickoff_with_empty_response_retry) -- one immediate retry, no +# backoff, since nothing has run yet this turn and so nothing is duplicated. +_EMPTY_RESPONSE_FIRST_CALL_RETRIES = 1 + def _is_empty_llm_response(exc: Exception) -> bool: """Whether ``exc`` is CrewAI reporting that an LLM call came back empty. @@ -405,7 +410,9 @@ async def _process_message( # "already handled" / "act on this now" markers, matches what # CrewAI actually does with the input either way. prompt = "\n\n".join(sections) - result = await self._crewai_agent.kickoff_async(prompt) + result = await self._kickoff_with_empty_response_retry( + self._crewai_agent, prompt, reply_tracker, room_id + ) except Exception as e: # An empty response is benign only once some tool ran this turn -- @@ -461,6 +468,43 @@ async def on_cleanup(self, room_id: str) -> None: del self._message_history[room_id] logger.debug("Room %s: Cleaned up CrewAI session", room_id) + async def _kickoff_with_empty_response_retry( + self, + agent: "CrewAIAgent", + prompt: str, + reply_tracker: ReplyTracker, + room_id: str, + ) -> Any: + """Retry a first-call empty completion once before giving up. + + CrewAI raises the identical ValueError whether the model correctly has + nothing left to say (fine once a tool has run -- handled by the + caller) or the turn's very first call came back empty. The second case + has run no tool yet, so there is nothing to duplicate: one immediate + retry absorbs a single-call fluke instead of failing the whole + delivery and leaving CrewAI to improvise on a cold redelivery. + """ + attempt = 0 + while True: + try: + return await agent.kickoff_async(prompt) + except Exception as e: + attempt += 1 + retryable = ( + attempt <= _EMPTY_RESPONSE_FIRST_CALL_RETRIES + and _is_empty_llm_response(e) + and not reply_tracker.any_tool_ran + ) + if not retryable: + raise + logger.info( + "Room %s: CrewAI's first LLM call came back empty before " + "any tool ran; retrying (%s/%s)", + room_id, + attempt, + _EMPTY_RESPONSE_FIRST_CALL_RETRIES, + ) + async def _report_error(self, tools: AgentToolsProtocol, error: str) -> None: """Send error event (best effort).""" try: diff --git a/tests/adapters/test_crewai_adapter.py b/tests/adapters/test_crewai_adapter.py index 92c53642a..588ce0d55 100644 --- a/tests/adapters/test_crewai_adapter.py +++ b/tests/adapters/test_crewai_adapter.py @@ -741,6 +741,54 @@ async def test_empty_answer_with_no_tool_call_still_raises( ) mock_tools.send_event.assert_awaited_once() + assert mock_crewai_agent.kickoff_async.call_count == 2 + + @pytest.mark.asyncio + async def test_empty_first_call_recovers_on_retry( + self, CrewAIAdapter, sample_message, mock_tools, mock_crewai_agent + ): + """A turn's first LLM call coming back empty is retried once before + giving up -- a successful retry must complete the turn normally. + + Nothing ran before the empty completion, so there is no side effect + for the retry to duplicate; the retry either behaves exactly like the + first attempt would have or, as here, recovers and replies. + """ + module = importlib.import_module("band.adapters.crewai") + + mock_result = MagicMock() + mock_result.raw = "Hello! I'm here to help." + + calls = 0 + + async def _kickoff(_prompt): + nonlocal calls + calls += 1 + if calls == 1: + raise ValueError(EMPTY_LLM_RESPONSE_ERROR) + tracker = module._reply_tracker_var.get() + if tracker is not None: + tracker.replied = True + return mock_result + + mock_crewai_agent.kickoff_async = AsyncMock(side_effect=_kickoff) + + adapter = CrewAIAdapter() + await adapter.on_started("TestBot", "Test bot") + adapter._crewai_agent = mock_crewai_agent + + await adapter.on_message( + msg=sample_message, + tools=mock_tools, + history=[], + participants_msg=None, + contacts_msg=None, + is_session_bootstrap=True, + room_id="room-123", + ) + + assert error_events(mock_tools) == [] + assert mock_crewai_agent.kickoff_async.call_count == 2 @pytest.mark.asyncio async def test_no_missing_reply_error_after_clean_tool_only_return( From da383aa707199b32a6763c5c8c82abebad1e24e2 Mon Sep 17 00:00:00 2001 From: Alexander Zaikman Date: Wed, 9 Sep 2026 12:55:00 +0300 Subject: [PATCH 2/4] fix: use a top-level import instead of importlib in the new retry test band.adapters.crewai is already safely imported at module scope in this file (it defers its own crewai imports to TYPE_CHECKING/function-local), so importlib.import_module inside the test just re-fetched the same cached module object for no reason. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01Fw8AZwnoa3Fen6HBhHEQ7H --- tests/adapters/test_crewai_adapter.py | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/tests/adapters/test_crewai_adapter.py b/tests/adapters/test_crewai_adapter.py index 588ce0d55..520dc65de 100644 --- a/tests/adapters/test_crewai_adapter.py +++ b/tests/adapters/test_crewai_adapter.py @@ -24,6 +24,7 @@ import pytest from pydantic import BaseModel, Field +import band.adapters.crewai as crewai_adapter from band.adapters.crewai import EMPTY_LLM_RESPONSE_MARKER from band.core.types import Capability, Emit, PlatformMessage from band.runtime.prompts import render_system_prompt @@ -754,8 +755,6 @@ async def test_empty_first_call_recovers_on_retry( for the retry to duplicate; the retry either behaves exactly like the first attempt would have or, as here, recovers and replies. """ - module = importlib.import_module("band.adapters.crewai") - mock_result = MagicMock() mock_result.raw = "Hello! I'm here to help." @@ -766,7 +765,7 @@ async def _kickoff(_prompt): calls += 1 if calls == 1: raise ValueError(EMPTY_LLM_RESPONSE_ERROR) - tracker = module._reply_tracker_var.get() + tracker = crewai_adapter._reply_tracker_var.get() if tracker is not None: tracker.replied = True return mock_result From 3327f2732d1f42e3936d9d0196db4203349fc848 Mon Sep 17 00:00:00 2001 From: Alexander Zaikman Date: Wed, 9 Sep 2026 13:09:47 +0300 Subject: [PATCH 3/4] fix: simplify the empty-response retry and disambiguate its failure logs Code review findings on PR #624: - The retry loop's while/attempt-counter machinery was sized for a general N-retry problem the ticket doesn't have (bound is fixed at 1) -- collapsed to a single inline try/except. The now-unused _EMPTY_RESPONSE_FIRST_CALL_RETRIES constant and the comment restating the docstring's rationale go with it. - The retry's own failure had no distinct log line, so an operator filtering at WARNING+ couldn't tell a delivery failed after one attempt from failing after the retry also came back empty -- added a warning log at the point the retry itself fails. - Test assertions on the resulting call count no longer reference a governing constant (removed above), so they stay literal with a short comment instead of drifting toward a fake single source of truth. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01Fw8AZwnoa3Fen6HBhHEQ7H --- src/band/adapters/crewai.py | 35 +++++++++++---------------- tests/adapters/test_crewai_adapter.py | 3 +++ 2 files changed, 17 insertions(+), 21 deletions(-) diff --git a/src/band/adapters/crewai.py b/src/band/adapters/crewai.py index 74a69b32e..f070f599a 100644 --- a/src/band/adapters/crewai.py +++ b/src/band/adapters/crewai.py @@ -56,11 +56,6 @@ # is the only discriminator. One definition, matched here and faked in tests. EMPTY_LLM_RESPONSE_MARKER = "Invalid response from LLM call" -# Bound on the in-process retry for a turn's very first LLM call coming back -# empty (see _kickoff_with_empty_response_retry) -- one immediate retry, no -# backoff, since nothing has run yet this turn and so nothing is duplicated. -_EMPTY_RESPONSE_FIRST_CALL_RETRIES = 1 - def _is_empty_llm_response(exc: Exception) -> bool: """Whether ``exc`` is CrewAI reporting that an LLM call came back empty. @@ -484,26 +479,24 @@ async def _kickoff_with_empty_response_retry( retry absorbs a single-call fluke instead of failing the whole delivery and leaving CrewAI to improvise on a cold redelivery. """ - attempt = 0 - while True: + try: + return await agent.kickoff_async(prompt) + except Exception as e: + if not (_is_empty_llm_response(e) and not reply_tracker.any_tool_ran): + raise + logger.info( + "Room %s: CrewAI's first LLM call came back empty before " + "any tool ran; retrying", + room_id, + ) try: return await agent.kickoff_async(prompt) - except Exception as e: - attempt += 1 - retryable = ( - attempt <= _EMPTY_RESPONSE_FIRST_CALL_RETRIES - and _is_empty_llm_response(e) - and not reply_tracker.any_tool_ran - ) - if not retryable: - raise - logger.info( - "Room %s: CrewAI's first LLM call came back empty before " - "any tool ran; retrying (%s/%s)", + except Exception: + logger.warning( + "Room %s: CrewAI's retry also came back empty; giving up", room_id, - attempt, - _EMPTY_RESPONSE_FIRST_CALL_RETRIES, ) + raise async def _report_error(self, tools: AgentToolsProtocol, error: str) -> None: """Send error event (best effort).""" diff --git a/tests/adapters/test_crewai_adapter.py b/tests/adapters/test_crewai_adapter.py index 520dc65de..4aeb37d18 100644 --- a/tests/adapters/test_crewai_adapter.py +++ b/tests/adapters/test_crewai_adapter.py @@ -742,6 +742,8 @@ async def test_empty_answer_with_no_tool_call_still_raises( ) mock_tools.send_event.assert_awaited_once() + # First call plus the one retry -- proves the retry actually happens + # before the final raise, not a bare pass-through of the first failure. assert mock_crewai_agent.kickoff_async.call_count == 2 @pytest.mark.asyncio @@ -787,6 +789,7 @@ async def _kickoff(_prompt): ) assert error_events(mock_tools) == [] + # First call plus the one retry that recovered it. assert mock_crewai_agent.kickoff_async.call_count == 2 @pytest.mark.asyncio From 1abc78ecf1e55116ba9d531d6ff9dcf1e982ae76 Mon Sep 17 00:00:00 2001 From: Alexander Zaikman Date: Wed, 9 Sep 2026 13:21:27 +0300 Subject: [PATCH 4/4] fix: don't mislabel a genuinely different retry failure as empty /simplify pass on PR #624: - The retry's second attempt logged "came back empty" unconditionally on any exception, not just another empty-response ValueError -- a real failure (timeout, auth, etc.) during the retry would be mislabeled. Now re-checks _is_empty_llm_response on the retry's own exception before choosing the log wording. - The new test's manual `calls` counter duplicated state the mock already tracks (kickoff_async.call_count); reads that instead. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01Fw8AZwnoa3Fen6HBhHEQ7H --- src/band/adapters/crewai.py | 17 ++++++++++++----- tests/adapters/test_crewai_adapter.py | 6 +----- 2 files changed, 13 insertions(+), 10 deletions(-) diff --git a/src/band/adapters/crewai.py b/src/band/adapters/crewai.py index f070f599a..ba7047123 100644 --- a/src/band/adapters/crewai.py +++ b/src/band/adapters/crewai.py @@ -491,11 +491,18 @@ async def _kickoff_with_empty_response_retry( ) try: return await agent.kickoff_async(prompt) - except Exception: - logger.warning( - "Room %s: CrewAI's retry also came back empty; giving up", - room_id, - ) + except Exception as retry_exc: + if _is_empty_llm_response(retry_exc): + logger.warning( + "Room %s: CrewAI's retry also came back empty; giving up", + room_id, + ) + else: + logger.warning( + "Room %s: CrewAI's retry failed with a different error: %s", + room_id, + retry_exc, + ) raise async def _report_error(self, tools: AgentToolsProtocol, error: str) -> None: diff --git a/tests/adapters/test_crewai_adapter.py b/tests/adapters/test_crewai_adapter.py index 4aeb37d18..48ccae12a 100644 --- a/tests/adapters/test_crewai_adapter.py +++ b/tests/adapters/test_crewai_adapter.py @@ -760,12 +760,8 @@ async def test_empty_first_call_recovers_on_retry( mock_result = MagicMock() mock_result.raw = "Hello! I'm here to help." - calls = 0 - async def _kickoff(_prompt): - nonlocal calls - calls += 1 - if calls == 1: + if mock_crewai_agent.kickoff_async.call_count == 1: raise ValueError(EMPTY_LLM_RESPONSE_ERROR) tracker = crewai_adapter._reply_tracker_var.get() if tracker is not None: