Skip to content

Commit d09d488

Browse files
fix: report model failures returned as data and use single-token E2E markers
Agno, Google ADK and Gemini return a failed model run as a value instead of raising. A shared ProviderRunError and generic_provider_failure fail the turn with the coarse code (SAFETY, PROHIBITED_CONTENT, ...) even after the turn did real work; provider text stays in the agent log. E2E markers become one uppercase alphanumeric token, since models dropped the word-like "note-" prefix. The add-participant smoke now reports the agent's tool calls and failed results when the add is missing. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017jhEdA7VR9ng3nZpKW2Z2X
1 parent a7e69e8 commit d09d488

18 files changed

Lines changed: 504 additions & 73 deletions

‎docs/turn-outcome.md‎

Lines changed: 12 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -228,6 +228,18 @@ returns record nothing; self-hosted tools record their own effects.
228228
file with the reason it carries the model's words, so a new one fails until it
229229
is justified.
230230

231+
## Failed runs returned as data: `ProviderRunError`
232+
233+
Some frameworks return a failed model run as a value instead of raising: Agno
234+
sets an error status, Google ADK yields an event with `error_code`, and a Gemini
235+
response carries a safety `finish_reason` or a blocked prompt. The judge cannot
236+
see such a failure when the turn already did work or replied, so the adapter
237+
raises `ProviderRunError(code, detail)` from `band.core.exceptions` inside its
238+
turn. Its failure path reports `generic_provider_failure(provider, error)` from
239+
`band.core.protocols`: the generic message plus the coarse `code` (such as
240+
`SAFETY`). The provider's `detail` can echo the prompt, so it goes only to the
241+
agent log. Agno, Google ADK and Gemini follow this rule.
242+
231243
## Detached turns
232244

233245
An adapter whose turn outlives `on_message` (one parked on a human approval)

‎src/band/adapters/agno.py‎

Lines changed: 7 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -12,12 +12,12 @@
1212

1313
from agno.media import Image
1414
from agno.tools.function import ToolResult
15-
from band_sdk_core import AgentFailure
1615
from typing_extensions import Unpack
1716

1817
from band.converters.agno import AgnoHistoryConverter, AgnoMessages
1918
from band.core.adapterconfig import BaseAdapterConfig
20-
from band.core.protocols import GENERIC_PROVIDER_FAILURE_MESSAGE, AgentToolsProtocol
19+
from band.core.exceptions import ProviderRunError
20+
from band.core.protocols import AgentToolsProtocol, generic_provider_failure
2121
from band.core.simple_adapter import SimpleAdapter
2222
from band.core.tool_filter import filter_tool_schemas
2323
from band.core.types import (
@@ -73,7 +73,7 @@
7373
)
7474

7575

76-
class AgnoRunError(RuntimeError):
76+
class AgnoRunError(ProviderRunError):
7777
"""An Agno run finished in an error state instead of raising.
7878
7979
Agno catches exceptions inside ``Agent.arun`` (model/API failures included)
@@ -84,6 +84,9 @@ class AgnoRunError(RuntimeError):
8484
that produced no output.
8585
"""
8686

87+
def __init__(self, detail: str) -> None:
88+
super().__init__(RunStatus.error.value, detail)
89+
8790

8891
def _error_summary(detail: str | None) -> str:
8992
"""A bounded, log-safe summary of Agno's swallowed error text."""
@@ -485,18 +488,10 @@ async def _run_agent(
485488
if response is not None and response.status == RunStatus.error:
486489
raise AgnoRunError(_error_summary(response.content))
487490
except Exception as e:
488-
# Keep the user-facing payload generic; the full traceback is in the
489-
# agent log via logger.exception. Exception text can include DB
490-
# strings, paths, and tokens that must not surface in chat. Only
491-
# the coarse RunStatus.error code -- never response.content -- is
492-
# safe to attach.
493491
logger.exception(
494492
"Room %s msg %s: error running Agno agent", room_id, msg_id
495493
)
496-
code = RunStatus.error.value if isinstance(e, AgnoRunError) else None
497-
await tools.send_failure(
498-
AgentFailure("agno", GENERIC_PROVIDER_FAILURE_MESSAGE, code)
499-
)
494+
await tools.send_failure(generic_provider_failure("agno", e))
500495
raise
501496

502497
if response is None:

‎src/band/adapters/gemini.py‎

Lines changed: 25 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -31,7 +31,8 @@
3131

3232
from band.converters.gemini import GeminiHistoryConverter, GeminiMessages
3333
from band.core.adapterconfig import BaseAdapterConfig
34-
from band.core.protocols import GENERIC_PROVIDER_FAILURE_MESSAGE, AgentToolsProtocol
34+
from band.core.exceptions import ProviderRunError
35+
from band.core.protocols import AgentToolsProtocol, generic_provider_failure
3536
from band.core.simple_adapter import SimpleAdapter
3637
from band.core.tool_filter import sanitize_tool_schema
3738
from band.core.types import (
@@ -88,7 +89,26 @@ def _to_agent_failure(e: Exception) -> AgentFailure:
8889
if isinstance(e, ServerError):
8990
status = e.status if e.status is None else str(e.status)
9091
return AgentFailure(_PROVIDER, str(e), status, e.message)
91-
return AgentFailure(_PROVIDER, GENERIC_PROVIDER_FAILURE_MESSAGE)
92+
return generic_provider_failure(_PROVIDER, e)
93+
94+
95+
def _model_failure(response: types.GenerateContentResponse) -> ProviderRunError | None:
96+
"""The failure a response carries as data (a blocked prompt, a safety stop),
97+
read the way google-adk's ``LlmResponse.create`` reads it. A reply with
98+
content, even one cut off at MAX_TOKENS, and an empty normal finish are
99+
not failures."""
100+
if response.candidates:
101+
candidate = response.candidates[0]
102+
if candidate.content and candidate.content.parts:
103+
return None
104+
reason = candidate.finish_reason
105+
if reason is None or reason == types.FinishReason.STOP:
106+
return None
107+
return ProviderRunError(reason.value, candidate.finish_message)
108+
feedback = response.prompt_feedback
109+
if feedback is None or feedback.block_reason is None:
110+
return None
111+
return ProviderRunError(feedback.block_reason.value, feedback.block_reason_message)
92112

93113

94114
class GeminiAdapterConfig(BaseAdapterConfig):
@@ -258,13 +278,14 @@ async def on_message(
258278
response = await self._call_gemini(
259279
contents=self._message_history[room_id], tools=gemini_tools
260280
)
281+
turn_usage = turn_usage + self._usage_from_response(response)
282+
if failure := _model_failure(response):
283+
raise failure
261284
except Exception as e:
262285
logger.exception("Error calling Gemini")
263286
await tools.send_failure(_to_agent_failure(e))
264287
raise
265288

266-
turn_usage = turn_usage + self._usage_from_response(response)
267-
268289
candidate_content = self._extract_candidate_content(response)
269290
if candidate_content is not None:
270291
self._message_history[room_id].append(candidate_content)

‎src/band/adapters/google_adk.py‎

Lines changed: 16 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -15,13 +15,13 @@
1515
import uuid
1616
from typing import TYPE_CHECKING, Any, ClassVar, cast
1717

18-
from band_sdk_core import AgentFailure
1918
from pydantic import PositiveInt, ValidationError
2019
from typing_extensions import Unpack
2120

2221
from band.converters.google_adk import GoogleADKHistoryConverter, GoogleADKMessages
2322
from band.core.adapterconfig import BaseAdapterConfig
24-
from band.core.protocols import GENERIC_PROVIDER_FAILURE_MESSAGE, AgentToolsProtocol
23+
from band.core.exceptions import ProviderRunError
24+
from band.core.protocols import AgentToolsProtocol, generic_provider_failure
2525
from band.core.simple_adapter import SimpleAdapter
2626
from band.core.tool_filter import sanitize_tool_schema
2727
from band.core.types import (
@@ -589,6 +589,7 @@ async def on_message(
589589
# reported per model response on the event stream, so sum across
590590
# the loop into one per-turn TurnUsage.
591591
final_response_text = ""
592+
model_failure: ProviderRunError | None = None
592593
async for event in runner.run_async(
593594
user_id=room_id,
594595
session_id=session_id,
@@ -597,6 +598,15 @@ async def on_message(
597598
if Emit.USAGE in self.features.emit:
598599
turn_usage = turn_usage + self._usage_from_event(event)
599600

601+
# ADK returns a model failure as an event. Its flow ends the
602+
# run after one, so drain rather than break: breaking leaves
603+
# ADK's nested generators open inside their tracing spans.
604+
# ADK <= 1.10 reports an empty normal finish as "STOP".
605+
if event.error_code and event.error_code != types.FinishReason.STOP:
606+
model_failure = ProviderRunError(
607+
event.error_code, event.error_message
608+
)
609+
600610
# Report tool calls/results if enabled
601611
if Emit.TOOL_CALLS in self.features.emit:
602612
try:
@@ -611,11 +621,11 @@ async def on_message(
611621
"Room %s: ADK agent completed with final response",
612622
room_id,
613623
)
614-
except Exception:
624+
if model_failure is not None:
625+
raise model_failure
626+
except Exception as e:
615627
logger.exception("Error running ADK agent in room %s", room_id)
616-
await tools.send_failure(
617-
AgentFailure("google_adk", GENERIC_PROVIDER_FAILURE_MESSAGE)
618-
)
628+
await tools.send_failure(generic_provider_failure("google_adk", e))
619629
raise
620630
finally:
621631
# Emit before close so a close() failure can't drop the usage, but

‎src/band/core/exceptions.py‎

Lines changed: 15 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -54,6 +54,21 @@ class BandToolError(BandError):
5454
"""Tool execution failures. Actionable by adapter/LLM."""
5555

5656

57+
class ProviderRunError(BandError):
58+
"""A provider or framework returned a failed run as data instead of raising.
59+
Actionable by the adapter: raise it inside the turn so the turn is reported
60+
failed.
61+
62+
``code`` is coarse and safe to post; ``detail`` is provider text, which can
63+
echo the prompt, so it belongs only in logs.
64+
"""
65+
66+
def __init__(self, code: str, detail: str | None = None) -> None:
67+
super().__init__(f"{code}: {detail}" if detail else code)
68+
self.code = code
69+
self.detail = detail
70+
71+
5772
def _levenshtein(a: str, b: str) -> int:
5873
"""Iterative Levenshtein distance. Pure Python, no dependencies."""
5974
if a == b:

‎src/band/core/protocols.py‎

Lines changed: 8 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -9,6 +9,7 @@
99
from band_sdk_core import AgentFailure
1010

1111
from band.core.content import has_visible_content
12+
from band.core.exceptions import ProviderRunError
1213

1314
logger = logging.getLogger(__name__)
1415

@@ -54,6 +55,13 @@
5455
)
5556

5657

58+
def generic_provider_failure(provider: str, error: BaseException) -> AgentFailure:
59+
"""The room-safe failure for a caught provider error: the generic message,
60+
plus the coarse code when the provider returned the failure as data."""
61+
code = error.code if isinstance(error, ProviderRunError) else None
62+
return AgentFailure(provider, GENERIC_PROVIDER_FAILURE_MESSAGE, code)
63+
64+
5765
# ``AgentFailure.provider`` for a turn failure the runtime reports on the
5866
# adapter's behalf.
5967
TURN_FAILURE_PROVIDER = "band-runtime"

‎tests/adapters/genaikit.py‎

Lines changed: 84 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,84 @@
1+
"""Real google-genai responses for scripting the Google adapters' turns.
2+
3+
Gemini reads a ``GenerateContentResponse`` directly and Google ADK reads the
4+
same object through ``LlmResponse.create``, so both adapters' tests script
5+
turns from these builders. Every response carries ``SCRIPTED_USAGE``.
6+
"""
7+
8+
from __future__ import annotations
9+
10+
from typing import Any
11+
from unittest.mock import MagicMock
12+
13+
from google.genai import types
14+
15+
from band.runtime.tools import AgentTools
16+
from band.testing import FakeAgentTools
17+
18+
SCRIPTED_USAGE = types.GenerateContentResponseUsageMetadata(
19+
prompt_token_count=11, candidates_token_count=3
20+
)
21+
22+
# Provider text that must reach only the agent log, never the room.
23+
PROVIDER_DETAIL = "blocked: the prompt asked for the secret"
24+
25+
26+
class PlatformSchemaFakeTools(FakeAgentTools):
27+
"""A ``FakeAgentTools`` advertising the real platform tool schemas, so a
28+
scripted function call goes through the adapter's own tool bridge."""
29+
30+
def get_openai_tool_schemas(self, **kwargs: Any) -> list[dict[str, Any]]:
31+
# Schema building reads the tool registry, never the REST link.
32+
return AgentTools("room-123", MagicMock()).get_openai_tool_schemas(**kwargs)
33+
34+
35+
def tool_call(name: str, args: dict[str, Any]) -> types.GenerateContentResponse:
36+
return _candidate(
37+
types.Content(
38+
role="model", parts=[types.Part.from_function_call(name=name, args=args)]
39+
),
40+
types.FinishReason.STOP,
41+
)
42+
43+
44+
def text_reply(
45+
text: str, finish: types.FinishReason = types.FinishReason.STOP
46+
) -> types.GenerateContentResponse:
47+
return _candidate(
48+
types.Content(role="model", parts=[types.Part.from_text(text=text)]), finish
49+
)
50+
51+
52+
def stopped(reason: types.FinishReason) -> types.GenerateContentResponse:
53+
"""A candidate with no content, ended for ``reason``."""
54+
return _candidate(None, reason, finish_message=PROVIDER_DETAIL)
55+
56+
57+
def prompt_blocked(reason: types.BlockedReason) -> types.GenerateContentResponse:
58+
"""No candidates: the prompt itself was refused."""
59+
return types.GenerateContentResponse(
60+
prompt_feedback=types.GenerateContentResponsePromptFeedback(
61+
block_reason=reason, block_reason_message=PROVIDER_DETAIL
62+
),
63+
usage_metadata=SCRIPTED_USAGE,
64+
)
65+
66+
67+
def no_candidates() -> types.GenerateContentResponse:
68+
return types.GenerateContentResponse(usage_metadata=SCRIPTED_USAGE)
69+
70+
71+
def _candidate(
72+
content: types.Content | None,
73+
finish: types.FinishReason,
74+
*,
75+
finish_message: str | None = None,
76+
) -> types.GenerateContentResponse:
77+
return types.GenerateContentResponse(
78+
candidates=[
79+
types.Candidate(
80+
content=content, finish_reason=finish, finish_message=finish_message
81+
)
82+
],
83+
usage_metadata=SCRIPTED_USAGE,
84+
)

0 commit comments

Comments
 (0)