Skip to content

Commit dea8110

Browse files
DeanChensjcopybara-github
authored andcommitted
fix(flows): detect thought-only and whitespace turns as empty content
When models return turns that contain only reasoning thoughts (parts with thought=True) or whitespace text with finish_reason=STOP after tool execution or under streaming, BaseLlmFlow previously treated them as complete final responses because parts was not empty. This caused the agent loop to silently terminate without answering the user or surfacing an error. - Add has_meaningful_content(llm_response) helper in _finalizer.py to verify if an LLM response contains actionable content (tools, code, data, or non-thought non-whitespace text). - Update apply_empty_response_policy in _model_call.py to stamp MODEL_RETURNED_NO_CONTENT when an LLM response has finish_reason=STOP but lacks meaningful content. - Ensure LiteLlm.generate_content_async streaming always yields a terminal partial=False response on normal termination even when no content chunks were emitted. - Add unit tests covering has_meaningful_content, BaseLlmFlow error events for thought-only and whitespace-only turns, and LiteLlm empty stream fallback. Co-authored-by: Shangjie Chen <deanchen@google.com> PiperOrigin-RevId: 988479382
1 parent cb79908 commit dea8110

7 files changed

Lines changed: 639 additions & 26 deletions

File tree

‎src/google/adk/flows/llm_flows/core/_finalizer.py‎

Lines changed: 46 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -79,6 +79,52 @@ def finalize_model_response_event(
7979
return finalized_event
8080

8181

82+
def has_meaningful_content(llm_response: Optional[LlmResponse]) -> bool:
83+
"""Returns whether the LLM response contains meaningful, actionable content.
84+
85+
A response is considered to have meaningful content if it contains at least
86+
one part with:
87+
- An active function call or function response
88+
- Executable code or a code execution result
89+
- Inline data or file data
90+
- Non-thought, non-whitespace text
91+
92+
Responses that are None, have no content, have empty parts, or contain only
93+
thought parts (reasoning tokens) or whitespace-only text return False.
94+
95+
Args:
96+
llm_response: The LLM response to check.
97+
98+
Returns:
99+
True if the response contains meaningful content, False otherwise.
100+
"""
101+
if (
102+
not llm_response
103+
or not llm_response.content
104+
or not llm_response.content.parts
105+
):
106+
return False
107+
108+
for part in llm_response.content.parts:
109+
if part.function_call is not None:
110+
return True
111+
if part.function_response is not None:
112+
return True
113+
if part.executable_code is not None:
114+
return True
115+
if part.code_execution_result is not None:
116+
return True
117+
if part.inline_data is not None:
118+
return True
119+
if part.file_data is not None:
120+
return True
121+
is_thought = getattr(part, 'thought', False) or False
122+
if not is_thought and part.text and part.text.strip():
123+
return True
124+
125+
return False
126+
127+
82128
async def handle_before_model_callback(
83129
invocation_context: InvocationContext,
84130
llm_request: LlmRequest,

‎src/google/adk/flows/llm_flows/core/_model_call.py‎

Lines changed: 32 additions & 11 deletions
Original file line numberDiff line numberDiff line change
@@ -27,13 +27,16 @@
2727
from ....agents.invocation_context import InvocationContext
2828
from ....agents.readonly_context import ReadonlyContext
2929
from ....events.event import Event
30+
from ....features import FeatureName
31+
from ....features import is_feature_enabled
3032
from ....live.live_request_queue import LiveRequestQueue
3133
from ....models.llm_request import LlmRequest
3234
from ....models.llm_response import LlmResponse
3335
from ....telemetry.tracing import trace_call_llm
3436
from ....telemetry.tracing import tracer
3537
from ....utils._runner_utils import _with_caller_context
3638
from ....utils.context_utils import Aclosing
39+
from ._finalizer import has_meaningful_content
3740
from ._utils import as_llm_agent as _as_llm_agent
3841
from ._utils import require_run_config as _require_run_config
3942

@@ -47,19 +50,27 @@
4750
NO_CONTENT_ERROR_MESSAGE = (
4851
'The model returned no content (finish_reason=STOP with empty parts).'
4952
)
53+
NO_MEANINGFUL_CONTENT_ERROR_MESSAGE = (
54+
'The model returned no actionable content (finish_reason=STOP with'
55+
' thought-only or whitespace-only parts).'
56+
)
5057

5158

5259
def apply_empty_response_policy(
5360
invocation_context: InvocationContext,
5461
llm_response: LlmResponse,
5562
) -> None:
56-
"""Marks non-streaming empty STOP responses with NO_CONTENT_ERROR_CODE.
57-
58-
A non-streaming turn that finishes with STOP but has no content parts would
59-
otherwise be skipped and become a silent empty final response; surface it as
60-
an actionable error instead. Streaming is excluded because a terminal
61-
finish-only chunk legitimately follows content already streamed in earlier
62-
chunks.
63+
"""Marks terminal STOP responses that lack meaningful content as errors.
64+
65+
A turn that finishes with STOP but has no meaningful content (empty parts,
66+
thought-only parts, or whitespace-only text) would otherwise be skipped or
67+
treated as a complete answer; surface it as an actionable error instead.
68+
In SSE streaming, progressive SSE aggregates the entire turn into a single
69+
non-partial response, so a non-partial response carrying only thought or
70+
whitespace parts represents a completed turn with no answer, whereas empty
71+
parts (a terminal finish-only chunk) and non-progressive SSE (where the
72+
aggregator emits a non-partial thought-only chunk before a function call)
73+
are excluded.
6374
6475
This must run before the response processors. Emptiness is a property of
6576
what the model returned, so it can only be judged before local processing
@@ -68,17 +79,27 @@ def apply_empty_response_policy(
6879
once it has run the code and emitted its result.
6980
"""
7081
run_config = _require_run_config(invocation_context)
82+
has_parts = bool(llm_response.content and llm_response.content.parts)
7183
if (
7284
not llm_response.partial
7385
and llm_response.error_code is None
7486
and llm_response.finish_reason == types.FinishReason.STOP
75-
and (not llm_response.content or not llm_response.content.parts)
76-
and run_config.streaming_mode != StreamingMode.SSE
87+
and not has_meaningful_content(llm_response)
88+
and (
89+
run_config.streaming_mode != StreamingMode.SSE
90+
or (
91+
has_parts
92+
and is_feature_enabled(FeatureName.PROGRESSIVE_SSE_STREAMING)
93+
)
94+
)
7795
):
7896
llm_response.error_code = NO_CONTENT_ERROR_CODE
79-
llm_response.error_message = (
80-
llm_response.error_message or NO_CONTENT_ERROR_MESSAGE
97+
default_message = (
98+
NO_MEANINGFUL_CONTENT_ERROR_MESSAGE
99+
if has_parts
100+
else NO_CONTENT_ERROR_MESSAGE
81101
)
102+
llm_response.error_message = llm_response.error_message or default_message
82103

83104

84105
async def resolve_llm(invocation_context: InvocationContext) -> BaseLlm:

‎src/google/adk/models/lite_llm.py‎

Lines changed: 15 additions & 15 deletions
Original file line numberDiff line numberDiff line change
@@ -3573,6 +3573,7 @@ async def generate_content_async(
35733573
usage_metadata = None
35743574
grounding_metadata = None
35753575
last_finish_reason: str | None = None
3576+
last_model_version: str | None = None
35763577
fallback_index = 0
35773578
multiple_choices_logged = False
35783579

@@ -3688,6 +3689,8 @@ def _reset_stream_buffers() -> None:
36883689
last_finish_reason = None
36893690

36903691
async for part in await self.llm_client.acompletion(**completion_args):
3692+
if getattr(part, "model", None):
3693+
last_model_version = part.model
36913694
part_choices = part.get("choices") or []
36923695
if not multiple_choices_logged and (
36933696
len(part_choices) > 1
@@ -3828,36 +3831,33 @@ def _reset_stream_buffers() -> None:
38283831
# the provider actually sent rather than assuming a clean stop, so a
38293832
# filtered stream reports the same finish_reason and error_code that the
38303833
# non-streaming path reports.
3834+
resolved_model_version = last_model_version or effective_model
38313835
if function_calls and not aggregated_llm_response_with_tool_call:
38323836
aggregated_llm_response_with_tool_call = _finalize_tool_call_response(
3833-
model_version=part.model,
3837+
model_version=resolved_model_version,
38343838
finish_reason=last_finish_reason or "tool_calls",
38353839
)
38363840
_reset_stream_buffers()
38373841

38383842
if (text_parts or reasoning_parts) and not aggregated_llm_response:
38393843
aggregated_llm_response = _finalize_text_response(
3840-
model_version=part.model,
3844+
model_version=resolved_model_version,
38413845
finish_reason=last_finish_reason or "stop",
38423846
)
38433847
_reset_stream_buffers()
38443848
elif (
38453849
not aggregated_llm_response
38463850
and not aggregated_llm_response_with_tool_call
38473851
):
3848-
# The stream ended abnormally without ever producing content (an
3849-
# immediate content filter, or truncation before the first token).
3850-
# Non-streaming reports that as an error response; without this the
3851-
# generator ends having yielded nothing at all, so the reason, the
3852-
# error and the usage are all dropped and the caller sees a silent stop.
3853-
trailing_finish_reason = last_finish_reason or ""
3854-
if trailing_finish_reason and _map_finish_reason(
3855-
trailing_finish_reason
3856-
) not in (None, types.FinishReason.STOP):
3857-
aggregated_llm_response = _finalize_text_response(
3858-
model_version=part.model,
3859-
finish_reason=trailing_finish_reason,
3860-
)
3852+
# The stream ended without ever producing content or tool calls (an
3853+
# immediate content filter, truncation before the first token, or a
3854+
# normal stop/tool_calls/function_call/EOF with empty deltas). Finalize
3855+
# an empty response so the finish_reason, error, and usage are
3856+
# preserved rather than ending the generator having yielded nothing.
3857+
aggregated_llm_response = _finalize_text_response(
3858+
model_version=resolved_model_version,
3859+
finish_reason=last_finish_reason or "stop",
3860+
)
38613861

38623862
# waiting until streaming ends to yield the llm_response as litellm tends
38633863
# to send chunk that contains usage_metadata after the chunk with

‎tests/unittests/flows/llm_flows/core/test_finalizer.py‎

Lines changed: 165 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -456,3 +456,168 @@ async def failing_generator():
456456
failing_generator(), ctx, llm_request, event
457457
):
458458
pass
459+
460+
461+
# --- Tests for has_meaningful_content ---
462+
463+
464+
def test_has_meaningful_content_none_response():
465+
assert not _model_response_finalizer.has_meaningful_content(None)
466+
467+
468+
def test_has_meaningful_content_none_content():
469+
resp = LlmResponse(content=None)
470+
assert not _model_response_finalizer.has_meaningful_content(resp)
471+
472+
473+
def test_has_meaningful_content_empty_parts():
474+
resp = LlmResponse(content=types.Content(role="model", parts=[]))
475+
assert not _model_response_finalizer.has_meaningful_content(resp)
476+
477+
478+
def test_has_meaningful_content_thought_only():
479+
resp = LlmResponse(
480+
content=types.Content(
481+
role="model",
482+
parts=[types.Part(text="Thinking about this...", thought=True)],
483+
)
484+
)
485+
assert not _model_response_finalizer.has_meaningful_content(resp)
486+
487+
488+
def test_has_meaningful_content_multiple_thoughts_only():
489+
resp = LlmResponse(
490+
content=types.Content(
491+
role="model",
492+
parts=[
493+
types.Part(text="Step 1...", thought=True),
494+
types.Part(text="Step 2...", thought=True),
495+
],
496+
)
497+
)
498+
assert not _model_response_finalizer.has_meaningful_content(resp)
499+
500+
501+
def test_has_meaningful_content_whitespace_only():
502+
resp = LlmResponse(
503+
content=types.Content(
504+
role="model",
505+
parts=[types.Part.from_text(text=" \n\t ")],
506+
)
507+
)
508+
assert not _model_response_finalizer.has_meaningful_content(resp)
509+
510+
511+
def test_has_meaningful_content_empty_string():
512+
resp = LlmResponse(
513+
content=types.Content(
514+
role="model",
515+
parts=[types.Part.from_text(text="")],
516+
)
517+
)
518+
assert not _model_response_finalizer.has_meaningful_content(resp)
519+
520+
521+
def test_has_meaningful_content_valid_text():
522+
resp = LlmResponse(
523+
content=types.Content(
524+
role="model",
525+
parts=[types.Part.from_text(text="Hello world")],
526+
)
527+
)
528+
assert _model_response_finalizer.has_meaningful_content(resp)
529+
530+
531+
def test_has_meaningful_content_thought_and_valid_text():
532+
resp = LlmResponse(
533+
content=types.Content(
534+
role="model",
535+
parts=[
536+
types.Part(text="Thinking...", thought=True),
537+
types.Part.from_text(text="Final answer."),
538+
],
539+
)
540+
)
541+
assert _model_response_finalizer.has_meaningful_content(resp)
542+
543+
544+
def test_has_meaningful_content_function_call():
545+
resp = LlmResponse(
546+
content=types.Content(
547+
role="model",
548+
parts=[types.Part.from_function_call(name="search", args={})],
549+
)
550+
)
551+
assert _model_response_finalizer.has_meaningful_content(resp)
552+
553+
554+
def test_has_meaningful_content_function_response():
555+
resp = LlmResponse(
556+
content=types.Content(
557+
role="model",
558+
parts=[types.Part.from_function_response(name="search", response={})],
559+
)
560+
)
561+
assert _model_response_finalizer.has_meaningful_content(resp)
562+
563+
564+
def test_has_meaningful_content_executable_code():
565+
resp = LlmResponse(
566+
content=types.Content(
567+
role="model",
568+
parts=[
569+
types.Part(
570+
executable_code=types.ExecutableCode(
571+
code="print(1)", language=types.Language.PYTHON
572+
)
573+
)
574+
],
575+
)
576+
)
577+
assert _model_response_finalizer.has_meaningful_content(resp)
578+
579+
580+
def test_has_meaningful_content_code_execution_result():
581+
resp = LlmResponse(
582+
content=types.Content(
583+
role="model",
584+
parts=[
585+
types.Part(
586+
code_execution_result=types.CodeExecutionResult(
587+
outcome=types.Outcome.OUTCOME_OK, output="1"
588+
)
589+
)
590+
],
591+
)
592+
)
593+
assert _model_response_finalizer.has_meaningful_content(resp)
594+
595+
596+
def test_has_meaningful_content_inline_data():
597+
resp = LlmResponse(
598+
content=types.Content(
599+
role="model",
600+
parts=[
601+
types.Part(
602+
inline_data=types.Blob(data=b"data", mime_type="image/png")
603+
)
604+
],
605+
)
606+
)
607+
assert _model_response_finalizer.has_meaningful_content(resp)
608+
609+
610+
def test_has_meaningful_content_file_data():
611+
resp = LlmResponse(
612+
content=types.Content(
613+
role="model",
614+
parts=[
615+
types.Part(
616+
file_data=types.FileData(
617+
file_uri="gs://bucket/file", mime_type="application/pdf"
618+
)
619+
)
620+
],
621+
)
622+
)
623+
assert _model_response_finalizer.has_meaningful_content(resp)

0 commit comments

Comments
 (0)