From 2cf41d63a6eaa5abcfc7ba4f91c33e6b971ae21a Mon Sep 17 00:00:00 2001 From: Cesar Garcia <128240629+Chesars@users.noreply.github.com> Date: Fri, 5 Dec 2025 20:51:51 -0300 Subject: [PATCH] fix(gemini): use thought:true instead of thoughtSignature to detect thinking blocks (#17266) The previous implementation incorrectly used `thoughtSignature` as the criterion to detect thinking blocks. However, per Google's docs: - `thought: true` indicates that a part contains reasoning/thinking content - `thoughtSignature` is just a token for multi-turn context preservation (a part can have thoughtSignature without thought:true, e.g., function calls) This caused functionCall data to leak into reasoning_content when using Gemini 2.5 Pro with streaming + tools enabled. Changes: - _extract_thinking_blocks_from_parts now checks `part.get("thought") is True` - Extract actual text content instead of json.dumps(part) - Include signature only when present (optional in Gemini 2.5) Refs: - https://ai.google.dev/gemini-api/docs/thinking - https://ai.google.dev/gemini-api/docs/thought-signatures --- .../vertex_and_google_ai_studio_gemini.py | 34 +++++---- ...test_vertex_and_google_ai_studio_gemini.py | 73 +++++++++++++++++-- 2 files changed, 83 insertions(+), 24 deletions(-) diff --git a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py index e604bd392a..106074811f 100644 --- a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py +++ b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py @@ -1085,24 +1085,26 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): def _extract_thinking_blocks_from_parts( self, parts: List[HttpxPartType] ) -> List[ChatCompletionThinkingBlock]: - """Extract thinking blocks from parts if present""" + """Extract thinking blocks from parts if present. + + Per Google's docs (https://ai.google.dev/gemini-api/docs/thinking): + - Parts with `thought: true` contain thinking/reasoning content + - `thoughtSignature` is a separate token for multi-turn context preservation, + it does NOT indicate that the content is thinking (a part can have + thoughtSignature without thought: true, e.g., function calls) + """ thinking_blocks: List[ChatCompletionThinkingBlock] = [] for part in parts: - if "thoughtSignature" in part: - part_copy = part.copy() - part_copy.pop("thoughtSignature") - - text_content = part_copy.get("text") - if isinstance(text_content, str) and text_content.strip() == "": - continue - - thinking_blocks.append( - ChatCompletionThinkingBlock( - type="thinking", - thinking=json.dumps(part_copy), - signature=part["thoughtSignature"], - ) - ) + if part.get("thought") is True: + thinking_text = part.get("text", "") + block: ChatCompletionThinkingBlock = { + "type": "thinking", + "thinking": thinking_text, + } + signature = part.get("thoughtSignature") + if signature is not None: + block["signature"] = signature + thinking_blocks.append(block) return thinking_blocks def _extract_image_response_from_parts( diff --git a/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py b/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py index 41afec9cd1..7d45ce4091 100644 --- a/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py +++ b/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py @@ -390,13 +390,13 @@ def test_streaming_chunk_includes_reasoning_content(): ) -def test_streaming_chunk_with_tool_calls_includes_reasoning_content(): +def test_streaming_chunk_with_tool_calls_and_thought_includes_reasoning_content(): """ - Test for issue #16805: Ensure that when Gemini returns a streaming chunk with - tool calls AND thoughtSignature, the reasoning_content is included in the delta. + Test that when Gemini returns a streaming chunk with both thought: true parts + AND tool calls, the reasoning_content is correctly extracted from the thought parts. - Previously, thinking_blocks were only added to non-streaming responses, causing - reasoning_content to be missing in streaming mode when tools were enabled. + Per Google's docs: thought: true indicates reasoning content, NOT thoughtSignature. + thoughtSignature is just a token for multi-turn context preservation. """ from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import ( ModelResponseIterator, @@ -409,12 +409,16 @@ def test_streaming_chunk_with_tool_calls_includes_reasoning_content(): { "content": { "parts": [ + { + "text": "Let me think about how to get the time...", + "thought": True, # This indicates reasoning content + }, { "functionCall": { "name": "get_current_time", "args": {"timezone": "America/New_York"}, }, - "thoughtSignature": "EsEDCr4DAdHtim...", # Base64 signature + "thoughtSignature": "EsEDCr4DAdHtim...", # Just a token, not reasoning } ] }, @@ -433,8 +437,8 @@ def test_streaming_chunk_with_tool_calls_includes_reasoning_content(): ) streaming_chunk = iterator.chunk_parser(chunk) - # Verify that reasoning_content is present in the streaming delta - assert streaming_chunk.choices[0].delta.reasoning_content is not None + # Verify reasoning_content comes from the thought: true part + assert streaming_chunk.choices[0].delta.reasoning_content == "Let me think about how to get the time..." # Verify tool calls are also present assert streaming_chunk.choices[0].delta.tool_calls is not None @@ -442,6 +446,59 @@ def test_streaming_chunk_with_tool_calls_includes_reasoning_content(): assert streaming_chunk.choices[0].delta.tool_calls[0].function.name == "get_current_time" +def test_streaming_chunk_with_tool_calls_no_thought_no_reasoning_content(): + """ + Test that when Gemini returns tool calls with thoughtSignature but WITHOUT + thought: true, there is NO reasoning_content. + + This is a regression test for the bug where functionCall data was incorrectly + being placed into reasoning_content when thoughtSignature was present. + Per Google's docs: thoughtSignature is just a token for multi-turn, not reasoning. + """ + from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import ( + ModelResponseIterator, + ) + + litellm_logging = MagicMock() + + chunk = { + "candidates": [ + { + "content": { + "parts": [ + { + "functionCall": { + "name": "get_current_time", + "args": {"timezone": "America/New_York"}, + }, + "thoughtSignature": "EsEDCr4DAdHtim...", # Just a token, NOT thought: true + } + ] + }, + "finishReason": "STOP", + } + ], + "usageMetadata": { + "promptTokenCount": 68, + "candidatesTokenCount": 120, + "totalTokenCount": 188, + }, + } + + iterator = ModelResponseIterator( + streaming_response=[], sync_stream=True, logging_obj=litellm_logging + ) + streaming_chunk = iterator.chunk_parser(chunk) + + # reasoning_content should be None - thoughtSignature alone does NOT mean reasoning + assert getattr(streaming_chunk.choices[0].delta, 'reasoning_content', None) is None + + # Tool calls should still work + assert streaming_chunk.choices[0].delta.tool_calls is not None + assert len(streaming_chunk.choices[0].delta.tool_calls) == 1 + assert streaming_chunk.choices[0].delta.tool_calls[0].function.name == "get_current_time" + + def test_check_finish_reason(): finish_reason_mappings = VertexGeminiConfig.get_finish_reason_mapping() for k, v in finish_reason_mappings.items():