From b6f792f301e4f2b344fa8e12722f5f5efca6f22b Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Sat, 8 Nov 2025 08:49:16 +0530 Subject: [PATCH] Added support for desabling thoughts by setting budget to 0 (#16347) --- .../vertex_and_google_ai_studio_gemini.py | 47 ++++++----- tests/llm_translation/test_gemini.py | 79 ++++++++++++++++++- 2 files changed, 103 insertions(+), 23 deletions(-) diff --git a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py index b8370d5fef..91a32d981c 100644 --- a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py +++ b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py @@ -543,33 +543,40 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): else: budget = DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET - return { - "thinkingBudget": budget, - "includeThoughts": True, - } + return VertexGeminiConfig._build_thinking_config(budget) elif reasoning_effort == "low": - return { - "thinkingBudget": DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET, - "includeThoughts": True, - } + return VertexGeminiConfig._build_thinking_config( + DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET + ) elif reasoning_effort == "medium": - return { - "thinkingBudget": DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET, - "includeThoughts": True, - } + return VertexGeminiConfig._build_thinking_config( + DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET + ) elif reasoning_effort == "high": - return { - "thinkingBudget": DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET, - "includeThoughts": True, - } + return VertexGeminiConfig._build_thinking_config( + DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET + ) elif reasoning_effort == "disable": - return { - "thinkingBudget": DEFAULT_REASONING_EFFORT_DISABLE_THINKING_BUDGET, - "includeThoughts": False, - } + return VertexGeminiConfig._build_thinking_config( + DEFAULT_REASONING_EFFORT_DISABLE_THINKING_BUDGET, + include_thoughts=False, + ) else: raise ValueError(f"Invalid reasoning effort: {reasoning_effort}") + @staticmethod + def _build_thinking_config( + budget: int, include_thoughts: Optional[bool] = None + ) -> GeminiThinkingConfig: + safe_budget = max(budget, 0) + include_thoughts = ( + safe_budget > 0 if include_thoughts is None else include_thoughts + ) + return { + "thinkingBudget": safe_budget, + "includeThoughts": include_thoughts, + } + @staticmethod def _is_thinking_budget_zero(thinking_budget: Optional[int]) -> bool: return thinking_budget is not None and thinking_budget == 0 diff --git a/tests/llm_translation/test_gemini.py b/tests/llm_translation/test_gemini.py index f4100d9e8a..5df4ad23a2 100644 --- a/tests/llm_translation/test_gemini.py +++ b/tests/llm_translation/test_gemini.py @@ -948,10 +948,14 @@ def test_gemini_reasoning_effort_minimal(): actual_budget == expected_min_budget ), f"Model {model} should map 'minimal' to {expected_min_budget} tokens, got {actual_budget}" - # Verify that includeThoughts is True for minimal reasoning effort + # Verify that includeThoughts aligns with whether the budget is greater than zero + expected_include_thoughts = expected_min_budget > 0 assert thinking_config.get( - "includeThoughts", True - ), f"Model {model} should have includeThoughts=True for minimal reasoning effort" + "includeThoughts" + ) == expected_include_thoughts, ( + f"Model {model} should set includeThoughts={expected_include_thoughts} " + f"when reasoning_effort is 'minimal'" + ) # Test with unknown model (should use generic fallback) try: @@ -978,6 +982,75 @@ def test_gemini_reasoning_effort_minimal(): pass +def test_gemini_reasoning_effort_zero_budget_disables_thoughts(monkeypatch): + """Ensure zero thinking budget turns off includeThoughts.""" + from litellm.utils import return_raw_request + from litellm.types.utils import CallTypes + + monkeypatch.setattr( + "litellm.litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini.DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET_GEMINI_2_5_FLASH", + 0, + ) + + raw_request = return_raw_request( + endpoint=CallTypes.completion, + kwargs={ + "model": "gemini/gemini-2.5-flash", + "messages": [{"role": "user", "content": "Hello"}], + "reasoning_effort": "minimal", + }, + ) + + request_body = raw_request["raw_request_body"] + generation_config = request_body["generationConfig"] + thinking_config = generation_config["thinkingConfig"] + + assert ( + thinking_config.get("thinkingBudget") == 0 + ), "Zero reasoning budget should be preserved in request" + assert ( + thinking_config.get("includeThoughts") is False + ), "includeThoughts should be False when thinking budget is zero" + + +def test_gemini_reasoning_effort_env_override(monkeypatch): + """Verify env vars override default minimal thinking budget.""" + import importlib + + monkeypatch.setenv( + "DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET_GEMINI_2_5_FLASH", "0" + ) + + import litellm.litellm.constants as constants_module + import litellm.litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini as gemini_module + + importlib.reload(constants_module) + importlib.reload(gemini_module) + + from litellm.utils import return_raw_request + from litellm.types.utils import CallTypes + + raw_request = return_raw_request( + endpoint=CallTypes.completion, + kwargs={ + "model": "gemini/gemini-2.5-flash", + "messages": [{"role": "user", "content": "Hello"}], + "reasoning_effort": "minimal", + }, + ) + + request_body = raw_request["raw_request_body"] + generation_config = request_body["generationConfig"] + thinking_config = generation_config["thinkingConfig"] + + assert ( + thinking_config.get("thinkingBudget") == 0 + ), "Env override should set thinkingBudget to 0" + assert ( + thinking_config.get("includeThoughts") is False + ), "Env override should disable includeThoughts" + + def test_gemini_exception_message_format(): """ Test that Gemini provider exceptions show as 'GeminiException' not 'VertexAIException'.