From 8d9db7b4a6b7de9c47b95372bf2ce39c9695ec8f Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Wed, 15 Apr 2026 15:01:30 +0530 Subject: [PATCH] fix(gemini): follow provider defaults for Gemini 3 thinking Stop forcing Gemini 3 thinkingLevel for Anthropic-style thinking params by default, and gate legacy low/minimal mapping behind an explicit feature flag to avoid provider-default confusion. Made-with: Cursor --- litellm/__init__.py | 3 + .../vertex_and_google_ai_studio_gemini.py | 11 +- tests/llm_translation/test_gemini.py | 156 +++++++++++------- 3 files changed, 97 insertions(+), 73 deletions(-) diff --git a/litellm/__init__.py b/litellm/__init__.py index 77fa48625d..d02d39cae1 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -330,6 +330,9 @@ enable_model_config_credential_overrides: bool = False enable_key_alias_format_validation: bool = ( False # opt-in validation of key_alias format on /key/generate and /key/update ) +enable_gemini_default_thinking_level_low: bool = ( + False # opt-in: force thinkingLevel low/minimal for Gemini 3 thinking param mapping +) #################### logging: bool = True enable_loadbalancing_on_batch_endpoints: Optional[bool] = None diff --git a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py index 474ddb402a..6278de662f 100644 --- a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py +++ b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py @@ -979,15 +979,8 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): params["includeThoughts"] = False else: params["includeThoughts"] = True - if thinking_budget >= 10000: - is_gemini3flash = ( - "gemini-3-flash-preview" in model.lower() - or "gemini-3-flash" in model.lower() - ) - params["thinkingLevel"] = ( - "minimal" if is_gemini3flash else "low" - ) - else: + # Follow provider defaults unless explicitly opted into legacy behavior. + if litellm.enable_gemini_default_thinking_level_low is True: is_gemini3flash = ( "gemini-3-flash-preview" in model.lower() or "gemini-3-flash" in model.lower() diff --git a/tests/llm_translation/test_gemini.py b/tests/llm_translation/test_gemini.py index a945a5c1ea..97b0aaee86 100644 --- a/tests/llm_translation/test_gemini.py +++ b/tests/llm_translation/test_gemini.py @@ -1331,14 +1331,14 @@ def test_gemini_function_args_preserve_unicode(): assert "José" in arguments_str -def test_anthropic_thinking_param_to_gemini_3_thinkingLevel(): +def test_anthropic_thinking_param_to_gemini_3_provider_defaults(): """ - Test that Anthropic thinking parameters are correctly transformed to Gemini 3 thinkingLevel - instead of thinkingBudget. + Test that Anthropic thinking parameters for Gemini 3+ follow provider defaults + unless force-low behavior is explicitly enabled. For Gemini 3+ models (gemini-3-flash, gemini-3-pro, gemini-3-flash-preview): - - Should use thinkingLevel instead of thinkingBudget - - budget_tokens should map to thinkingLevel + - Should not force thinkingLevel by default + - Should still set includeThoughts correctly Related issue: https://github.com/BerriAI/litellm/issues/XXXX """ @@ -1347,72 +1347,102 @@ def test_anthropic_thinking_param_to_gemini_3_thinkingLevel(): ) from litellm.types.llms.anthropic import AnthropicThinkingParam + original_force_low_flag = litellm.enable_gemini_default_thinking_level_low + litellm.enable_gemini_default_thinking_level_low = False + # Test 1: Anthropic thinking enabled with budget_tokens for Gemini 3 model thinking_param: AnthropicThinkingParam = { "type": "enabled", "budget_tokens": 10000, } + try: + result = VertexGeminiConfig._map_thinking_param( + thinking_param=thinking_param, + model="gemini-3-flash", + ) - result = VertexGeminiConfig._map_thinking_param( - thinking_param=thinking_param, - model="gemini-3-flash", + # For Gemini 3, should not force thinkingLevel by default + assert "thinkingLevel" not in result, "Should not force thinkingLevel for Gemini 3" + assert "thinkingBudget" not in result, "Should NOT have thinkingBudget for Gemini 3" + assert result["includeThoughts"] is True + + # Test 2: Anthropic thinking disabled for Gemini 3 + thinking_param_disabled: AnthropicThinkingParam = { + "type": "disabled", + "budget_tokens": None, + } + + result_disabled = VertexGeminiConfig._map_thinking_param( + thinking_param=thinking_param_disabled, + model="gemini-3-pro-preview", + ) + + assert result_disabled.get("includeThoughts") is False + assert ( + "thinkingLevel" not in result_disabled + or result_disabled.get("thinkingLevel") is None + ) + + # Test 3: Budget tokens = 0 for Gemini 3 + thinking_param_zero: AnthropicThinkingParam = { + "type": "enabled", + "budget_tokens": 0, + } + + result_zero = VertexGeminiConfig._map_thinking_param( + thinking_param=thinking_param_zero, + model="gemini-3-flash", + ) + + assert result_zero["includeThoughts"] is False + assert "thinkingLevel" not in result_zero or result_zero.get("thinkingLevel") is None + + # Test 4: Gemini 3 flash-preview should also follow provider defaults by default + result_gemini3flashpreview = VertexGeminiConfig._map_thinking_param( + thinking_param=thinking_param, + model="gemini-3-flash-preview", + ) + + assert "thinkingLevel" not in result_gemini3flashpreview + assert "thinkingBudget" not in result_gemini3flashpreview + assert result_gemini3flashpreview["includeThoughts"] is True + finally: + litellm.enable_gemini_default_thinking_level_low = original_force_low_flag + + +def test_anthropic_thinking_param_to_gemini_3_force_low_feature_flag(): + """ + Test that Gemini 3 thinkingLevel forced mapping is available behind a feature flag. + """ + from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import ( + VertexGeminiConfig, ) + from litellm.types.llms.anthropic import AnthropicThinkingParam - # For Gemini 3, should use thinkingLevel, not thinkingBudget - assert "thinkingLevel" in result, "Should have thinkingLevel for Gemini 3" - assert "thinkingBudget" not in result, "Should NOT have thinkingBudget for Gemini 3" - assert result["includeThoughts"] is True - assert result["thinkingLevel"] in [ - "minimal", - "low", - ], "thinkingLevel should be 'minimal' or 'low'" + original_force_low_flag = litellm.enable_gemini_default_thinking_level_low + litellm.enable_gemini_default_thinking_level_low = True - # Test 2: Anthropic thinking disabled for Gemini 3 - thinking_param_disabled: AnthropicThinkingParam = { - "type": "disabled", - "budget_tokens": None, - } - - result_disabled = VertexGeminiConfig._map_thinking_param( - thinking_param=thinking_param_disabled, - model="gemini-3-pro-preview", - ) - - assert result_disabled.get("includeThoughts") is False - assert ( - "thinkingLevel" not in result_disabled - or result_disabled.get("thinkingLevel") is None - ) - - # Test 3: Budget tokens = 0 for Gemini 3 - thinking_param_zero: AnthropicThinkingParam = { + thinking_param: AnthropicThinkingParam = { "type": "enabled", - "budget_tokens": 0, + "budget_tokens": 10000, } - result_zero = VertexGeminiConfig._map_thinking_param( - thinking_param=thinking_param_zero, - model="gemini-3-flash", - ) + try: + result_flash = VertexGeminiConfig._map_thinking_param( + thinking_param=thinking_param, + model="gemini-3-flash", + ) + assert result_flash["thinkingLevel"] == "minimal" + assert result_flash["includeThoughts"] is True - assert result_zero["includeThoughts"] is False - assert ( - "thinkingLevel" not in result_zero or result_zero.get("thinkingLevel") is None - ) - - # Test 4: Fiercefalcon model (Gemini 3 Flash checkpoint) should use thinkingLevel - result_gemini3flashpreview = VertexGeminiConfig._map_thinking_param( - thinking_param=thinking_param, - model="gemini-3-flash-preview", - ) - - assert ( - "thinkingLevel" in result_gemini3flashpreview - ), "Should have thinkingLevel for gemini-3-flash-preview" - assert ( - "thinkingBudget" not in result_gemini3flashpreview - ), "Should NOT have thinkingBudget for gemini-3-flash-preview" - assert result_gemini3flashpreview["includeThoughts"] is True + result_pro = VertexGeminiConfig._map_thinking_param( + thinking_param=thinking_param, + model="gemini-3-pro-preview", + ) + assert result_pro["thinkingLevel"] == "low" + assert result_pro["includeThoughts"] is True + finally: + litellm.enable_gemini_default_thinking_level_low = original_force_low_flag def test_anthropic_thinking_param_to_gemini_2_thinkingBudget(): @@ -1465,7 +1495,7 @@ def test_anthropic_thinking_param_to_gemini_2_thinkingBudget(): def test_anthropic_thinking_param_via_map_openai_params(): """ Test that the thinking parameter is correctly transformed through the full map_openai_params flow - for Gemini 3 models, resulting in thinkingConfig with thinkingLevel. + for Gemini 3 models, without forcing thinkingLevel by default. This tests the full integration from Anthropic API format to Gemini format. """ @@ -1492,13 +1522,11 @@ def test_anthropic_thinking_param_via_map_openai_params(): drop_params=False, ) - # Check that thinkingConfig was created with thinkingLevel + # Check that thinkingConfig was created without forced thinkingLevel assert "thinkingConfig" in result, "Should have thinkingConfig in optional_params" thinking_config = result["thinkingConfig"] - assert "thinkingLevel" in thinking_config, "Should have thinkingLevel for Gemini 3" - assert ( - "thinkingBudget" not in thinking_config - ), "Should NOT have thinkingBudget for Gemini 3" + assert "thinkingLevel" not in thinking_config, "Should not force thinkingLevel for Gemini 3 by default" + assert "thinkingBudget" not in thinking_config, "Should NOT have thinkingBudget for Gemini 3" assert thinking_config["includeThoughts"] is True # Test with Gemini 2 model