Merge pull request #27074 from BerriAI/litellm_fix_reasoning_effort_followup-0f97

fix(anthropic,bedrock,vertex): forward output_config.effort + 400 on garbage reasoning_effort
This commit is contained in:
Mateo Wang
2026-05-04 14:59:27 -07:00
committed by GitHub
31 changed files with 1904 additions and 326 deletions
+2 -2
View File
@@ -72,7 +72,7 @@
"computer-use-2025-11-24": "computer-use-2025-11-24",
"context-1m-2025-08-07": "context-1m-2025-08-07",
"context-management-2025-06-27": null,
"effort-2025-11-24": null,
"effort-2025-11-24": "effort-2025-11-24",
"fast-mode-2026-02-01": null,
"files-api-2025-04-14": null,
"fine-grained-tool-streaming-2025-05-14": null,
@@ -103,7 +103,7 @@
"computer-use-2025-11-24": "computer-use-2025-11-24",
"context-1m-2025-08-07": "context-1m-2025-08-07",
"context-management-2025-06-27": null,
"effort-2025-11-24": null,
"effort-2025-11-24": "effort-2025-11-24",
"fast-mode-2026-02-01": null,
"files-api-2025-04-14": null,
"fine-grained-tool-streaming-2025-05-14": null,
+8
View File
@@ -202,6 +202,12 @@ DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET = int(
DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET = int(
os.getenv("DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET", 4096)
)
DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET = int(
os.getenv("DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET", 8192)
)
DEFAULT_REASONING_EFFORT_MAX_THINKING_BUDGET = int(
os.getenv("DEFAULT_REASONING_EFFORT_MAX_THINKING_BUDGET", 16384)
)
MAX_TOKEN_TRIMMING_ATTEMPTS = int(
os.getenv("MAX_TOKEN_TRIMMING_ATTEMPTS", 10)
) # Maximum number of attempts to trim the message
@@ -399,6 +405,8 @@ BEDROCK_MAX_POLICY_SIZE = int(os.getenv("BEDROCK_MAX_POLICY_SIZE", 75))
BEDROCK_MIN_THINKING_BUDGET_TOKENS = int(
os.getenv("BEDROCK_MIN_THINKING_BUDGET_TOKENS", 1024)
)
# Anthropic's Messages API rejects thinking.budget_tokens < 1024.
ANTHROPIC_MIN_THINKING_BUDGET_TOKENS = 1024
REPLICATE_POLLING_DELAY_SECONDS = float(
os.getenv("REPLICATE_POLLING_DELAY_SECONDS", 0.5)
)
+175 -47
View File
@@ -1,18 +1,31 @@
import json
import re
import time
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Tuple, Union, cast
from typing import (
TYPE_CHECKING,
Any,
Dict,
List,
NoReturn,
Optional,
Tuple,
Union,
cast,
)
import httpx
import litellm
from litellm.constants import (
ANTHROPIC_MIN_THINKING_BUDGET_TOKENS,
ANTHROPIC_WEB_SEARCH_TOOL_MAX_USES,
DEFAULT_ANTHROPIC_CHAT_MAX_TOKENS,
DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET,
DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET,
DEFAULT_REASONING_EFFORT_MAX_THINKING_BUDGET,
DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET,
DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET,
DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET,
RESPONSE_FORMAT_TOOL_NAME,
)
from litellm.litellm_core_utils.core_helpers import map_finish_reason
@@ -92,6 +105,22 @@ else:
LoggingClass = Any
REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT: Dict[str, str] = {
"low": "low",
"minimal": "low",
"medium": "medium",
"high": "high",
"xhigh": "xhigh",
"max": "max",
}
DROP_UNSUPPORTED_OUTPUT_CONFIG_WARNING = (
"Dropping unsupported `output_config` for model=%s "
"(drop_params=True). Effort is only supported on Opus 4.5+, "
"Sonnet 4.6+, and Mythos Preview."
)
class AnthropicConfig(AnthropicModelInfo, BaseConfig):
"""
Reference: https://docs.anthropic.com/claude/reference/messages_post
@@ -202,17 +231,96 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
def _supports_effort_level(model: str, level: str) -> bool:
"""Check ``supports_{level}_reasoning_effort`` in the model map.
Mirrors the pattern used in ``openai/chat/gpt_5_transformation.py`` so
that adding support for a new effort level is a pure model-map change.
Strips bedrock/vertex prefixes so a provider-routed Claude still
resolves to the Anthropic model-map entry.
"""
key = f"supports_{level}_reasoning_effort"
try:
return _supports_factory(
if _supports_factory(
model=model,
custom_llm_provider="anthropic",
key=f"supports_{level}_reasoning_effort",
)
key=key,
):
return True
except Exception:
return False
pass
candidates = [model]
for prefix in (
"bedrock/converse/",
"bedrock/invoke/",
"bedrock/",
"vertex_ai/",
):
if model.startswith(prefix):
candidates.append(model[len(prefix) :])
try:
from litellm.llms.bedrock.common_utils import BedrockModelInfo
base = BedrockModelInfo.get_base_model(model)
if base:
candidates.append(base)
candidates.append(f"bedrock/{base}")
except Exception:
pass
try:
import litellm
for cand in candidates:
if cand in litellm.model_cost and (
litellm.model_cost[cand].get(key) is True
):
return True
except Exception:
pass
return False
@staticmethod
def _validate_effort_for_model(model: str, effort: Optional[str]) -> Optional[str]:
"""Return ``None`` if ``effort`` is allowed on ``model``, else an error message."""
if effort == "max" and not (
AnthropicConfig._is_claude_4_6_model(model)
or AnthropicConfig._is_claude_4_7_model(model)
or AnthropicConfig._supports_effort_level(model, "max")
):
return f"effort='max' is not supported by this model. Got model: {model}"
if effort == "xhigh" and not AnthropicConfig._supports_effort_level(
model, "xhigh"
):
return f"effort='xhigh' is not supported by this model. Got model: {model}"
return None
@staticmethod
def _model_supports_effort_param(model: str) -> bool:
"""Whether the model accepts ``output_config.effort`` at all."""
return any(
AnthropicConfig._supports_effort_level(model, level)
for level in ("low", "minimal", "medium", "high", "xhigh", "max")
)
@staticmethod
def _raise_invalid_reasoning_effort(
model: str, value: Any, llm_provider: str
) -> NoReturn:
"""Raise a ``BadRequestError`` for an unrecognised ``reasoning_effort``.
Args:
model: The model id the request was routed to (surfaced in the error).
value: The offending ``reasoning_effort`` value supplied by the caller.
llm_provider: Provider tag for the raised exception (``"anthropic"``,
``"bedrock_converse"``, ``"databricks"``, ...).
Raises:
litellm.exceptions.BadRequestError: Always.
"""
raise litellm.exceptions.BadRequestError(
message=(
f"Invalid reasoning_effort: {value!r}. "
f"Must be one of: 'minimal', 'low', 'medium', "
f"'high', 'xhigh', 'max', 'none'"
),
model=model,
llm_provider=llm_provider,
)
def get_supported_openai_params(self, model: str):
params = [
@@ -794,12 +902,11 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
def _map_reasoning_effort(
reasoning_effort: Optional[Union[REASONING_EFFORT, str]],
model: str,
llm_provider: str = "anthropic",
) -> Optional[AnthropicThinkingParam]:
if reasoning_effort is None or reasoning_effort == "none":
return None
if AnthropicConfig._is_claude_4_6_model(
model
) or AnthropicConfig._is_claude_4_7_model(model):
if AnthropicConfig._is_adaptive_thinking_model(model):
return AnthropicThinkingParam(
type="adaptive",
)
@@ -818,13 +925,34 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
type="enabled",
budget_tokens=DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET,
)
elif reasoning_effort == "xhigh":
return AnthropicThinkingParam(
type="enabled",
budget_tokens=DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET,
)
elif reasoning_effort == "max":
return AnthropicThinkingParam(
type="enabled",
budget_tokens=DEFAULT_REASONING_EFFORT_MAX_THINKING_BUDGET,
)
elif reasoning_effort == "minimal":
return AnthropicThinkingParam(
type="enabled",
budget_tokens=DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET,
budget_tokens=max(
DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET,
ANTHROPIC_MIN_THINKING_BUDGET_TOKENS,
),
)
else:
raise ValueError(f"Unmapped reasoning effort: {reasoning_effort}")
raise litellm.exceptions.BadRequestError(
message=(
f"Unmapped reasoning effort: {reasoning_effort!r}. "
f"Must be one of: 'minimal', 'low', 'medium', 'high', "
f"'xhigh', 'max', 'none'."
),
model=model,
llm_provider=llm_provider,
)
def _extract_json_schema_from_response_format(
self, value: Optional[dict]
@@ -1089,27 +1217,25 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
optional_params["thinking"] = value
elif param == "reasoning_effort" and isinstance(value, str):
mapped_thinking = AnthropicConfig._map_reasoning_effort(
reasoning_effort=value, model=model
reasoning_effort=value,
model=model,
llm_provider=self.custom_llm_provider or "anthropic",
)
if mapped_thinking is None:
optional_params.pop("thinking", None)
optional_params.pop("output_config", None)
else:
optional_params["thinking"] = mapped_thinking
# For Claude 4.6+ models, effort is controlled via output_config,
# not thinking budget_tokens. Map reasoning_effort to output_config.
if AnthropicConfig._is_claude_4_6_model(
model
) or AnthropicConfig._is_claude_4_7_model(model):
effort_map = {
"low": "low",
"minimal": "low",
"medium": "medium",
"high": "high",
"xhigh": "xhigh",
"max": "max",
}
mapped_effort = effort_map.get(value, value)
if AnthropicConfig._is_adaptive_thinking_model(model):
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
value
)
if mapped_effort is None:
AnthropicConfig._raise_invalid_reasoning_effort(
model=model,
value=value,
llm_provider=self.custom_llm_provider or "anthropic",
)
optional_params["output_config"] = {"effort": mapped_effort}
elif param == "web_search_options" and isinstance(value, dict):
hosted_web_search_tool = self.map_web_search_tool(
@@ -1532,29 +1658,31 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
output_config = optional_params.get("output_config")
if not output_config or not isinstance(output_config, dict):
return
if litellm.drop_params is True and not self._model_supports_effort_param(model):
litellm.verbose_logger.warning(
DROP_UNSUPPORTED_OUTPUT_CONFIG_WARNING,
model,
)
optional_params.pop("output_config", None)
data.pop("output_config", None)
return
effort = output_config.get("effort")
valid_efforts = ["high", "medium", "low", "xhigh", "max"]
if effort and effort not in valid_efforts:
raise ValueError(
f"Invalid effort value: {effort}. Must be one of: "
f"'high', 'medium', 'low', 'xhigh', 'max'"
if effort is not None and effort not in valid_efforts:
raise litellm.exceptions.BadRequestError(
message=(
f"Invalid effort value: {effort!r}. Must be one of: "
f"'high', 'medium', 'low', 'xhigh', 'max'"
),
model=model,
llm_provider=self.custom_llm_provider or "anthropic",
)
# ``max`` is for Opus 4.6+ output effort (not Sonnet 4.6, not Opus 4.5).
# Accept known Opus 4.6/4.7 id patterns and/or ``supports_max_reasoning_effort``
# in the model map (same pattern as ``xhigh`` below).
if effort == "max" and not (
self._is_opus_4_6_model(model)
or self._is_opus_4_7_model(model)
or self._supports_effort_level(model, "max")
):
raise ValueError(
f"effort='max' is not supported by this model. Got model: {model}"
)
# ``xhigh`` is data-driven via ``supports_xhigh_reasoning_effort`` so
# enabling it for a new model is a pure model-map change.
if effort == "xhigh" and not self._supports_effort_level(model, "xhigh"):
raise ValueError(
f"effort='xhigh' is not supported by this model. Got model: {model}"
gate_error = self._validate_effort_for_model(model, effort)
if gate_error is not None:
raise litellm.exceptions.BadRequestError(
message=gate_error,
model=model,
llm_provider=self.custom_llm_provider or "anthropic",
)
data["output_config"] = output_config
+12 -1
View File
@@ -273,7 +273,18 @@ class AnthropicModelInfo(BaseLLMModelInfo):
@staticmethod
def _is_adaptive_thinking_model(model: str) -> bool:
"""Claude 4.6+ models use adaptive thinking with output_config effort."""
"""Claude 4.6+ models use adaptive thinking with ``output_config.effort``."""
from litellm.utils import _supports_factory
try:
if _supports_factory(
model=model,
custom_llm_provider=None,
key="supports_adaptive_thinking",
):
return True
except Exception:
pass
return AnthropicModelInfo._is_claude_4_6_model(
model
) or AnthropicModelInfo._is_claude_4_7_model(model)
@@ -47,6 +47,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
"inference_geo",
"speed",
"output_config",
"reasoning_effort",
# TODO: Add Anthropic `metadata` support
# "metadata",
]
@@ -166,6 +167,62 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
return headers, api_base
@staticmethod
def _translate_reasoning_effort_to_anthropic(
model: str, optional_params: Dict
) -> None:
"""Map OpenAI-style ``reasoning_effort`` to native Anthropic params.
Caller-supplied ``thinking`` / ``output_config`` win over the alias.
``effort='none'`` clears both. Invalid efforts raise a 400.
"""
from litellm.exceptions import BadRequestError as _BadRequestError
from litellm.llms.anthropic.chat.transformation import (
REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT,
AnthropicConfig,
)
reasoning_effort = optional_params.pop("reasoning_effort", None)
if not isinstance(reasoning_effort, str):
return
try:
mapped_thinking = AnthropicConfig._map_reasoning_effort(
reasoning_effort=reasoning_effort, model=model
)
except _BadRequestError as e:
raise AnthropicError(message=str(e.message), status_code=400)
if mapped_thinking is None:
optional_params.pop("thinking", None)
optional_params.pop("output_config", None)
return
optional_params.setdefault("thinking", mapped_thinking)
if AnthropicModelInfo._is_adaptive_thinking_model(model):
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
reasoning_effort
)
if mapped_effort is None:
raise AnthropicError(
message=(
f"Invalid reasoning_effort: {reasoning_effort!r}. "
f"Must be one of: 'minimal', 'low', 'medium', 'high', "
f"'xhigh', 'max', 'none'"
),
status_code=400,
)
gate_error = AnthropicConfig._validate_effort_for_model(
model, mapped_effort
)
if gate_error is not None:
raise AnthropicError(message=gate_error, status_code=400)
existing_output_config = optional_params.get("output_config")
if not isinstance(existing_output_config, dict):
existing_output_config = {}
existing_output_config.setdefault("effort", mapped_effort)
optional_params["output_config"] = existing_output_config
@staticmethod
def _translate_legacy_thinking_for_adaptive_model(
model: str, optional_params: Dict
@@ -217,6 +274,11 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
status_code=400,
)
self._translate_reasoning_effort_to_anthropic(
model=model,
optional_params=anthropic_messages_optional_request_params,
)
self._translate_legacy_thinking_for_adaptive_model(
model=model,
optional_params=anthropic_messages_optional_request_params,
@@ -12,6 +12,23 @@ if TYPE_CHECKING:
pass
def _promote_extra_body_to_optional_params(optional_params: dict) -> None:
"""Promote anthropic-native passthrough keys out of ``extra_body``.
``azure_ai`` is an OpenAI-compatible provider, so non-OpenAI kwargs like
``output_config`` get auto-routed into ``extra_body`` by
``add_provider_specific_params_to_optional_params``. For the AzureAnthropic
route those keys must reach the request body and be validated, so promote
them. ``setdefault`` keeps explicit top-level values authoritative.
"""
extra_body = optional_params.get("extra_body")
if not isinstance(extra_body, dict) or not extra_body:
return
for k, v in extra_body.items():
optional_params.setdefault(k, v)
optional_params.pop("extra_body", None)
class AzureAnthropicConfig(AnthropicConfig):
"""
Azure Anthropic configuration that extends AnthropicConfig.
@@ -39,6 +56,8 @@ class AzureAnthropicConfig(AnthropicConfig):
1. API key via 'api-key' header
2. Azure AD token via 'Authorization: Bearer <token>' header
"""
_promote_extra_body_to_optional_params(optional_params)
# Convert dict to GenericLiteLLMParams if needed
if isinstance(litellm_params, dict):
# Ensure api_key is included if provided
@@ -101,7 +120,8 @@ class AzureAnthropicConfig(AnthropicConfig):
Transform request using parent AnthropicConfig, then remove unsupported params.
Azure Anthropic doesn't support extra_body, max_retries, or stream_options parameters.
"""
# Call parent transform_request
_promote_extra_body_to_optional_params(optional_params)
data = super().transform_request(
model=model,
messages=messages,
+1 -3
View File
@@ -87,9 +87,7 @@ class BaseConfig(ABC):
return {
k: v
for k, v in cls.__dict__.items()
if not k.startswith("__")
and not k.startswith("_abc")
and not k.startswith("_is_base_class")
if not k.startswith("_")
and not isinstance(
v,
(
@@ -31,7 +31,11 @@ from litellm.litellm_core_utils.prompt_templates.factory import (
_bedrock_converse_messages_pt,
_bedrock_tools_pt,
)
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
from litellm.llms.anthropic.chat.transformation import (
DROP_UNSUPPORTED_OUTPUT_CONFIG_WARNING,
REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT,
AnthropicConfig,
)
from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException
from litellm.types.llms.bedrock import *
from litellm.types.llms.openai import (
@@ -189,7 +193,7 @@ class AmazonConverseConfig(BaseConfig):
return {
k: v
for k, v in cls.__dict__.items()
if not k.startswith("__")
if not k.startswith("_")
and not isinstance(
v,
(
@@ -410,52 +414,65 @@ class AmazonConverseConfig(BaseConfig):
"""
Handle the reasoning_effort parameter based on the model type.
Different model families handle reasoning effort differently:
- GPT-OSS models: Keep reasoning_effort as-is (passed to additionalModelRequestFields)
- Nova 2 models: Transform to reasoningConfig structure
- Other models (Anthropic, etc.): Convert to thinking parameter
Args:
model: The model identifier
reasoning_effort: The reasoning effort value
optional_params: Dictionary of optional parameters to update in-place
Examples:
>>> config = AmazonConverseConfig()
>>> params = {}
>>> config._handle_reasoning_effort_parameter("gpt-oss-model", "high", params)
>>> params
{'reasoning_effort': 'high'}
>>> params = {}
>>> config._handle_reasoning_effort_parameter("amazon.nova-2-lite-v1:0", "high", params)
>>> params
{'reasoningConfig': {'type': 'enabled', 'maxReasoningEffort': 'high'}}
>>> params = {}
>>> config._handle_reasoning_effort_parameter("anthropic.claude-3", "high", params)
>>> params
{'thinking': {'type': 'enabled', 'budget_tokens': 10000}}
- GPT-OSS models: passed through unchanged via additionalModelRequestFields.
- Nova 2 models: transformed to reasoningConfig.
- Anthropic models: mapped to ``thinking`` (and ``output_config.effort`` on
adaptive Claude 4.6 / 4.7).
"""
if "gpt-oss" in model:
# GPT-OSS models: keep reasoning_effort as-is
# It will be passed through to additionalModelRequestFields
optional_params["reasoning_effort"] = reasoning_effort
elif self._is_nova_2_model(model):
# Nova 2 models: transform to reasoningConfig
reasoning_config = self._transform_reasoning_effort_to_reasoning_config(
reasoning_effort
)
optional_params.update(reasoning_config)
else:
# Anthropic and other models: convert to thinking parameter
mapped_thinking = AnthropicConfig._map_reasoning_effort(
reasoning_effort=reasoning_effort, model=model
reasoning_effort=reasoning_effort,
model=model,
llm_provider="bedrock_converse",
)
if mapped_thinking is None:
optional_params.pop("thinking", None)
optional_params.pop("output_config", None)
else:
optional_params["thinking"] = mapped_thinking
if AnthropicConfig._is_adaptive_thinking_model(model):
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
reasoning_effort
)
if mapped_effort is None:
AnthropicConfig._raise_invalid_reasoning_effort(
model=model,
value=reasoning_effort,
llm_provider="bedrock_converse",
)
self._validate_anthropic_adaptive_effort(
model=model, effort=mapped_effort
)
optional_params["output_config"] = {"effort": mapped_effort}
@staticmethod
def _validate_anthropic_adaptive_effort(model: str, effort: str) -> None:
"""Validate ``output_config.effort`` for adaptive-thinking Claude 4.6/4.7."""
valid_efforts = {"high", "medium", "low", "xhigh", "max"}
if effort not in valid_efforts:
raise litellm.exceptions.BadRequestError(
message=(
f"Invalid reasoning_effort/output_config.effort value: "
f"{effort!r}. Must be one of: 'low', 'medium', 'high', "
f"'xhigh', or 'max'."
),
model=model,
llm_provider="bedrock_converse",
)
error = AnthropicConfig._validate_effort_for_model(model=model, effort=effort)
if error is not None:
raise litellm.exceptions.BadRequestError(
message=error,
model=model,
llm_provider="bedrock_converse",
)
@staticmethod
def _clamp_thinking_budget_tokens(optional_params: dict) -> None:
@@ -1196,9 +1213,11 @@ class AmazonConverseConfig(BaseConfig):
+ supported_config_params
)
inference_params.pop("json_mode", None) # used for handling json_schema
# Anthropic-only key. Bedrock expects `outputConfig` (camelCase) and
# will reject `output_config` if it leaks through pass-through routes.
inference_params.pop("output_config", None)
# Anthropic-only ``output_config`` (snake_case) — re-attached to
# ``additionalModelRequestFields`` for Anthropic models below. The
# Bedrock-native ``outputConfig`` (camelCase) is handled separately.
anthropic_output_config = inference_params.pop("output_config", None)
# Extract requestMetadata before processing other parameters
request_metadata = inference_params.pop("requestMetadata", None)
@@ -1208,9 +1227,6 @@ class AmazonConverseConfig(BaseConfig):
output_config: Optional[OutputConfigBlock] = inference_params.pop(
"outputConfig", None
)
inference_params.pop(
"output_config", None
) # Bedrock Converse doesn't support it
# keep supported params in 'inference_params', and set all model-specific params in 'additional_request_params'
additional_request_params = {
@@ -1253,6 +1269,27 @@ class AmazonConverseConfig(BaseConfig):
additional_request_params
)
if anthropic_output_config is not None and isinstance(
anthropic_output_config, dict
):
base_model = BedrockModelInfo.get_base_model(model)
if base_model.startswith("anthropic"):
if (
litellm.drop_params is True
and not AnthropicConfig._model_supports_effort_param(model)
):
litellm.verbose_logger.warning(
DROP_UNSUPPORTED_OUTPUT_CONFIG_WARNING,
model,
)
else:
effort = anthropic_output_config.get("effort")
if effort is not None:
self._validate_anthropic_adaptive_effort(
model=model, effort=effort
)
additional_request_params["output_config"] = anthropic_output_config
return (
inference_params,
additional_request_params,
@@ -1376,9 +1413,25 @@ class AmazonConverseConfig(BaseConfig):
# Append pre-formatted tools (systemTool etc.) after transformation
bedrock_tools.extend(pre_formatted_tools)
# Opus 4.5 gates ``output_config.effort`` behind a beta header;
# Claude 4.6/4.7 accept it without one.
base_model = BedrockModelInfo.get_base_model(model)
if base_model.startswith("anthropic"):
output_config = additional_request_params.get("output_config")
if (
isinstance(output_config, dict)
and output_config.get("effort") is not None
and not AnthropicConfig._is_adaptive_thinking_model(model)
):
from litellm.types.llms.anthropic import (
ANTHROPIC_EFFORT_BETA_HEADER,
)
if ANTHROPIC_EFFORT_BETA_HEADER not in anthropic_beta_list:
anthropic_beta_list.append(ANTHROPIC_EFFORT_BETA_HEADER)
# Set anthropic_beta in additional_request_params if we have any beta features
# ONLY apply to Anthropic/Claude models - other models (e.g., Qwen, Llama) don't support this field
base_model = BedrockModelInfo.get_base_model(model)
if anthropic_beta_list and base_model.startswith("anthropic"):
additional_request_params["anthropic_beta"] = anthropic_beta_list
@@ -169,7 +169,6 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
anthropic_request.pop("model", None)
anthropic_request.pop("stream", None)
anthropic_request.pop("output_format", None)
anthropic_request.pop("output_config", None)
if "anthropic_version" not in anthropic_request:
anthropic_request["anthropic_version"] = self.anthropic_version
@@ -12,9 +12,14 @@ from typing import (
import httpx
import litellm
from litellm.anthropic_beta_headers_manager import filter_and_transform_beta_headers
from litellm.constants import BEDROCK_MIN_THINKING_BUDGET_TOKENS
from litellm.litellm_core_utils.litellm_logging import verbose_logger
from litellm.llms.anthropic.chat.transformation import (
DROP_UNSUPPORTED_OUTPUT_CONFIG_WARNING,
AnthropicConfig,
)
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
from litellm.llms.anthropic.experimental_pass_through.messages.transformation import (
AnthropicMessagesConfig,
@@ -580,6 +585,17 @@ class AmazonAnthropicClaudeMessagesConfig(
if filtered_betas:
anthropic_messages_request["anthropic_beta"] = filtered_betas
if (
litellm.drop_params is True
and "output_config" in anthropic_messages_request
and not AnthropicConfig._model_supports_effort_param(model)
):
verbose_logger.warning(
DROP_UNSUPPORTED_OUTPUT_CONFIG_WARNING,
model,
)
anthropic_messages_request.pop("output_config", None)
# 7. Final safety net: filter top-level fields to the Bedrock Invoke allowlist.
# Catches Anthropic-only extensions (context_management, output_config, speed,
# mcp_servers, ...) and any future additions Claude Code may start sending.
+27 -3
View File
@@ -56,7 +56,10 @@ from litellm.types.utils import (
Usage,
)
from ...anthropic.chat.transformation import AnthropicConfig
from ...anthropic.chat.transformation import (
REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT,
AnthropicConfig,
)
from ...openai_like.chat.transformation import OpenAILikeChatConfig
from ..common_utils import DatabricksBase, DatabricksException
@@ -330,9 +333,30 @@ class DatabricksConfig(DatabricksBase, OpenAILikeChatConfig, AnthropicConfig):
) # unsupported for claude models - if json_schema -> convert to tool call
if "reasoning_effort" in non_default_params and "claude" in model:
optional_params["thinking"] = AnthropicConfig._map_reasoning_effort(
reasoning_effort=non_default_params.get("reasoning_effort"), model=model
reasoning_effort_value = non_default_params.get("reasoning_effort")
mapped_thinking = AnthropicConfig._map_reasoning_effort(
reasoning_effort=reasoning_effort_value,
model=model,
llm_provider="databricks",
)
if mapped_thinking is None:
optional_params.pop("thinking", None)
optional_params.pop("output_config", None)
else:
optional_params["thinking"] = mapped_thinking
if AnthropicConfig._is_adaptive_thinking_model(model):
mapped_effort: Optional[str] = None
if isinstance(reasoning_effort_value, str):
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
reasoning_effort_value
)
if mapped_effort is None:
AnthropicConfig._raise_invalid_reasoning_effort(
model=model,
value=reasoning_effort_value,
llm_provider="databricks",
)
optional_params["output_config"] = {"effort": mapped_effort}
optional_params.pop("reasoning_effort", None)
## handle thinking tokens
self.update_optional_params_with_thinking_tokens(
@@ -159,10 +159,6 @@ class VertexAIPartnerModelsAnthropicMessagesConfig(AnthropicMessagesConfig, Vert
"model", None
) # do not pass model in request body to vertex ai
# Vertex AI Claude accepts ``output_config.format`` (structured outputs)
# and ``output_format``, but rejects ``output_config.effort`` with 400
# "Extra inputs are not permitted". Sanitize in place so the supported
# bits flow through.
sanitize_vertex_anthropic_output_params(anthropic_messages_request)
return anthropic_messages_request
@@ -11,11 +11,9 @@ keeps the parent module's import surface narrow.
"""
# Keys inside ``output_config`` that Vertex AI Claude does not accept.
# Today only ``effort`` triggers "Extra inputs are not permitted"; add new
# entries here as Vertex parity drifts. Keep this list narrow — anything
# Vertex DOES accept (e.g. ``format`` for structured outputs) must be
# preserved so callers can rely on Anthropic-native features.
VERTEX_UNSUPPORTED_OUTPUT_CONFIG_KEYS: frozenset = frozenset({"effort"})
# Add an entry only when a 400 "Extra inputs are not permitted" is
# reproducible against the live Vertex endpoint.
VERTEX_UNSUPPORTED_OUTPUT_CONFIG_KEYS: frozenset = frozenset()
def sanitize_vertex_anthropic_output_params(data: dict) -> None:
@@ -106,11 +106,6 @@ class VertexAIAnthropicConfig(AnthropicConfig):
data.pop("model", None) # vertex anthropic doesn't accept 'model' parameter
# Vertex AI Claude accepts ``output_config.format`` (structured outputs /
# JSON Schema) but NOT ``output_config.effort`` — sending ``effort`` to
# Vertex returns 400 "Extra inputs are not permitted". Sanitize in place:
# forward the structured-output bits, drop the unsupported keys.
# Same treatment for the legacy top-level ``output_format`` field.
sanitize_vertex_anthropic_output_params(data)
tools = optional_params.get("tools")
+35
View File
@@ -223,8 +223,43 @@ class XAIChatConfig(OpenAIGPTConfig):
self._enhance_usage_with_xai_web_search_fields(response, raw_response_json)
except Exception as e:
verbose_logger.debug(f"Error extracting X.AI web search usage: {e}")
self._fold_reasoning_tokens_into_completion(response)
return response
@staticmethod
def _fold_reasoning_tokens_into_completion(model_response: ModelResponse) -> None:
"""Reconcile xAI Usage to the OpenAI invariant.
xAI accounts ``reasoning_tokens`` separately from
``completion_tokens`` while still summing them into ``total_tokens``.
OpenAI's contract (o1/o3) folds reasoning into ``completion_tokens``,
so fold here to keep ``total = prompt + completion``. Idempotent.
"""
usage = getattr(model_response, "usage", None)
if usage is None:
return
details = getattr(usage, "completion_tokens_details", None)
reasoning_tokens = (
int(getattr(details, "reasoning_tokens", 0) or 0) if details else 0
)
if reasoning_tokens <= 0:
return
prompt_tokens = int(getattr(usage, "prompt_tokens", 0) or 0)
completion_tokens = int(getattr(usage, "completion_tokens", 0) or 0)
total_tokens = int(getattr(usage, "total_tokens", 0) or 0)
if total_tokens == prompt_tokens + completion_tokens:
return
# Guard against double-counting if xAI changes accounting.
if total_tokens != prompt_tokens + completion_tokens + reasoning_tokens:
return
usage.completion_tokens = completion_tokens + reasoning_tokens
def _enhance_usage_with_xai_web_search_fields(
self, model_response: ModelResponse, raw_response_json: dict
) -> None:
+12 -3
View File
@@ -25,16 +25,25 @@ def cost_per_token(model: str, usage: Usage) -> Tuple[float, float]:
Returns:
Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd
"""
# XAI-specific completion cost calculation
# For XAI models, completion is billed as (visible completion tokens + reasoning tokens)
# XAI-specific completion cost: completion is billed as visible + reasoning
# tokens. Detect when the transformation layer already folded them so we
# don't double-count; fall back to raw xAI shape for callers that bypass
# the transformation (e.g. proxy logs replayed into cost calc).
prompt_tokens = int(getattr(usage, "prompt_tokens", 0) or 0)
completion_tokens = int(getattr(usage, "completion_tokens", 0) or 0)
total_tokens = int(getattr(usage, "total_tokens", 0) or 0)
reasoning_tokens = 0
if hasattr(usage, "completion_tokens_details") and usage.completion_tokens_details:
reasoning_tokens = int(
getattr(usage.completion_tokens_details, "reasoning_tokens", 0) or 0
)
total_completion_tokens = completion_tokens + reasoning_tokens
already_normalised = total_tokens == prompt_tokens + completion_tokens
total_completion_tokens = (
completion_tokens
if already_normalised
else completion_tokens + reasoning_tokens
)
modified_usage = Usage(
prompt_tokens=usage.prompt_tokens,
@@ -977,6 +977,7 @@
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
@@ -1162,6 +1163,21 @@
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"anthropic.claude-mythos-preview": {
"input_cost_per_token": 0,
"output_cost_per_token": 0,
"litellm_provider": "bedrock",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"supports_function_calling": true,
"supports_vision": true,
"supports_prompt_caching": false,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": true,
"supports_tool_choice": true
},
"global.anthropic.claude-opus-4-7": {
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_1hr": 1e-05,
@@ -1307,6 +1323,7 @@
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
@@ -1336,6 +1353,7 @@
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
@@ -1365,6 +1383,7 @@
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
@@ -1393,6 +1412,7 @@
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
@@ -1421,6 +1441,7 @@
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
@@ -1915,6 +1936,7 @@
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true
@@ -2038,6 +2060,7 @@
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
@@ -9212,6 +9235,7 @@
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
@@ -9347,6 +9371,7 @@
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
@@ -9374,6 +9399,7 @@
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
@@ -9477,7 +9503,6 @@
"us": 1.1,
"fast": 6.0
},
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"claude-opus-4-7-20260416": {
@@ -9512,7 +9537,6 @@
"us": 1.1,
"fast": 6.0
},
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"claude-sonnet-4-20250514": {
@@ -10790,6 +10814,7 @@
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": true,
"supports_tool_choice": true
},
"databricks/databricks-claude-sonnet-4": {
@@ -15548,7 +15573,7 @@
"mode": "embedding",
"output_cost_per_token": 0,
"output_vector_size": 3072,
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing",
"source": "https://ai.google.dev/gemini-api/docs/embeddings#multimodal",
"supports_multimodal": true,
"uses_embed_content": true
},
@@ -17150,7 +17175,8 @@
],
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_vision": true
"supports_vision": true,
"supports_minimal_reasoning_effort": true
},
"github_copilot/claude-opus-4.6-fast": {
"litellm_provider": "github_copilot",
@@ -17663,7 +17689,8 @@
"mode": "chat",
"output_cost_per_token": 2.5e-05,
"supports_function_calling": true,
"supports_vision": true
"supports_vision": true,
"supports_minimal_reasoning_effort": true
},
"gmi/anthropic/claude-sonnet-4.5": {
"input_cost_per_token": 3e-06,
@@ -21142,7 +21169,7 @@
},
"gradient_ai/alibaba-qwen3-32b": {
"litellm_provider": "gradient_ai",
"max_tokens": 2048,
"max_tokens": 40960,
"mode": "chat",
"supported_endpoints": [
"/v1/chat/completions"
@@ -21150,7 +21177,9 @@
"supported_modalities": [
"text"
],
"supports_tool_choice": false
"supports_tool_choice": false,
"max_input_tokens": 131072,
"max_output_tokens": 40960
},
"gradient_ai/anthropic-claude-3-opus": {
"input_cost_per_token": 1.5e-05,
@@ -21164,7 +21193,9 @@
"supported_modalities": [
"text"
],
"supports_tool_choice": false
"supports_tool_choice": false,
"max_input_tokens": 200000,
"max_output_tokens": 1024
},
"gradient_ai/anthropic-claude-3.5-haiku": {
"input_cost_per_token": 8e-07,
@@ -21178,7 +21209,9 @@
"supported_modalities": [
"text"
],
"supports_tool_choice": false
"supports_tool_choice": false,
"max_input_tokens": 200000,
"max_output_tokens": 1024
},
"gradient_ai/anthropic-claude-3.5-sonnet": {
"input_cost_per_token": 3e-06,
@@ -21192,7 +21225,9 @@
"supported_modalities": [
"text"
],
"supports_tool_choice": false
"supports_tool_choice": false,
"max_input_tokens": 200000,
"max_output_tokens": 1024
},
"gradient_ai/anthropic-claude-3.7-sonnet": {
"input_cost_per_token": 3e-06,
@@ -21206,7 +21241,9 @@
"supported_modalities": [
"text"
],
"supports_tool_choice": false
"supports_tool_choice": false,
"max_input_tokens": 200000,
"max_output_tokens": 1024
},
"gradient_ai/deepseek-r1-distill-llama-70b": {
"input_cost_per_token": 9.9e-07,
@@ -21220,7 +21257,9 @@
"supported_modalities": [
"text"
],
"supports_tool_choice": false
"supports_tool_choice": false,
"max_input_tokens": 32768,
"max_output_tokens": 8000
},
"gradient_ai/llama3-8b-instruct": {
"input_cost_per_token": 2e-07,
@@ -21234,7 +21273,9 @@
"supported_modalities": [
"text"
],
"supports_tool_choice": false
"supports_tool_choice": false,
"max_input_tokens": 8192,
"max_output_tokens": 512
},
"gradient_ai/llama3.3-70b-instruct": {
"input_cost_per_token": 6.5e-07,
@@ -21248,7 +21289,9 @@
"supported_modalities": [
"text"
],
"supports_tool_choice": false
"supports_tool_choice": false,
"max_input_tokens": 128000,
"max_output_tokens": 2048
},
"gradient_ai/mistral-nemo-instruct-2407": {
"input_cost_per_token": 3e-07,
@@ -21262,7 +21305,9 @@
"supported_modalities": [
"text"
],
"supports_tool_choice": false
"supports_tool_choice": false,
"max_input_tokens": 128000,
"max_output_tokens": 512
},
"gradient_ai/openai-gpt-4o": {
"litellm_provider": "gradient_ai",
@@ -21274,7 +21319,9 @@
"supported_modalities": [
"text"
],
"supports_tool_choice": false
"supports_tool_choice": false,
"max_input_tokens": 128000,
"max_output_tokens": 16384
},
"gradient_ai/openai-gpt-4o-mini": {
"litellm_provider": "gradient_ai",
@@ -21286,7 +21333,9 @@
"supported_modalities": [
"text"
],
"supports_tool_choice": false
"supports_tool_choice": false,
"max_input_tokens": 128000,
"max_output_tokens": 16384
},
"gradient_ai/openai-o3": {
"input_cost_per_token": 2e-06,
@@ -21300,7 +21349,9 @@
"supported_modalities": [
"text"
],
"supports_tool_choice": false
"supports_tool_choice": false,
"max_input_tokens": 200000,
"max_output_tokens": 100000
},
"gradient_ai/openai-o3-mini": {
"input_cost_per_token": 1.1e-06,
@@ -21314,7 +21365,9 @@
"supported_modalities": [
"text"
],
"supports_tool_choice": false
"supports_tool_choice": false,
"max_input_tokens": 200000,
"max_output_tokens": 100000
},
"lemonade/Qwen3-Coder-30B-A3B-Instruct-GGUF": {
"input_cost_per_token": 0,
@@ -21614,11 +21667,13 @@
},
"heroku/claude-3-5-haiku": {
"litellm_provider": "heroku",
"max_tokens": 4096,
"max_tokens": 8192,
"mode": "chat",
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true
"supports_tool_choice": true,
"max_input_tokens": 200000,
"max_output_tokens": 8192
},
"heroku/claude-3-5-sonnet-latest": {
"litellm_provider": "heroku",
@@ -21626,7 +21681,9 @@
"mode": "chat",
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true
"supports_tool_choice": true,
"max_input_tokens": 200000,
"max_output_tokens": 8192
},
"heroku/claude-3-7-sonnet": {
"litellm_provider": "heroku",
@@ -21634,7 +21691,9 @@
"mode": "chat",
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true
"supports_tool_choice": true,
"max_input_tokens": 200000,
"max_output_tokens": 8192
},
"heroku/claude-4-sonnet": {
"litellm_provider": "heroku",
@@ -21642,7 +21701,9 @@
"mode": "chat",
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true
"supports_tool_choice": true,
"max_input_tokens": 200000,
"max_output_tokens": 8192
},
"high/1024-x-1024/gpt-image-1": {
"input_cost_per_image": 0.167,
@@ -22450,48 +22511,6 @@
"/v1/images/generations"
]
},
"luminous-base": {
"input_cost_per_token": 3e-05,
"litellm_provider": "aleph_alpha",
"max_tokens": 2048,
"mode": "completion",
"output_cost_per_token": 3.3e-05
},
"luminous-base-control": {
"input_cost_per_token": 3.75e-05,
"litellm_provider": "aleph_alpha",
"max_tokens": 2048,
"mode": "chat",
"output_cost_per_token": 4.125e-05
},
"luminous-extended": {
"input_cost_per_token": 4.5e-05,
"litellm_provider": "aleph_alpha",
"max_tokens": 2048,
"mode": "completion",
"output_cost_per_token": 4.95e-05
},
"luminous-extended-control": {
"input_cost_per_token": 5.625e-05,
"litellm_provider": "aleph_alpha",
"max_tokens": 2048,
"mode": "chat",
"output_cost_per_token": 6.1875e-05
},
"luminous-supreme": {
"input_cost_per_token": 0.000175,
"litellm_provider": "aleph_alpha",
"max_tokens": 2048,
"mode": "completion",
"output_cost_per_token": 0.0001925
},
"luminous-supreme-control": {
"input_cost_per_token": 0.00021875,
"litellm_provider": "aleph_alpha",
"max_tokens": 2048,
"mode": "chat",
"output_cost_per_token": 0.000240625
},
"max-x-max/50-steps/stability.stable-diffusion-xl-v0": {
"litellm_provider": "bedrock",
"max_input_tokens": 77,
@@ -25963,12 +25982,14 @@
"input_cost_per_image": 0.0004,
"input_cost_per_token": 2.5e-07,
"litellm_provider": "openrouter",
"max_tokens": 200000,
"max_tokens": 4096,
"mode": "chat",
"output_cost_per_token": 1.25e-06,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_vision": true
"supports_vision": true,
"max_input_tokens": 200000,
"max_output_tokens": 4096
},
"openrouter/anthropic/claude-3.5-sonnet": {
"input_cost_per_token": 3e-06,
@@ -26087,6 +26108,7 @@
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 159,
@@ -26105,6 +26127,7 @@
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_minimal_reasoning_effort": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true,
@@ -26126,6 +26149,7 @@
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
@@ -26174,6 +26198,29 @@
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
},
"openrouter/anthropic/claude-opus-4.7": {
"cache_creation_input_token_cost": 6.25e-06,
"cache_read_input_token_cost": 5e-07,
"input_cost_per_token": 5e-06,
"litellm_provider": "openrouter",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 2.5e-05,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"supports_xhigh_reasoning_effort": true,
"tool_use_system_prompt_tokens": 346
},
"openrouter/bytedance/ui-tars-1.5-7b": {
"input_cost_per_token": 1e-07,
"litellm_provider": "openrouter",
@@ -26539,18 +26586,22 @@
"openrouter/mancer/weaver": {
"input_cost_per_token": 5.625e-06,
"litellm_provider": "openrouter",
"max_tokens": 8000,
"max_tokens": 2000,
"mode": "chat",
"output_cost_per_token": 5.625e-06,
"supports_tool_choice": true
"supports_tool_choice": true,
"max_input_tokens": 8000,
"max_output_tokens": 2000
},
"openrouter/meta-llama/llama-3-70b-instruct": {
"input_cost_per_token": 5.9e-07,
"litellm_provider": "openrouter",
"max_tokens": 8192,
"max_tokens": 8000,
"mode": "chat",
"output_cost_per_token": 7.9e-07,
"supports_tool_choice": true
"supports_tool_choice": true,
"max_input_tokens": 8192,
"max_output_tokens": 8000
},
"openrouter/minimax/minimax-m2": {
"input_cost_per_token": 2.55e-07,
@@ -26638,34 +26689,42 @@
"openrouter/mistralai/mistral-7b-instruct": {
"input_cost_per_token": 1.3e-07,
"litellm_provider": "openrouter",
"max_tokens": 8192,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_token": 1.3e-07,
"supports_tool_choice": true
"supports_tool_choice": true,
"max_input_tokens": 32768,
"max_output_tokens": 8191
},
"openrouter/mistralai/mistral-large": {
"input_cost_per_token": 8e-06,
"litellm_provider": "openrouter",
"max_tokens": 32000,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_token": 2.4e-05,
"supports_tool_choice": true
"supports_tool_choice": true,
"max_input_tokens": 128000,
"max_output_tokens": 8191
},
"openrouter/mistralai/mistral-small-3.1-24b-instruct": {
"input_cost_per_token": 1e-07,
"litellm_provider": "openrouter",
"max_tokens": 32000,
"max_tokens": 131072,
"mode": "chat",
"output_cost_per_token": 3e-07,
"supports_tool_choice": true
"supports_tool_choice": true,
"max_input_tokens": 131072,
"max_output_tokens": 131072
},
"openrouter/mistralai/mistral-small-3.2-24b-instruct": {
"input_cost_per_token": 1e-07,
"litellm_provider": "openrouter",
"max_tokens": 32000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3e-07,
"supports_tool_choice": true
"supports_tool_choice": true,
"max_input_tokens": 128000,
"max_output_tokens": 128000
},
"openrouter/mistralai/mixtral-8x22b-instruct": {
"input_cost_per_token": 6.5e-07,
@@ -26673,7 +26732,9 @@
"max_tokens": 65536,
"mode": "chat",
"output_cost_per_token": 6.5e-07,
"supports_tool_choice": true
"supports_tool_choice": true,
"max_input_tokens": 65536,
"max_output_tokens": 65536
},
"openrouter/moonshotai/kimi-k2.5": {
"cache_read_input_token_cost": 1e-07,
@@ -26693,26 +26754,32 @@
"openrouter/openai/gpt-3.5-turbo": {
"input_cost_per_token": 1.5e-06,
"litellm_provider": "openrouter",
"max_tokens": 4095,
"max_tokens": 4096,
"mode": "chat",
"output_cost_per_token": 2e-06,
"supports_tool_choice": true
"supports_tool_choice": true,
"max_input_tokens": 16385,
"max_output_tokens": 4096
},
"openrouter/openai/gpt-3.5-turbo-16k": {
"input_cost_per_token": 3e-06,
"litellm_provider": "openrouter",
"max_tokens": 16383,
"max_tokens": 4096,
"mode": "chat",
"output_cost_per_token": 4e-06,
"supports_tool_choice": true
"supports_tool_choice": true,
"max_input_tokens": 16385,
"max_output_tokens": 4096
},
"openrouter/openai/gpt-4": {
"input_cost_per_token": 3e-05,
"litellm_provider": "openrouter",
"max_tokens": 8192,
"max_tokens": 4096,
"mode": "chat",
"output_cost_per_token": 6e-05,
"supports_tool_choice": true
"supports_tool_choice": true,
"max_input_tokens": 8191,
"max_output_tokens": 4096
},
"openrouter/openai/gpt-4.1": {
"cache_read_input_token_cost": 5e-07,
@@ -27220,10 +27287,12 @@
"openrouter/undi95/remm-slerp-l2-13b": {
"input_cost_per_token": 1.875e-06,
"litellm_provider": "openrouter",
"max_tokens": 6144,
"max_tokens": 4096,
"mode": "chat",
"output_cost_per_token": 1.875e-06,
"supports_tool_choice": true
"supports_tool_choice": true,
"max_input_tokens": 6144,
"max_output_tokens": 4096
},
"openrouter/x-ai/grok-4": {
"input_cost_per_token": 3e-06,
@@ -28026,7 +28095,8 @@
"mode": "responses",
"supports_web_search": true,
"supports_reasoning": false,
"supports_function_calling": true
"supports_function_calling": true,
"supports_minimal_reasoning_effort": true
},
"perplexity/anthropic/claude-sonnet-4-5": {
"litellm_provider": "perplexity",
@@ -29844,14 +29914,16 @@
"together_ai/deepseek-ai/DeepSeek-V3.1": {
"input_cost_per_token": 6e-07,
"litellm_provider": "together_ai",
"max_tokens": 128000,
"max_tokens": 16384,
"mode": "chat",
"output_cost_per_token": 1.7e-06,
"source": "https://www.together.ai/models/deepseek-v3-1",
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true
"supports_tool_choice": true,
"max_input_tokens": 128000,
"max_output_tokens": 16384
},
"together_ai/meta-llama/Llama-3.2-3B-Instruct-Turbo": {
"litellm_provider": "together_ai",
@@ -30503,6 +30575,7 @@
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_minimal_reasoning_effort": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
@@ -30531,6 +30604,7 @@
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_minimal_reasoning_effort": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
@@ -30558,6 +30632,7 @@
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_minimal_reasoning_effort": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
@@ -31138,6 +31213,7 @@
"output_cost_per_token": 2.5e-05,
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_minimal_reasoning_effort": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
@@ -32330,6 +32406,7 @@
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_minimal_reasoning_effort": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
@@ -32356,6 +32433,7 @@
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_minimal_reasoning_effort": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
@@ -32522,6 +32600,7 @@
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
@@ -34426,6 +34505,7 @@
"output_cost_per_token": 1.5e-05,
"source": "https://x.ai/api#pricing",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_response_schema": false,
"supports_tool_choice": true,
"supports_web_search": true
@@ -34441,6 +34521,7 @@
"output_cost_per_token": 1.5e-05,
"source": "https://x.ai/api#pricing",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_response_schema": false,
"supports_tool_choice": true,
"supports_web_search": true
@@ -34456,6 +34537,7 @@
"output_cost_per_token": 2.5e-05,
"source": "https://x.ai/api#pricing",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_response_schema": false,
"supports_tool_choice": true,
"supports_web_search": true
@@ -34471,6 +34553,7 @@
"output_cost_per_token": 2.5e-05,
"source": "https://x.ai/api#pricing",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_response_schema": false,
"supports_tool_choice": true,
"supports_web_search": true
@@ -34486,6 +34569,7 @@
"output_cost_per_token": 1.5e-05,
"source": "https://x.ai/api#pricing",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_response_schema": false,
"supports_tool_choice": true,
"supports_web_search": true
@@ -34502,6 +34586,7 @@
"output_cost_per_token": 5e-07,
"source": "https://x.ai/api#pricing",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": false,
"supports_tool_choice": true,
@@ -34519,6 +34604,7 @@
"output_cost_per_token": 5e-07,
"source": "https://x.ai/api#pricing",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": false,
"supports_tool_choice": true,
@@ -34535,6 +34621,7 @@
"output_cost_per_token": 4e-06,
"source": "https://x.ai/api#pricing",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": false,
"supports_tool_choice": true,
@@ -34551,6 +34638,7 @@
"output_cost_per_token": 4e-06,
"source": "https://x.ai/api#pricing",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": false,
"supports_tool_choice": true,
@@ -34567,6 +34655,7 @@
"output_cost_per_token": 4e-06,
"source": "https://x.ai/api#pricing",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": false,
"supports_tool_choice": true,
@@ -34583,6 +34672,7 @@
"output_cost_per_token": 5e-07,
"source": "https://x.ai/api#pricing",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": false,
"supports_tool_choice": true,
@@ -34598,38 +34688,41 @@
"output_cost_per_token": 1.5e-05,
"source": "https://docs.x.ai/docs/models",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_tool_choice": true,
"supports_web_search": true
},
"xai/grok-4-fast-reasoning": {
"cache_read_input_token_cost": 5e-08,
"input_cost_per_token": 2e-07,
"input_cost_per_token_above_128k_tokens": 4e-07,
"litellm_provider": "xai",
"max_input_tokens": 2000000.0,
"max_output_tokens": 2000000.0,
"max_tokens": 2000000.0,
"mode": "chat",
"input_cost_per_token": 2e-07,
"input_cost_per_token_above_128k_tokens": 4e-07,
"output_cost_per_token": 5e-07,
"output_cost_per_token_above_128k_tokens": 1e-06,
"cache_read_input_token_cost": 5e-08,
"source": "https://docs.x.ai/docs/models",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_tool_choice": true,
"supports_web_search": true
},
"xai/grok-4-fast-non-reasoning": {
"cache_read_input_token_cost": 5e-08,
"input_cost_per_token": 2e-07,
"input_cost_per_token_above_128k_tokens": 4e-07,
"litellm_provider": "xai",
"max_input_tokens": 2000000.0,
"max_output_tokens": 2000000.0,
"cache_read_input_token_cost": 5e-08,
"max_tokens": 2000000.0,
"mode": "chat",
"input_cost_per_token": 2e-07,
"input_cost_per_token_above_128k_tokens": 4e-07,
"output_cost_per_token": 5e-07,
"output_cost_per_token_above_128k_tokens": 1e-06,
"source": "https://docs.x.ai/docs/models",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_tool_choice": true,
"supports_web_search": true
},
@@ -34645,6 +34738,7 @@
"output_cost_per_token_above_128k_tokens": 3e-05,
"source": "https://docs.x.ai/docs/models",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_tool_choice": true,
"supports_web_search": true
},
@@ -34660,6 +34754,7 @@
"output_cost_per_token_above_128k_tokens": 3e-05,
"source": "https://docs.x.ai/docs/models",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_tool_choice": true,
"supports_web_search": true
},
@@ -34677,6 +34772,7 @@
"source": "https://docs.x.ai/docs/models/grok-4-1-fast-reasoning",
"supports_audio_input": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
@@ -34697,6 +34793,7 @@
"source": "https://docs.x.ai/docs/models/grok-4-1-fast-reasoning",
"supports_audio_input": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
@@ -34717,6 +34814,7 @@
"source": "https://docs.x.ai/docs/models/grok-4-1-fast-reasoning",
"supports_audio_input": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
@@ -34737,6 +34835,7 @@
"source": "https://docs.x.ai/docs/models/grok-4-1-fast-non-reasoning",
"supports_audio_input": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
@@ -34756,6 +34855,7 @@
"source": "https://docs.x.ai/docs/models/grok-4-1-fast-non-reasoning",
"supports_audio_input": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
@@ -34772,6 +34872,7 @@
"output_cost_per_token": 6e-06,
"source": "https://docs.x.ai/docs/models",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"supports_vision": true,
@@ -34788,6 +34889,7 @@
"output_cost_per_token": 6e-06,
"source": "https://docs.x.ai/docs/models",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"supports_vision": true,
@@ -34820,6 +34922,7 @@
"output_cost_per_token": 6e-06,
"source": "https://docs.x.ai/docs/models",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_tool_choice": true,
"supports_vision": true,
"supports_web_search": true
@@ -34848,6 +34951,7 @@
"output_cost_per_token": 1.5e-06,
"source": "https://docs.x.ai/docs/models",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
@@ -34862,6 +34966,7 @@
"output_cost_per_token": 1.5e-06,
"source": "https://docs.x.ai/docs/models",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
@@ -34876,6 +34981,7 @@
"output_cost_per_token": 1.5e-06,
"source": "https://docs.x.ai/docs/models",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
@@ -34907,6 +35013,20 @@
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"zai.glm-5": {
"input_cost_per_token": 1e-06,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 200000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.2e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"zai.glm-4.7-flash": {
"input_cost_per_token": 7e-08,
"litellm_provider": "bedrock_converse",
@@ -39492,6 +39612,7 @@
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
@@ -39716,6 +39837,87 @@
}
]
},
"zai.glm-5": {
"input_cost_per_token": 1e-06,
"output_cost_per_token": 3.2e-06,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 200000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-east-1/zai.glm-5": {
"input_cost_per_token": 1e-06,
"output_cost_per_token": 3.2e-06,
"litellm_provider": "bedrock",
"max_input_tokens": 200000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-west-2/zai.glm-5": {
"input_cost_per_token": 1e-06,
"output_cost_per_token": 3.2e-06,
"litellm_provider": "bedrock",
"max_input_tokens": 200000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"minimax.minimax-m2.5": {
"input_cost_per_token": 3e-07,
"output_cost_per_token": 1.2e-06,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 1000000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-east-1/minimax.minimax-m2.5": {
"input_cost_per_token": 3e-07,
"output_cost_per_token": 1.2e-06,
"litellm_provider": "bedrock",
"max_input_tokens": 1000000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-west-2/minimax.minimax-m2.5": {
"input_cost_per_token": 3e-07,
"output_cost_per_token": 1.2e-06,
"litellm_provider": "bedrock",
"max_input_tokens": 1000000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-gov-east-1/anthropic.claude-haiku-4-5-20251001-v1:0": {
"cache_creation_input_token_cost": 1.5e-06,
"cache_read_input_token_cost": 1.2e-07,
+1
View File
@@ -393,6 +393,7 @@ class AnthropicMessagesRequestOptionalParams(TypedDict, total=False):
AnthropicOutputConfig
] # Configuration for Claude's output behavior
cache_control: Optional[Dict[str, Any]] # Automatic prompt caching
reasoning_effort: Optional[str]
class AnthropicMessagesRequest(AnthropicMessagesRequestOptionalParams, total=False):
+1
View File
@@ -1041,3 +1041,4 @@ class BedrockInvokeAnthropicMessagesRequest(TypedDict, total=False):
# `metadata` is part of the common Anthropic Messages API shape.
thinking: dict
metadata: dict
output_config: dict
+36 -5
View File
@@ -977,6 +977,7 @@
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
@@ -1174,6 +1175,7 @@
"supports_vision": true,
"supports_prompt_caching": false,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": true,
"supports_tool_choice": true
},
"global.anthropic.claude-opus-4-7": {
@@ -1321,6 +1323,7 @@
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
@@ -1350,6 +1353,7 @@
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
@@ -1379,6 +1383,7 @@
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
@@ -1407,6 +1412,7 @@
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
@@ -1435,6 +1441,7 @@
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
@@ -1929,6 +1936,7 @@
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true
@@ -2052,6 +2060,7 @@
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
@@ -9219,6 +9228,7 @@
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"supports_adaptive_thinking": true,
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,
@@ -9226,6 +9236,7 @@
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
@@ -9361,6 +9372,7 @@
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
@@ -9388,6 +9400,7 @@
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
@@ -9409,6 +9422,7 @@
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"supports_adaptive_thinking": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
@@ -9442,6 +9456,7 @@
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"supports_adaptive_thinking": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
@@ -9475,6 +9490,7 @@
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"supports_adaptive_thinking": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
@@ -9491,7 +9507,6 @@
"us": 1.1,
"fast": 6.0
},
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"claude-opus-4-7-20260416": {
@@ -9510,6 +9525,7 @@
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"supports_adaptive_thinking": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
@@ -9526,7 +9542,6 @@
"us": 1.1,
"fast": 6.0
},
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"claude-sonnet-4-20250514": {
@@ -10804,6 +10819,7 @@
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": true,
"supports_tool_choice": true
},
"databricks/databricks-claude-sonnet-4": {
@@ -17164,7 +17180,8 @@
],
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_vision": true
"supports_vision": true,
"supports_minimal_reasoning_effort": true
},
"github_copilot/claude-opus-4.6-fast": {
"litellm_provider": "github_copilot",
@@ -17677,7 +17694,8 @@
"mode": "chat",
"output_cost_per_token": 2.5e-05,
"supports_function_calling": true,
"supports_vision": true
"supports_vision": true,
"supports_minimal_reasoning_effort": true
},
"gmi/anthropic/claude-sonnet-4.5": {
"input_cost_per_token": 3e-06,
@@ -26095,6 +26113,7 @@
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 159,
@@ -26113,6 +26132,7 @@
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_minimal_reasoning_effort": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true,
@@ -26134,6 +26154,7 @@
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
@@ -26199,6 +26220,7 @@
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"supports_xhigh_reasoning_effort": true,
@@ -28078,7 +28100,8 @@
"mode": "responses",
"supports_web_search": true,
"supports_reasoning": false,
"supports_function_calling": true
"supports_function_calling": true,
"supports_minimal_reasoning_effort": true
},
"perplexity/anthropic/claude-sonnet-4-5": {
"litellm_provider": "perplexity",
@@ -30557,6 +30580,7 @@
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_minimal_reasoning_effort": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
@@ -30585,6 +30609,7 @@
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_minimal_reasoning_effort": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
@@ -30612,6 +30637,7 @@
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_minimal_reasoning_effort": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
@@ -31192,6 +31218,7 @@
"output_cost_per_token": 2.5e-05,
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_minimal_reasoning_effort": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
@@ -32384,6 +32411,7 @@
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_minimal_reasoning_effort": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
@@ -32410,6 +32438,7 @@
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_minimal_reasoning_effort": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
@@ -32576,6 +32605,7 @@
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
@@ -39587,6 +39617,7 @@
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_max_reasoning_effort": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
@@ -0,0 +1,23 @@
"""Force ``litellm.model_cost`` to load from the PR-local JSON for these tests.
By default ``litellm.model_cost`` is fetched from the main branch on GitHub,
which lags behind PR-branch flag additions. This fixture loads the local
file so per-model flag tests pass in CI as well as locally.
See https://github.com/BerriAI/litellm/issues/27122.
"""
import pytest
import litellm
from litellm.litellm_core_utils.get_model_cost_map import get_model_cost_map
@pytest.fixture(autouse=True)
def _use_pr_local_model_cost_map(monkeypatch):
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(
litellm,
"model_cost",
get_model_cost_map(url=litellm.model_cost_map_url),
)
yield
@@ -8,6 +8,7 @@ sys.path.insert(
) # Adds the parent directory to the system path
from unittest.mock import MagicMock, patch
import litellm
from litellm.constants import RESPONSE_FORMAT_TOOL_NAME
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
from litellm.llms.anthropic.experimental_pass_through.messages.transformation import (
@@ -1631,8 +1632,9 @@ def test_effort_validation():
)
assert result["output_config"]["effort"] == effort
# Invalid value should raise error
with pytest.raises(ValueError, match="Invalid effort value"):
with pytest.raises(
litellm.exceptions.BadRequestError, match="Invalid effort value"
):
optional_params = {"output_config": {"effort": "invalid"}}
config.transform_request(
model="claude-opus-4-5-20251101",
@@ -1687,7 +1689,10 @@ def test_max_effort_rejected_for_opus_45():
messages = [{"role": "user", "content": "Test"}]
with pytest.raises(ValueError, match="effort='max' is not supported by this model"):
with pytest.raises(
litellm.exceptions.BadRequestError,
match="effort='max' is not supported by this model",
):
optional_params = {"output_config": {"effort": "max"}}
config.transform_request(
model="claude-opus-4-5-20251101",
@@ -1739,6 +1744,97 @@ def test_effort_with_other_features():
assert "thinking" in result
def test_anthropic_drop_params_strips_output_config_for_pre_4_5_models():
"""``drop_params=True`` strips unsupported ``output_config`` for pre-4.5 models."""
config = AnthropicConfig()
messages = [{"role": "user", "content": "Hello"}]
original = litellm.drop_params
litellm.drop_params = True
try:
result = config.transform_request(
model="claude-3-haiku-20240307",
messages=messages,
optional_params={"output_config": {"effort": "low"}},
litellm_params={},
headers={},
)
finally:
litellm.drop_params = original
assert "output_config" not in result
def test_anthropic_drop_params_keeps_output_config_for_supporting_models():
"""``drop_params=True`` must not strip on models that support effort."""
config = AnthropicConfig()
messages = [{"role": "user", "content": "Hello"}]
original = litellm.drop_params
litellm.drop_params = True
try:
result = config.transform_request(
model="claude-opus-4-7",
messages=messages,
optional_params={"output_config": {"effort": "high"}},
litellm_params={},
headers={},
)
finally:
litellm.drop_params = original
assert result.get("output_config") == {"effort": "high"}
def test_anthropic_drop_params_false_forwards_to_unsupported_model():
"""Default ``drop_params=False`` forwards ``output_config`` and lets the provider 400."""
config = AnthropicConfig()
messages = [{"role": "user", "content": "Hello"}]
original = litellm.drop_params
litellm.drop_params = False
try:
result = config.transform_request(
model="claude-3-haiku-20240307",
messages=messages,
optional_params={"output_config": {"effort": "low"}},
litellm_params={},
headers={},
)
finally:
litellm.drop_params = original
assert result.get("output_config") == {"effort": "low"}
@pytest.mark.parametrize(
"model",
[
"claude-opus-4-5-20251101",
"claude-opus-4-6",
"claude-opus-4-7",
"claude-sonnet-4-6",
"anthropic.claude-mythos-preview",
"bedrock/anthropic.claude-mythos-preview",
],
)
def test_anthropic_model_supports_effort_param_recognizes_supporting_models(model):
assert AnthropicConfig._model_supports_effort_param(model) is True
@pytest.mark.parametrize(
"model",
[
"claude-3-haiku-20240307",
"claude-3-5-sonnet-20241022",
"claude-3-opus-20240229",
"claude-sonnet-4-20250514",
],
)
def test_anthropic_model_supports_effort_param_rejects_non_supporting_models(model):
assert AnthropicConfig._model_supports_effort_param(model) is False
def test_translate_system_message_skips_empty_string_content():
"""
Test that translate_system_message skips system messages with empty string content.
@@ -1950,6 +2046,67 @@ def test_get_config_without_model_uses_fallback():
assert config["max_tokens"] == 4096
def test_get_config_does_not_leak_module_constants():
"""``get_config`` must not leak the reasoning-effort lookup table onto the wire."""
cfg = AnthropicConfig.get_config(model="claude-opus-4-7")
for forbidden in (
"REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT",
"_REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT",
):
assert forbidden not in cfg
@pytest.mark.parametrize(
"model,level,expected",
[
("claude-opus-4-7", "max", True),
("claude-opus-4-7", "xhigh", True),
("claude-opus-4-6", "max", True),
("claude-opus-4-6", "xhigh", False),
("claude-sonnet-4-6", "max", True),
("claude-sonnet-4-6", "xhigh", False),
("bedrock/invoke/us.anthropic.claude-opus-4-7", "max", True),
("bedrock/invoke/us.anthropic.claude-opus-4-7", "xhigh", True),
("bedrock/invoke/us.anthropic.claude-opus-4-6-v1", "max", True),
("bedrock/invoke/us.anthropic.claude-opus-4-6-v1", "xhigh", False),
("bedrock/invoke/us.anthropic.claude-sonnet-4-6", "max", True),
("vertex_ai/claude-opus-4-7", "xhigh", True),
("azure_ai/claude-opus-4-7", "xhigh", True),
],
)
def test_supports_effort_level_handles_provider_prefixes(model, level, expected):
"""``_supports_effort_level`` resolves bedrock/vertex/azure-prefixed model ids."""
assert AnthropicConfig._supports_effort_level(model, level) is expected
@pytest.mark.parametrize(
"model,effort,expect_error",
[
("claude-opus-4-6", "max", False),
("claude-sonnet-4-6", "max", False),
("claude-opus-4-7", "max", False),
("claude-opus-4-5-20251101", "max", True),
("claude-sonnet-4-5", "max", True),
("claude-opus-4-7", "xhigh", False),
("claude-opus-4-6", "xhigh", True),
("claude-sonnet-4-6", "xhigh", True),
("claude-opus-4-5-20251101", "high", False),
("claude-haiku-4-5", "low", False),
("claude-opus-4-5-20251101", None, False),
],
)
def test_validate_effort_for_model_centralises_per_model_gating(
model, effort, expect_error
):
err = AnthropicConfig._validate_effort_for_model(model, effort)
if expect_error:
assert err is not None
assert effort in err
assert model in err
else:
assert err is None
def test_transform_request_uses_dynamic_max_tokens():
"""
Test that transform_request uses dynamic max_tokens based on model
@@ -2153,12 +2310,12 @@ def test_reasoning_effort_maps_to_budget_thinking_for_non_opus_4_6():
"""
config = AnthropicConfig()
# Test with Claude Sonnet 4.5 (non-Opus 4.6 model)
# ``minimal`` floors at ANTHROPIC_MIN_THINKING_BUDGET_TOKENS (1024).
test_cases = [
("low", 1024), # DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET
("medium", 2048), # DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET
("high", 4096), # DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET
("minimal", 128), # DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET
("low", 1024),
("medium", 2048),
("high", 4096),
("minimal", 1024),
]
for effort, expected_budget in test_cases:
@@ -2244,19 +2401,31 @@ def test_reasoning_effort_does_not_set_output_config_for_older_models():
), f"output_config should not be set for {model}"
def test_max_effort_rejected_for_sonnet_46():
"""Test that effort='max' is rejected for Sonnet 4.6 (Opus-only effort level)."""
@pytest.mark.parametrize(
"model",
[
"claude-sonnet-4-6",
"claude-sonnet-4-6-20260219",
"us.anthropic.claude-sonnet-4-6",
"bedrock/converse/us.anthropic.claude-sonnet-4-6",
"vertex_ai/claude-sonnet-4-6",
"openrouter/anthropic/claude-sonnet-4.6",
],
)
def test_max_effort_accepted_for_sonnet_46_variants(model):
"""``effort='max'`` is supported on Claude 4.6 (Opus + Sonnet) and 4.7."""
config = AnthropicConfig()
messages = [{"role": "user", "content": "Test"}]
with pytest.raises(ValueError, match="effort='max' is not supported by this model"):
config.transform_request(
model="claude-sonnet-4-6-20260219",
messages=messages,
optional_params={"output_config": {"effort": "max"}},
litellm_params={},
headers={},
)
result = config.transform_request(
model=model,
messages=messages,
optional_params={"output_config": {"effort": "max"}},
litellm_params={},
headers={},
)
assert result["output_config"]["effort"] == "max"
def test_max_effort_accepted_for_opus_46():
@@ -2335,6 +2504,74 @@ def test_reasoning_effort_none_omits_thinking_and_output_config(model):
assert "output_config" not in result
@pytest.mark.parametrize(
"effort",
["disabled", "invalid", ""],
)
def test_reasoning_effort_garbage_raises_bad_request(effort):
"""Unmapped reasoning_effort raises BadRequestError (clean 400, not a 500)."""
config = AnthropicConfig()
with pytest.raises(litellm.exceptions.BadRequestError):
config.map_openai_params(
non_default_params={"reasoning_effort": effort},
optional_params={},
model="claude-sonnet-4-5-20250929",
drop_params=False,
)
@pytest.mark.parametrize(
"effort,expected_budget",
[("xhigh", 8192), ("max", 16384)],
)
def test_reasoning_effort_xhigh_max_maps_to_budget_on_budget_model(
effort, expected_budget
):
"""``xhigh`` / ``max`` extend the budget_tokens progression on budget-mode models."""
config = AnthropicConfig()
result = config.map_openai_params(
non_default_params={"reasoning_effort": effort},
optional_params={},
model="claude-sonnet-4-5-20250929",
drop_params=False,
)
assert result["thinking"]["type"] == "enabled"
assert result["thinking"]["budget_tokens"] == expected_budget
assert "output_config" not in result
def test_output_config_effort_empty_string_raises_bad_request():
"""``output_config={"effort": ""}`` is rejected with a 400."""
config = AnthropicConfig()
with pytest.raises(litellm.exceptions.BadRequestError, match="Invalid effort"):
config.transform_request(
model="claude-opus-4-7",
messages=[{"role": "user", "content": "hi"}],
optional_params={"output_config": {"effort": ""}, "max_tokens": 32},
litellm_params={},
headers={},
)
def test_reasoning_effort_minimal_floors_at_anthropic_provider_minimum():
"""``minimal`` floors at the Anthropic provider minimum (1024)."""
config = AnthropicConfig()
result = config.map_openai_params(
non_default_params={"reasoning_effort": "minimal"},
optional_params={},
model="claude-sonnet-4-5-20250929",
drop_params=False,
)
assert result["thinking"]["type"] == "enabled"
assert result["thinking"]["budget_tokens"] >= 1024
def test_effort_beta_header_still_injected_for_older_models():
"""
Test that is_effort_used still returns True for pre-4.6 models
@@ -0,0 +1,193 @@
"""Tests for ``reasoning_effort`` translation on the Anthropic /v1/messages route."""
import pytest
from litellm.llms.anthropic.common_utils import AnthropicError
from litellm.llms.anthropic.experimental_pass_through.messages.transformation import (
AnthropicMessagesConfig,
)
@pytest.mark.parametrize(
"reasoning_effort,expected_effort",
[
("minimal", "low"),
("low", "low"),
("medium", "medium"),
("high", "high"),
("xhigh", "xhigh"),
("max", "max"),
],
)
def test_reasoning_effort_maps_to_output_config_for_adaptive_model(
reasoning_effort, expected_effort
):
config = AnthropicMessagesConfig()
optional_params = {"max_tokens": 1024, "reasoning_effort": reasoning_effort}
result = config.transform_anthropic_messages_request(
model="claude-opus-4-7",
messages=[{"role": "user", "content": "Hello"}],
anthropic_messages_optional_request_params=optional_params,
litellm_params={},
headers={},
)
assert "reasoning_effort" not in result
assert result.get("thinking") == {"type": "adaptive"}
assert result.get("output_config") == {"effort": expected_effort}
def test_reasoning_effort_none_clears_thinking_and_output_config():
config = AnthropicMessagesConfig()
optional_params = {
"max_tokens": 1024,
"reasoning_effort": "none",
"thinking": {"type": "adaptive"},
"output_config": {"effort": "high"},
}
result = config.transform_anthropic_messages_request(
model="claude-opus-4-7",
messages=[{"role": "user", "content": "Hello"}],
anthropic_messages_optional_request_params=optional_params,
litellm_params={},
headers={},
)
assert "reasoning_effort" not in result
assert "thinking" not in result
assert "output_config" not in result
def test_reasoning_effort_on_non_adaptive_model_uses_thinking_budget():
config = AnthropicMessagesConfig()
optional_params = {"max_tokens": 1024, "reasoning_effort": "high"}
result = config.transform_anthropic_messages_request(
model="claude-opus-4-5",
messages=[{"role": "user", "content": "Hello"}],
anthropic_messages_optional_request_params=optional_params,
litellm_params={},
headers={},
)
assert "reasoning_effort" not in result
assert "output_config" not in result
thinking = result.get("thinking")
assert isinstance(thinking, dict)
assert thinking.get("type") == "enabled"
assert isinstance(thinking.get("budget_tokens"), int)
assert thinking["budget_tokens"] >= 1024
@pytest.mark.parametrize("bad_effort", ["invalid", "disabled", ""])
def test_invalid_reasoning_effort_raises_400(bad_effort):
config = AnthropicMessagesConfig()
optional_params = {"max_tokens": 1024, "reasoning_effort": bad_effort}
with pytest.raises(AnthropicError) as exc_info:
config.transform_anthropic_messages_request(
model="claude-opus-4-7",
messages=[{"role": "user", "content": "Hello"}],
anthropic_messages_optional_request_params=optional_params,
litellm_params={},
headers={},
)
assert exc_info.value.status_code == 400
@pytest.mark.parametrize(
"model,bad_effort",
[
("claude-opus-4-6", "xhigh"),
("bedrock/invoke/us.anthropic.claude-opus-4-6-v1", "xhigh"),
("claude-sonnet-4-6", "xhigh"),
],
)
def test_reasoning_effort_unsupported_tier_raises_400_messages(model, bad_effort):
config = AnthropicMessagesConfig()
optional_params = {"max_tokens": 1024, "reasoning_effort": bad_effort}
with pytest.raises(AnthropicError) as exc_info:
config.transform_anthropic_messages_request(
model=model,
messages=[{"role": "user", "content": "Hello"}],
anthropic_messages_optional_request_params=optional_params,
litellm_params={},
headers={},
)
assert exc_info.value.status_code == 400
assert "not supported by this model" in str(exc_info.value)
@pytest.mark.parametrize(
"model",
[
"claude-sonnet-4-6",
"bedrock/invoke/us.anthropic.claude-sonnet-4-6",
],
)
def test_reasoning_effort_max_accepted_on_sonnet_46_messages(model):
config = AnthropicMessagesConfig()
optional_params = {"max_tokens": 1024, "reasoning_effort": "max"}
result = config.transform_anthropic_messages_request(
model=model,
messages=[{"role": "user", "content": "Hello"}],
anthropic_messages_optional_request_params=optional_params,
litellm_params={},
headers={},
)
output_config = result.get("output_config")
assert isinstance(output_config, dict) and output_config.get("effort") == "max"
def test_explicit_output_config_wins_over_reasoning_effort():
config = AnthropicMessagesConfig()
optional_params = {
"max_tokens": 1024,
"reasoning_effort": "low",
"output_config": {"effort": "max"},
}
result = config.transform_anthropic_messages_request(
model="claude-opus-4-7",
messages=[{"role": "user", "content": "Hello"}],
anthropic_messages_optional_request_params=optional_params,
litellm_params={},
headers={},
)
assert "reasoning_effort" not in result
assert result.get("output_config") == {"effort": "max"}
def test_explicit_thinking_wins_over_reasoning_effort():
config = AnthropicMessagesConfig()
optional_params = {
"max_tokens": 1024,
"reasoning_effort": "low",
"thinking": {"type": "enabled", "budget_tokens": 8000},
}
result = config.transform_anthropic_messages_request(
model="claude-opus-4-5",
messages=[{"role": "user", "content": "Hello"}],
anthropic_messages_optional_request_params=optional_params,
litellm_params={},
headers={},
)
assert "reasoning_effort" not in result
assert result.get("thinking") == {"type": "enabled", "budget_tokens": 8000}
def test_reasoning_effort_in_supported_params():
config = AnthropicMessagesConfig()
assert "reasoning_effort" in config.get_supported_anthropic_messages_params(
"claude-opus-4-7"
)
@@ -291,6 +291,100 @@ class TestAzureAnthropicConfig:
assert "anthropic-beta" in headers
assert "compact-2026-01-12" in headers["anthropic-beta"]
def test_output_config_promoted_from_extra_body(self):
"""Anthropic ``output_config`` routed via ``extra_body`` reaches the request body."""
config = AzureAnthropicConfig()
messages = [{"role": "user", "content": "Hello"}]
optional_params = {
"max_tokens": 100,
"extra_body": {"output_config": {"effort": "low"}},
}
litellm_params = {"api_key": "test-key"}
headers = {"api-key": "test-key", "anthropic-version": "2023-06-01"}
result = config.transform_request(
model="claude-opus-4-6",
messages=messages,
optional_params=optional_params,
litellm_params=litellm_params,
headers=headers,
)
assert result["output_config"] == {"effort": "low"}
assert "extra_body" not in result
def test_invalid_output_config_effort_raises_via_extra_body(self):
"""Invalid ``effort`` via ``extra_body`` raises BadRequestError."""
import litellm
config = AzureAnthropicConfig()
messages = [{"role": "user", "content": "Hello"}]
optional_params = {
"max_tokens": 100,
"extra_body": {"output_config": {"effort": "invalid"}},
}
litellm_params = {"api_key": "test-key"}
headers = {"api-key": "test-key", "anthropic-version": "2023-06-01"}
with pytest.raises(litellm.exceptions.BadRequestError) as exc_info:
config.transform_request(
model="claude-opus-4-6",
messages=messages,
optional_params=optional_params,
litellm_params=litellm_params,
headers=headers,
)
assert "Invalid effort value" in str(exc_info.value)
def test_unsupported_effort_xhigh_raises_via_extra_body(self):
"""Unsupported ``effort='xhigh'`` via ``extra_body`` raises BadRequestError."""
import litellm
config = AzureAnthropicConfig()
messages = [{"role": "user", "content": "Hello"}]
optional_params = {
"max_tokens": 100,
"extra_body": {"output_config": {"effort": "xhigh"}},
}
litellm_params = {"api_key": "test-key"}
headers = {"api-key": "test-key", "anthropic-version": "2023-06-01"}
with pytest.raises(litellm.exceptions.BadRequestError) as exc_info:
config.transform_request(
model="claude-sonnet-4-6",
messages=messages,
optional_params=optional_params,
litellm_params=litellm_params,
headers=headers,
)
assert "xhigh" in str(exc_info.value)
def test_extra_body_promotion_does_not_clobber_top_level(self):
"""Top-level ``optional_params`` wins over duplicates in ``extra_body``."""
config = AzureAnthropicConfig()
messages = [{"role": "user", "content": "Hello"}]
optional_params = {
"max_tokens": 100,
"output_config": {"effort": "low"},
"extra_body": {"output_config": {"effort": "high"}},
}
litellm_params = {"api_key": "test-key"}
headers = {"api-key": "test-key", "anthropic-version": "2023-06-01"}
result = config.transform_request(
model="claude-opus-4-6",
messages=messages,
optional_params=optional_params,
litellm_params=litellm_params,
headers=headers,
)
assert result["output_config"] == {"effort": "low"}
def test_context_management_mixed_edits_beta_headers(self):
"""Test that context_management with both compact and other edits adds both beta headers"""
config = AzureAnthropicConfig()
@@ -406,37 +406,26 @@ def test_opus_4_5_model_detection():
# f"computer-use beta should be kept, got: {anthropic_beta}"
def test_output_config_removed_from_bedrock_chat_invoke_request():
"""
Test that output_config parameter is stripped from Bedrock Chat Invoke requests.
Bedrock Invoke API doesn't support the output_config parameter (Anthropic-only).
Ensures the chat/invoke path mirrors the messages/invoke path fix.
Fixes: https://github.com/BerriAI/litellm/issues/22797
"""
def test_output_config_forwarded_for_bedrock_chat_invoke_request():
"""Bedrock Invoke chat path forwards ``output_config`` for adaptive Claude models."""
config = AmazonAnthropicClaudeConfig()
messages = [{"role": "user", "content": "test"}]
# Inject output_config into optional_params (simulates Anthropic SDK forwarding it)
optional_params = {
"max_tokens": 100,
"output_config": {"effort": "high"},
}
result = config.transform_request(
model="anthropic.claude-sonnet-4-20250514-v1:0",
model="anthropic.claude-opus-4-7",
messages=messages,
optional_params=optional_params,
litellm_params={},
headers={},
)
assert (
"output_config" not in result
), f"output_config should be stripped for Bedrock Chat Invoke, got keys: {list(result.keys())}"
# Verify normal params survive
assert result.get("output_config") == {"effort": "high"}
assert result["max_tokens"] == 100
@@ -310,6 +310,111 @@ def test_reasoning_effort_none_omits_thinking_for_anthropic_converse(model):
assert "thinking" not in optional_params
@pytest.mark.parametrize(
"model,effort,expected_effort",
[
("bedrock/converse/us.anthropic.claude-opus-4-7", "low", "low"),
("bedrock/converse/us.anthropic.claude-opus-4-7", "medium", "medium"),
("bedrock/converse/us.anthropic.claude-opus-4-7", "high", "high"),
("bedrock/converse/us.anthropic.claude-opus-4-7", "xhigh", "xhigh"),
("bedrock/converse/us.anthropic.claude-opus-4-7", "max", "max"),
("bedrock/converse/us.anthropic.claude-opus-4-6-v1", "max", "max"),
("bedrock/converse/us.anthropic.claude-sonnet-4-6", "high", "high"),
("bedrock/converse/us.anthropic.claude-sonnet-4-6", "minimal", "low"),
],
)
def test_reasoning_effort_sets_output_config_for_adaptive_models_converse(
model, effort, expected_effort
):
"""Adaptive Claude 4.6 / 4.7 on Bedrock Converse routes the tier via ``output_config.effort``."""
config = AmazonConverseConfig()
optional_params = config.map_openai_params(
non_default_params={"reasoning_effort": effort},
optional_params={},
model=model,
drop_params=False,
)
assert optional_params["thinking"]["type"] == "adaptive"
assert optional_params["output_config"] == {"effort": expected_effort}
@pytest.mark.parametrize(
"model",
[
"bedrock/converse/us.anthropic.claude-opus-4-7",
"bedrock/converse/us.anthropic.claude-opus-4-6-v1",
"bedrock/converse/us.anthropic.claude-sonnet-4-6",
],
)
def test_output_config_effort_forwarded_into_additional_request_fields(model):
"""``output_config`` rides along inside ``additionalModelRequestFields``."""
config = AmazonConverseConfig()
messages = [{"role": "user", "content": "hi"}]
result = config._transform_request(
model=model,
messages=messages,
optional_params={
"maxTokens": 256,
"thinking": {"type": "adaptive"},
"output_config": {"effort": "high"},
},
litellm_params={},
headers={},
)
additional = result.get("additionalModelRequestFields", {})
assert additional.get("output_config") == {"effort": "high"}
@pytest.mark.parametrize(
"effort",
["disabled", "invalid", ""],
)
def test_reasoning_effort_garbage_raises_bad_request_converse(effort):
"""Unmapped reasoning_effort on Bedrock Converse Anthropic raises BadRequestError."""
config = AmazonConverseConfig()
with pytest.raises(litellm.exceptions.BadRequestError):
config.map_openai_params(
non_default_params={"reasoning_effort": effort},
optional_params={},
model="bedrock/converse/us.anthropic.claude-opus-4-7",
drop_params=False,
)
@pytest.mark.parametrize(
"model",
[
"bedrock/converse/us.anthropic.claude-sonnet-4-6",
"bedrock/converse/global.anthropic.claude-sonnet-4-6",
"bedrock/converse/eu.anthropic.claude-sonnet-4-6",
"bedrock/converse/au.anthropic.claude-sonnet-4-6",
],
)
def test_output_config_effort_max_passes_through_on_sonnet_46_variants(model):
"""``effort='max'`` flows through for every Bedrock Converse Sonnet 4.6 id."""
config = AmazonConverseConfig()
messages = [{"role": "user", "content": "hi"}]
result = config._transform_request(
model=model,
messages=messages,
optional_params={
"maxTokens": 256,
"output_config": {"effort": "max"},
},
litellm_params={},
headers={},
)
additional = result.get("additionalModelRequestFields", {})
assert additional.get("output_config") == {"effort": "max"}
def test_get_supported_openai_params():
config = AmazonConverseConfig()
supported_params = config.get_supported_openai_params(
@@ -3329,6 +3434,57 @@ def test_transform_request_strips_anthropic_output_config():
assert "output_config" not in additional_fields
def test_converse_drop_params_strips_output_config_for_pre_4_5_anthropic():
"""``drop_params=True`` strips unsupported ``output_config`` on Bedrock Converse."""
config = AmazonConverseConfig()
messages = [{"role": "user", "content": "hi"}]
original = litellm.drop_params
litellm.drop_params = True
try:
result = config._transform_request(
model="bedrock/converse/anthropic.claude-3-haiku-20240307-v1:0",
messages=messages,
optional_params={
"maxTokens": 256,
"output_config": {"effort": "low"},
},
litellm_params={},
headers={},
)
finally:
litellm.drop_params = original
additional = result.get("additionalModelRequestFields", {})
assert "output_config" not in additional
def test_converse_drop_params_keeps_output_config_for_supporting_anthropic():
"""``drop_params=True`` does not strip on models that support ``output_config``."""
config = AmazonConverseConfig()
messages = [{"role": "user", "content": "hi"}]
original = litellm.drop_params
litellm.drop_params = True
try:
result = config._transform_request(
model="bedrock/converse/us.anthropic.claude-opus-4-7",
messages=messages,
optional_params={
"maxTokens": 256,
"thinking": {"type": "adaptive"},
"output_config": {"effort": "high"},
},
litellm_params={},
headers={},
)
finally:
litellm.drop_params = original
additional = result.get("additionalModelRequestFields", {})
assert additional.get("output_config") == {"effort": "high"}
def test_transform_response_native_structured_output():
"""Test response handling when model returns JSON as text content (native structured output)."""
response_json = {
@@ -592,13 +592,8 @@ def test_remove_scope_from_cache_control():
assert request["messages"][0]["content"][0]["cache_control"]["type"] == "ephemeral"
def test_bedrock_messages_strips_output_config():
"""
Ensure output_config is stripped from the request before sending to
Bedrock Invoke, which doesn't support this Anthropic-specific parameter.
Regression test for: https://github.com/BerriAI/litellm/issues/22797
"""
def test_bedrock_messages_forwards_output_config():
"""Bedrock Invoke /v1/messages forwards ``output_config`` for adaptive Claude models."""
from litellm.types.router import GenericLiteLLMParams
cfg = AmazonAnthropicClaudeMessagesConfig()
@@ -611,26 +606,20 @@ def test_bedrock_messages_strips_output_config():
}
result = cfg.transform_anthropic_messages_request(
model="anthropic.claude-3-haiku-20240307-v1:0",
model="anthropic.claude-opus-4-7",
messages=messages,
anthropic_messages_optional_request_params=optional_params,
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert (
"output_config" not in result
), "output_config should be stripped — Bedrock Invoke rejects it"
assert result.get("output_config") == {"effort": "high"}
# Other params should be preserved
assert result.get("max_tokens") == 4096
def test_bedrock_messages_strips_output_config_with_output_format():
"""
When both output_config and output_format are present, both should be
stripped (output_format is converted to inline schema, output_config
is simply dropped).
"""
def test_bedrock_messages_forwards_output_config_with_output_format():
"""``output_config`` is forwarded; ``output_format`` is converted to inline schema."""
from litellm.types.router import GenericLiteLLMParams
cfg = AmazonAnthropicClaudeMessagesConfig()
@@ -647,6 +636,29 @@ def test_bedrock_messages_strips_output_config_with_output_format():
},
}
result = cfg.transform_anthropic_messages_request(
model="anthropic.claude-opus-4-7",
messages=messages,
anthropic_messages_optional_request_params=optional_params,
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert result.get("output_config") == {"effort": "low"}
assert "output_format" not in result
def test_bedrock_messages_forwards_output_config_for_non_adaptive_model():
"""``output_config`` is forwarded for non-adaptive models so the provider's error surfaces."""
from litellm.types.router import GenericLiteLLMParams
cfg = AmazonAnthropicClaudeMessagesConfig()
messages = [{"role": "user", "content": [{"type": "text", "text": "Hello"}]}]
optional_params = {
"max_tokens": 4096,
"output_config": {"effort": "high"},
}
result = cfg.transform_anthropic_messages_request(
model="anthropic.claude-3-haiku-20240307-v1:0",
messages=messages,
@@ -655,8 +667,201 @@ def test_bedrock_messages_strips_output_config_with_output_format():
headers={},
)
assert result.get("output_config") == {"effort": "high"}
assert result.get("max_tokens") == 4096
def test_bedrock_messages_drop_params_strips_output_config_for_pre_4_5():
"""``drop_params=True`` strips ``output_config`` for pre-4.5 Anthropic on /v1/messages."""
import litellm
from litellm.types.router import GenericLiteLLMParams
cfg = AmazonAnthropicClaudeMessagesConfig()
messages = [{"role": "user", "content": [{"type": "text", "text": "Hello"}]}]
optional_params = {
"max_tokens": 4096,
"output_config": {"effort": "low"},
}
original = litellm.drop_params
litellm.drop_params = True
try:
result = cfg.transform_anthropic_messages_request(
model="anthropic.claude-3-haiku-20240307-v1:0",
messages=messages,
anthropic_messages_optional_request_params=optional_params,
litellm_params=GenericLiteLLMParams(),
headers={},
)
finally:
litellm.drop_params = original
assert "output_config" not in result
assert "output_format" not in result
def test_bedrock_messages_drop_params_keeps_output_config_for_4_7():
"""``drop_params=True`` does not strip on opus-4-7 (supports effort)."""
import litellm
from litellm.types.router import GenericLiteLLMParams
cfg = AmazonAnthropicClaudeMessagesConfig()
messages = [{"role": "user", "content": [{"type": "text", "text": "Hello"}]}]
optional_params = {
"max_tokens": 4096,
"output_config": {"effort": "high"},
}
original = litellm.drop_params
litellm.drop_params = True
try:
result = cfg.transform_anthropic_messages_request(
model="anthropic.claude-opus-4-7",
messages=messages,
anthropic_messages_optional_request_params=optional_params,
litellm_params=GenericLiteLLMParams(),
headers={},
)
finally:
litellm.drop_params = original
assert result.get("output_config") == {"effort": "high"}
@pytest.mark.parametrize(
"reasoning_effort,expected_effort",
[
("minimal", "low"),
("low", "low"),
("medium", "medium"),
("high", "high"),
("xhigh", "xhigh"),
("max", "max"),
],
)
def test_bedrock_messages_maps_reasoning_effort_for_adaptive_model(
reasoning_effort, expected_effort
):
"""``reasoning_effort`` maps to ``thinking`` + ``output_config.effort`` on /v1/messages."""
from litellm.types.router import GenericLiteLLMParams
cfg = AmazonAnthropicClaudeMessagesConfig()
messages = [{"role": "user", "content": [{"type": "text", "text": "Hello"}]}]
optional_params = {
"max_tokens": 4096,
"reasoning_effort": reasoning_effort,
}
result = cfg.transform_anthropic_messages_request(
model="anthropic.claude-opus-4-7",
messages=messages,
anthropic_messages_optional_request_params=optional_params,
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert "reasoning_effort" not in result
assert result.get("thinking") == {"type": "adaptive"}
assert result.get("output_config") == {"effort": expected_effort}
def test_bedrock_messages_reasoning_effort_on_non_adaptive_uses_thinking_budget():
"""Non-adaptive models map ``reasoning_effort`` to ``thinking.budget_tokens``."""
from litellm.types.router import GenericLiteLLMParams
cfg = AmazonAnthropicClaudeMessagesConfig()
messages = [{"role": "user", "content": [{"type": "text", "text": "Hello"}]}]
optional_params = {
"max_tokens": 4096,
"reasoning_effort": "medium",
}
result = cfg.transform_anthropic_messages_request(
model="anthropic.claude-opus-4-5-20251101-v1:0",
messages=messages,
anthropic_messages_optional_request_params=optional_params,
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert "reasoning_effort" not in result
assert "output_config" not in result
thinking = result.get("thinking")
assert isinstance(thinking, dict)
assert thinking.get("type") == "enabled"
assert isinstance(thinking.get("budget_tokens"), int)
assert thinking["budget_tokens"] >= 1024
def test_bedrock_messages_reasoning_effort_none_clears_thinking():
"""``reasoning_effort='none'`` clears both ``thinking`` and ``output_config``."""
from litellm.types.router import GenericLiteLLMParams
cfg = AmazonAnthropicClaudeMessagesConfig()
messages = [{"role": "user", "content": [{"type": "text", "text": "Hello"}]}]
optional_params = {
"max_tokens": 4096,
"reasoning_effort": "none",
"output_config": {"effort": "high"},
"thinking": {"type": "adaptive"},
}
result = cfg.transform_anthropic_messages_request(
model="anthropic.claude-opus-4-7",
messages=messages,
anthropic_messages_optional_request_params=optional_params,
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert "reasoning_effort" not in result
assert "thinking" not in result
assert "output_config" not in result
def test_bedrock_messages_invalid_reasoning_effort_raises_400():
"""Garbage ``reasoning_effort`` raises AnthropicError (400)."""
from litellm.llms.anthropic.common_utils import AnthropicError
from litellm.types.router import GenericLiteLLMParams
cfg = AmazonAnthropicClaudeMessagesConfig()
messages = [{"role": "user", "content": [{"type": "text", "text": "Hello"}]}]
for bad_effort in ("invalid", "disabled", ""):
with pytest.raises(AnthropicError):
cfg.transform_anthropic_messages_request(
model="anthropic.claude-opus-4-7",
messages=messages,
anthropic_messages_optional_request_params={
"max_tokens": 4096,
"reasoning_effort": bad_effort,
},
litellm_params=GenericLiteLLMParams(),
headers={},
)
def test_bedrock_messages_explicit_output_config_wins_over_reasoning_effort():
"""Explicit ``output_config.effort`` wins over the ``reasoning_effort`` alias."""
from litellm.types.router import GenericLiteLLMParams
cfg = AmazonAnthropicClaudeMessagesConfig()
messages = [{"role": "user", "content": [{"type": "text", "text": "Hello"}]}]
optional_params = {
"max_tokens": 4096,
"reasoning_effort": "low",
"output_config": {"effort": "max"},
}
result = cfg.transform_anthropic_messages_request(
model="anthropic.claude-opus-4-7",
messages=messages,
anthropic_messages_optional_request_params=optional_params,
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert "reasoning_effort" not in result
assert result.get("output_config") == {"effort": "max"}
def test_bedrock_messages_strips_context_management():
@@ -728,13 +933,13 @@ def test_bedrock_messages_allowlist_filters_anthropic_only_fields():
"mcp_servers",
"container",
"inference_geo",
"output_config",
"context_management",
"model",
"stream",
):
assert bad not in result, f"{bad} should be stripped by the allowlist"
assert result.get("output_config") == {"effort": "low"}
# Supported fields pass through.
assert result["max_tokens"] == 4096
assert result["temperature"] == 0.5
@@ -882,7 +1087,10 @@ async def test_promote_message_start_cache_when_message_stop_omits_cache_fields(
"delta": {"stop_reason": "end_turn", "stop_sequence": None},
"usage": {"input_tokens": 10, "output_tokens": 181},
}
yield {"type": "message_stop", "usage": {"input_tokens": 10, "output_tokens": 181}}
yield {
"type": "message_stop",
"usage": {"input_tokens": 10, "output_tokens": 181},
}
merged: list[dict] = []
async for chunk in cfg._promote_message_stop_usage(_stream()):
@@ -936,7 +1144,11 @@ async def test_unified_bedrock_messages_cache_on_start_only_never_negative_cost(
},
},
}
yield {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}
yield {
"type": "content_block_start",
"index": 0,
"content_block": {"type": "text", "text": ""},
}
yield {
"type": "content_block_delta",
"index": 0,
@@ -948,7 +1160,10 @@ async def test_unified_bedrock_messages_cache_on_start_only_never_negative_cost(
"delta": {"stop_reason": "end_turn", "stop_sequence": None},
"usage": {"output_tokens": 181, "input_tokens": 10},
}
yield {"type": "message_stop", "usage": {"input_tokens": 10, "output_tokens": 181}}
yield {
"type": "message_stop",
"usage": {"input_tokens": 10, "output_tokens": 181},
}
logging_obj = LiteLLMLoggingObj(
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
@@ -498,12 +498,8 @@ def test_vertex_ai_partner_models_anthropic_remove_prompt_caching_scope_beta_hea
), "Header should be removed if no supported values remain"
def test_vertex_ai_anthropic_output_config_effort_only_dropped():
"""
``output_config`` containing only ``effort`` (an Anthropic-only key Vertex
rejects with "Extra inputs are not permitted") is dropped entirely so the
request body has no empty dict.
"""
def test_vertex_ai_anthropic_output_config_effort_only_forwarded():
"""Vertex AI Claude 4.6/4.7 accept ``output_config.effort`` on rawPredict."""
config = VertexAIAnthropicConfig()
messages = [{"role": "user", "content": "What is 2+2?"}]
@@ -515,16 +511,14 @@ def test_vertex_ai_anthropic_output_config_effort_only_dropped():
}
result = config.transform_request(
model="claude-3-5-sonnet-20241022",
model="claude-opus-4-6",
messages=messages,
optional_params=optional_params,
litellm_params={},
headers=headers,
)
assert (
"output_config" not in result
), "output_config containing only effort must be dropped"
assert result["output_config"] == {"effort": "high"}
assert result["max_tokens"] == 1024
assert "messages" in result
@@ -566,14 +560,8 @@ def test_vertex_ai_anthropic_output_config_format_passes_through():
assert result["output_config"] == output_config
def test_vertex_ai_anthropic_output_config_format_plus_effort_strips_only_effort():
"""
Greptile P1 on PR #23396: when ``output_config`` contains BOTH ``format``
and ``effort``, the prior conditional-passthrough logic forwarded the
full dict including the unsupported ``effort`` key, reproducing the
400 error the fix was meant to resolve. Only ``effort`` (and any future
Vertex-unsupported keys) should be filtered; ``format`` must survive.
"""
def test_vertex_ai_anthropic_output_config_format_plus_effort_preserved():
"""Both ``format`` and ``effort`` ride along on Vertex Claude 4.6/4.7."""
config = VertexAIAnthropicConfig()
messages = [{"role": "user", "content": "Return a person object."}]
@@ -591,7 +579,7 @@ def test_vertex_ai_anthropic_output_config_format_plus_effort_strips_only_effort
optional_params = {"max_tokens": 1024, "output_config": output_config}
result = config.transform_request(
model="claude-3-5-sonnet-20241022",
model="claude-opus-4-6",
messages=messages,
optional_params=optional_params,
litellm_params={},
@@ -599,9 +587,7 @@ def test_vertex_ai_anthropic_output_config_format_plus_effort_strips_only_effort
)
assert "output_config" in result
assert (
"effort" not in result["output_config"]
), "effort must be stripped — Vertex returns 400 on unknown keys"
assert result["output_config"]["effort"] == "high"
assert result["output_config"]["format"] == output_config["format"]
@@ -623,15 +609,8 @@ def test_vertex_ai_anthropic_output_config_non_dict_dropped():
assert "output_config" not in result
def test_vertex_ai_anthropic_output_format_preserved_output_config_effort_dropped():
"""
When the request carries both ``output_format`` (top-level structured
outputs) AND an ``output_config`` whose only useful key for Vertex is
``effort``: ``output_format`` must be forwarded (Vertex accepts it),
while ``output_config`` is dropped because Vertex returns 400 on
``effort``. This replaces the old "drop both" behavior, which was the
silent strip the bug report flagged.
"""
def test_vertex_ai_anthropic_output_format_and_output_config_effort_preserved():
"""Both ``output_format`` and ``output_config.effort`` are forwarded on Vertex 4.6/4.7."""
config = VertexAIAnthropicConfig()
messages = [{"role": "user", "content": "Extract structured data"}]
@@ -653,7 +632,7 @@ def test_vertex_ai_anthropic_output_format_preserved_output_config_effort_droppe
}
test_data = {
"model": "claude-3-5-sonnet-20241022",
"model": "claude-opus-4-6",
"messages": messages,
"max_tokens": 2048,
"output_format": output_format,
@@ -671,7 +650,7 @@ def test_vertex_ai_anthropic_output_format_preserved_output_config_effort_droppe
try:
result = config.transform_request(
model="claude-3-5-sonnet-20241022",
model="claude-opus-4-6",
messages=messages,
optional_params=optional_params,
litellm_params={},
@@ -680,9 +659,8 @@ def test_vertex_ai_anthropic_output_format_preserved_output_config_effort_droppe
# output_format flows through unchanged — Vertex AI Claude accepts it.
assert result["output_format"] == output_format
# output_config containing only ``effort`` is dropped to avoid the
# 400 "Extra inputs are not permitted" the silent strip used to mask.
assert "output_config" not in result
# output_config.effort now flows through (Vertex accepts it on 4.6/4.7).
assert result["output_config"] == {"effort": "high"}
assert result["max_tokens"] == 2048
assert "model" not in result, "model is still stripped (Vertex routes by URL)"
finally:
@@ -702,10 +680,10 @@ def test_sanitize_vertex_anthropic_output_params_unit():
sanitize_vertex_anthropic_output_params(data)
assert data == {"max_tokens": 8}
# Effort-only → dropped entirely.
# Effort-only → preserved (Vertex 4.6/4.7 accept it on rawPredict).
data = {"output_config": {"effort": "high"}}
sanitize_vertex_anthropic_output_params(data)
assert "output_config" not in data
assert data["output_config"] == {"effort": "high"}
# Format-only → preserved unchanged.
fmt = {"format": {"type": "json_schema", "schema": {"type": "object"}}}
@@ -713,10 +691,10 @@ def test_sanitize_vertex_anthropic_output_params_unit():
sanitize_vertex_anthropic_output_params(data)
assert data["output_config"] == fmt
# Mixed → effort filtered, format kept.
# Mixed → both effort and format kept (no current Vertex-unsupported keys).
data = {"output_config": {"format": fmt["format"], "effort": "high"}}
sanitize_vertex_anthropic_output_params(data)
assert data["output_config"] == fmt
assert data["output_config"] == {"format": fmt["format"], "effort": "high"}
# Non-dict → dropped defensively.
data = {"output_config": "garbage"}
@@ -6,6 +6,90 @@ sys.path.insert(
) # Adds the parent directory to the system path
from litellm.llms.xai.chat.transformation import XAIChatConfig
from litellm.types.utils import (
CompletionTokensDetailsWrapper,
ModelResponse,
Usage,
)
class TestXAIReasoningTokenFolding:
"""``_fold_reasoning_tokens_into_completion`` re-aligns xAI Usage to the OpenAI invariant."""
@staticmethod
def _make_response(
prompt_tokens: int,
completion_tokens: int,
total_tokens: int,
reasoning_tokens: int = 0,
) -> ModelResponse:
details = (
CompletionTokensDetailsWrapper(reasoning_tokens=reasoning_tokens)
if reasoning_tokens
else None
)
usage = Usage(
prompt_tokens=prompt_tokens,
completion_tokens=completion_tokens,
total_tokens=total_tokens,
completion_tokens_details=details,
)
response = ModelResponse()
setattr(response, "usage", usage)
return response
def test_should_fold_when_total_explained_by_reasoning_gap(self):
# xAI live shape: 14 + 10 + 312 == 336.
response = self._make_response(
prompt_tokens=14,
completion_tokens=10,
total_tokens=336,
reasoning_tokens=312,
)
XAIChatConfig._fold_reasoning_tokens_into_completion(response)
usage = response.usage
assert usage.completion_tokens == 322
assert usage.total_tokens == usage.prompt_tokens + usage.completion_tokens
def test_should_not_fold_when_already_normalised(self):
response = self._make_response(
prompt_tokens=14,
completion_tokens=322,
total_tokens=336,
reasoning_tokens=312,
)
XAIChatConfig._fold_reasoning_tokens_into_completion(response)
assert response.usage.completion_tokens == 322
def test_should_skip_when_no_reasoning_tokens(self):
response = self._make_response(
prompt_tokens=14,
completion_tokens=10,
total_tokens=24,
reasoning_tokens=0,
)
XAIChatConfig._fold_reasoning_tokens_into_completion(response)
assert response.usage.completion_tokens == 10
def test_should_skip_when_gap_does_not_match_reasoning(self):
# Refuse to fold if xAI accounting changes (gap != reasoning_tokens).
response = self._make_response(
prompt_tokens=14,
completion_tokens=10,
total_tokens=999,
reasoning_tokens=312,
)
XAIChatConfig._fold_reasoning_tokens_into_completion(response)
assert response.usage.completion_tokens == 10
assert response.usage.total_tokens == 999
class TestXAIParallelToolCalls:
@@ -14,9 +98,7 @@ class TestXAIParallelToolCalls:
def test_get_supported_openai_params_includes_parallel_tool_calls(self):
"""Test that parallel_tool_calls is in supported parameters."""
config = XAIChatConfig()
supported_params = config.get_supported_openai_params(
"xai/grok-4.20"
)
supported_params = config.get_supported_openai_params("xai/grok-4.20")
assert "parallel_tool_calls" in supported_params
def test_transform_request_preserves_parallel_tool_calls(self):
@@ -110,7 +110,7 @@ class TestXAICostCalculator:
usage = Usage(
prompt_tokens=10,
completion_tokens=200,
total_tokens=210,
total_tokens=360,
completion_tokens_details=CompletionTokensDetailsWrapper(
accepted_prediction_tokens=0,
audio_tokens=0,
@@ -136,7 +136,7 @@ class TestXAICostCalculator:
usage = Usage(
prompt_tokens=20,
completion_tokens=300,
total_tokens=320,
total_tokens=520,
completion_tokens_details=CompletionTokensDetailsWrapper(
accepted_prediction_tokens=0,
audio_tokens=0,
@@ -177,7 +177,7 @@ class TestXAICostCalculator:
usage = Usage(
prompt_tokens=12,
completion_tokens=50, # Less than reasoning_tokens
total_tokens=62,
total_tokens=162,
completion_tokens_details=CompletionTokensDetailsWrapper(
accepted_prediction_tokens=0,
audio_tokens=0,
@@ -204,7 +204,7 @@ class TestXAICostCalculator:
usage = Usage(
prompt_tokens=150000, # Above 128k threshold
completion_tokens=100000, # Above 128k threshold
total_tokens=250000,
total_tokens=300000,
completion_tokens_details=CompletionTokensDetailsWrapper(
accepted_prediction_tokens=0,
audio_tokens=0,
@@ -233,7 +233,7 @@ class TestXAICostCalculator:
usage = Usage(
prompt_tokens=100000, # Below 128k threshold
completion_tokens=50000,
total_tokens=150000,
total_tokens=160000,
completion_tokens_details=CompletionTokensDetailsWrapper(
accepted_prediction_tokens=0,
audio_tokens=0,
@@ -261,7 +261,7 @@ class TestXAICostCalculator:
usage = Usage(
prompt_tokens=200000, # Above 128k threshold
completion_tokens=100000,
total_tokens=300000,
total_tokens=350000,
completion_tokens_details=CompletionTokensDetailsWrapper(
accepted_prediction_tokens=0,
audio_tokens=0,
@@ -289,7 +289,7 @@ class TestXAICostCalculator:
usage = Usage(
prompt_tokens=150000, # Above 128k threshold
completion_tokens=50000, # Below 128k threshold
total_tokens=200000,
total_tokens=210000,
completion_tokens_details=CompletionTokensDetailsWrapper(
accepted_prediction_tokens=0,
audio_tokens=0,
@@ -331,6 +331,29 @@ class TestXAICostCalculator:
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
def test_already_normalised_usage_does_not_double_count_reasoning(self):
"""Cost calc must not double-bill when Usage is already OpenAI-normalised."""
usage = Usage(
prompt_tokens=12,
completion_tokens=200,
total_tokens=212,
completion_tokens_details=CompletionTokensDetailsWrapper(
accepted_prediction_tokens=0,
audio_tokens=0,
reasoning_tokens=100,
rejected_prediction_tokens=0,
text_tokens=None,
),
)
prompt_cost, completion_cost = cost_per_token(model="grok-3-mini", usage=usage)
expected_prompt_cost = 12 * 3e-7
expected_completion_cost = 200 * 5e-7
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
def test_web_search_cost_calculation(self):
"""Test web search cost calculation for X.AI models."""
# Test with web_search_requests in prompt_tokens_details (primary path)
+4 -3
View File
@@ -852,6 +852,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"supports_none_reasoning_effort": {"type": "boolean"},
"supports_xhigh_reasoning_effort": {"type": "boolean"},
"supports_max_reasoning_effort": {"type": "boolean"},
"supports_adaptive_thinking": {"type": "boolean"},
"supports_service_tier": {"type": "boolean"},
"supports_preset": {"type": "boolean"},
"tool_use_system_prompt_tokens": {"type": "number"},
@@ -2904,9 +2905,9 @@ def test_gemini_embedding_2_ga_in_cost_map():
assert info.get("input_cost_per_audio_per_second") == 0.00016
assert info.get("input_cost_per_video_per_second") == 0.00079
if provider in ("vertex_ai-embedding-models", "vertex_ai"):
assert info.get("uses_embed_content") is True, (
f"{key} must have uses_embed_content=true for correct Vertex AI routing"
)
assert (
info.get("uses_embed_content") is True
), f"{key} must have uses_embed_content=true for correct Vertex AI routing"
def test_gemini_lyria_3_preview_models_in_cost_map():