Fix converse anthropic usage object according to v1/messages specs

This commit is contained in:
Sameer Kankute
2026-02-16 14:03:31 +05:30
parent 44bb1dafdb
commit 01cdec5771
3 changed files with 125 additions and 3 deletions
@@ -239,8 +239,13 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
merged_chunk["delta"] = {}
# Add usage to the held chunk
uncached_input_tokens = chunk.usage.prompt_tokens or 0
if hasattr(chunk.usage, "prompt_tokens_details") and chunk.usage.prompt_tokens_details:
cached_tokens = getattr(chunk.usage.prompt_tokens_details, "cached_tokens", 0) or 0
uncached_input_tokens -= cached_tokens
usage_dict: UsageDelta = {
"input_tokens": chunk.usage.prompt_tokens or 0,
"input_tokens": uncached_input_tokens,
"output_tokens": chunk.usage.completion_tokens or 0,
}
# Add cache tokens if available (for prompt caching support)
@@ -412,6 +417,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
if block_type == "tool_use":
# Type narrowing: content_block_start is ToolUseBlock when block_type is "tool_use"
from typing import cast
from litellm.types.llms.anthropic import ToolUseBlock
tool_block = cast(ToolUseBlock, content_block_start)
@@ -430,6 +436,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
# if we get a function name since it signals a new tool call
if block_type == "tool_use":
from typing import cast
from litellm.types.llms.anthropic import ToolUseBlock
tool_block = cast(ToolUseBlock, content_block_start)
@@ -1070,8 +1070,13 @@ class LiteLLMAnthropicMessagesAdapter:
)
# extract usage
usage: Usage = getattr(response, "usage")
uncached_input_tokens = usage.prompt_tokens or 0
if hasattr(usage, "prompt_tokens_details") and usage.prompt_tokens_details:
cached_tokens = getattr(usage.prompt_tokens_details, "cached_tokens", 0) or 0
uncached_input_tokens -= cached_tokens
anthropic_usage = AnthropicUsage(
input_tokens=usage.prompt_tokens or 0,
input_tokens=uncached_input_tokens,
output_tokens=usage.completion_tokens or 0,
)
# Add cache tokens if available (for prompt caching support)
@@ -1230,8 +1235,13 @@ class LiteLLMAnthropicMessagesAdapter:
else:
litellm_usage_chunk = None
if litellm_usage_chunk is not None:
uncached_input_tokens = litellm_usage_chunk.prompt_tokens or 0
if hasattr(litellm_usage_chunk, "prompt_tokens_details") and litellm_usage_chunk.prompt_tokens_details:
cached_tokens = getattr(litellm_usage_chunk.prompt_tokens_details, "cached_tokens", 0) or 0
uncached_input_tokens -= cached_tokens
usage_delta = UsageDelta(
input_tokens=litellm_usage_chunk.prompt_tokens or 0,
input_tokens=uncached_input_tokens,
output_tokens=litellm_usage_chunk.completion_tokens or 0,
)
# Add cache tokens if available (for prompt caching support)
@@ -1706,3 +1706,108 @@ def test_translate_openai_response_restores_tool_names():
assert len(tool_use_blocks) == 1
# Name should be restored to original
assert tool_use_blocks[0]["name"] == original_name
def test_translate_openai_response_to_anthropic_input_tokens_excludes_cached_tokens():
"""
Regression test: input_tokens in Anthropic format should NOT include cached tokens.
Issue: v1/messages API was returning incorrect input_token count when using prompt caching.
The OpenAI format includes cached tokens in prompt_tokens, but Anthropic format should not.
According to Anthropic's spec:
- input_tokens = uncached input tokens only
- cache_read_input_tokens = tokens read from cache
In OpenAI format:
- prompt_tokens = all input tokens (including cached)
- prompt_tokens_details.cached_tokens = cached tokens
Expected: anthropic.input_tokens = openai.prompt_tokens - openai.prompt_tokens_details.cached_tokens
"""
from litellm.types.utils import PromptTokensDetailsWrapper
# Create OpenAI format response with cached tokens
# Scenario: 100 total prompt tokens, 30 of which are cached
usage = Usage(
prompt_tokens=100,
completion_tokens=50,
total_tokens=150,
prompt_tokens_details=PromptTokensDetailsWrapper(
cached_tokens=30
),
cache_read_input_tokens=30, # Anthropic format cache info
)
response = ModelResponse(
id="test-id",
choices=[
Choices(
index=0,
finish_reason="stop",
message=Message(
role="assistant",
content="Test response",
),
)
],
model="claude-3-sonnet-20240229",
usage=usage,
)
# Convert to Anthropic format
adapter = LiteLLMAnthropicMessagesAdapter()
anthropic_response = adapter.translate_openai_response_to_anthropic(
response=response,
tool_name_mapping=None,
)
# Validate: input_tokens should be 70 (100 - 30 cached), not 100
assert anthropic_response["usage"]["input_tokens"] == 70, (
f"Expected input_tokens=70 (100 total - 30 cached), "
f"but got {anthropic_response['usage']['input_tokens']}. "
f"input_tokens should NOT include cached tokens per Anthropic spec."
)
assert anthropic_response["usage"]["output_tokens"] == 50
assert anthropic_response["usage"]["cache_read_input_tokens"] == 30
def test_translate_openai_response_to_anthropic_input_tokens_no_cache():
"""
Regression test: input_tokens should equal prompt_tokens when there are no cached tokens.
"""
from litellm.types.utils import PromptTokensDetailsWrapper
# Create OpenAI format response without cached tokens
usage = Usage(
prompt_tokens=100,
completion_tokens=50,
total_tokens=150,
)
response = ModelResponse(
id="test-id",
choices=[
Choices(
index=0,
finish_reason="stop",
message=Message(
role="assistant",
content="Test response",
),
)
],
model="claude-3-sonnet-20240229",
usage=usage,
)
# Convert to Anthropic format
adapter = LiteLLMAnthropicMessagesAdapter()
anthropic_response = adapter.translate_openai_response_to_anthropic(
response=response,
tool_name_mapping=None,
)
# Validate: input_tokens should equal prompt_tokens when no caching
assert anthropic_response["usage"]["input_tokens"] == 100
assert anthropic_response["usage"]["output_tokens"] == 50