mirror of
https://github.com/tiennm99/litellm.git
synced 2026-08-02 10:21:52 +00:00
Fix converse anthropic usage object according to v1/messages specs
This commit is contained in:
@@ -239,8 +239,13 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
|
||||
merged_chunk["delta"] = {}
|
||||
|
||||
# Add usage to the held chunk
|
||||
uncached_input_tokens = chunk.usage.prompt_tokens or 0
|
||||
if hasattr(chunk.usage, "prompt_tokens_details") and chunk.usage.prompt_tokens_details:
|
||||
cached_tokens = getattr(chunk.usage.prompt_tokens_details, "cached_tokens", 0) or 0
|
||||
uncached_input_tokens -= cached_tokens
|
||||
|
||||
usage_dict: UsageDelta = {
|
||||
"input_tokens": chunk.usage.prompt_tokens or 0,
|
||||
"input_tokens": uncached_input_tokens,
|
||||
"output_tokens": chunk.usage.completion_tokens or 0,
|
||||
}
|
||||
# Add cache tokens if available (for prompt caching support)
|
||||
@@ -412,6 +417,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
|
||||
if block_type == "tool_use":
|
||||
# Type narrowing: content_block_start is ToolUseBlock when block_type is "tool_use"
|
||||
from typing import cast
|
||||
|
||||
from litellm.types.llms.anthropic import ToolUseBlock
|
||||
|
||||
tool_block = cast(ToolUseBlock, content_block_start)
|
||||
@@ -430,6 +436,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
|
||||
# if we get a function name since it signals a new tool call
|
||||
if block_type == "tool_use":
|
||||
from typing import cast
|
||||
|
||||
from litellm.types.llms.anthropic import ToolUseBlock
|
||||
|
||||
tool_block = cast(ToolUseBlock, content_block_start)
|
||||
|
||||
@@ -1070,8 +1070,13 @@ class LiteLLMAnthropicMessagesAdapter:
|
||||
)
|
||||
# extract usage
|
||||
usage: Usage = getattr(response, "usage")
|
||||
uncached_input_tokens = usage.prompt_tokens or 0
|
||||
if hasattr(usage, "prompt_tokens_details") and usage.prompt_tokens_details:
|
||||
cached_tokens = getattr(usage.prompt_tokens_details, "cached_tokens", 0) or 0
|
||||
uncached_input_tokens -= cached_tokens
|
||||
|
||||
anthropic_usage = AnthropicUsage(
|
||||
input_tokens=usage.prompt_tokens or 0,
|
||||
input_tokens=uncached_input_tokens,
|
||||
output_tokens=usage.completion_tokens or 0,
|
||||
)
|
||||
# Add cache tokens if available (for prompt caching support)
|
||||
@@ -1230,8 +1235,13 @@ class LiteLLMAnthropicMessagesAdapter:
|
||||
else:
|
||||
litellm_usage_chunk = None
|
||||
if litellm_usage_chunk is not None:
|
||||
uncached_input_tokens = litellm_usage_chunk.prompt_tokens or 0
|
||||
if hasattr(litellm_usage_chunk, "prompt_tokens_details") and litellm_usage_chunk.prompt_tokens_details:
|
||||
cached_tokens = getattr(litellm_usage_chunk.prompt_tokens_details, "cached_tokens", 0) or 0
|
||||
uncached_input_tokens -= cached_tokens
|
||||
|
||||
usage_delta = UsageDelta(
|
||||
input_tokens=litellm_usage_chunk.prompt_tokens or 0,
|
||||
input_tokens=uncached_input_tokens,
|
||||
output_tokens=litellm_usage_chunk.completion_tokens or 0,
|
||||
)
|
||||
# Add cache tokens if available (for prompt caching support)
|
||||
|
||||
+105
@@ -1706,3 +1706,108 @@ def test_translate_openai_response_restores_tool_names():
|
||||
assert len(tool_use_blocks) == 1
|
||||
# Name should be restored to original
|
||||
assert tool_use_blocks[0]["name"] == original_name
|
||||
|
||||
|
||||
def test_translate_openai_response_to_anthropic_input_tokens_excludes_cached_tokens():
|
||||
"""
|
||||
Regression test: input_tokens in Anthropic format should NOT include cached tokens.
|
||||
|
||||
Issue: v1/messages API was returning incorrect input_token count when using prompt caching.
|
||||
The OpenAI format includes cached tokens in prompt_tokens, but Anthropic format should not.
|
||||
|
||||
According to Anthropic's spec:
|
||||
- input_tokens = uncached input tokens only
|
||||
- cache_read_input_tokens = tokens read from cache
|
||||
|
||||
In OpenAI format:
|
||||
- prompt_tokens = all input tokens (including cached)
|
||||
- prompt_tokens_details.cached_tokens = cached tokens
|
||||
|
||||
Expected: anthropic.input_tokens = openai.prompt_tokens - openai.prompt_tokens_details.cached_tokens
|
||||
"""
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper
|
||||
|
||||
# Create OpenAI format response with cached tokens
|
||||
# Scenario: 100 total prompt tokens, 30 of which are cached
|
||||
usage = Usage(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=50,
|
||||
total_tokens=150,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
cached_tokens=30
|
||||
),
|
||||
cache_read_input_tokens=30, # Anthropic format cache info
|
||||
)
|
||||
|
||||
response = ModelResponse(
|
||||
id="test-id",
|
||||
choices=[
|
||||
Choices(
|
||||
index=0,
|
||||
finish_reason="stop",
|
||||
message=Message(
|
||||
role="assistant",
|
||||
content="Test response",
|
||||
),
|
||||
)
|
||||
],
|
||||
model="claude-3-sonnet-20240229",
|
||||
usage=usage,
|
||||
)
|
||||
|
||||
# Convert to Anthropic format
|
||||
adapter = LiteLLMAnthropicMessagesAdapter()
|
||||
anthropic_response = adapter.translate_openai_response_to_anthropic(
|
||||
response=response,
|
||||
tool_name_mapping=None,
|
||||
)
|
||||
|
||||
# Validate: input_tokens should be 70 (100 - 30 cached), not 100
|
||||
assert anthropic_response["usage"]["input_tokens"] == 70, (
|
||||
f"Expected input_tokens=70 (100 total - 30 cached), "
|
||||
f"but got {anthropic_response['usage']['input_tokens']}. "
|
||||
f"input_tokens should NOT include cached tokens per Anthropic spec."
|
||||
)
|
||||
assert anthropic_response["usage"]["output_tokens"] == 50
|
||||
assert anthropic_response["usage"]["cache_read_input_tokens"] == 30
|
||||
|
||||
|
||||
def test_translate_openai_response_to_anthropic_input_tokens_no_cache():
|
||||
"""
|
||||
Regression test: input_tokens should equal prompt_tokens when there are no cached tokens.
|
||||
"""
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper
|
||||
|
||||
# Create OpenAI format response without cached tokens
|
||||
usage = Usage(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=50,
|
||||
total_tokens=150,
|
||||
)
|
||||
|
||||
response = ModelResponse(
|
||||
id="test-id",
|
||||
choices=[
|
||||
Choices(
|
||||
index=0,
|
||||
finish_reason="stop",
|
||||
message=Message(
|
||||
role="assistant",
|
||||
content="Test response",
|
||||
),
|
||||
)
|
||||
],
|
||||
model="claude-3-sonnet-20240229",
|
||||
usage=usage,
|
||||
)
|
||||
|
||||
# Convert to Anthropic format
|
||||
adapter = LiteLLMAnthropicMessagesAdapter()
|
||||
anthropic_response = adapter.translate_openai_response_to_anthropic(
|
||||
response=response,
|
||||
tool_name_mapping=None,
|
||||
)
|
||||
|
||||
# Validate: input_tokens should equal prompt_tokens when no caching
|
||||
assert anthropic_response["usage"]["input_tokens"] == 100
|
||||
assert anthropic_response["usage"]["output_tokens"] == 50
|
||||
|
||||
Reference in New Issue
Block a user