mirror of
https://github.com/tiennm99/litellm.git
synced 2026-08-02 12:21:10 +00:00
[Feat] Add Cost Estimator for AI Gateway (#18643)
* add estimate_cost endpoint * TestCostEstimateEndpoint * fix estimate_cost * add /cost/estimate to spend tracking routes * fix code QA checks * fixes endpoint
This commit is contained in:
@@ -11640,6 +11640,7 @@
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"gemini-1.5-flash": {
|
||||
"deprecation_date": "2025-09-29",
|
||||
"input_cost_per_audio_per_second": 2e-06,
|
||||
"input_cost_per_audio_per_second_above_128k_tokens": 4e-06,
|
||||
"input_cost_per_character": 1.875e-08,
|
||||
@@ -11744,6 +11745,7 @@
|
||||
"supports_vision": true
|
||||
},
|
||||
"gemini-1.5-flash-exp-0827": {
|
||||
"deprecation_date": "2025-09-29",
|
||||
"input_cost_per_audio_per_second": 2e-06,
|
||||
"input_cost_per_audio_per_second_above_128k_tokens": 4e-06,
|
||||
"input_cost_per_character": 1.875e-08,
|
||||
@@ -11778,6 +11780,7 @@
|
||||
"supports_vision": true
|
||||
},
|
||||
"gemini-1.5-flash-preview-0514": {
|
||||
"deprecation_date": "2025-09-29",
|
||||
"input_cost_per_audio_per_second": 2e-06,
|
||||
"input_cost_per_audio_per_second_above_128k_tokens": 4e-06,
|
||||
"input_cost_per_character": 1.875e-08,
|
||||
@@ -11811,6 +11814,7 @@
|
||||
"supports_vision": true
|
||||
},
|
||||
"gemini-1.5-pro": {
|
||||
"deprecation_date": "2025-09-29",
|
||||
"input_cost_per_audio_per_second": 3.125e-05,
|
||||
"input_cost_per_audio_per_second_above_128k_tokens": 6.25e-05,
|
||||
"input_cost_per_character": 3.125e-07,
|
||||
@@ -11898,6 +11902,7 @@
|
||||
"supports_vision": true
|
||||
},
|
||||
"gemini-1.5-pro-preview-0215": {
|
||||
"deprecation_date": "2025-09-29",
|
||||
"input_cost_per_audio_per_second": 3.125e-05,
|
||||
"input_cost_per_audio_per_second_above_128k_tokens": 6.25e-05,
|
||||
"input_cost_per_character": 3.125e-07,
|
||||
@@ -11925,6 +11930,7 @@
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"gemini-1.5-pro-preview-0409": {
|
||||
"deprecation_date": "2025-09-29",
|
||||
"input_cost_per_audio_per_second": 3.125e-05,
|
||||
"input_cost_per_audio_per_second_above_128k_tokens": 6.25e-05,
|
||||
"input_cost_per_character": 3.125e-07,
|
||||
@@ -11951,6 +11957,7 @@
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"gemini-1.5-pro-preview-0514": {
|
||||
"deprecation_date": "2025-09-29",
|
||||
"input_cost_per_audio_per_second": 3.125e-05,
|
||||
"input_cost_per_audio_per_second_above_128k_tokens": 6.25e-05,
|
||||
"input_cost_per_character": 3.125e-07,
|
||||
@@ -12222,6 +12229,7 @@
|
||||
"tpm": 250000
|
||||
},
|
||||
"gemini-2.0-flash-preview-image-generation": {
|
||||
"deprecation_date": "2025-11-14",
|
||||
"cache_read_input_token_cost": 2.5e-08,
|
||||
"input_cost_per_audio_token": 7e-07,
|
||||
"input_cost_per_token": 1e-07,
|
||||
@@ -12260,6 +12268,7 @@
|
||||
"supports_web_search": true
|
||||
},
|
||||
"gemini-2.0-flash-thinking-exp": {
|
||||
"deprecation_date": "2025-12-02",
|
||||
"cache_read_input_token_cost": 0.0,
|
||||
"input_cost_per_audio_per_second": 0,
|
||||
"input_cost_per_audio_per_second_above_128k_tokens": 0,
|
||||
@@ -12308,6 +12317,7 @@
|
||||
"supports_web_search": true
|
||||
},
|
||||
"gemini-2.0-flash-thinking-exp-01-21": {
|
||||
"deprecation_date": "2025-12-02",
|
||||
"cache_read_input_token_cost": 0.0,
|
||||
"input_cost_per_audio_per_second": 0,
|
||||
"input_cost_per_audio_per_second_above_128k_tokens": 0,
|
||||
@@ -12494,6 +12504,7 @@
|
||||
"tpm": 8000000
|
||||
},
|
||||
"gemini-2.5-flash-image-preview": {
|
||||
"deprecation_date": "2026-01-15",
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"input_cost_per_audio_token": 1e-06,
|
||||
"input_cost_per_token": 3e-07,
|
||||
@@ -12804,6 +12815,7 @@
|
||||
"tpm": 8000000
|
||||
},
|
||||
"gemini-2.5-flash-lite-preview-06-17": {
|
||||
"deprecation_date": "2025-11-18",
|
||||
"cache_read_input_token_cost": 2.5e-08,
|
||||
"input_cost_per_audio_token": 5e-07,
|
||||
"input_cost_per_token": 1e-07,
|
||||
@@ -12893,6 +12905,7 @@
|
||||
"supports_web_search": true
|
||||
},
|
||||
"gemini-2.5-flash-preview-05-20": {
|
||||
"deprecation_date": "2025-11-18",
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"input_cost_per_audio_token": 1e-06,
|
||||
"input_cost_per_token": 3e-07,
|
||||
@@ -13164,6 +13177,7 @@
|
||||
"supports_web_search": true
|
||||
},
|
||||
"gemini-2.5-pro-preview-03-25": {
|
||||
"deprecation_date": "2025-12-02",
|
||||
"cache_read_input_token_cost": 3.125e-07,
|
||||
"input_cost_per_audio_token": 1.25e-06,
|
||||
"input_cost_per_token": 1.25e-06,
|
||||
@@ -13209,6 +13223,7 @@
|
||||
"supports_web_search": true
|
||||
},
|
||||
"gemini-2.5-pro-preview-05-06": {
|
||||
"deprecation_date": "2025-12-02",
|
||||
"cache_read_input_token_cost": 3.125e-07,
|
||||
"input_cost_per_audio_token": 1.25e-06,
|
||||
"input_cost_per_token": 1.25e-06,
|
||||
@@ -13424,6 +13439,7 @@
|
||||
"tpm": 10000000
|
||||
},
|
||||
"gemini/gemini-1.5-flash": {
|
||||
"deprecation_date": "2025-09-29",
|
||||
"input_cost_per_token": 7.5e-08,
|
||||
"input_cost_per_token_above_128k_tokens": 1.5e-07,
|
||||
"litellm_provider": "gemini",
|
||||
@@ -13507,6 +13523,7 @@
|
||||
"tpm": 4000000
|
||||
},
|
||||
"gemini/gemini-1.5-flash-8b": {
|
||||
"deprecation_date": "2025-09-29",
|
||||
"input_cost_per_token": 0,
|
||||
"input_cost_per_token_above_128k_tokens": 0,
|
||||
"litellm_provider": "gemini",
|
||||
@@ -13533,6 +13550,7 @@
|
||||
"tpm": 4000000
|
||||
},
|
||||
"gemini/gemini-1.5-flash-8b-exp-0827": {
|
||||
"deprecation_date": "2025-09-29",
|
||||
"input_cost_per_token": 0,
|
||||
"input_cost_per_token_above_128k_tokens": 0,
|
||||
"litellm_provider": "gemini",
|
||||
@@ -13558,6 +13576,7 @@
|
||||
"tpm": 4000000
|
||||
},
|
||||
"gemini/gemini-1.5-flash-8b-exp-0924": {
|
||||
"deprecation_date": "2025-09-29",
|
||||
"input_cost_per_token": 0,
|
||||
"input_cost_per_token_above_128k_tokens": 0,
|
||||
"litellm_provider": "gemini",
|
||||
@@ -13584,6 +13603,7 @@
|
||||
"tpm": 4000000
|
||||
},
|
||||
"gemini/gemini-1.5-flash-exp-0827": {
|
||||
"deprecation_date": "2025-09-29",
|
||||
"input_cost_per_token": 0,
|
||||
"input_cost_per_token_above_128k_tokens": 0,
|
||||
"litellm_provider": "gemini",
|
||||
@@ -13609,6 +13629,7 @@
|
||||
"tpm": 4000000
|
||||
},
|
||||
"gemini/gemini-1.5-flash-latest": {
|
||||
"deprecation_date": "2025-09-29",
|
||||
"input_cost_per_token": 7.5e-08,
|
||||
"input_cost_per_token_above_128k_tokens": 1.5e-07,
|
||||
"litellm_provider": "gemini",
|
||||
@@ -13635,6 +13656,7 @@
|
||||
"tpm": 4000000
|
||||
},
|
||||
"gemini/gemini-1.5-pro": {
|
||||
"deprecation_date": "2025-09-29",
|
||||
"input_cost_per_token": 3.5e-06,
|
||||
"input_cost_per_token_above_128k_tokens": 7e-06,
|
||||
"litellm_provider": "gemini",
|
||||
@@ -13696,6 +13718,7 @@
|
||||
"tpm": 4000000
|
||||
},
|
||||
"gemini/gemini-1.5-pro-exp-0801": {
|
||||
"deprecation_date": "2025-09-29",
|
||||
"input_cost_per_token": 3.5e-06,
|
||||
"input_cost_per_token_above_128k_tokens": 7e-06,
|
||||
"litellm_provider": "gemini",
|
||||
@@ -13715,6 +13738,7 @@
|
||||
"tpm": 4000000
|
||||
},
|
||||
"gemini/gemini-1.5-pro-exp-0827": {
|
||||
"deprecation_date": "2025-09-29",
|
||||
"input_cost_per_token": 0,
|
||||
"input_cost_per_token_above_128k_tokens": 0,
|
||||
"litellm_provider": "gemini",
|
||||
@@ -13734,6 +13758,7 @@
|
||||
"tpm": 4000000
|
||||
},
|
||||
"gemini/gemini-1.5-pro-latest": {
|
||||
"deprecation_date": "2025-09-29",
|
||||
"input_cost_per_token": 3.5e-06,
|
||||
"input_cost_per_token_above_128k_tokens": 7e-06,
|
||||
"litellm_provider": "gemini",
|
||||
@@ -13916,6 +13941,7 @@
|
||||
"tpm": 4000000
|
||||
},
|
||||
"gemini/gemini-2.0-flash-lite-preview-02-05": {
|
||||
"deprecation_date": "2025-12-02",
|
||||
"cache_read_input_token_cost": 1.875e-08,
|
||||
"input_cost_per_audio_token": 7.5e-08,
|
||||
"input_cost_per_token": 7.5e-08,
|
||||
@@ -13953,6 +13979,7 @@
|
||||
"tpm": 10000000
|
||||
},
|
||||
"gemini/gemini-2.0-flash-live-001": {
|
||||
"deprecation_date": "2025-12-09",
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"input_cost_per_audio_token": 2.1e-06,
|
||||
"input_cost_per_image": 2.1e-06,
|
||||
@@ -14001,6 +14028,7 @@
|
||||
"tpm": 250000
|
||||
},
|
||||
"gemini/gemini-2.0-flash-preview-image-generation": {
|
||||
"deprecation_date": "2025-11-14",
|
||||
"cache_read_input_token_cost": 2.5e-08,
|
||||
"input_cost_per_audio_token": 7e-07,
|
||||
"input_cost_per_token": 1e-07,
|
||||
@@ -14040,6 +14068,7 @@
|
||||
"tpm": 10000000
|
||||
},
|
||||
"gemini/gemini-2.0-flash-thinking-exp": {
|
||||
"deprecation_date": "2025-12-02",
|
||||
"cache_read_input_token_cost": 0.0,
|
||||
"input_cost_per_audio_per_second": 0,
|
||||
"input_cost_per_audio_per_second_above_128k_tokens": 0,
|
||||
@@ -14089,6 +14118,7 @@
|
||||
"tpm": 4000000
|
||||
},
|
||||
"gemini/gemini-2.0-flash-thinking-exp-01-21": {
|
||||
"deprecation_date": "2025-12-02",
|
||||
"cache_read_input_token_cost": 0.0,
|
||||
"input_cost_per_audio_per_second": 0,
|
||||
"input_cost_per_audio_per_second_above_128k_tokens": 0,
|
||||
@@ -14277,6 +14307,7 @@
|
||||
"tpm": 8000000
|
||||
},
|
||||
"gemini/gemini-2.5-flash-image-preview": {
|
||||
"deprecation_date": "2026-01-15",
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"input_cost_per_audio_token": 1e-06,
|
||||
"input_cost_per_token": 3e-07,
|
||||
@@ -14597,6 +14628,7 @@
|
||||
"tpm": 250000
|
||||
},
|
||||
"gemini/gemini-2.5-flash-lite-preview-06-17": {
|
||||
"deprecation_date": "2025-11-18",
|
||||
"cache_read_input_token_cost": 2.5e-08,
|
||||
"input_cost_per_audio_token": 5e-07,
|
||||
"input_cost_per_token": 1e-07,
|
||||
@@ -14688,6 +14720,7 @@
|
||||
"tpm": 250000
|
||||
},
|
||||
"gemini/gemini-2.5-flash-preview-05-20": {
|
||||
"deprecation_date": "2025-11-18",
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"input_cost_per_audio_token": 1e-06,
|
||||
"input_cost_per_token": 3e-07,
|
||||
@@ -15034,6 +15067,7 @@
|
||||
"tpm": 250000
|
||||
},
|
||||
"gemini/gemini-2.5-pro-preview-03-25": {
|
||||
"deprecation_date": "2025-12-02",
|
||||
"cache_read_input_token_cost": 3.125e-07,
|
||||
"input_cost_per_audio_token": 7e-07,
|
||||
"input_cost_per_token": 1.25e-06,
|
||||
@@ -15074,6 +15108,7 @@
|
||||
"tpm": 10000000
|
||||
},
|
||||
"gemini/gemini-2.5-pro-preview-05-06": {
|
||||
"deprecation_date": "2025-12-02",
|
||||
"cache_read_input_token_cost": 3.125e-07,
|
||||
"input_cost_per_audio_token": 7e-07,
|
||||
"input_cost_per_token": 1.25e-06,
|
||||
@@ -15349,6 +15384,7 @@
|
||||
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing"
|
||||
},
|
||||
"gemini/imagen-3.0-generate-002": {
|
||||
"deprecation_date": "2025-11-10",
|
||||
"litellm_provider": "gemini",
|
||||
"mode": "image_generation",
|
||||
"output_cost_per_image": 0.04,
|
||||
@@ -15415,6 +15451,7 @@
|
||||
]
|
||||
},
|
||||
"gemini/veo-3.0-fast-generate-preview": {
|
||||
"deprecation_date": "2025-11-12",
|
||||
"litellm_provider": "gemini",
|
||||
"max_input_tokens": 1024,
|
||||
"max_tokens": 1024,
|
||||
@@ -15429,6 +15466,7 @@
|
||||
]
|
||||
},
|
||||
"gemini/veo-3.0-generate-preview": {
|
||||
"deprecation_date": "2025-11-12",
|
||||
"litellm_provider": "gemini",
|
||||
"max_input_tokens": 1024,
|
||||
"max_tokens": 1024,
|
||||
@@ -25126,6 +25164,7 @@
|
||||
"source": "https://docs.mistral.ai/capabilities/code_generation/"
|
||||
},
|
||||
"text-embedding-004": {
|
||||
"deprecation_date": "2026-01-14",
|
||||
"input_cost_per_character": 2.5e-08,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"litellm_provider": "vertex_ai-embedding-models",
|
||||
@@ -27896,6 +27935,7 @@
|
||||
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing"
|
||||
},
|
||||
"vertex_ai/imagen-3.0-generate-002": {
|
||||
"deprecation_date": "2025-11-10",
|
||||
"litellm_provider": "vertex_ai-image-models",
|
||||
"mode": "image_generation",
|
||||
"output_cost_per_image": 0.04,
|
||||
@@ -28406,6 +28446,7 @@
|
||||
]
|
||||
},
|
||||
"vertex_ai/veo-3.0-fast-generate-preview": {
|
||||
"deprecation_date": "2025-11-12",
|
||||
"litellm_provider": "vertex_ai-video-models",
|
||||
"max_input_tokens": 1024,
|
||||
"max_tokens": 1024,
|
||||
@@ -28420,6 +28461,7 @@
|
||||
]
|
||||
},
|
||||
"vertex_ai/veo-3.0-generate-preview": {
|
||||
"deprecation_date": "2025-11-12",
|
||||
"litellm_provider": "vertex_ai-video-models",
|
||||
"max_input_tokens": 1024,
|
||||
"max_tokens": 1024,
|
||||
|
||||
@@ -522,6 +522,7 @@ class LiteLLMRoutes(enum.Enum):
|
||||
"/spend/tags",
|
||||
"/spend/calculate",
|
||||
"/spend/logs",
|
||||
"/cost/estimate",
|
||||
]
|
||||
|
||||
global_spend_tracking_routes = [
|
||||
@@ -3825,3 +3826,46 @@ class LiteLLM_ManagedVectorStoresTable(LiteLLMPydanticObjectBase):
|
||||
|
||||
class ResponseLiteLLM_ManagedVectorStore(TypedDict, total=False):
|
||||
vector_store: LiteLLM_ManagedVectorStoresTable
|
||||
|
||||
|
||||
class CostEstimateRequest(LiteLLMPydanticObjectBase):
|
||||
"""Request body for /cost/estimate endpoint."""
|
||||
|
||||
model: str = Field(description="Model name (from /model_group/info)")
|
||||
input_tokens: int = Field(description="Expected input tokens per request", ge=0)
|
||||
output_tokens: int = Field(description="Expected output tokens per request", ge=0)
|
||||
num_requests_per_day: Optional[int] = Field(
|
||||
default=None, description="Number of requests per day", ge=0
|
||||
)
|
||||
num_requests_per_month: Optional[int] = Field(
|
||||
default=None, description="Number of requests per month", ge=0
|
||||
)
|
||||
|
||||
|
||||
class CostEstimateResponse(LiteLLMPydanticObjectBase):
|
||||
"""Response body for /cost/estimate endpoint."""
|
||||
|
||||
model: str
|
||||
input_tokens: int
|
||||
output_tokens: int
|
||||
num_requests_per_day: Optional[int] = None
|
||||
num_requests_per_month: Optional[int] = None
|
||||
# Per-request costs
|
||||
cost_per_request: float = Field(description="Total cost per request (includes margin)")
|
||||
input_cost_per_request: float = Field(description="Input token cost per request (before margin)")
|
||||
output_cost_per_request: float = Field(description="Output token cost per request (before margin)")
|
||||
margin_cost_per_request: float = Field(default=0.0, description="Margin/fee added per request")
|
||||
# Daily costs (if num_requests_per_day provided)
|
||||
daily_cost: Optional[float] = Field(default=None, description="Total daily cost (includes margin)")
|
||||
daily_input_cost: Optional[float] = Field(default=None, description="Daily input token cost")
|
||||
daily_output_cost: Optional[float] = Field(default=None, description="Daily output token cost")
|
||||
daily_margin_cost: Optional[float] = Field(default=None, description="Daily margin/fee")
|
||||
# Monthly costs (if num_requests_per_month provided)
|
||||
monthly_cost: Optional[float] = Field(default=None, description="Total monthly cost (includes margin)")
|
||||
monthly_input_cost: Optional[float] = Field(default=None, description="Monthly input token cost")
|
||||
monthly_output_cost: Optional[float] = Field(default=None, description="Monthly output token cost")
|
||||
monthly_margin_cost: Optional[float] = Field(default=None, description="Monthly margin/fee")
|
||||
# Pricing info
|
||||
input_cost_per_token: Optional[float] = None
|
||||
output_cost_per_token: Optional[float] = None
|
||||
provider: Optional[str] = None
|
||||
|
||||
@@ -7,6 +7,7 @@ GET /config/cost_discount_config - Get current cost discount configuration
|
||||
PATCH /config/cost_discount_config - Update cost discount configuration
|
||||
GET /config/cost_margin_config - Get current cost margin configuration
|
||||
PATCH /config/cost_margin_config - Update cost margin configuration
|
||||
POST /cost/estimate - Estimate cost for a given model and token counts
|
||||
"""
|
||||
|
||||
from typing import Dict, Union
|
||||
@@ -15,13 +16,37 @@ from fastapi import APIRouter, Depends, HTTPException
|
||||
|
||||
import litellm
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.proxy._types import CommonProxyErrors, UserAPIKeyAuth
|
||||
from litellm.cost_calculator import completion_cost
|
||||
from litellm.proxy._types import (
|
||||
CommonProxyErrors,
|
||||
CostEstimateRequest,
|
||||
CostEstimateResponse,
|
||||
UserAPIKeyAuth,
|
||||
)
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
from litellm.types.utils import LlmProvidersSet
|
||||
|
||||
router = APIRouter()
|
||||
|
||||
|
||||
def _calculate_period_costs(
|
||||
num_requests, cost_per_request, input_cost, output_cost, margin_cost
|
||||
):
|
||||
"""
|
||||
Calculate costs for a given number of requests.
|
||||
|
||||
Returns tuple of (total_cost, input_cost, output_cost, margin_cost) or all None if num_requests is None/0.
|
||||
"""
|
||||
if not num_requests:
|
||||
return None, None, None, None
|
||||
return (
|
||||
cost_per_request * num_requests,
|
||||
input_cost * num_requests,
|
||||
output_cost * num_requests,
|
||||
margin_cost * num_requests,
|
||||
)
|
||||
|
||||
|
||||
@router.get(
|
||||
"/config/cost_discount_config",
|
||||
tags=["Cost Tracking"],
|
||||
@@ -347,3 +372,144 @@ async def update_cost_margin_config(
|
||||
detail={"error": f"Failed to update cost margin config: {str(e)}"}
|
||||
)
|
||||
|
||||
|
||||
@router.post(
|
||||
"/cost/estimate",
|
||||
tags=["Cost Tracking"],
|
||||
dependencies=[Depends(user_api_key_auth)],
|
||||
response_model=CostEstimateResponse,
|
||||
)
|
||||
async def estimate_cost(
|
||||
request: CostEstimateRequest,
|
||||
user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
|
||||
) -> CostEstimateResponse:
|
||||
"""
|
||||
Estimate cost for a given model and token counts.
|
||||
|
||||
This endpoint uses the same cost calculation logic as actual requests,
|
||||
including any configured margins and discounts.
|
||||
|
||||
Parameters:
|
||||
- model: Model name (e.g., "gpt-4", "claude-3-opus")
|
||||
- input_tokens: Expected input tokens per request
|
||||
- output_tokens: Expected output tokens per request
|
||||
- num_requests_per_day: Number of requests per day (optional)
|
||||
- num_requests_per_month: Number of requests per month (optional)
|
||||
|
||||
Returns cost breakdown including:
|
||||
- Per-request costs (input, output, margin)
|
||||
- Daily costs (if num_requests_per_day provided)
|
||||
- Monthly costs (if num_requests_per_month provided)
|
||||
|
||||
Example:
|
||||
```json
|
||||
{
|
||||
"model": "gpt-4",
|
||||
"input_tokens": 1000,
|
||||
"output_tokens": 500,
|
||||
"num_requests_per_day": 100,
|
||||
"num_requests_per_month": 3000
|
||||
}
|
||||
```
|
||||
"""
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
from litellm.types.utils import Usage
|
||||
from litellm.utils import ModelResponse
|
||||
|
||||
# Create a mock response with usage for completion_cost
|
||||
mock_response = ModelResponse(
|
||||
model=request.model,
|
||||
usage=Usage(
|
||||
prompt_tokens=request.input_tokens,
|
||||
completion_tokens=request.output_tokens,
|
||||
total_tokens=request.input_tokens + request.output_tokens,
|
||||
),
|
||||
)
|
||||
|
||||
# Create a logging object to capture cost breakdown
|
||||
litellm_logging_obj = LiteLLMLoggingObj(
|
||||
model=request.model,
|
||||
messages=[],
|
||||
stream=False,
|
||||
call_type="completion",
|
||||
start_time=None,
|
||||
litellm_call_id="cost-estimate",
|
||||
function_id="cost-estimate",
|
||||
)
|
||||
|
||||
# Use completion_cost which handles all the logic including margins/discounts
|
||||
try:
|
||||
cost_per_request = completion_cost(
|
||||
completion_response=mock_response,
|
||||
model=request.model,
|
||||
litellm_logging_obj=litellm_logging_obj,
|
||||
)
|
||||
except Exception as e:
|
||||
raise HTTPException(
|
||||
status_code=404,
|
||||
detail={
|
||||
"error": f"Could not calculate cost for model '{request.model}': {str(e)}"
|
||||
},
|
||||
)
|
||||
|
||||
# Get cost breakdown from the logging object
|
||||
cost_breakdown = litellm_logging_obj.cost_breakdown
|
||||
|
||||
input_cost = cost_breakdown.get("input_cost", 0.0) if cost_breakdown else 0.0
|
||||
output_cost = cost_breakdown.get("output_cost", 0.0) if cost_breakdown else 0.0
|
||||
margin_cost = cost_breakdown.get("margin_total_amount", 0.0) if cost_breakdown else 0.0
|
||||
|
||||
# Get model info for per-token pricing display
|
||||
try:
|
||||
model_info = litellm.get_model_info(model=request.model)
|
||||
input_cost_per_token = model_info.get("input_cost_per_token")
|
||||
output_cost_per_token = model_info.get("output_cost_per_token")
|
||||
custom_llm_provider = model_info.get("litellm_provider")
|
||||
except Exception:
|
||||
input_cost_per_token = None
|
||||
output_cost_per_token = None
|
||||
custom_llm_provider = None
|
||||
|
||||
# Calculate daily and monthly costs
|
||||
daily_cost, daily_input_cost, daily_output_cost, daily_margin_cost = (
|
||||
_calculate_period_costs(
|
||||
num_requests=request.num_requests_per_day,
|
||||
cost_per_request=cost_per_request,
|
||||
input_cost=input_cost,
|
||||
output_cost=output_cost,
|
||||
margin_cost=margin_cost,
|
||||
)
|
||||
)
|
||||
monthly_cost, monthly_input_cost, monthly_output_cost, monthly_margin_cost = (
|
||||
_calculate_period_costs(
|
||||
num_requests=request.num_requests_per_month,
|
||||
cost_per_request=cost_per_request,
|
||||
input_cost=input_cost,
|
||||
output_cost=output_cost,
|
||||
margin_cost=margin_cost,
|
||||
)
|
||||
)
|
||||
|
||||
return CostEstimateResponse(
|
||||
model=request.model,
|
||||
input_tokens=request.input_tokens,
|
||||
output_tokens=request.output_tokens,
|
||||
num_requests_per_day=request.num_requests_per_day,
|
||||
num_requests_per_month=request.num_requests_per_month,
|
||||
cost_per_request=cost_per_request,
|
||||
input_cost_per_request=input_cost,
|
||||
output_cost_per_request=output_cost,
|
||||
margin_cost_per_request=margin_cost,
|
||||
daily_cost=daily_cost,
|
||||
daily_input_cost=daily_input_cost,
|
||||
daily_output_cost=daily_output_cost,
|
||||
daily_margin_cost=daily_margin_cost,
|
||||
monthly_cost=monthly_cost,
|
||||
monthly_input_cost=monthly_input_cost,
|
||||
monthly_output_cost=monthly_output_cost,
|
||||
monthly_margin_cost=monthly_margin_cost,
|
||||
input_cost_per_token=input_cost_per_token,
|
||||
output_cost_per_token=output_cost_per_token,
|
||||
provider=custom_llm_provider,
|
||||
)
|
||||
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
"""
|
||||
Tests for the /cost/estimate endpoint in cost_tracking_settings.py
|
||||
"""
|
||||
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.proxy._types import CostEstimateRequest, CostEstimateResponse
|
||||
from litellm.proxy.management_endpoints.cost_tracking_settings import estimate_cost
|
||||
|
||||
|
||||
class TestCostEstimateEndpoint:
|
||||
"""Tests for the cost estimation endpoint."""
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_estimate_cost_daily_and_monthly(self):
|
||||
"""
|
||||
Test that cost estimation calculates daily and monthly costs correctly.
|
||||
"""
|
||||
request = CostEstimateRequest(
|
||||
model="gpt-4",
|
||||
input_tokens=1000,
|
||||
output_tokens=500,
|
||||
num_requests_per_day=100,
|
||||
num_requests_per_month=3000,
|
||||
)
|
||||
|
||||
with patch(
|
||||
"litellm.proxy.management_endpoints.cost_tracking_settings.completion_cost"
|
||||
) as mock_completion_cost:
|
||||
mock_completion_cost.return_value = 0.06
|
||||
|
||||
with patch("litellm.get_model_info") as mock_get_model_info:
|
||||
mock_get_model_info.return_value = {
|
||||
"input_cost_per_token": 0.00003,
|
||||
"output_cost_per_token": 0.00006,
|
||||
"litellm_provider": "openai",
|
||||
}
|
||||
|
||||
response = await estimate_cost(
|
||||
request=request,
|
||||
user_api_key_dict=MagicMock(),
|
||||
)
|
||||
|
||||
assert response.model == "gpt-4"
|
||||
assert response.cost_per_request == 0.06
|
||||
assert response.daily_cost == pytest.approx(6.0) # 0.06 * 100
|
||||
assert response.monthly_cost == pytest.approx(180.0) # 0.06 * 3000
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_estimate_cost_model_not_found(self):
|
||||
"""
|
||||
Test that 404 is raised when model cost calculation fails.
|
||||
"""
|
||||
request = CostEstimateRequest(
|
||||
model="nonexistent-model",
|
||||
input_tokens=1000,
|
||||
output_tokens=500,
|
||||
)
|
||||
|
||||
with patch(
|
||||
"litellm.proxy.management_endpoints.cost_tracking_settings.completion_cost"
|
||||
) as mock_completion_cost:
|
||||
mock_completion_cost.side_effect = Exception("Model not found in cost map")
|
||||
|
||||
from fastapi import HTTPException
|
||||
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await estimate_cost(
|
||||
request=request,
|
||||
user_api_key_dict=MagicMock(),
|
||||
)
|
||||
|
||||
assert exc_info.value.status_code == 404
|
||||
Reference in New Issue
Block a user