diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index d32adf54b5..81b4469f24 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -11640,6 +11640,7 @@ "supports_tool_choice": true }, "gemini-1.5-flash": { + "deprecation_date": "2025-09-29", "input_cost_per_audio_per_second": 2e-06, "input_cost_per_audio_per_second_above_128k_tokens": 4e-06, "input_cost_per_character": 1.875e-08, @@ -11744,6 +11745,7 @@ "supports_vision": true }, "gemini-1.5-flash-exp-0827": { + "deprecation_date": "2025-09-29", "input_cost_per_audio_per_second": 2e-06, "input_cost_per_audio_per_second_above_128k_tokens": 4e-06, "input_cost_per_character": 1.875e-08, @@ -11778,6 +11780,7 @@ "supports_vision": true }, "gemini-1.5-flash-preview-0514": { + "deprecation_date": "2025-09-29", "input_cost_per_audio_per_second": 2e-06, "input_cost_per_audio_per_second_above_128k_tokens": 4e-06, "input_cost_per_character": 1.875e-08, @@ -11811,6 +11814,7 @@ "supports_vision": true }, "gemini-1.5-pro": { + "deprecation_date": "2025-09-29", "input_cost_per_audio_per_second": 3.125e-05, "input_cost_per_audio_per_second_above_128k_tokens": 6.25e-05, "input_cost_per_character": 3.125e-07, @@ -11898,6 +11902,7 @@ "supports_vision": true }, "gemini-1.5-pro-preview-0215": { + "deprecation_date": "2025-09-29", "input_cost_per_audio_per_second": 3.125e-05, "input_cost_per_audio_per_second_above_128k_tokens": 6.25e-05, "input_cost_per_character": 3.125e-07, @@ -11925,6 +11930,7 @@ "supports_tool_choice": true }, "gemini-1.5-pro-preview-0409": { + "deprecation_date": "2025-09-29", "input_cost_per_audio_per_second": 3.125e-05, "input_cost_per_audio_per_second_above_128k_tokens": 6.25e-05, "input_cost_per_character": 3.125e-07, @@ -11951,6 +11957,7 @@ "supports_tool_choice": true }, "gemini-1.5-pro-preview-0514": { + "deprecation_date": "2025-09-29", "input_cost_per_audio_per_second": 3.125e-05, "input_cost_per_audio_per_second_above_128k_tokens": 6.25e-05, "input_cost_per_character": 3.125e-07, @@ -12222,6 +12229,7 @@ "tpm": 250000 }, "gemini-2.0-flash-preview-image-generation": { + "deprecation_date": "2025-11-14", "cache_read_input_token_cost": 2.5e-08, "input_cost_per_audio_token": 7e-07, "input_cost_per_token": 1e-07, @@ -12260,6 +12268,7 @@ "supports_web_search": true }, "gemini-2.0-flash-thinking-exp": { + "deprecation_date": "2025-12-02", "cache_read_input_token_cost": 0.0, "input_cost_per_audio_per_second": 0, "input_cost_per_audio_per_second_above_128k_tokens": 0, @@ -12308,6 +12317,7 @@ "supports_web_search": true }, "gemini-2.0-flash-thinking-exp-01-21": { + "deprecation_date": "2025-12-02", "cache_read_input_token_cost": 0.0, "input_cost_per_audio_per_second": 0, "input_cost_per_audio_per_second_above_128k_tokens": 0, @@ -12494,6 +12504,7 @@ "tpm": 8000000 }, "gemini-2.5-flash-image-preview": { + "deprecation_date": "2026-01-15", "cache_read_input_token_cost": 7.5e-08, "input_cost_per_audio_token": 1e-06, "input_cost_per_token": 3e-07, @@ -12804,6 +12815,7 @@ "tpm": 8000000 }, "gemini-2.5-flash-lite-preview-06-17": { + "deprecation_date": "2025-11-18", "cache_read_input_token_cost": 2.5e-08, "input_cost_per_audio_token": 5e-07, "input_cost_per_token": 1e-07, @@ -12893,6 +12905,7 @@ "supports_web_search": true }, "gemini-2.5-flash-preview-05-20": { + "deprecation_date": "2025-11-18", "cache_read_input_token_cost": 7.5e-08, "input_cost_per_audio_token": 1e-06, "input_cost_per_token": 3e-07, @@ -13164,6 +13177,7 @@ "supports_web_search": true }, "gemini-2.5-pro-preview-03-25": { + "deprecation_date": "2025-12-02", "cache_read_input_token_cost": 3.125e-07, "input_cost_per_audio_token": 1.25e-06, "input_cost_per_token": 1.25e-06, @@ -13209,6 +13223,7 @@ "supports_web_search": true }, "gemini-2.5-pro-preview-05-06": { + "deprecation_date": "2025-12-02", "cache_read_input_token_cost": 3.125e-07, "input_cost_per_audio_token": 1.25e-06, "input_cost_per_token": 1.25e-06, @@ -13424,6 +13439,7 @@ "tpm": 10000000 }, "gemini/gemini-1.5-flash": { + "deprecation_date": "2025-09-29", "input_cost_per_token": 7.5e-08, "input_cost_per_token_above_128k_tokens": 1.5e-07, "litellm_provider": "gemini", @@ -13507,6 +13523,7 @@ "tpm": 4000000 }, "gemini/gemini-1.5-flash-8b": { + "deprecation_date": "2025-09-29", "input_cost_per_token": 0, "input_cost_per_token_above_128k_tokens": 0, "litellm_provider": "gemini", @@ -13533,6 +13550,7 @@ "tpm": 4000000 }, "gemini/gemini-1.5-flash-8b-exp-0827": { + "deprecation_date": "2025-09-29", "input_cost_per_token": 0, "input_cost_per_token_above_128k_tokens": 0, "litellm_provider": "gemini", @@ -13558,6 +13576,7 @@ "tpm": 4000000 }, "gemini/gemini-1.5-flash-8b-exp-0924": { + "deprecation_date": "2025-09-29", "input_cost_per_token": 0, "input_cost_per_token_above_128k_tokens": 0, "litellm_provider": "gemini", @@ -13584,6 +13603,7 @@ "tpm": 4000000 }, "gemini/gemini-1.5-flash-exp-0827": { + "deprecation_date": "2025-09-29", "input_cost_per_token": 0, "input_cost_per_token_above_128k_tokens": 0, "litellm_provider": "gemini", @@ -13609,6 +13629,7 @@ "tpm": 4000000 }, "gemini/gemini-1.5-flash-latest": { + "deprecation_date": "2025-09-29", "input_cost_per_token": 7.5e-08, "input_cost_per_token_above_128k_tokens": 1.5e-07, "litellm_provider": "gemini", @@ -13635,6 +13656,7 @@ "tpm": 4000000 }, "gemini/gemini-1.5-pro": { + "deprecation_date": "2025-09-29", "input_cost_per_token": 3.5e-06, "input_cost_per_token_above_128k_tokens": 7e-06, "litellm_provider": "gemini", @@ -13696,6 +13718,7 @@ "tpm": 4000000 }, "gemini/gemini-1.5-pro-exp-0801": { + "deprecation_date": "2025-09-29", "input_cost_per_token": 3.5e-06, "input_cost_per_token_above_128k_tokens": 7e-06, "litellm_provider": "gemini", @@ -13715,6 +13738,7 @@ "tpm": 4000000 }, "gemini/gemini-1.5-pro-exp-0827": { + "deprecation_date": "2025-09-29", "input_cost_per_token": 0, "input_cost_per_token_above_128k_tokens": 0, "litellm_provider": "gemini", @@ -13734,6 +13758,7 @@ "tpm": 4000000 }, "gemini/gemini-1.5-pro-latest": { + "deprecation_date": "2025-09-29", "input_cost_per_token": 3.5e-06, "input_cost_per_token_above_128k_tokens": 7e-06, "litellm_provider": "gemini", @@ -13916,6 +13941,7 @@ "tpm": 4000000 }, "gemini/gemini-2.0-flash-lite-preview-02-05": { + "deprecation_date": "2025-12-02", "cache_read_input_token_cost": 1.875e-08, "input_cost_per_audio_token": 7.5e-08, "input_cost_per_token": 7.5e-08, @@ -13953,6 +13979,7 @@ "tpm": 10000000 }, "gemini/gemini-2.0-flash-live-001": { + "deprecation_date": "2025-12-09", "cache_read_input_token_cost": 7.5e-08, "input_cost_per_audio_token": 2.1e-06, "input_cost_per_image": 2.1e-06, @@ -14001,6 +14028,7 @@ "tpm": 250000 }, "gemini/gemini-2.0-flash-preview-image-generation": { + "deprecation_date": "2025-11-14", "cache_read_input_token_cost": 2.5e-08, "input_cost_per_audio_token": 7e-07, "input_cost_per_token": 1e-07, @@ -14040,6 +14068,7 @@ "tpm": 10000000 }, "gemini/gemini-2.0-flash-thinking-exp": { + "deprecation_date": "2025-12-02", "cache_read_input_token_cost": 0.0, "input_cost_per_audio_per_second": 0, "input_cost_per_audio_per_second_above_128k_tokens": 0, @@ -14089,6 +14118,7 @@ "tpm": 4000000 }, "gemini/gemini-2.0-flash-thinking-exp-01-21": { + "deprecation_date": "2025-12-02", "cache_read_input_token_cost": 0.0, "input_cost_per_audio_per_second": 0, "input_cost_per_audio_per_second_above_128k_tokens": 0, @@ -14277,6 +14307,7 @@ "tpm": 8000000 }, "gemini/gemini-2.5-flash-image-preview": { + "deprecation_date": "2026-01-15", "cache_read_input_token_cost": 7.5e-08, "input_cost_per_audio_token": 1e-06, "input_cost_per_token": 3e-07, @@ -14597,6 +14628,7 @@ "tpm": 250000 }, "gemini/gemini-2.5-flash-lite-preview-06-17": { + "deprecation_date": "2025-11-18", "cache_read_input_token_cost": 2.5e-08, "input_cost_per_audio_token": 5e-07, "input_cost_per_token": 1e-07, @@ -14688,6 +14720,7 @@ "tpm": 250000 }, "gemini/gemini-2.5-flash-preview-05-20": { + "deprecation_date": "2025-11-18", "cache_read_input_token_cost": 7.5e-08, "input_cost_per_audio_token": 1e-06, "input_cost_per_token": 3e-07, @@ -15034,6 +15067,7 @@ "tpm": 250000 }, "gemini/gemini-2.5-pro-preview-03-25": { + "deprecation_date": "2025-12-02", "cache_read_input_token_cost": 3.125e-07, "input_cost_per_audio_token": 7e-07, "input_cost_per_token": 1.25e-06, @@ -15074,6 +15108,7 @@ "tpm": 10000000 }, "gemini/gemini-2.5-pro-preview-05-06": { + "deprecation_date": "2025-12-02", "cache_read_input_token_cost": 3.125e-07, "input_cost_per_audio_token": 7e-07, "input_cost_per_token": 1.25e-06, @@ -15349,6 +15384,7 @@ "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" }, "gemini/imagen-3.0-generate-002": { + "deprecation_date": "2025-11-10", "litellm_provider": "gemini", "mode": "image_generation", "output_cost_per_image": 0.04, @@ -15415,6 +15451,7 @@ ] }, "gemini/veo-3.0-fast-generate-preview": { + "deprecation_date": "2025-11-12", "litellm_provider": "gemini", "max_input_tokens": 1024, "max_tokens": 1024, @@ -15429,6 +15466,7 @@ ] }, "gemini/veo-3.0-generate-preview": { + "deprecation_date": "2025-11-12", "litellm_provider": "gemini", "max_input_tokens": 1024, "max_tokens": 1024, @@ -25126,6 +25164,7 @@ "source": "https://docs.mistral.ai/capabilities/code_generation/" }, "text-embedding-004": { + "deprecation_date": "2026-01-14", "input_cost_per_character": 2.5e-08, "input_cost_per_token": 1e-07, "litellm_provider": "vertex_ai-embedding-models", @@ -27896,6 +27935,7 @@ "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" }, "vertex_ai/imagen-3.0-generate-002": { + "deprecation_date": "2025-11-10", "litellm_provider": "vertex_ai-image-models", "mode": "image_generation", "output_cost_per_image": 0.04, @@ -28406,6 +28446,7 @@ ] }, "vertex_ai/veo-3.0-fast-generate-preview": { + "deprecation_date": "2025-11-12", "litellm_provider": "vertex_ai-video-models", "max_input_tokens": 1024, "max_tokens": 1024, @@ -28420,6 +28461,7 @@ ] }, "vertex_ai/veo-3.0-generate-preview": { + "deprecation_date": "2025-11-12", "litellm_provider": "vertex_ai-video-models", "max_input_tokens": 1024, "max_tokens": 1024, diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index fa32f60c07..b77cc40d6d 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -522,6 +522,7 @@ class LiteLLMRoutes(enum.Enum): "/spend/tags", "/spend/calculate", "/spend/logs", + "/cost/estimate", ] global_spend_tracking_routes = [ @@ -3825,3 +3826,46 @@ class LiteLLM_ManagedVectorStoresTable(LiteLLMPydanticObjectBase): class ResponseLiteLLM_ManagedVectorStore(TypedDict, total=False): vector_store: LiteLLM_ManagedVectorStoresTable + + +class CostEstimateRequest(LiteLLMPydanticObjectBase): + """Request body for /cost/estimate endpoint.""" + + model: str = Field(description="Model name (from /model_group/info)") + input_tokens: int = Field(description="Expected input tokens per request", ge=0) + output_tokens: int = Field(description="Expected output tokens per request", ge=0) + num_requests_per_day: Optional[int] = Field( + default=None, description="Number of requests per day", ge=0 + ) + num_requests_per_month: Optional[int] = Field( + default=None, description="Number of requests per month", ge=0 + ) + + +class CostEstimateResponse(LiteLLMPydanticObjectBase): + """Response body for /cost/estimate endpoint.""" + + model: str + input_tokens: int + output_tokens: int + num_requests_per_day: Optional[int] = None + num_requests_per_month: Optional[int] = None + # Per-request costs + cost_per_request: float = Field(description="Total cost per request (includes margin)") + input_cost_per_request: float = Field(description="Input token cost per request (before margin)") + output_cost_per_request: float = Field(description="Output token cost per request (before margin)") + margin_cost_per_request: float = Field(default=0.0, description="Margin/fee added per request") + # Daily costs (if num_requests_per_day provided) + daily_cost: Optional[float] = Field(default=None, description="Total daily cost (includes margin)") + daily_input_cost: Optional[float] = Field(default=None, description="Daily input token cost") + daily_output_cost: Optional[float] = Field(default=None, description="Daily output token cost") + daily_margin_cost: Optional[float] = Field(default=None, description="Daily margin/fee") + # Monthly costs (if num_requests_per_month provided) + monthly_cost: Optional[float] = Field(default=None, description="Total monthly cost (includes margin)") + monthly_input_cost: Optional[float] = Field(default=None, description="Monthly input token cost") + monthly_output_cost: Optional[float] = Field(default=None, description="Monthly output token cost") + monthly_margin_cost: Optional[float] = Field(default=None, description="Monthly margin/fee") + # Pricing info + input_cost_per_token: Optional[float] = None + output_cost_per_token: Optional[float] = None + provider: Optional[str] = None diff --git a/litellm/proxy/management_endpoints/cost_tracking_settings.py b/litellm/proxy/management_endpoints/cost_tracking_settings.py index 86433a232c..0622393ec8 100644 --- a/litellm/proxy/management_endpoints/cost_tracking_settings.py +++ b/litellm/proxy/management_endpoints/cost_tracking_settings.py @@ -7,6 +7,7 @@ GET /config/cost_discount_config - Get current cost discount configuration PATCH /config/cost_discount_config - Update cost discount configuration GET /config/cost_margin_config - Get current cost margin configuration PATCH /config/cost_margin_config - Update cost margin configuration +POST /cost/estimate - Estimate cost for a given model and token counts """ from typing import Dict, Union @@ -15,13 +16,37 @@ from fastapi import APIRouter, Depends, HTTPException import litellm from litellm._logging import verbose_proxy_logger -from litellm.proxy._types import CommonProxyErrors, UserAPIKeyAuth +from litellm.cost_calculator import completion_cost +from litellm.proxy._types import ( + CommonProxyErrors, + CostEstimateRequest, + CostEstimateResponse, + UserAPIKeyAuth, +) from litellm.proxy.auth.user_api_key_auth import user_api_key_auth from litellm.types.utils import LlmProvidersSet router = APIRouter() +def _calculate_period_costs( + num_requests, cost_per_request, input_cost, output_cost, margin_cost +): + """ + Calculate costs for a given number of requests. + + Returns tuple of (total_cost, input_cost, output_cost, margin_cost) or all None if num_requests is None/0. + """ + if not num_requests: + return None, None, None, None + return ( + cost_per_request * num_requests, + input_cost * num_requests, + output_cost * num_requests, + margin_cost * num_requests, + ) + + @router.get( "/config/cost_discount_config", tags=["Cost Tracking"], @@ -347,3 +372,144 @@ async def update_cost_margin_config( detail={"error": f"Failed to update cost margin config: {str(e)}"} ) + +@router.post( + "/cost/estimate", + tags=["Cost Tracking"], + dependencies=[Depends(user_api_key_auth)], + response_model=CostEstimateResponse, +) +async def estimate_cost( + request: CostEstimateRequest, + user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), +) -> CostEstimateResponse: + """ + Estimate cost for a given model and token counts. + + This endpoint uses the same cost calculation logic as actual requests, + including any configured margins and discounts. + + Parameters: + - model: Model name (e.g., "gpt-4", "claude-3-opus") + - input_tokens: Expected input tokens per request + - output_tokens: Expected output tokens per request + - num_requests_per_day: Number of requests per day (optional) + - num_requests_per_month: Number of requests per month (optional) + + Returns cost breakdown including: + - Per-request costs (input, output, margin) + - Daily costs (if num_requests_per_day provided) + - Monthly costs (if num_requests_per_month provided) + + Example: + ```json + { + "model": "gpt-4", + "input_tokens": 1000, + "output_tokens": 500, + "num_requests_per_day": 100, + "num_requests_per_month": 3000 + } + ``` + """ + from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj + from litellm.types.utils import Usage + from litellm.utils import ModelResponse + + # Create a mock response with usage for completion_cost + mock_response = ModelResponse( + model=request.model, + usage=Usage( + prompt_tokens=request.input_tokens, + completion_tokens=request.output_tokens, + total_tokens=request.input_tokens + request.output_tokens, + ), + ) + + # Create a logging object to capture cost breakdown + litellm_logging_obj = LiteLLMLoggingObj( + model=request.model, + messages=[], + stream=False, + call_type="completion", + start_time=None, + litellm_call_id="cost-estimate", + function_id="cost-estimate", + ) + + # Use completion_cost which handles all the logic including margins/discounts + try: + cost_per_request = completion_cost( + completion_response=mock_response, + model=request.model, + litellm_logging_obj=litellm_logging_obj, + ) + except Exception as e: + raise HTTPException( + status_code=404, + detail={ + "error": f"Could not calculate cost for model '{request.model}': {str(e)}" + }, + ) + + # Get cost breakdown from the logging object + cost_breakdown = litellm_logging_obj.cost_breakdown + + input_cost = cost_breakdown.get("input_cost", 0.0) if cost_breakdown else 0.0 + output_cost = cost_breakdown.get("output_cost", 0.0) if cost_breakdown else 0.0 + margin_cost = cost_breakdown.get("margin_total_amount", 0.0) if cost_breakdown else 0.0 + + # Get model info for per-token pricing display + try: + model_info = litellm.get_model_info(model=request.model) + input_cost_per_token = model_info.get("input_cost_per_token") + output_cost_per_token = model_info.get("output_cost_per_token") + custom_llm_provider = model_info.get("litellm_provider") + except Exception: + input_cost_per_token = None + output_cost_per_token = None + custom_llm_provider = None + + # Calculate daily and monthly costs + daily_cost, daily_input_cost, daily_output_cost, daily_margin_cost = ( + _calculate_period_costs( + num_requests=request.num_requests_per_day, + cost_per_request=cost_per_request, + input_cost=input_cost, + output_cost=output_cost, + margin_cost=margin_cost, + ) + ) + monthly_cost, monthly_input_cost, monthly_output_cost, monthly_margin_cost = ( + _calculate_period_costs( + num_requests=request.num_requests_per_month, + cost_per_request=cost_per_request, + input_cost=input_cost, + output_cost=output_cost, + margin_cost=margin_cost, + ) + ) + + return CostEstimateResponse( + model=request.model, + input_tokens=request.input_tokens, + output_tokens=request.output_tokens, + num_requests_per_day=request.num_requests_per_day, + num_requests_per_month=request.num_requests_per_month, + cost_per_request=cost_per_request, + input_cost_per_request=input_cost, + output_cost_per_request=output_cost, + margin_cost_per_request=margin_cost, + daily_cost=daily_cost, + daily_input_cost=daily_input_cost, + daily_output_cost=daily_output_cost, + daily_margin_cost=daily_margin_cost, + monthly_cost=monthly_cost, + monthly_input_cost=monthly_input_cost, + monthly_output_cost=monthly_output_cost, + monthly_margin_cost=monthly_margin_cost, + input_cost_per_token=input_cost_per_token, + output_cost_per_token=output_cost_per_token, + provider=custom_llm_provider, + ) + diff --git a/tests/litellm/proxy/management_endpoints/test_cost_estimate_endpoint.py b/tests/litellm/proxy/management_endpoints/test_cost_estimate_endpoint.py new file mode 100644 index 0000000000..f2d8d87855 --- /dev/null +++ b/tests/litellm/proxy/management_endpoints/test_cost_estimate_endpoint.py @@ -0,0 +1,75 @@ +""" +Tests for the /cost/estimate endpoint in cost_tracking_settings.py +""" + +from unittest.mock import MagicMock, patch + +import pytest + +from litellm.proxy._types import CostEstimateRequest, CostEstimateResponse +from litellm.proxy.management_endpoints.cost_tracking_settings import estimate_cost + + +class TestCostEstimateEndpoint: + """Tests for the cost estimation endpoint.""" + + @pytest.mark.asyncio + async def test_estimate_cost_daily_and_monthly(self): + """ + Test that cost estimation calculates daily and monthly costs correctly. + """ + request = CostEstimateRequest( + model="gpt-4", + input_tokens=1000, + output_tokens=500, + num_requests_per_day=100, + num_requests_per_month=3000, + ) + + with patch( + "litellm.proxy.management_endpoints.cost_tracking_settings.completion_cost" + ) as mock_completion_cost: + mock_completion_cost.return_value = 0.06 + + with patch("litellm.get_model_info") as mock_get_model_info: + mock_get_model_info.return_value = { + "input_cost_per_token": 0.00003, + "output_cost_per_token": 0.00006, + "litellm_provider": "openai", + } + + response = await estimate_cost( + request=request, + user_api_key_dict=MagicMock(), + ) + + assert response.model == "gpt-4" + assert response.cost_per_request == 0.06 + assert response.daily_cost == pytest.approx(6.0) # 0.06 * 100 + assert response.monthly_cost == pytest.approx(180.0) # 0.06 * 3000 + + @pytest.mark.asyncio + async def test_estimate_cost_model_not_found(self): + """ + Test that 404 is raised when model cost calculation fails. + """ + request = CostEstimateRequest( + model="nonexistent-model", + input_tokens=1000, + output_tokens=500, + ) + + with patch( + "litellm.proxy.management_endpoints.cost_tracking_settings.completion_cost" + ) as mock_completion_cost: + mock_completion_cost.side_effect = Exception("Model not found in cost map") + + from fastapi import HTTPException + + with pytest.raises(HTTPException) as exc_info: + await estimate_cost( + request=request, + user_api_key_dict=MagicMock(), + ) + + assert exc_info.value.status_code == 404