From 9edf463c3bc01495d2505a967aa5a9bbc45acd6c Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Thu, 2 May 2024 16:02:52 -0700 Subject: [PATCH] fix - revert init langfuse client on slack alerts --- litellm/integrations/slack_alerting.py | 75 +------------------ ...odel_prices_and_context_window_backup.json | 34 +++++++-- 2 files changed, 30 insertions(+), 79 deletions(-) diff --git a/litellm/integrations/slack_alerting.py b/litellm/integrations/slack_alerting.py index 8f8ce712e9..a9aba2f1c6 100644 --- a/litellm/integrations/slack_alerting.py +++ b/litellm/integrations/slack_alerting.py @@ -48,19 +48,6 @@ class SlackAlerting: self.internal_usage_cache = DualCache() self.async_http_handler = AsyncHTTPHandler() self.alert_to_webhook_url = alert_to_webhook_url - self.langfuse_logger = None - - try: - from litellm.integrations.langfuse import LangFuseLogger - - self.langfuse_logger = LangFuseLogger( - os.getenv("LANGFUSE_PUBLIC_KEY"), - os.getenv("LANGFUSE_SECRET_KEY"), - flush_interval=1, - ) - except: - pass - pass def update_values( @@ -110,62 +97,8 @@ class SlackAlerting: start_time: Optional[datetime.datetime] = None, end_time: Optional[datetime.datetime] = None, ): - import uuid - - # For now: do nothing as we're debugging why this is not working as expected - if request_data is not None: - trace_id = request_data.get("metadata", {}).get( - "trace_id", None - ) # get langfuse trace id - if trace_id is None: - trace_id = "litellm-alert-trace-" + str(uuid.uuid4()) - request_data["metadata"]["trace_id"] = trace_id - elif kwargs is not None: - _litellm_params = kwargs.get("litellm_params", {}) - trace_id = _litellm_params.get("metadata", {}).get( - "trace_id", None - ) # get langfuse trace id - if trace_id is None: - trace_id = "litellm-alert-trace-" + str(uuid.uuid4()) - _litellm_params["metadata"]["trace_id"] = trace_id - - # Log hanging request as an error on langfuse - if type == "hanging_request": - if self.langfuse_logger is not None: - _logging_kwargs = copy.deepcopy(request_data) - if _logging_kwargs is None: - _logging_kwargs = {} - _logging_kwargs["litellm_params"] = {} - request_data = request_data or {} - _logging_kwargs["litellm_params"]["metadata"] = request_data.get( - "metadata", {} - ) - # log to langfuse in a separate thread - import threading - - threading.Thread( - target=self.langfuse_logger.log_event, - args=( - _logging_kwargs, - None, - start_time, - end_time, - None, - print, - "ERROR", - "Requests is hanging", - ), - ).start() - - _langfuse_host = os.environ.get("LANGFUSE_HOST", "https://cloud.langfuse.com") - _langfuse_project_id = os.environ.get("LANGFUSE_PROJECT_ID") - - # langfuse urls look like: https://us.cloud.langfuse.com/project/************/traces/litellm-alert-trace-ididi9dk-09292-************ - - _langfuse_url = ( - f"{_langfuse_host}/project/{_langfuse_project_id}/traces/{trace_id}" - ) - request_info += f"\n🪢 Langfuse Trace: {_langfuse_url}" + # do nothing for now + pass return request_info def _response_taking_too_long_callback( @@ -242,10 +175,6 @@ class SlackAlerting: request_info = f"\nRequest Model: `{model}`\nAPI Base: `{api_base}`\nMessages: `{messages}`" slow_message = f"`Responses are slow - {round(time_difference_float,2)}s response time > Alerting threshold: {self.alerting_threshold}s`" if time_difference_float > self.alerting_threshold: - if "langfuse" in litellm.success_callback: - request_info = self._add_langfuse_trace_id_to_alert( - request_info=request_info, kwargs=kwargs, type="slow_response" - ) # add deployment latencies to alert if ( kwargs is not None diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index ce6f9b800f..7fcd425bb5 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -338,6 +338,18 @@ "output_cost_per_second": 0.0001, "litellm_provider": "azure" }, + "azure/gpt-4-turbo-2024-04-09": { + "max_tokens": 4096, + "max_input_tokens": 128000, + "max_output_tokens": 4096, + "input_cost_per_token": 0.00001, + "output_cost_per_token": 0.00003, + "litellm_provider": "azure", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_vision": true + }, "azure/gpt-4-0125-preview": { "max_tokens": 4096, "max_input_tokens": 128000, @@ -813,6 +825,7 @@ "litellm_provider": "anthropic", "mode": "chat", "supports_function_calling": true, + "supports_vision": true, "tool_use_system_prompt_tokens": 264 }, "claude-3-opus-20240229": { @@ -824,6 +837,7 @@ "litellm_provider": "anthropic", "mode": "chat", "supports_function_calling": true, + "supports_vision": true, "tool_use_system_prompt_tokens": 395 }, "claude-3-sonnet-20240229": { @@ -835,6 +849,7 @@ "litellm_provider": "anthropic", "mode": "chat", "supports_function_calling": true, + "supports_vision": true, "tool_use_system_prompt_tokens": 159 }, "text-bison": { @@ -1142,7 +1157,8 @@ "output_cost_per_token": 0.000015, "litellm_provider": "vertex_ai-anthropic_models", "mode": "chat", - "supports_function_calling": true + "supports_function_calling": true, + "supports_vision": true }, "vertex_ai/claude-3-haiku@20240307": { "max_tokens": 4096, @@ -1152,7 +1168,8 @@ "output_cost_per_token": 0.00000125, "litellm_provider": "vertex_ai-anthropic_models", "mode": "chat", - "supports_function_calling": true + "supports_function_calling": true, + "supports_vision": true }, "vertex_ai/claude-3-opus@20240229": { "max_tokens": 4096, @@ -1162,7 +1179,8 @@ "output_cost_per_token": 0.0000075, "litellm_provider": "vertex_ai-anthropic_models", "mode": "chat", - "supports_function_calling": true + "supports_function_calling": true, + "supports_vision": true }, "textembedding-gecko": { "max_tokens": 3072, @@ -1581,6 +1599,7 @@ "litellm_provider": "openrouter", "mode": "chat", "supports_function_calling": true, + "supports_vision": true, "tool_use_system_prompt_tokens": 395 }, "openrouter/google/palm-2-chat-bison": { @@ -1929,7 +1948,8 @@ "output_cost_per_token": 0.000015, "litellm_provider": "bedrock", "mode": "chat", - "supports_function_calling": true + "supports_function_calling": true, + "supports_vision": true }, "anthropic.claude-3-haiku-20240307-v1:0": { "max_tokens": 4096, @@ -1939,7 +1959,8 @@ "output_cost_per_token": 0.00000125, "litellm_provider": "bedrock", "mode": "chat", - "supports_function_calling": true + "supports_function_calling": true, + "supports_vision": true }, "anthropic.claude-3-opus-20240229-v1:0": { "max_tokens": 4096, @@ -1949,7 +1970,8 @@ "output_cost_per_token": 0.000075, "litellm_provider": "bedrock", "mode": "chat", - "supports_function_calling": true + "supports_function_calling": true, + "supports_vision": true }, "anthropic.claude-v1": { "max_tokens": 8191,