diff --git a/litellm/proxy/_experimental/out/model_hub_table.html b/litellm/proxy/_experimental/out/model_hub_table.html deleted file mode 100644 index 155cf9b802..0000000000 --- a/litellm/proxy/_experimental/out/model_hub_table.html +++ /dev/null @@ -1 +0,0 @@ -LiteLLM Dashboard \ No newline at end of file diff --git a/litellm/proxy/_experimental/out/onboarding.html b/litellm/proxy/_experimental/out/onboarding.html deleted file mode 100644 index 8a3400b40a..0000000000 --- a/litellm/proxy/_experimental/out/onboarding.html +++ /dev/null @@ -1 +0,0 @@ -LiteLLM Dashboard \ No newline at end of file diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index ab1c52c9ae..329393c873 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -730,8 +730,11 @@ class ProxyBaseLLMRequestProcessing: """ Anthropic /messages and Google /generateContent streaming data generator require SSE events """ + from litellm.types.utils import ModelResponse, ModelResponseStream + verbose_proxy_logger.debug("inside generator") try: + str_so_far = "" async for chunk in response: verbose_proxy_logger.debug( "async_data_generator: received streaming chunk - {}".format(chunk) @@ -741,8 +744,13 @@ class ProxyBaseLLMRequestProcessing: user_api_key_dict=user_api_key_dict, response=chunk, data=request_data, + str_so_far=str_so_far, ) + if isinstance(chunk, (ModelResponse, ModelResponseStream)): + response_str = litellm.get_response_string(response_obj=chunk) + str_so_far += response_str + # Format chunk using helper function yield ProxyBaseLLMRequestProcessing.return_sse_chunk(chunk) except Exception as e: diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 7fca29f2bf..63f72d4b17 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -147,7 +147,6 @@ from litellm.litellm_core_utils.credential_accessor import CredentialAccessor from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.litellm_core_utils.sensitive_data_masker import SensitiveDataMasker from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler -from litellm.types.realtime import RealtimeQueryParams from litellm.proxy._experimental.mcp_server.rest_endpoints import ( router as mcp_rest_endpoints_router, ) @@ -367,6 +366,7 @@ from litellm.types.proxy.management_endpoints.ui_sso import ( DefaultTeamSSOParams, LiteLLM_UpperboundKeyGenerateParams, ) +from litellm.types.realtime import RealtimeQueryParams from litellm.types.router import DeploymentTypedDict from litellm.types.router import ModelInfo as RouterModelInfo from litellm.types.router import RouterGeneralSettings, updateDeployment @@ -3235,6 +3235,7 @@ async def async_data_generator( ): verbose_proxy_logger.debug("inside generator") try: + str_so_far = "" async for chunk in proxy_logging_obj.async_post_call_streaming_iterator_hook( user_api_key_dict=user_api_key_dict, response=response, @@ -3243,13 +3244,21 @@ async def async_data_generator( verbose_proxy_logger.debug( "async_data_generator: received streaming chunk - {}".format(chunk) ) + + ### CALL HOOKS ### - modify outgoing data chunk = await proxy_logging_obj.async_post_call_streaming_hook( user_api_key_dict=user_api_key_dict, response=chunk, data=request_data, + str_so_far=str_so_far, ) + + if isinstance(chunk, (ModelResponse, ModelResponseStream)): + response_str = litellm.get_response_string(response_obj=chunk) + str_so_far += response_str + if isinstance(chunk, BaseModel): chunk = chunk.model_dump_json(exclude_none=True, exclude_unset=True) @@ -3538,7 +3547,7 @@ class ProxyStartupEvent: timezone=ZoneInfo("America/Los_Angeles"), # Pacific Time ) await proxy_logging_obj.slack_alerting_instance.send_fallback_stats_from_prometheus() - + await cls._initialize_spend_tracking_background_jobs(scheduler=scheduler) ### SPEND LOG CLEANUP ### @@ -3587,12 +3596,14 @@ class ProxyStartupEvent: scheduler.start() @classmethod - async def _initialize_spend_tracking_background_jobs(cls, scheduler: AsyncIOScheduler): + async def _initialize_spend_tracking_background_jobs( + cls, scheduler: AsyncIOScheduler + ): """ Initialize the spend tracking background jobs 1. CloudZero Background Job 2. Prometheus Background Job - + Args: scheduler: The scheduler to add the background jobs to """ @@ -3606,7 +3617,7 @@ class ProxyStartupEvent: if await is_cloudzero_setup_in_db(): await init_cloudzero_background_job() - + ######################################################## # Prometheus Background Job ######################################################## @@ -3617,7 +3628,6 @@ class ProxyStartupEvent: PrometheusLogger.initialize_budget_metrics_cron_job(scheduler=scheduler) except Exception: PrometheusLogger = None - @classmethod async def _setup_prisma_client( @@ -4681,7 +4691,9 @@ from litellm import _arealtime async def websocket_endpoint( websocket: WebSocket, model: str, - intent: str = fastapi.Query(None, description="The intent of the websocket connection."), + intent: str = fastapi.Query( + None, description="The intent of the websocket connection." + ), user_api_key_dict=Depends(user_api_key_auth_websocket), ): import websockets diff --git a/litellm/proxy/utils.py b/litellm/proxy/utils.py index 02b7e14229..fa5eee7197 100644 --- a/litellm/proxy/utils.py +++ b/litellm/proxy/utils.py @@ -1070,6 +1070,7 @@ class ProxyLogging: ModelResponse, EmbeddingResponse, ImageResponse, ModelResponseStream ], user_api_key_dict: UserAPIKeyAuth, + str_so_far: Optional[str] = None, ): """ Allow user to modify outgoing streaming data -> per chunk @@ -1110,8 +1111,13 @@ class ProxyLogging: else: _callback = callback # type: ignore if _callback is not None and isinstance(_callback, CustomLogger): + if str_so_far is not None: + complete_response = str_so_far + response_str + else: + complete_response = response_str await _callback.async_post_call_streaming_hook( - user_api_key_dict=user_api_key_dict, response=response_str + user_api_key_dict=user_api_key_dict, + response=complete_response, ) except Exception as e: raise e