From 5d22229d354e40a40debbec7af9f2e97a429fec5 Mon Sep 17 00:00:00 2001 From: Alexsander Hamir Date: Sat, 4 Oct 2025 09:10:37 -0700 Subject: [PATCH] [Fix] Cache - Avoiding expensive operations when cache isn't available (#15182) * Optimize cache performance by avoiding expensive operations when caching is disabled - Moved cache availability checks before expensive operations to improve performance for non-cached requests - Updated client code to handle None responses from caching handler * clean hot path * Fix TypeError with isinstance check for CustomStreamWrapper in caching Fixed `TypeError: typing.Any cannot be used with isinstance()` that was occurring in the caching handler when checking cached streaming responses. The issue was caused by CustomStreamWrapper being aliased to `typing.Any` at runtime through the TYPE_CHECKING conditional import pattern. When the code attempted to use isinstance(cached_result, CustomStreamWrapper) at lines 222 and 338, it failed because Python's isinstance() cannot be used with typing.Any. Solution: Import CustomStreamWrapper at runtime separately from the TYPE_CHECKING block, while keeping a type alias for static type checking. This allows isinstance checks to work properly while maintaining type hints. * fix: remove unnecessary type checking --- litellm/caching/caching_handler.py | 85 +++++++++++++++++------------- litellm/utils.py | 18 ++++--- 2 files changed, 57 insertions(+), 46 deletions(-) diff --git a/litellm/caching/caching_handler.py b/litellm/caching/caching_handler.py index b151ebd651..17cd50f75a 100644 --- a/litellm/caching/caching_handler.py +++ b/litellm/caching/caching_handler.py @@ -14,6 +14,7 @@ It utilizes the (RedisCache, s3Cache, RedisSemanticCache, QdrantSemanticCache, I In each method it will call the appropriate method from caching.py """ +import time import asyncio import datetime import inspect @@ -57,10 +58,16 @@ from litellm.types.utils import ( if TYPE_CHECKING: from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj - from litellm.utils import CustomStreamWrapper else: LiteLLMLoggingObj = Any - CustomStreamWrapper = Any + + +from litellm.litellm_core_utils.streaming_handler import CustomStreamWrapper + + +from litellm.litellm_core_utils.core_helpers import ( +_get_parent_otel_span_from_kwargs, +) class CachingHandlerResponse(BaseModel): @@ -112,7 +119,7 @@ class LLMCachingHandler: call_type: str, kwargs: Dict[str, Any], args: Optional[Tuple[Any, ...]] = None, - ) -> CachingHandlerResponse: + ) -> Optional[CachingHandlerResponse]: """ Internal method to get from the cache. Handles different call types (embeddings, chat/completions, text_completion, transcription) @@ -133,32 +140,27 @@ class LLMCachingHandler: Raises: None """ - from litellm.litellm_core_utils.core_helpers import ( - _get_parent_otel_span_from_kwargs, - ) - from litellm.utils import CustomStreamWrapper - - kwargs = kwargs.copy() - args = args or () - ######################################################### - # Init cache timing metrics - ######################################################### - cache_check_start_time = datetime.datetime.now() - cache_check_end_time = None - ######################################################### - - - parent_otel_span = _get_parent_otel_span_from_kwargs(kwargs) - kwargs["parent_otel_span"] = parent_otel_span - final_embedding_cached_response: Optional[EmbeddingResponse] = None - embedding_all_elements_cache_hit: bool = False - cached_result: Optional[Any] = None + # Check if caching should be performed BEFORE doing expensive operations if ( (kwargs.get("caching", None) is None and litellm.cache is not None) or kwargs.get("caching", False) is True ) and ( kwargs.get("cache", {}).get("no-cache", False) is not True ): # allow users to control returning cached responses from the completion function + args = args or () + final_embedding_cached_response: Optional[EmbeddingResponse] = None + embedding_all_elements_cache_hit: bool = False + cached_result: Optional[Any] = None + kwargs = kwargs.copy() + ######################################################### + # Init cache timing metrics + ######################################################### + cache_check_start_time = time.perf_counter() + cache_check_end_time: Optional[float] = None + ######################################################### + parent_otel_span = _get_parent_otel_span_from_kwargs(kwargs) + kwargs["parent_otel_span"] = parent_otel_span + if litellm.cache is not None and self._is_call_type_supported_by_cache( original_function=original_function ): @@ -168,7 +170,7 @@ class LLMCachingHandler: kwargs=kwargs, args=args, ) - cache_check_end_time = datetime.datetime.now() + cache_check_end_time = time.perf_counter() if cached_result is not None and not isinstance(cached_result, list): verbose_logger.debug("Cache Hit!") @@ -180,7 +182,7 @@ class LLMCachingHandler: api_base=kwargs.get("api_base", None), api_key=kwargs.get("api_key", None), ) - cache_duration_ms = (cache_check_end_time - cache_check_start_time).total_seconds() * 1000 + cache_duration_ms = (cache_check_end_time - cache_check_start_time) * 1000 self._update_litellm_logging_obj_environment( logging_obj=logging_obj, model=model, @@ -245,11 +247,14 @@ class LLMCachingHandler: final_embedding_cached_response=final_embedding_cached_response, embedding_all_elements_cache_hit=embedding_all_elements_cache_hit, ) - verbose_logger.debug(f"CACHE RESULT: {cached_result}") - return CachingHandlerResponse( - cached_result=cached_result, - final_embedding_cached_response=final_embedding_cached_response, - ) + + verbose_logger.debug(f"CACHE RESULT: {cached_result}") + return CachingHandlerResponse( + cached_result=cached_result, + final_embedding_cached_response=final_embedding_cached_response, + ) + # Caching disabled - return None to indicate no caching attempted + return None def _sync_get_cache( self, @@ -263,18 +268,22 @@ class LLMCachingHandler: ) -> CachingHandlerResponse: from litellm.utils import CustomStreamWrapper - args = args or () - new_kwargs = kwargs.copy() - new_kwargs.update( - convert_args_to_kwargs( - self.original_function, - args, - ) - ) + cached_result: Optional[Any] = None + + # Check if caching should be performed BEFORE doing expensive kwargs copy if litellm.cache is not None and self._is_call_type_supported_by_cache( original_function=original_function ): + args = args or () + # Now that we confirmed caching will happen, prepare kwargs + new_kwargs = kwargs.copy() + new_kwargs.update( + convert_args_to_kwargs( + self.original_function, + args, + ) + ) print_verbose("Checking Sync Cache") cached_result = litellm.cache.get_cache(**new_kwargs) if cached_result is not None: diff --git a/litellm/utils.py b/litellm/utils.py index f963e8c944..3c6c3ac86e 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -1402,7 +1402,7 @@ def client(original_function): # noqa: PLR0915 print_verbose( f"ASYNC kwargs[caching]: {kwargs.get('caching', False)}; litellm.cache: {litellm.cache}; kwargs.get('cache'): {kwargs.get('cache', None)}" ) - _caching_handler_response: CachingHandlerResponse = ( + _caching_handler_response: Optional[CachingHandlerResponse] = ( await _llm_caching_handler._async_get_cache( model=model or "", original_function=original_function, @@ -1414,14 +1414,15 @@ def client(original_function): # noqa: PLR0915 ) ) - if ( - _caching_handler_response.cached_result is not None - and _caching_handler_response.final_embedding_cached_response is None - ): - return _caching_handler_response.cached_result + if _caching_handler_response is not None: + if ( + _caching_handler_response.cached_result is not None + and _caching_handler_response.final_embedding_cached_response is None + ): + return _caching_handler_response.cached_result - elif _caching_handler_response.embedding_all_elements_cache_hit is True: - return _caching_handler_response.final_embedding_cached_response + elif _caching_handler_response.embedding_all_elements_cache_hit is True: + return _caching_handler_response.final_embedding_cached_response # CHECK MAX TOKENS if ( @@ -1524,6 +1525,7 @@ def client(original_function): # noqa: PLR0915 # REBUILD EMBEDDING CACHING if ( isinstance(result, EmbeddingResponse) + and _caching_handler_response is not None and _caching_handler_response.final_embedding_cached_response is not None ):