From f353bb6dbad5ae04adad67dcc572d477405fa9bb Mon Sep 17 00:00:00 2001 From: Alexsander Hamir Date: Thu, 18 Dec 2025 08:49:23 -0800 Subject: [PATCH] [Fix] CI/CD - litellm_mapped_tests_llms | litellm_mapped_tests_core | caching_unit_tests (#18197) --- litellm/caching/redis_cache.py | 50 ++++++++++++++++++- .../litellm_core_utils/get_model_cost_map.py | 26 +++++----- .../prompt_templates/factory.py | 23 ++++++++- litellm/llms/custom_httpx/http_handler.py | 2 +- .../llms/fireworks_ai/chat/transformation.py | 41 +++++++++++++-- litellm/llms/watsonx/chat/transformation.py | 28 +++++++++-- ...odel_prices_and_context_window_backup.json | 39 ++++++++------- .../management_endpoints/team_endpoints.py | 21 ++++++++ model_prices_and_context_window.json | 46 +++++++++-------- pyproject.toml | 2 +- requirements.txt | 3 +- .../test_anthropic_prompt_caching.py | 4 +- tests/local_testing/test_ollama.py | 2 +- .../interactions/test_openapi_compliance.py | 40 +++++++++------ .../llms/custom_httpx/test_http_handler.py | 2 +- 15 files changed, 243 insertions(+), 86 deletions(-) diff --git a/litellm/caching/redis_cache.py b/litellm/caching/redis_cache.py index 8d6a729638..a988a81b6f 100644 --- a/litellm/caching/redis_cache.py +++ b/litellm/caching/redis_cache.py @@ -10,6 +10,7 @@ Has 4 primary methods: import ast import asyncio +import hashlib import inspect import json import time @@ -159,6 +160,22 @@ class RedisCache(BaseCache): "Error connecting to Async Redis client - {}".format(str(e)), extra={"error": str(e)}, ) + # Non-blocking service health monitoring for async Redis init failures + try: + loop = asyncio.get_running_loop() + start_time = time.time() + end_time = start_time + loop.create_task( + self.service_logger_obj.async_service_failure_hook( + service=ServiceTypes.REDIS, + duration=end_time - start_time, + error=e, + call_type="redis_async_ping", + ) + ) + except Exception: + # If no event loop or hook fails, ignore to keep init non-blocking + pass ### SYNC HEALTH PING ### try: @@ -168,12 +185,40 @@ class RedisCache(BaseCache): verbose_logger.error( "Error connecting to Sync Redis client", extra={"error": str(e)} ) + # Non-blocking service health monitoring for sync Redis init failures + try: + loop = asyncio.get_running_loop() + start_time = time.time() + end_time = start_time + loop.create_task( + self.service_logger_obj.async_service_failure_hook( + service=ServiceTypes.REDIS, + duration=end_time - start_time, + error=e, + call_type="redis_sync_ping", + ) + ) + except Exception: + # If no event loop or hook fails, ignore to keep init non-blocking + pass if litellm.default_redis_ttl is not None: super().__init__(default_ttl=int(litellm.default_redis_ttl)) else: super().__init__() # defaults to 60s + def _get_async_client_cache_key(self) -> str: + """ + Generate a cache key for the async Redis client based on connection parameters. + This ensures different Redis configurations use different cached clients. + """ + # Create a stable representation of redis_kwargs for hashing + # Sort keys to ensure consistent hash regardless of parameter order + sorted_kwargs = sorted(self.redis_kwargs.items()) + kwargs_str = json.dumps(sorted_kwargs, sort_keys=True) + kwargs_hash = hashlib.sha256(kwargs_str.encode()).hexdigest()[:16] + return f"async-redis-client-{kwargs_hash}" + def init_async_client( self, ) -> Union[async_redis_client, async_redis_cluster_client]: @@ -181,7 +226,8 @@ class RedisCache(BaseCache): from .._redis import get_redis_async_client, get_redis_connection_pool - cached_client = in_memory_llm_clients_cache.get_cache(key="async-redis-client") + cache_key = self._get_async_client_cache_key() + cached_client = in_memory_llm_clients_cache.get_cache(key=cache_key) if cached_client is not None: redis_async_client = cast( Union[async_redis_client, async_redis_cluster_client], cached_client @@ -193,7 +239,7 @@ class RedisCache(BaseCache): connection_pool=self.async_redis_conn_pool, **self.redis_kwargs ) in_memory_llm_clients_cache.set_cache( - key="async-redis-client", value=redis_async_client + key=cache_key, value=redis_async_client ) self.redis_async_client = redis_async_client # type: ignore diff --git a/litellm/litellm_core_utils/get_model_cost_map.py b/litellm/litellm_core_utils/get_model_cost_map.py index b6a3a243c4..9b86f4ca2f 100644 --- a/litellm/litellm_core_utils/get_model_cost_map.py +++ b/litellm/litellm_core_utils/get_model_cost_map.py @@ -18,14 +18,15 @@ def get_model_cost_map(url: str) -> dict: os.getenv("LITELLM_LOCAL_MODEL_COST_MAP", False) or os.getenv("LITELLM_LOCAL_MODEL_COST_MAP", False) == "True" ): - import importlib.resources + from importlib.resources import files import json - with importlib.resources.open_text( - "litellm", "model_prices_and_context_window_backup.json" - ) as f: - content = json.load(f) - return content + content = json.loads( + files("litellm") + .joinpath("model_prices_and_context_window_backup.json") + .read_text(encoding="utf-8") + ) + return content try: response = httpx.get( @@ -35,11 +36,12 @@ def get_model_cost_map(url: str) -> dict: content = response.json() return content except Exception: - import importlib.resources + from importlib.resources import files import json - with importlib.resources.open_text( - "litellm", "model_prices_and_context_window_backup.json" - ) as f: - content = json.load(f) - return content + content = json.loads( + files("litellm") + .joinpath("model_prices_and_context_window_backup.json") + .read_text(encoding="utf-8") + ) + return content diff --git a/litellm/litellm_core_utils/prompt_templates/factory.py b/litellm/litellm_core_utils/prompt_templates/factory.py index 652692c7b8..6cc6c229f5 100644 --- a/litellm/litellm_core_utils/prompt_templates/factory.py +++ b/litellm/litellm_core_utils/prompt_templates/factory.py @@ -1572,6 +1572,21 @@ def convert_to_gemini_tool_call_result( return _part +def _sanitize_anthropic_tool_use_id(tool_use_id: str) -> str: + """ + Sanitize tool_use_id to match Anthropic's required pattern: ^[a-zA-Z0-9_-]+$ + + Anthropic requires tool_use_id to only contain alphanumeric characters, underscores, and hyphens. + This function replaces any invalid characters with underscores. + """ + # Replace any character that's not alphanumeric, underscore, or hyphen with underscore + sanitized = re.sub(r'[^a-zA-Z0-9_-]', '_', tool_use_id) + # Ensure it's not empty (fallback to a default if needed) + if not sanitized: + sanitized = "tool_use_id" + return sanitized + + def convert_to_anthropic_tool_result( message: Union[ChatCompletionToolMessage, ChatCompletionFunctionMessage], ) -> AnthropicMessagesToolResultParam: @@ -1639,18 +1654,22 @@ def convert_to_anthropic_tool_result( if message["role"] == "tool": tool_message: ChatCompletionToolMessage = message tool_call_id: str = tool_message["tool_call_id"] + # Sanitize tool_use_id to match Anthropic's pattern requirement: ^[a-zA-Z0-9_-]+$ + sanitized_tool_use_id = _sanitize_anthropic_tool_use_id(tool_call_id) # We can't determine from openai message format whether it's a successful or # error call result so default to the successful result template anthropic_tool_result = AnthropicMessagesToolResultParam( - type="tool_result", tool_use_id=tool_call_id, content=anthropic_content + type="tool_result", tool_use_id=sanitized_tool_use_id, content=anthropic_content ) if message["role"] == "function": function_message: ChatCompletionFunctionMessage = message tool_call_id = function_message.get("tool_call_id") or str(uuid.uuid4()) + # Sanitize tool_use_id to match Anthropic's pattern requirement: ^[a-zA-Z0-9_-]+$ + sanitized_tool_use_id = _sanitize_anthropic_tool_use_id(tool_call_id) anthropic_tool_result = AnthropicMessagesToolResultParam( - type="tool_result", tool_use_id=tool_call_id, content=anthropic_content + type="tool_result", tool_use_id=sanitized_tool_use_id, content=anthropic_content ) if anthropic_tool_result is None: diff --git a/litellm/llms/custom_httpx/http_handler.py b/litellm/llms/custom_httpx/http_handler.py index 1b4d04af80..7fdb78c167 100644 --- a/litellm/llms/custom_httpx/http_handler.py +++ b/litellm/llms/custom_httpx/http_handler.py @@ -769,7 +769,7 @@ class AsyncHTTPHandler: connector_kwargs["ssl"] = ssl_context elif ssl_verify is False: # Priority 2: Explicitly disable SSL verification - connector_kwargs["verify_ssl"] = False + connector_kwargs["ssl"] = False return connector_kwargs diff --git a/litellm/llms/fireworks_ai/chat/transformation.py b/litellm/llms/fireworks_ai/chat/transformation.py index 81b34f80ea..86bcd94450 100644 --- a/litellm/llms/fireworks_ai/chat/transformation.py +++ b/litellm/llms/fireworks_ai/chat/transformation.py @@ -240,12 +240,43 @@ class FireworksAIConfig(OpenAIGPTConfig): return messages def get_provider_info(self, model: str) -> ProviderSpecificModelInfo: - provider_specific_model_info = ProviderSpecificModelInfo( - supports_function_calling=True, - supports_prompt_caching=True, # https://docs.fireworks.ai/guides/prompt-caching - supports_pdf_input=True, # via document inlining - supports_vision=True, # via document inlining + # Models that support reasoning_effort + reasoning_supported_models = [ + "qwen3-8b", + "qwen3-32b", + "qwen3-coder-480b-a35b-instruct", + "deepseek-v3p1", + "deepseek-v3p2", + "glm-4p5", + "glm-4p5-air", + "glm-4p6", + "gpt-oss-120b", + "gpt-oss-20b", + ] + + # Normalize model name - remove prefix if present + normalized_model = model + if model.startswith("fireworks_ai/"): + normalized_model = model.replace("fireworks_ai/", "") + if normalized_model.startswith("accounts/fireworks/models/"): + normalized_model = normalized_model.replace("accounts/fireworks/models/", "") + + # Check if model supports reasoning + supports_reasoning_value = any( + reasoning_model in normalized_model for reasoning_model in reasoning_supported_models ) + + provider_specific_model_info: ProviderSpecificModelInfo = { + "supports_function_calling": True, + "supports_prompt_caching": True, # https://docs.fireworks.ai/guides/prompt-caching + "supports_pdf_input": True, # via document inlining + "supports_vision": True, # via document inlining + } + + # Only include supports_reasoning if True + if supports_reasoning_value: + provider_specific_model_info["supports_reasoning"] = True + return provider_specific_model_info def transform_request( diff --git a/litellm/llms/watsonx/chat/transformation.py b/litellm/llms/watsonx/chat/transformation.py index 917f7d89a2..0bb96673ef 100644 --- a/litellm/llms/watsonx/chat/transformation.py +++ b/litellm/llms/watsonx/chat/transformation.py @@ -6,6 +6,7 @@ Docs: https://cloud.ibm.com/apidocs/watsonx-ai#text-chat from typing import Dict, List, Optional, Tuple, Union +from litellm import verbose_logger from litellm.secret_managers.main import get_secret_str from litellm.types.llms.watsonx import ( WatsonXAIEndpoint, @@ -150,8 +151,13 @@ class IBMWatsonXChatConfig(IBMWatsonXMixin, OpenAIGPTConfig): else: hf_model = model try: - return hf_template_fn(model=hf_model, messages=messages) + result = hf_template_fn(model=hf_model, messages=messages) + # Return result if it's truthy (not None and not empty string) + # The caller will handle None/empty by falling back to default + if result: + return result except Exception: + # Silently fall through to return None - caller will handle fallback pass elif WatsonXModelPattern.LLAMA3_INSTRUCT.value in model: return custom_prompt( @@ -204,11 +210,23 @@ class IBMWatsonXChatConfig(IBMWatsonXMixin, OpenAIGPTConfig): try: # Use sync if cached, async if not if hf_model in litellm.known_tokenizer_config: - return hf_chat_template(model=hf_model, messages=messages) + result = hf_chat_template(model=hf_model, messages=messages) else: - return await ahf_chat_template(model=hf_model, messages=messages) - except Exception: - pass + result = await ahf_chat_template(model=hf_model, messages=messages) + # Return result if it's truthy (not None and not empty string) + # The caller (_aconvert_watsonx_messages_core) will handle None/empty by falling back to default + if result: + return result + except Exception as e: + # Log the exception for debugging but don't raise it + # The caller will fall back to default prompt factory + try: + verbose_logger.debug( + f"Failed to apply HuggingFace template for model {hf_model}: {e}" + ) + except Exception: + # If logging fails, silently continue - don't break the flow + pass elif WatsonXModelPattern.LLAMA3_INSTRUCT.value in model: return custom_prompt( role_dict={ diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 5ddb7cf862..fc72dc8a8e 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -15366,7 +15366,7 @@ "max_tokens": 16000, "mode": "chat", "supported_endpoints": [ - "/chat/completions" + "/v1/chat/completions" ], "supports_function_calling": true, "supports_parallel_function_calling": true, @@ -15379,7 +15379,7 @@ "max_tokens": 16000, "mode": "chat", "supported_endpoints": [ - "/chat/completions" + "/v1/chat/completions" ], "supports_function_calling": true, "supports_parallel_function_calling": true, @@ -15392,7 +15392,7 @@ "max_tokens": 16000, "mode": "chat", "supported_endpoints": [ - "/chat/completions" + "/v1/chat/completions" ], "supports_vision": true }, @@ -15403,7 +15403,7 @@ "max_tokens": 16000, "mode": "chat", "supported_endpoints": [ - "/chat/completions" + "/v1/chat/completions" ], "supports_function_calling": true, "supports_parallel_function_calling": true, @@ -15416,7 +15416,7 @@ "max_tokens": 16000, "mode": "chat", "supported_endpoints": [ - "/chat/completions" + "/v1/chat/completions" ], "supports_function_calling": true, "supports_parallel_function_calling": true, @@ -15573,8 +15573,8 @@ "max_tokens": 128000, "mode": "chat", "supported_endpoints": [ - "/chat/completions", - "/responses" + "/v1/chat/completions", + "/v1/responses" ], "supports_function_calling": true, "supports_parallel_function_calling": true, @@ -15599,8 +15599,8 @@ "max_tokens": 64000, "mode": "chat", "supported_endpoints": [ - "/chat/completions", - "/responses" + "/v1/chat/completions", + "/v1/responses" ], "supports_function_calling": true, "supports_parallel_function_calling": true, @@ -15614,7 +15614,7 @@ "max_tokens": 128000, "mode": "responses", "supported_endpoints": [ - "/responses" + "/v1/responses" ], "supports_function_calling": true, "supports_parallel_function_calling": true, @@ -15628,8 +15628,8 @@ "max_tokens": 64000, "mode": "chat", "supported_endpoints": [ - "/chat/completions", - "/responses" + "/v1/chat/completions", + "/v1/responses" ], "supports_function_calling": true, "supports_parallel_function_calling": true, @@ -16831,7 +16831,8 @@ "supported_endpoints": [ "/v1/images/generations" ], - "supports_vision": true + "supports_vision": true, + "supports_pdf_input": true }, "gpt-image-1.5-2025-12-16": { "cache_read_input_image_token_cost": 2e-06, @@ -16845,7 +16846,8 @@ "supported_endpoints": [ "/v1/images/generations" ], - "supports_vision": true + "supports_vision": true, + "supports_pdf_input": true }, "gpt-5": { "cache_read_input_token_cost": 1.25e-07, @@ -22386,7 +22388,7 @@ "input_cost_per_token": 0, "litellm_provider": "openrouter", "max_input_tokens": 262144, - "max_output_tokens": null, + "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 0, @@ -22428,7 +22430,7 @@ "input_cost_per_token": 1.5e-07, "litellm_provider": "openrouter", "max_input_tokens": 262144, - "max_output_tokens": null, + "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 1.5e-07, @@ -22456,7 +22458,7 @@ "input_cost_per_token": 5e-07, "litellm_provider": "openrouter", "max_input_tokens": 262144, - "max_output_tokens": null, + "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 1.5e-06, @@ -31070,7 +31072,8 @@ "input_cost_per_token": 9e-07, "output_cost_per_token": 9e-07, "litellm_provider": "fireworks_ai", - "mode": "chat" + "mode": "chat", + "supports_reasoning": true }, "fireworks_ai/accounts/fireworks/models/qwen3-4b": { "max_tokens": 40960, diff --git a/litellm/proxy/management_endpoints/team_endpoints.py b/litellm/proxy/management_endpoints/team_endpoints.py index 911f42e752..c6fab9a73f 100644 --- a/litellm/proxy/management_endpoints/team_endpoints.py +++ b/litellm/proxy/management_endpoints/team_endpoints.py @@ -2434,6 +2434,27 @@ def validate_membership( ): # allow team keys to check their info return + # Handle case where user_id is None (e.g., team key accessing different team) + if user_api_key_dict.user_id is None: + if user_api_key_dict.team_id is not None: + raise HTTPException( + status_code=403, + detail={ + "error": "Team key for team={} not authorized to access this team={}".format( + user_api_key_dict.team_id, team_table.team_id + ) + }, + ) + else: + raise HTTPException( + status_code=403, + detail={ + "error": "API key not authorized to access this team={}. No user_id or team_id associated with this key.".format( + team_table.team_id + ) + }, + ) + if user_api_key_dict.user_id not in [ m.user_id for m in team_table.members_with_roles ]: diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 78f1b37fba..7cb397a496 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -15371,7 +15371,7 @@ "max_tokens": 16000, "mode": "chat", "supported_endpoints": [ - "/chat/completions" + "/v1/chat/completions" ], "supports_function_calling": true, "supports_parallel_function_calling": true, @@ -15384,7 +15384,7 @@ "max_tokens": 16000, "mode": "chat", "supported_endpoints": [ - "/chat/completions" + "/v1/chat/completions" ], "supports_function_calling": true, "supports_parallel_function_calling": true, @@ -15397,7 +15397,7 @@ "max_tokens": 16000, "mode": "chat", "supported_endpoints": [ - "/chat/completions" + "/v1/chat/completions" ], "supports_vision": true }, @@ -15408,7 +15408,7 @@ "max_tokens": 16000, "mode": "chat", "supported_endpoints": [ - "/chat/completions" + "/v1/chat/completions" ], "supports_function_calling": true, "supports_parallel_function_calling": true, @@ -15421,7 +15421,7 @@ "max_tokens": 16000, "mode": "chat", "supported_endpoints": [ - "/chat/completions" + "/v1/chat/completions" ], "supports_function_calling": true, "supports_parallel_function_calling": true, @@ -15578,8 +15578,8 @@ "max_tokens": 128000, "mode": "chat", "supported_endpoints": [ - "/chat/completions", - "/responses" + "/v1/chat/completions", + "/v1/responses" ], "supports_function_calling": true, "supports_parallel_function_calling": true, @@ -15604,8 +15604,8 @@ "max_tokens": 64000, "mode": "chat", "supported_endpoints": [ - "/chat/completions", - "/responses" + "/v1/chat/completions", + "/v1/responses" ], "supports_function_calling": true, "supports_parallel_function_calling": true, @@ -15619,7 +15619,7 @@ "max_tokens": 128000, "mode": "responses", "supported_endpoints": [ - "/responses" + "/v1/responses" ], "supports_function_calling": true, "supports_parallel_function_calling": true, @@ -15633,8 +15633,8 @@ "max_tokens": 64000, "mode": "chat", "supported_endpoints": [ - "/chat/completions", - "/responses" + "/v1/chat/completions", + "/v1/responses" ], "supports_function_calling": true, "supports_parallel_function_calling": true, @@ -16836,7 +16836,8 @@ "supported_endpoints": [ "/v1/images/generations" ], - "supports_vision": true + "supports_vision": true, + "supports_pdf_input": true }, "gpt-image-1.5-2025-12-16": { "cache_read_input_image_token_cost": 2e-06, @@ -16850,7 +16851,8 @@ "supported_endpoints": [ "/v1/images/generations" ], - "supports_vision": true + "supports_vision": true, + "supports_pdf_input": true }, "gpt-5": { "cache_read_input_token_cost": 1.25e-07, @@ -22391,7 +22393,7 @@ "input_cost_per_token": 0, "litellm_provider": "openrouter", "max_input_tokens": 262144, - "max_output_tokens": null, + "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 0, @@ -22419,7 +22421,7 @@ "input_cost_per_token": 1e-07, "litellm_provider": "openrouter", "max_input_tokens": 131072, - "max_output_tokens": null, + "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", "output_cost_per_token": 1e-07, @@ -22433,7 +22435,7 @@ "input_cost_per_token": 1.5e-07, "litellm_provider": "openrouter", "max_input_tokens": 262144, - "max_output_tokens": null, + "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 1.5e-07, @@ -22447,7 +22449,7 @@ "input_cost_per_token": 2e-07, "litellm_provider": "openrouter", "max_input_tokens": 262144, - "max_output_tokens": null, + "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 2e-07, @@ -22461,7 +22463,7 @@ "input_cost_per_token": 5e-07, "litellm_provider": "openrouter", "max_input_tokens": 262144, - "max_output_tokens": null, + "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 1.5e-06, @@ -29374,7 +29376,8 @@ "input_cost_per_token": 4.5e-07, "output_cost_per_token": 1.8e-06, "litellm_provider": "fireworks_ai", - "mode": "chat" + "mode": "chat", + "supports_reasoning": true }, "fireworks_ai/accounts/fireworks/models/flux-kontext-pro": { "max_tokens": 4096, @@ -31076,7 +31079,8 @@ "input_cost_per_token": 9e-07, "output_cost_per_token": 9e-07, "litellm_provider": "fireworks_ai", - "mode": "chat" + "mode": "chat", + "supports_reasoning": true }, "fireworks_ai/accounts/fireworks/models/qwen3-4b": { "max_tokens": 40960, diff --git a/pyproject.toml b/pyproject.toml index 6020e9b6eb..9623e326db 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -31,7 +31,7 @@ click = "*" jinja2 = "^3.1.2" aiohttp = ">=3.10" pydantic = "^2.5.0" -jsonschema = "^4.22.0" +jsonschema = ">=4.23.0,<5.0.0" numpydoc = {version = "*", optional = true} # used in utils.py uvicorn = {version = "^0.31.1", optional = true} diff --git a/requirements.txt b/requirements.txt index c36f94f075..239eb707e7 100644 --- a/requirements.txt +++ b/requirements.txt @@ -60,9 +60,10 @@ aiohttp==3.12.14 # for network calls aioboto3==13.4.0 # for async sagemaker calls tenacity==8.5.0 # for retrying requests, when litellm.num_retries set pydantic>=2.11,<3 # proxy + openai req. + mcp -jsonschema==4.22.0 # validating json schema +jsonschema>=4.23.0,<5.0.0 # validating json schema - aligned with openapi-core + mcp websockets==13.1.0 # for realtime API soundfile==0.12.1 # for audio file processing +openapi-core==0.21.0 # for OpenAPI compliance tests ######################## # LITELLM ENTERPRISE DEPENDENCIES diff --git a/tests/local_testing/test_anthropic_prompt_caching.py b/tests/local_testing/test_anthropic_prompt_caching.py index 4d01a9269e..0926bd17b7 100644 --- a/tests/local_testing/test_anthropic_prompt_caching.py +++ b/tests/local_testing/test_anthropic_prompt_caching.py @@ -153,7 +153,7 @@ async def test_litellm_anthropic_prompt_caching_tools(): }, } ], - "max_tokens": 4096, + "max_tokens": 64000, "model": "claude-3-7-sonnet-20250219", } @@ -684,7 +684,7 @@ async def test_litellm_anthropic_prompt_caching_system(): ], } ], - "max_tokens": 4096, + "max_tokens": 64000, "model": "claude-3-7-sonnet-20250219", } diff --git a/tests/local_testing/test_ollama.py b/tests/local_testing/test_ollama.py index e7660ddc24..1269296e73 100644 --- a/tests/local_testing/test_ollama.py +++ b/tests/local_testing/test_ollama.py @@ -285,7 +285,7 @@ async def test_async_ollama_ssl_verify(stream): # create aiohttp transport with ssl_verify=False import aiohttp - aiohttp_session = aiohttp.ClientSession(connector=aiohttp.TCPConnector(verify_ssl=False)) + aiohttp_session = aiohttp.ClientSession(connector=aiohttp.TCPConnector(ssl=False)) print("aiohttp_session ssl=", aiohttp_session.connector._ssl) assert litellm_created_session.connector._ssl is False diff --git a/tests/test_litellm/interactions/test_openapi_compliance.py b/tests/test_litellm/interactions/test_openapi_compliance.py index d18d52f96b..5b490777f0 100644 --- a/tests/test_litellm/interactions/test_openapi_compliance.py +++ b/tests/test_litellm/interactions/test_openapi_compliance.py @@ -15,26 +15,38 @@ from unittest.mock import MagicMock, patch import httpx import pytest from openapi_core import OpenAPI -from openapi_core.testing.mock import MockRequest, MockResponse OPENAPI_SPEC_URL = "https://ai.google.dev/static/api/interactions.openapi.json" +def _load_openapi_spec_dict() -> Dict[str, Any]: + """ + Load the OpenAPI spec JSON. + + In CI or offline environments, network access may not be available. + In that case, gracefully skip these tests instead of erroring. + """ + try: + response = httpx.get(OPENAPI_SPEC_URL, timeout=5.0) + response.raise_for_status() + return response.json() + except Exception as e: # pragma: no cover - defensive, env-dependent + pytest.skip( + f"Skipping Google Interactions OpenAPI compliance tests - " + f"unable to load spec from {OPENAPI_SPEC_URL}: {e}" + ) + + @pytest.fixture(scope="module") -def openapi_spec(): - """Load the OpenAPI spec.""" - response = httpx.get(OPENAPI_SPEC_URL) - response.raise_for_status() - spec_dict = response.json() - return OpenAPI.from_dict(spec_dict) - - -@pytest.fixture(scope="module") -def spec_dict(): +def spec_dict() -> Dict[str, Any]: """Load raw spec dict for manual validation.""" - response = httpx.get(OPENAPI_SPEC_URL) - response.raise_for_status() - return response.json() + return _load_openapi_spec_dict() + + +@pytest.fixture(scope="module") +def openapi_spec(spec_dict: Dict[str, Any]) -> OpenAPI: + """Load the OpenAPI spec as an OpenAPI object.""" + return OpenAPI.from_dict(spec_dict) class TestRequestCompliance: diff --git a/tests/test_litellm/llms/custom_httpx/test_http_handler.py b/tests/test_litellm/llms/custom_httpx/test_http_handler.py index 1a728caee7..0b154474d4 100644 --- a/tests/test_litellm/llms/custom_httpx/test_http_handler.py +++ b/tests/test_litellm/llms/custom_httpx/test_http_handler.py @@ -128,7 +128,7 @@ async def test_ssl_verification_with_aiohttp_transport(): assert isinstance(transport_connector, TCPConnector) aiohttp_session = aiohttp.ClientSession( - connector=aiohttp.TCPConnector(verify_ssl=False) + connector=aiohttp.TCPConnector(ssl=False) ) aiohttp_connector = aiohttp_session.connector assert isinstance(aiohttp_connector, aiohttp.TCPConnector)