diff --git a/docs/my-website/docs/providers/crusoe.md b/docs/my-website/docs/providers/crusoe.md new file mode 100644 index 0000000000..aa737cbdcd --- /dev/null +++ b/docs/my-website/docs/providers/crusoe.md @@ -0,0 +1,196 @@ +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# Crusoe + +## Overview + +| Property | Details | +|-------|-------| +| Description | Crusoe Cloud provides GPU-accelerated inference for open-source large language models, optimized for performance and cost efficiency. | +| Provider Route on LiteLLM | `crusoe/` | +| Link to Provider Doc | [Crusoe Managed Inference Documentation ↗](https://docs.crusoecloud.com/managed-inference/overview/index.html) | +| Base URL | `https://managed-inference-api-proxy.crusoecloud.com/v1` | +| Supported Operations | [`/chat/completions`](#sample-usage) | + +
+
+ +**We support ALL Crusoe models, just set `crusoe/` as a prefix when sending completion requests** + +## Available Models + +| Model | Description | Context Window | +|-------|-------------|----------------| +| `crusoe/deepseek-ai/DeepSeek-R1-0528` | DeepSeek R1 reasoning model (May 2025) | 163,840 tokens | +| `crusoe/deepseek-ai/DeepSeek-V3-0324` | DeepSeek V3 chat model (March 2025) | 163,840 tokens | +| `crusoe/google/gemma-3-12b-it` | Google Gemma 3 12B instruction-tuned | 131,072 tokens | +| `crusoe/meta-llama/Llama-3.3-70B-Instruct` | Llama 3.3 70B instruction-tuned | 131,072 tokens | +| `crusoe/moonshotai/Kimi-K2-Thinking` | Kimi K2 extended thinking model | 262,144 tokens | +| `crusoe/openai/gpt-oss-120b` | OpenAI 120B open-source model | 131,072 tokens | +| `crusoe/Qwen/Qwen3-235B-A22B-Instruct-2507` | Qwen3 235B MoE instruction-tuned | 262,144 tokens | + +## Required Variables + +```python showLineNumbers title="Environment Variables" +os.environ["CRUSOE_API_KEY"] = "" # your Crusoe API key +``` + +## Usage - LiteLLM Python SDK + +### Non-streaming + +```python showLineNumbers title="Crusoe Non-streaming Completion" +import os +import litellm +from litellm import completion + +os.environ["CRUSOE_API_KEY"] = "" # your Crusoe API key + +messages = [{"content": "Hello, how are you?", "role": "user"}] + +# Crusoe call +response = completion( + model="crusoe/meta-llama/Llama-3.3-70B-Instruct", + messages=messages +) + +print(response) +``` + +### Streaming + +```python showLineNumbers title="Crusoe Streaming Completion" +import os +import litellm +from litellm import completion + +os.environ["CRUSOE_API_KEY"] = "" # your Crusoe API key + +messages = [{"content": "Write a short story about AI", "role": "user"}] + +# Crusoe call with streaming +response = completion( + model="crusoe/meta-llama/Llama-3.3-70B-Instruct", + messages=messages, + stream=True +) + +for chunk in response: + print(chunk) +``` + +### Function Calling + +```python showLineNumbers title="Crusoe Function Calling" +import os +import litellm +from litellm import completion + +os.environ["CRUSOE_API_KEY"] = "" # your Crusoe API key + +tools = [{ + "type": "function", + "function": { + "name": "get_weather", + "description": "Get the current weather in a location", + "parameters": { + "type": "object", + "properties": { + "location": { + "type": "string", + "description": "The city and state, e.g. San Francisco, CA" + } + }, + "required": ["location"] + } + } +}] + +messages = [{"role": "user", "content": "What's the weather in Boston?"}] + +response = completion( + model="crusoe/meta-llama/Llama-3.3-70B-Instruct", + messages=messages, + tools=tools, + tool_choice="auto" +) + +print(response) +``` + +## Usage - LiteLLM Proxy Server + +```yaml showLineNumbers title="config.yaml" +model_list: + - model_name: llama-3.3-70b + litellm_params: + model: crusoe/meta-llama/Llama-3.3-70B-Instruct + api_key: os.environ/CRUSOE_API_KEY + - model_name: deepseek-r1 + litellm_params: + model: crusoe/deepseek-ai/DeepSeek-R1-0528 + api_key: os.environ/CRUSOE_API_KEY + - model_name: deepseek-v3 + litellm_params: + model: crusoe/deepseek-ai/DeepSeek-V3-0324 + api_key: os.environ/CRUSOE_API_KEY + - model_name: qwen3-235b + litellm_params: + model: crusoe/Qwen/Qwen3-235B-A22B-Instruct-2507 + api_key: os.environ/CRUSOE_API_KEY + - model_name: kimi-k2 + litellm_params: + model: crusoe/moonshotai/Kimi-K2-Thinking + api_key: os.environ/CRUSOE_API_KEY +``` + +## Custom API Base + +**Option 1: Environment variable** + +```python showLineNumbers title="Custom API Base via env var" +import os +from litellm import completion + +os.environ["CRUSOE_API_BASE"] = "https://custom.crusoecloud.com/v1" +os.environ["CRUSOE_API_KEY"] = "" # your API key + +response = completion( + model="crusoe/meta-llama/Llama-3.3-70B-Instruct", + messages=[{"content": "Hello!", "role": "user"}], +) +``` + +**Option 2: Pass directly** + +```python showLineNumbers title="Custom API Base via parameter" +from litellm import completion + +response = completion( + model="crusoe/meta-llama/Llama-3.3-70B-Instruct", + messages=[{"content": "Hello!", "role": "user"}], + api_base="https://custom.crusoecloud.com/v1", + api_key="your-api-key", +) +``` + +## Supported OpenAI Parameters + +- `temperature` +- `max_tokens` +- `max_completion_tokens` +- `top_p` +- `frequency_penalty` +- `presence_penalty` +- `stop` +- `n` +- `stream` +- `tools` +- `tool_choice` +- `response_format` +- `seed` +- `user` +- `logit_bias` +- `logprobs` +- `top_logprobs` diff --git a/litellm/_logging.py b/litellm/_logging.py index d072cc549d..5ddafd6c6a 100644 --- a/litellm/_logging.py +++ b/litellm/_logging.py @@ -1,12 +1,12 @@ import ast import logging import os -import re import sys from datetime import datetime from logging import Formatter -from typing import Any, Dict, List, Optional +from typing import Any, Dict, Optional +from litellm.litellm_core_utils.secret_redaction import redact_string from litellm.litellm_core_utils.safe_json_dumps import safe_dumps from litellm.litellm_core_utils.safe_json_loads import safe_json_loads @@ -21,74 +21,11 @@ _ENABLE_SECRET_REDACTION = ( os.getenv("LITELLM_DISABLE_REDACT_SECRETS", "").lower() != "true" ) -_REDACTED = "REDACTED" - - -def _build_secret_patterns() -> re.Pattern: - patterns: List[str] = [ - # ── PEM private key / certificate blocks ── - r"-----BEGIN[A-Z \-]*PRIVATE KEY-----[\s\S]*?-----END[A-Z \-]*PRIVATE KEY-----", - # ── GCP OAuth2 access tokens (ya29.*) ── - r"\bya29\.[A-Za-z0-9_.~+/-]+", - # ── Credential %s formatting (space separator, no key= prefix) ── - r"(?:client_secret|azure_password|azure_username)\s+[^\s,'\"})\]{}>]+", - # AWS access key IDs - r"(?:AKIA|ASIA)[0-9A-Z]{16}", - # AWS secrets / session tokens / access key IDs (key=value) - r"(?:aws_secret_access_key|aws_session_token|aws_access_key_id)" - r"\s*[:=]\s*[A-Za-z0-9/+=]{20,}", - # Bearer tokens (OAuth, JWT, etc.) - r"Bearer\s+[A-Za-z0-9\-._~+/]{10,}=*", - # Basic auth headers - r"Basic\s+[A-Za-z0-9+/]{10,}={0,2}", - # OpenAI / Anthropic sk- prefixed keys - r"sk-[A-Za-z0-9\-_]{20,}", - # Generic api_key / api-key / apikey (handles 'key': 'value' dict repr) - r"(?:api[_-]?key)['\"]?\s*[:=]\s*['\"]?[^\s,'\"})\]{}>]{8,}", - # x-api-key / api-key header values (handles 'key': 'value' dict repr) - r"(?:x-api-key|api-key)['\"]?\s*[:=]\s*['\"]?[^\s,'\"})\]{}>]+", - # Anthropic internal header keys - r"x-ak-[A-Za-z0-9\-_]{20,}", - # Google API keys - r"AIza[0-9A-Za-z\-_]{35}", - # Password / secret params (handles key=value and 'key': 'value') - # Word boundary prevents O(n^2) backtracking on long word-char runs. - r"(?:^|(?<=\W))\w*(?:password|passwd|client_secret|secret_key|_secret)" - r"['\"]?\s*[:=]\s*['\"]?[^\s,'\"})\]{}>]+", - # Database connection string credentials (scheme://user:pass@host) - r"(?<=://)[^\s'\"]*:[^\s'\"@]+(?=@)", - # Databricks personal access tokens - r"dapi[0-9a-f]{32}", - # ── Key-name-based redaction ── - # Catches secrets inside dicts/config dumps by matching on the KEY name - # regardless of what the value looks like. - # e.g. 'master_key': 'any-value-here', "database_url": "postgres://..." - # private_key with PEM-aware value capture - r"""private_key['\"]?\s*[:=]\s*['\"]?(?:-----BEGIN[A-Z \-]*PRIVATE KEY-----[\s\S]*?-----END[A-Z \-]*PRIVATE KEY-----|[^\s,'\"})\]{}>]+)""", - r"(?:master_key|database_url|db_url|connection_string|" - r"signing_key|encryption_key|" - r"auth_token|access_token|refresh_token|" - r"slack_webhook_url|webhook_url|" - r"database_connection_string|" - r"huggingface_token|jwt_secret)" - r"""['\"]?\s*[:=]\s*['\"]?[^\s,'\"})\]{}>]+""", - # ── Raw JWTs (without Bearer prefix) ── - r"\beyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]*", - # ── Azure SAS tokens in URLs ── - r"[?&]sig=[A-Za-z0-9%+/=]+", - # ── Full JSON service-account blobs (single-line and multi-line) ── - r'\{[^{}]*"type"\s*:\s*"service_account"[^{}]*(?:\{[^{}]*\}[^{}]*)*\}', - ] - return re.compile("|".join(patterns), re.IGNORECASE) - - -_SECRET_RE = _build_secret_patterns() - def _redact_string(value: str) -> str: if not _ENABLE_SECRET_REDACTION: return value - return _SECRET_RE.sub(_REDACTED, value) + return redact_string(value) def redact_secrets(value: str) -> str: diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 8a68d74be5..9b4dd80265 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -513,7 +513,10 @@ def cost_per_token( # noqa: PLR0915 return fireworks_ai_cost_per_token(model=model, usage=usage_block) elif custom_llm_provider == "azure": return azure_openai_cost_per_token( - model=model, usage=usage_block, response_time_ms=response_time_ms + model=model, + usage=usage_block, + response_time_ms=response_time_ms, + service_tier=service_tier, ) elif custom_llm_provider == "gemini": return gemini_cost_per_token( @@ -539,6 +542,7 @@ def cost_per_token( # noqa: PLR0915 usage=usage_block, response_time_ms=response_time_ms, request_model=request_model, + service_tier=service_tier, ) else: model_info = _cached_get_model_info_helper( diff --git a/litellm/litellm_core_utils/exception_mapping_utils.py b/litellm/litellm_core_utils/exception_mapping_utils.py index 5a7d4e33b6..2c1d92920a 100644 --- a/litellm/litellm_core_utils/exception_mapping_utils.py +++ b/litellm/litellm_core_utils/exception_mapping_utils.py @@ -6,7 +6,8 @@ from typing import Any, Optional import httpx import litellm -from litellm._logging import _redact_string, verbose_logger +from litellm._logging import _ENABLE_SECRET_REDACTION, _redact_string, verbose_logger +from litellm.litellm_core_utils.secret_redaction import redact_string from litellm.types.utils import LlmProviders from ..exceptions import ( @@ -261,10 +262,18 @@ def exception_type( # type: ignore # noqa: PLR0915 original_exception=original_exception ) try: - error_str = str(original_exception) + error_str = ( + redact_string(str(original_exception)) + if _ENABLE_SECRET_REDACTION + else str(original_exception) + ) if model: if hasattr(original_exception, "message"): - error_str = str(original_exception.message) + error_str = ( + redact_string(str(original_exception.message)) + if _ENABLE_SECRET_REDACTION + else str(original_exception.message) + ) if isinstance(original_exception, BaseException): exception_type = type(original_exception).__name__ else: @@ -2431,7 +2440,8 @@ def exception_type( # type: ignore # noqa: PLR0915 else: raise APIConnectionError( message="{}\n{}".format( - str(original_exception), _redact_string(traceback.format_exc()) + str(original_exception), + _redact_string(traceback.format_exc()), ), llm_provider=custom_llm_provider, model=model, @@ -2461,7 +2471,8 @@ def exception_type( # type: ignore # noqa: PLR0915 raise e # it's already mapped raised_exc = APIConnectionError( message="{}\n{}".format( - original_exception, _redact_string(traceback.format_exc()) + original_exception, + _redact_string(traceback.format_exc()), ), llm_provider="", model="", diff --git a/litellm/litellm_core_utils/secret_redaction.py b/litellm/litellm_core_utils/secret_redaction.py new file mode 100644 index 0000000000..5c4e3e3dac --- /dev/null +++ b/litellm/litellm_core_utils/secret_redaction.py @@ -0,0 +1,81 @@ +""" +Credential/secret redaction utilities. + +This module owns the compiled regex and the public `redact_string` helper so +that any part of the codebase (logging, exception mapping, etc.) can scrub +secrets from strings without depending on the logging-configuration module. +""" + +import re +from typing import List + +_REDACTED = "REDACTED" + + +def _build_secret_patterns() -> "re.Pattern[str]": + patterns: List[str] = [ + # PEM private key / certificate blocks + r"-----BEGIN[A-Z \-]*PRIVATE KEY-----[\s\S]*?-----END[A-Z \-]*PRIVATE KEY-----", + # GCP OAuth2 access tokens (ya29.*) + r"\bya29\.[A-Za-z0-9_.~+/-]+", + # Credential %s formatting (space separator, no key= prefix) + r"(?:client_secret|azure_password|azure_username)\s+[^\s,'\"})\]{}>]+", + # AWS access key IDs + r"(?:AKIA|ASIA)[0-9A-Z]{16}", + # AWS secrets / session tokens / access key IDs (key=value) + r"(?:aws_secret_access_key|aws_session_token|aws_access_key_id)" + r"\s*[:=]\s*[A-Za-z0-9/+=]{20,}", + # Bearer tokens (OAuth, JWT, etc.) + r"Bearer\s+[A-Za-z0-9\-._~+/]{10,}=*", + # Basic auth headers + r"Basic\s+[A-Za-z0-9+/]{10,}={0,2}", + # OpenAI / Anthropic sk- prefixed keys + r"sk-[A-Za-z0-9\-_]{20,}", + # Generic api_key / api-key / apikey (handles 'key': 'value' dict repr) + r"(?:api[_-]?key)['\"]?\s*[:=]\s*['\"]?[^\s,'\"})\]{}>]{8,}", + # x-api-key / api-key header values (handles 'key': 'value' dict repr) + r"(?:x-api-key|api-key)['\"]?\s*[:=]\s*['\"]?[^\s,'\"})\]{}>]+", + # Anthropic internal header keys + r"x-ak-[A-Za-z0-9\-_]{20,}", + # Google API keys (bare key value) + r"AIza[0-9A-Za-z\-_]{35}", + # URL query-param key=VALUE (e.g. ?key=AIza... or &key=...) — catches the + # full "key=" fragment so the value is redacted regardless of format. + r"(?<=[?&])key=[^\s&'\"]{8,}", + # Password / secret params (handles key=value and 'key': 'value') + # Word boundary prevents O(n^2) backtracking on long word-char runs. + r"(?:^|(?<=\W))\w*(?:password|passwd|client_secret|secret_key|_secret)" + r"['\"]?\s*[:=]\s*['\"]?[^\s,'\"})\]{}>]+", + # Database connection string credentials (scheme://user:pass@host) + r"(?<=://)[^\s'\"]*:[^\s'\"@]+(?=@)", + # Databricks personal access tokens + r"dapi[0-9a-f]{32}", + # ── Key-name-based redaction ── + # Catches secrets inside dicts/config dumps by matching on the KEY name + # regardless of what the value looks like. + # e.g. 'master_key': 'any-value-here', "database_url": "postgres://..." + # private_key with PEM-aware value capture + r"""private_key['\"]?\s*[:=]\s*['\"]?(?:-----BEGIN[A-Z \-]*PRIVATE KEY-----[\s\S]*?-----END[A-Z \-]*PRIVATE KEY-----|[^\s,'\"})\]{}>]+)""", + r"(?:master_key|database_url|db_url|connection_string|" + r"signing_key|encryption_key|" + r"auth_token|access_token|refresh_token|" + r"slack_webhook_url|webhook_url|" + r"database_connection_string|" + r"huggingface_token|jwt_secret)" + r"""['\"]?\s*[:=]\s*['\"]?[^\s,'\"})\]{}>]+""", + # Raw JWTs (without Bearer prefix) + r"\beyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]*", + # Azure SAS tokens in URLs + r"[?&]sig=[A-Za-z0-9%+/=]+", + # Full JSON service-account blobs (single-line and multi-line) + r'\{[^{}]*"type"\s*:\s*"service_account"[^{}]*(?:\{[^{}]*\}[^{}]*)*\}', + ] + return re.compile("|".join(patterns), re.IGNORECASE) + + +_SECRET_RE = _build_secret_patterns() + + +def redact_string(value: str) -> str: + """Scrub known secret/credential patterns from *value* and return the result.""" + return _SECRET_RE.sub(_REDACTED, value) diff --git a/litellm/llms/azure/azure.py b/litellm/llms/azure/azure.py index 877a7d3c84..c0e070b6c1 100644 --- a/litellm/llms/azure/azure.py +++ b/litellm/llms/azure/azure.py @@ -793,6 +793,7 @@ class AzureChatCompletion(BaseAzureLLM, BaseLLM): client=client, litellm_params=litellm_params, api_base=api_base, + api_version=api_version, ) azure_client = self.get_azure_openai_client( api_version=api_version, diff --git a/litellm/llms/azure/cost_calculation.py b/litellm/llms/azure/cost_calculation.py index 5b411095ea..2a20c55a6c 100644 --- a/litellm/llms/azure/cost_calculation.py +++ b/litellm/llms/azure/cost_calculation.py @@ -12,7 +12,10 @@ from litellm.utils import get_model_info def cost_per_token( - model: str, usage: Usage, response_time_ms: Optional[float] = 0.0 + model: str, + usage: Usage, + response_time_ms: Optional[float] = 0.0, + service_tier: Optional[str] = None, ) -> Tuple[float, float]: """ Calculates the cost per token for a given model, prompt tokens, and completion tokens. @@ -47,4 +50,5 @@ def cost_per_token( model=model, usage=usage, custom_llm_provider="azure", + service_tier=service_tier, ) diff --git a/litellm/llms/azure_ai/cost_calculator.py b/litellm/llms/azure_ai/cost_calculator.py index 067181b946..755d44fdef 100644 --- a/litellm/llms/azure_ai/cost_calculator.py +++ b/litellm/llms/azure_ai/cost_calculator.py @@ -65,6 +65,7 @@ def cost_per_token( usage: Usage, response_time_ms: Optional[float] = 0.0, request_model: Optional[str] = None, + service_tier: Optional[str] = None, ) -> Tuple[float, float]: """ Calculate the cost per token for Azure AI models. @@ -102,6 +103,7 @@ def cost_per_token( model=model, usage=usage, custom_llm_provider="azure_ai", + service_tier=service_tier, ) except Exception as e: # For Model Router, the model name (e.g., "azure-model-router") may not be in the cost map diff --git a/litellm/llms/openai_like/providers.json b/litellm/llms/openai_like/providers.json index 5dd1247001..b5e5aa4ea2 100644 --- a/litellm/llms/openai_like/providers.json +++ b/litellm/llms/openai_like/providers.json @@ -106,5 +106,13 @@ "base_url": "https://aihubmix.com/v1", "api_key_env": "AIHUBMIX_API_KEY", "api_base_env": "AIHUBMIX_API_BASE" + }, + "crusoe": { + "base_url": "https://managed-inference-api-proxy.crusoecloud.com/v1", + "api_key_env": "CRUSOE_API_KEY", + "api_base_env": "CRUSOE_API_BASE", + "param_mappings": { + "max_completion_tokens": "max_tokens" + } } } diff --git a/litellm/llms/vertex_ai/gemini/transformation.py b/litellm/llms/vertex_ai/gemini/transformation.py index 7fa220a59f..9afa5dec46 100644 --- a/litellm/llms/vertex_ai/gemini/transformation.py +++ b/litellm/llms/vertex_ai/gemini/transformation.py @@ -212,6 +212,22 @@ def _process_gemini_media( return _apply_gemini_metadata( part, model, media_resolution_enum, video_metadata ) + elif image_url.startswith( + "https://generativelanguage.googleapis.com/v1beta/files/" + ): + # Gemini Files API URIs — the file is already uploaded to Google's + # servers; pass the URI through as file_data without fetching it. + # These URLs return 403 when accessed directly, so we must not try + # to resolve their MIME type via HTTP. + if format: + file_data = FileDataType(mime_type=format, file_uri=image_url) + else: + # Gemini Files API references can be passed through as URI-only. + file_data = cast(FileDataType, {"file_uri": image_url}) + part = {"file_data": file_data} + return _apply_gemini_metadata( + part, model, media_resolution_enum, video_metadata + ) elif ( "https://" in image_url and (image_type := format or _get_image_mime_type_from_url(image_url)) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index a1e3e42a9c..6078d7e690 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -22061,6 +22061,98 @@ "tool_use_system_prompt_tokens": 346, "supports_native_structured_output": true }, + "crusoe/deepseek-ai/DeepSeek-R1-0528": { + "input_cost_per_token": 3e-06, + "litellm_provider": "crusoe", + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "max_tokens": 163840, + "mode": "chat", + "output_cost_per_token": 7e-06, + "supports_function_calling": false, + "supports_reasoning": true, + "supports_system_messages": true, + "supports_tool_choice": false + }, + "crusoe/deepseek-ai/DeepSeek-V3-0324": { + "input_cost_per_token": 1.5e-06, + "litellm_provider": "crusoe", + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "max_tokens": 163840, + "mode": "chat", + "output_cost_per_token": 1.5e-06, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "crusoe/google/gemma-3-12b-it": { + "input_cost_per_token": 1e-07, + "litellm_provider": "crusoe", + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 1e-07, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "crusoe/meta-llama/Llama-3.3-70B-Instruct": { + "input_cost_per_token": 2e-07, + "litellm_provider": "crusoe", + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 2e-07, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "crusoe/moonshotai/Kimi-K2-Thinking": { + "input_cost_per_token": 2.5e-06, + "litellm_provider": "crusoe", + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 2.5e-06, + "supports_function_calling": false, + "supports_reasoning": true, + "supports_system_messages": true, + "supports_tool_choice": false + }, + "crusoe/openai/gpt-oss-120b": { + "input_cost_per_token": 8e-07, + "litellm_provider": "crusoe", + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 8e-07, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "crusoe/Qwen/Qwen3-235B-A22B-Instruct-2507": { + "input_cost_per_token": 3e-06, + "litellm_provider": "crusoe", + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 3e-06, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, "lambda_ai/deepseek-llama3.3-70b": { "input_cost_per_token": 2e-07, "litellm_provider": "lambda_ai", diff --git a/litellm/proxy/guardrails/guardrail_hooks/qohash/__init__.py b/litellm/proxy/guardrails/guardrail_hooks/qohash/__init__.py new file mode 100644 index 0000000000..465f52db3d --- /dev/null +++ b/litellm/proxy/guardrails/guardrail_hooks/qohash/__init__.py @@ -0,0 +1,35 @@ +from typing import TYPE_CHECKING + +from litellm.types.guardrails import SupportedGuardrailIntegrations + +from .qohash import QostodianNexus + +if TYPE_CHECKING: + from litellm.types.guardrails import Guardrail, LitellmParams + + +def initialize_guardrail(litellm_params: "LitellmParams", guardrail: "Guardrail"): + import litellm + + _instance = QostodianNexus( + api_base=litellm_params.api_base, + guardrail_name=guardrail.get("guardrail_name", ""), + event_hook=litellm_params.mode, + default_on=litellm_params.default_on, + additional_provider_specific_params=litellm_params.additional_provider_specific_params, + extra_headers=getattr(litellm_params, "extra_headers", None), + ) + + litellm.logging_callback_manager.add_litellm_callback(_instance) + + return _instance + + +guardrail_initializer_registry = { + SupportedGuardrailIntegrations.QOSTODIAN_NEXUS.value: initialize_guardrail, +} + + +guardrail_class_registry = { + SupportedGuardrailIntegrations.QOSTODIAN_NEXUS.value: QostodianNexus, +} diff --git a/litellm/proxy/guardrails/guardrail_hooks/qohash/qohash.py b/litellm/proxy/guardrails/guardrail_hooks/qohash/qohash.py new file mode 100644 index 0000000000..a1bab6dbac --- /dev/null +++ b/litellm/proxy/guardrails/guardrail_hooks/qohash/qohash.py @@ -0,0 +1,81 @@ +""" +Qostodian Nexus (by Qohash) — LiteLLM guardrail integration. +""" + +import os +from typing import TYPE_CHECKING, Literal, Optional, Type + +from litellm.integrations.custom_guardrail import log_guardrail_information +from litellm.proxy.guardrails.guardrail_hooks.generic_guardrail_api.generic_guardrail_api import ( + GenericGuardrailAPI, +) +from litellm.types.proxy.guardrails.guardrail_hooks.qohash import ( + QostodianNexusConfigModel, +) +from litellm.types.utils import GenericGuardrailAPIInputs + +if TYPE_CHECKING: + from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj + +GUARDRAIL_NAME = "qostodian_nexus" + + +class QostodianNexus(GenericGuardrailAPI): + def __init__( + self, + api_base: Optional[str] = None, + **kwargs, + ): + api_base = api_base or os.environ.get( + "QOSTODIAN_NEXUS_API_BASE", "http://nexus:8800" + ) + + kwargs["guardrail_name"] = kwargs.get("guardrail_name", GUARDRAIL_NAME) + + # Merge built-in Qostodian Nexus identifier headers with any caller-supplied extra_headers + nexus_headers = [ + "x-qostodian-nexus-identifiers-trace", + "x-qostodian-nexus-identifiers-source", + "x-qostodian-nexus-identifiers-container", + "x-qostodian-nexus-identifiers-identity", + ] + + existing = kwargs.get("extra_headers") or [] + kwargs["extra_headers"] = nexus_headers + [ + h for h in existing if h not in nexus_headers + ] + + super().__init__( + api_base=api_base, + **kwargs, + ) + + @log_guardrail_information + async def apply_guardrail( + self, + inputs: GenericGuardrailAPIInputs, + request_data: dict, + input_type: Literal["request", "response"], + logging_obj: Optional["LiteLLMLoggingObj"] = None, + ) -> GenericGuardrailAPIInputs: + """ + Apply Qostodian Nexus to the given inputs. + + NOTE: This override is intentionally a pass-through. It must be present + directly in this class's __dict__ so that LiteLLM's unified guardrail + routing check (`"apply_guardrail" in type(callback).__dict__` in + litellm/proxy/utils.py) routes calls correctly. Do not remove. + """ + return await super().apply_guardrail( + inputs=inputs, + request_data=request_data, + input_type=input_type, + logging_obj=logging_obj, + ) + + @classmethod + def get_config_model(cls) -> Optional[Type[QostodianNexusConfigModel]]: + """ + Returns the config model for Qostodian Nexus. + """ + return QostodianNexusConfigModel diff --git a/litellm/proxy/utils.py b/litellm/proxy/utils.py index 8c5fce8409..69cd7b983e 100644 --- a/litellm/proxy/utils.py +++ b/litellm/proxy/utils.py @@ -3188,6 +3188,8 @@ class PrismaClient: t.organization_id as org_id, p.project_alias AS project_alias, tm.spend AS team_member_spend, + b_tm.tpm_limit AS team_member_tpm_limit, + b_tm.rpm_limit AS team_member_rpm_limit, m.aliases AS team_model_aliases, -- Added comma to separate b.* columns b.max_budget AS litellm_budget_table_max_budget, @@ -3203,6 +3205,7 @@ class PrismaClient: FROM "LiteLLM_VerificationToken" AS v LEFT JOIN "LiteLLM_TeamTable" AS t ON v.team_id = t.team_id LEFT JOIN "LiteLLM_TeamMembership" AS tm ON v.team_id = tm.team_id AND tm.user_id = v.user_id + LEFT JOIN "LiteLLM_BudgetTable" AS b_tm ON tm.budget_id = b_tm.budget_id LEFT JOIN "LiteLLM_ModelTable" m ON t.model_id = m.id LEFT JOIN "LiteLLM_BudgetTable" AS b ON v.budget_id = b.budget_id LEFT JOIN "LiteLLM_ProjectTable" AS p ON v.project_id = p.project_id diff --git a/litellm/types/guardrails.py b/litellm/types/guardrails.py index a98f9d666a..04347aebe3 100644 --- a/litellm/types/guardrails.py +++ b/litellm/types/guardrails.py @@ -38,6 +38,9 @@ from litellm.types.proxy.guardrails.guardrail_hooks.tool_permission import ( from litellm.types.proxy.guardrails.guardrail_hooks.hiddenlayer import ( HiddenlayerGuardrailConfigModel, ) +from litellm.types.proxy.guardrails.guardrail_hooks.qohash import ( + QostodianNexusConfigModel, +) """ Pydantic object defining how to set guardrails on litellm proxy @@ -96,6 +99,7 @@ class SupportedGuardrailIntegrations(Enum): AKTO = "akto" MCP_JWT_SIGNER = "mcp_jwt_signer" LLM_AS_A_JUDGE = "llm_as_a_judge" + QOSTODIAN_NEXUS = "qostodian_nexus" class Role(Enum): @@ -773,6 +777,7 @@ class LitellmParams( QualifireGuardrailConfigModel, BlockCodeExecutionGuardrailConfigModel, HiddenlayerGuardrailConfigModel, + QostodianNexusConfigModel, ): guardrail: str = Field(description="The type of guardrail integration to use") mode: Union[str, List[str], Mode] = Field( diff --git a/litellm/types/proxy/guardrails/guardrail_hooks/qohash.py b/litellm/types/proxy/guardrails/guardrail_hooks/qohash.py new file mode 100644 index 0000000000..5abfa69148 --- /dev/null +++ b/litellm/types/proxy/guardrails/guardrail_hooks/qohash.py @@ -0,0 +1,16 @@ +from typing import Optional + +from pydantic import Field + +from .base import GuardrailConfigModel + + +class QostodianNexusConfigModel(GuardrailConfigModel): + api_base: Optional[str] = Field( + default=None, + description="The API base URL for Qostodian Nexus. If not provided, the `QOSTODIAN_NEXUS_API_BASE` environment variable is checked. Defaults to http://nexus:8800.", + ) + + @staticmethod + def ui_friendly_name() -> str: + return "Qostodian Nexus" diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 8391fdb48f..c38c14a1f7 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -22109,6 +22109,98 @@ "tool_use_system_prompt_tokens": 346, "supports_native_structured_output": true }, + "crusoe/deepseek-ai/DeepSeek-R1-0528": { + "input_cost_per_token": 3e-06, + "litellm_provider": "crusoe", + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "max_tokens": 163840, + "mode": "chat", + "output_cost_per_token": 7e-06, + "supports_function_calling": false, + "supports_reasoning": true, + "supports_system_messages": true, + "supports_tool_choice": false + }, + "crusoe/deepseek-ai/DeepSeek-V3-0324": { + "input_cost_per_token": 1.5e-06, + "litellm_provider": "crusoe", + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "max_tokens": 163840, + "mode": "chat", + "output_cost_per_token": 1.5e-06, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "crusoe/google/gemma-3-12b-it": { + "input_cost_per_token": 1e-07, + "litellm_provider": "crusoe", + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 1e-07, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "crusoe/meta-llama/Llama-3.3-70B-Instruct": { + "input_cost_per_token": 2e-07, + "litellm_provider": "crusoe", + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 2e-07, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "crusoe/moonshotai/Kimi-K2-Thinking": { + "input_cost_per_token": 2.5e-06, + "litellm_provider": "crusoe", + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 2.5e-06, + "supports_function_calling": false, + "supports_reasoning": true, + "supports_system_messages": true, + "supports_tool_choice": false + }, + "crusoe/openai/gpt-oss-120b": { + "input_cost_per_token": 8e-07, + "litellm_provider": "crusoe", + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 8e-07, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "crusoe/Qwen/Qwen3-235B-A22B-Instruct-2507": { + "input_cost_per_token": 3e-06, + "litellm_provider": "crusoe", + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 3e-06, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, "lambda_ai/deepseek-llama3.3-70b": { "input_cost_per_token": 2e-07, "litellm_provider": "lambda_ai", diff --git a/provider_endpoints_support.json b/provider_endpoints_support.json index ed49c14621..3fc7cd4318 100644 --- a/provider_endpoints_support.json +++ b/provider_endpoints_support.json @@ -635,6 +635,24 @@ "interactions": true } }, + "crusoe": { + "display_name": "Crusoe (`crusoe`)", + "url": "https://docs.litellm.ai/docs/providers/crusoe", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": true, + "interactions": true + } + }, "custom": { "display_name": "Custom (`custom`)", "url": "https://docs.litellm.ai/docs/providers/custom_llm_server", diff --git a/tests/litellm/llms/azure/__init__.py b/tests/litellm/llms/azure/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/tests/litellm/llms/azure/test_azure_embedding.py b/tests/litellm/llms/azure/test_azure_embedding.py new file mode 100644 index 0000000000..22ee503ef0 --- /dev/null +++ b/tests/litellm/llms/azure/test_azure_embedding.py @@ -0,0 +1,94 @@ +import os +import sys +from unittest.mock import MagicMock, patch + +import pytest + +sys.path.insert( + 0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../..")) +) + +from litellm.llms.azure.azure import AzureChatCompletion +from litellm.types.utils import EmbeddingResponse, Usage + + +def _make_embedding_response() -> EmbeddingResponse: + return EmbeddingResponse( + model="text-embedding-3-large", + usage=Usage(prompt_tokens=3, completion_tokens=0, total_tokens=3), + data=[{"embedding": [0.1, 0.2, 0.3], "index": 0, "object": "embedding"}], + ) + + +def _make_logging_obj() -> MagicMock: + return MagicMock() + + +class TestAzureV1AsyncEmbedding: + def test_aembedding_receives_api_version(self): + """Regression: api_version must be forwarded to aembedding() when aembedding=True. + Without the fix, it was silently dropped, causing AsyncAzureOpenAI to be used + instead of AsyncOpenAI for Azure AI Foundry (v1) endpoints. Fixes #24848.""" + handler = AzureChatCompletion() + + with patch.object(handler, "aembedding") as mock_aembedding: + handler.embedding( + model="text-embedding-3-large", + input=["hello world"], + api_base="https://my-endpoint.openai.azure.com", + api_version="v1", + timeout=60.0, + logging_obj=_make_logging_obj(), + model_response=_make_embedding_response(), + optional_params={}, + api_key="fake-key", + aembedding=True, + litellm_params={}, + ) + + mock_aembedding.assert_called_once() + _, kwargs = mock_aembedding.call_args + assert kwargs.get("api_version") == "v1" + + def test_get_azure_openai_client_returns_async_openai_for_v1(self): + from openai import AsyncAzureOpenAI, AsyncOpenAI + + handler = AzureChatCompletion() + client = handler.get_azure_openai_client( + api_key="fake-key", + api_base="https://my-endpoint.openai.azure.com", + api_version="v1", + _is_async=True, + litellm_params={}, + ) + + assert isinstance(client, AsyncOpenAI) + assert not isinstance(client, AsyncAzureOpenAI) + + def test_get_azure_openai_client_uses_v1_base_url(self): + handler = AzureChatCompletion() + client = handler.get_azure_openai_client( + api_key="fake-key", + api_base="https://my-endpoint.openai.azure.com", + api_version="v1", + _is_async=True, + litellm_params={}, + ) + + assert client is not None + assert "/openai/v1/" in str(client.base_url) + + @pytest.mark.parametrize("api_version", ["v1", "latest", "preview"]) + def test_all_v1_variants_use_openai_client(self, api_version: str): + from openai import AsyncOpenAI + + handler = AzureChatCompletion() + client = handler.get_azure_openai_client( + api_key="fake-key", + api_base="https://my-endpoint.openai.azure.com", + api_version=api_version, + _is_async=True, + litellm_params={}, + ) + + assert isinstance(client, AsyncOpenAI) diff --git a/tests/llm_translation/test_crusoe.py b/tests/llm_translation/test_crusoe.py new file mode 100644 index 0000000000..56aa4e4cd4 --- /dev/null +++ b/tests/llm_translation/test_crusoe.py @@ -0,0 +1,108 @@ +""" +Tests for Crusoe provider integration +""" +import os +from unittest import mock + +import litellm + +CRUSOE_API_BASE = "https://managed-inference-api-proxy.crusoecloud.com/v1" + + +def test_crusoe_json_registry(): + """Test CrusoeChatConfig is loaded from JSON provider registry""" + from litellm.llms.openai_like.json_loader import JSONProviderRegistry + + assert JSONProviderRegistry.exists("crusoe") + config = JSONProviderRegistry.get("crusoe") + assert config is not None + assert config.base_url == CRUSOE_API_BASE + assert config.api_key_env == "CRUSOE_API_KEY" + assert config.api_base_env == "CRUSOE_API_BASE" + + +def test_crusoe_get_openai_compatible_provider_info(): + """Test Crusoe provider info retrieval""" + from litellm.llms.openai_like.dynamic_config import create_config_class + from litellm.llms.openai_like.json_loader import JSONProviderRegistry + + config = create_config_class(JSONProviderRegistry.get("crusoe"))() + + # Test with default values (no env vars set) + with mock.patch.dict(os.environ, {}, clear=True): + api_base, api_key = config._get_openai_compatible_provider_info(None, None) + assert api_base == CRUSOE_API_BASE + assert api_key is None + + # Test with environment variables + with mock.patch.dict( + os.environ, + { + "CRUSOE_API_KEY": "test-key", + "CRUSOE_API_BASE": "https://custom.crusoecloud.com/v1", + }, + ): + api_base, api_key = config._get_openai_compatible_provider_info(None, None) + assert api_base == "https://custom.crusoecloud.com/v1" + assert api_key == "test-key" + + # Test with explicit parameters (should override env vars) + with mock.patch.dict( + os.environ, + { + "CRUSOE_API_KEY": "env-key", + "CRUSOE_API_BASE": "https://env.crusoecloud.com/v1", + }, + ): + api_base, api_key = config._get_openai_compatible_provider_info( + "https://param.crusoecloud.com/v1", "param-key" + ) + assert api_base == "https://param.crusoecloud.com/v1" + assert api_key == "param-key" + + +def test_get_llm_provider_crusoe(): + """Test that get_llm_provider correctly identifies Crusoe""" + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + # Test with crusoe/model-name format + model, provider, api_key, api_base = get_llm_provider( + "crusoe/meta-llama/Llama-3.3-70B-Instruct" + ) + assert model == "meta-llama/Llama-3.3-70B-Instruct" + assert provider == "crusoe" + + +def test_crusoe_models_configuration(): + """Test that Crusoe models are configured correctly""" + from litellm import get_model_info + + original_model_cost = litellm.model_cost + original_env = os.environ.get("LITELLM_LOCAL_MODEL_COST_MAP") + try: + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" + litellm.model_cost = litellm.get_model_cost_map(url="") + + crusoe_models = [ + "crusoe/meta-llama/Llama-3.3-70B-Instruct", + "crusoe/deepseek-ai/DeepSeek-R1-0528", + "crusoe/deepseek-ai/DeepSeek-V3-0324", + "crusoe/Qwen/Qwen3-235B-A22B-Instruct-2507", + "crusoe/moonshotai/Kimi-K2-Thinking", + "crusoe/openai/gpt-oss-120b", + "crusoe/google/gemma-3-12b-it", + ] + + for model in crusoe_models: + model_info = get_model_info(model) + assert model_info is not None, f"Model info not found for {model}" + assert model_info.get("litellm_provider") == "crusoe", ( + f"{model} should have crusoe as provider" + ) + assert model_info.get("mode") == "chat", f"{model} should be in chat mode" + finally: + litellm.model_cost = original_model_cost + if original_env is None: + os.environ.pop("LITELLM_LOCAL_MODEL_COST_MAP", None) + else: + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = original_env diff --git a/tests/test_litellm/llms/azure/test_azure_cost_calculation.py b/tests/test_litellm/llms/azure/test_azure_cost_calculation.py new file mode 100644 index 0000000000..53c91032b3 --- /dev/null +++ b/tests/test_litellm/llms/azure/test_azure_cost_calculation.py @@ -0,0 +1,75 @@ +""" +Test Azure OpenAI cost calculator — service_tier pricing. +""" + +import pytest + +import litellm +from litellm.llms.azure.cost_calculation import cost_per_token +from litellm.types.utils import Usage + + +# Register a test model with tier-specific pricing +TEST_MODEL = "test-azure-gpt-4.1" +TEST_MODEL_COST = { + TEST_MODEL: { + "input_cost_per_token": 0.001, + "output_cost_per_token": 0.002, + "input_cost_per_token_priority": 0.01, + "output_cost_per_token_priority": 0.02, + "input_cost_per_token_flex": 0.0005, + "output_cost_per_token_flex": 0.001, + "litellm_provider": "azure", + "max_tokens": 8192, + } +} + + +class TestAzureServiceTierCostCalculation: + """Test that service_tier is passed through Azure cost calculation.""" + + @pytest.fixture(autouse=True) + def register_test_model(self): + litellm.register_model(model_cost=TEST_MODEL_COST) + + def test_service_tier_priority_higher_cost(self): + """Priority tier should cost more than standard.""" + usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500) + + standard_prompt, standard_completion = cost_per_token( + model=TEST_MODEL, usage=usage + ) + priority_prompt, priority_completion = cost_per_token( + model=TEST_MODEL, usage=usage, service_tier="priority" + ) + + assert priority_prompt > standard_prompt + assert priority_completion > standard_completion + + def test_service_tier_flex_lower_cost(self): + """Flex tier should cost less than standard.""" + usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500) + + standard_prompt, standard_completion = cost_per_token( + model=TEST_MODEL, usage=usage + ) + flex_prompt, flex_completion = cost_per_token( + model=TEST_MODEL, usage=usage, service_tier="flex" + ) + + assert flex_prompt < standard_prompt + assert flex_completion < standard_completion + + def test_service_tier_none_returns_standard(self): + """service_tier=None should return standard pricing.""" + usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500) + + none_prompt, none_completion = cost_per_token( + model=TEST_MODEL, usage=usage, service_tier=None + ) + standard_prompt, standard_completion = cost_per_token( + model=TEST_MODEL, usage=usage, service_tier="standard" + ) + + assert abs(none_prompt - standard_prompt) < 1e-10 + assert abs(none_completion - standard_completion) < 1e-10 diff --git a/tests/test_litellm/llms/azure_ai/test_azure_ai_cost_calculator.py b/tests/test_litellm/llms/azure_ai/test_azure_ai_cost_calculator.py index 37add41b83..20260c744f 100644 --- a/tests/test_litellm/llms/azure_ai/test_azure_ai_cost_calculator.py +++ b/tests/test_litellm/llms/azure_ai/test_azure_ai_cost_calculator.py @@ -451,3 +451,51 @@ class TestAzureModelRouterCostBreakdown: assert logging_obj.cost_breakdown["additional_costs"][ "Azure Model Router Flat Cost" ] == pytest.approx(expected_flat_cost, rel=1e-9) + + +class TestAzureAIServiceTierCostCalculation: + """Test that service_tier is passed through Azure AI cost calculation.""" + + @pytest.fixture(autouse=True) + def register_test_model(self): + import litellm + litellm.register_model(model_cost={ + "test-azure-ai-model": { + "input_cost_per_token": 0.001, + "output_cost_per_token": 0.002, + "input_cost_per_token_priority": 0.01, + "output_cost_per_token_priority": 0.02, + "input_cost_per_token_flex": 0.0005, + "output_cost_per_token_flex": 0.001, + "litellm_provider": "azure_ai", + "max_tokens": 8192, + } + }) + + def test_service_tier_priority_higher_cost(self): + """Priority tier should cost more than standard for azure_ai.""" + usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500) + + standard_prompt, standard_completion = cost_per_token( + model="test-azure-ai-model", usage=usage + ) + priority_prompt, priority_completion = cost_per_token( + model="test-azure-ai-model", usage=usage, service_tier="priority" + ) + + assert priority_prompt > standard_prompt + assert priority_completion > standard_completion + + def test_service_tier_flex_lower_cost(self): + """Flex tier should cost less than standard for azure_ai.""" + usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500) + + standard_prompt, standard_completion = cost_per_token( + model="test-azure-ai-model", usage=usage + ) + flex_prompt, flex_completion = cost_per_token( + model="test-azure-ai-model", usage=usage, service_tier="flex" + ) + + assert flex_prompt < standard_prompt + assert flex_completion < standard_completion diff --git a/tests/test_litellm/llms/crusoe/__init__.py b/tests/test_litellm/llms/crusoe/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/tests/test_litellm/llms/crusoe/test_crusoe.py b/tests/test_litellm/llms/crusoe/test_crusoe.py new file mode 100644 index 0000000000..0a05126919 --- /dev/null +++ b/tests/test_litellm/llms/crusoe/test_crusoe.py @@ -0,0 +1,135 @@ +import os +from unittest.mock import patch + +CRUSOE_API_BASE = "https://managed-inference-api-proxy.crusoecloud.com/v1" + + +def test_crusoe_json_registry(): + """Test Crusoe is registered in the JSON provider registry""" + from litellm.llms.openai_like.json_loader import JSONProviderRegistry + + assert JSONProviderRegistry.exists("crusoe") + config = JSONProviderRegistry.get("crusoe") + assert config is not None + assert config.base_url == CRUSOE_API_BASE + assert config.api_key_env == "CRUSOE_API_KEY" + assert config.api_base_env == "CRUSOE_API_BASE" + + +def test_crusoe_dynamic_config_defaults(): + """Test dynamic config returns correct default API base""" + from litellm.llms.openai_like.dynamic_config import create_config_class + from litellm.llms.openai_like.json_loader import JSONProviderRegistry + + config = create_config_class(JSONProviderRegistry.get("crusoe"))() + + with patch.dict(os.environ, {}, clear=True): + api_base, api_key = config._get_openai_compatible_provider_info(None, None) + + assert api_base == CRUSOE_API_BASE + assert api_key is None + + +def test_crusoe_dynamic_config_env_vars(): + """Test dynamic config reads CRUSOE_API_KEY and CRUSOE_API_BASE from env""" + from litellm.llms.openai_like.dynamic_config import create_config_class + from litellm.llms.openai_like.json_loader import JSONProviderRegistry + + config = create_config_class(JSONProviderRegistry.get("crusoe"))() + + with patch.dict( + os.environ, + {"CRUSOE_API_KEY": "test-key", "CRUSOE_API_BASE": "https://custom.crusoe.com/v1"}, + ): + api_base, api_key = config._get_openai_compatible_provider_info(None, None) + + assert api_base == "https://custom.crusoe.com/v1" + assert api_key == "test-key" + + +def test_crusoe_dynamic_config_explicit_params(): + """Test explicit params override env vars""" + from litellm.llms.openai_like.dynamic_config import create_config_class + from litellm.llms.openai_like.json_loader import JSONProviderRegistry + + config = create_config_class(JSONProviderRegistry.get("crusoe"))() + + with patch.dict(os.environ, {"CRUSOE_API_KEY": "env-key"}): + api_base, api_key = config._get_openai_compatible_provider_info( + "https://override.crusoe.com/v1", "override-key" + ) + + assert api_base == "https://override.crusoe.com/v1" + assert api_key == "override-key" + + +def test_crusoe_supported_params(): + """Test dynamic config returns standard OpenAI params""" + from litellm.llms.openai_like.dynamic_config import create_config_class + from litellm.llms.openai_like.json_loader import JSONProviderRegistry + + config = create_config_class(JSONProviderRegistry.get("crusoe"))() + params = config.get_supported_openai_params(model="meta-llama/Llama-3.3-70B-Instruct") + + assert isinstance(params, list) + assert len(params) > 0 + assert "temperature" in params + assert "max_tokens" in params + assert "stream" in params + + +def test_crusoe_param_mapping_max_completion_tokens(): + """Test max_completion_tokens is mapped to max_tokens for Crusoe""" + from litellm.llms.openai_like.dynamic_config import create_config_class + from litellm.llms.openai_like.json_loader import JSONProviderRegistry + + config = create_config_class(JSONProviderRegistry.get("crusoe"))() + optional_params = config.map_openai_params( + non_default_params={"max_completion_tokens": 1024}, + optional_params={}, + model="meta-llama/Llama-3.3-70B-Instruct", + drop_params=False, + ) + + assert "max_tokens" in optional_params, "max_completion_tokens should be mapped to max_tokens" + assert optional_params["max_tokens"] == 1024 + assert "max_completion_tokens" not in optional_params + + +def test_crusoe_provider_detection_by_prefix(): + """Test crusoe/model prefix is correctly routed""" + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + model, provider, _, _ = get_llm_provider("crusoe/meta-llama/Llama-3.3-70B-Instruct") + assert provider == "crusoe" + assert model == "meta-llama/Llama-3.3-70B-Instruct" + + +def test_crusoe_model_list_populated(): + """Test Crusoe models are present in model_prices_and_context_window.json""" + import litellm + + original_model_cost = litellm.model_cost + original_env = os.environ.get("LITELLM_LOCAL_MODEL_COST_MAP") + try: + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" + litellm.model_cost = litellm.get_model_cost_map(url="") + + expected = [ + "crusoe/meta-llama/Llama-3.3-70B-Instruct", + "crusoe/deepseek-ai/DeepSeek-R1-0528", + "crusoe/deepseek-ai/DeepSeek-V3-0324", + "crusoe/Qwen/Qwen3-235B-A22B-Instruct-2507", + "crusoe/moonshotai/Kimi-K2-Thinking", + "crusoe/openai/gpt-oss-120b", + "crusoe/google/gemma-3-12b-it", + ] + for model in expected: + assert model in litellm.model_cost, f"{model} not found in model_cost" + assert litellm.model_cost[model].get("litellm_provider") == "crusoe" + finally: + litellm.model_cost = original_model_cost + if original_env is None: + os.environ.pop("LITELLM_LOCAL_MODEL_COST_MAP", None) + else: + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = original_env diff --git a/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_ai_gemini_transformation.py b/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_ai_gemini_transformation.py index f219bd179a..977c53280a 100644 --- a/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_ai_gemini_transformation.py +++ b/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_ai_gemini_transformation.py @@ -1361,6 +1361,53 @@ def test_file_data_field_order_gcs_urls(): ), "mime_type must come before file_uri in the file_data dict" +def test_gemini_files_api_uri_without_format(): + """ + Test that Gemini Files API URIs work WITHOUT an explicit format/mime_type. + + When a user uploads a file via the Gemini Files API and then references it + by URI (https://generativelanguage.googleapis.com/v1beta/files/...), + the file is already on Google's servers. These URLs return 403 when + fetched directly, so _process_gemini_media must NOT try to resolve the + MIME type via HTTP. Instead it should pass the URI through as file_data + and let the Gemini API resolve the type from its stored metadata. + + Related issue: https://github.com/BerriAI/litellm/issues/24907 + """ + from litellm.llms.vertex_ai.gemini.transformation import _process_gemini_media + + file_url = "https://generativelanguage.googleapis.com/v1beta/files/37eh7rsw1vfe" + + # Should NOT raise — previously this hit the generic https:// handler + # which called _get_image_mime_type_from_url() and got a 403. + result = _process_gemini_media(image_url=file_url) + + assert "file_data" in result + file_data = result["file_data"] + assert file_data["file_uri"] == file_url + # When no format is provided, mime_type should be absent so the + # Gemini API infers it from the stored file metadata. + assert "mime_type" not in file_data + + +def test_gemini_files_api_uri_with_format(): + """ + Test that Gemini Files API URIs correctly forward an explicit format. + + Related issue: https://github.com/BerriAI/litellm/issues/24907 + """ + from litellm.llms.vertex_ai.gemini.transformation import _process_gemini_media + + file_url = "https://generativelanguage.googleapis.com/v1beta/files/n1vhxa28lyaw" + + result = _process_gemini_media(image_url=file_url, format="text/plain") + + assert "file_data" in result + file_data = result["file_data"] + assert file_data["file_uri"] == file_url + assert file_data["mime_type"] == "text/plain" + + def test_extract_file_data_with_path_object(): """ Test that filename is correctly extracted from Path objects for MIME type detection. diff --git a/tests/test_litellm/proxy/guardrails/test_qostodian_nexus_guardrail.py b/tests/test_litellm/proxy/guardrails/test_qostodian_nexus_guardrail.py new file mode 100644 index 0000000000..6daa3e1430 --- /dev/null +++ b/tests/test_litellm/proxy/guardrails/test_qostodian_nexus_guardrail.py @@ -0,0 +1,252 @@ +""" +Unit tests for Qostodian Nexus (by Qohash) integration. + +Tests verify: +1. QostodianNexus can be instantiated with default and custom values +2. Qostodian Nexus is registered in SupportedGuardrailIntegrations +3. Guardrail initializer and class registries contain Qostodian Nexus +4. Configuration parameters are properly passed through +5. QostodianNexusConfigModel works correctly +""" + +import os +import pytest +from unittest.mock import MagicMock + + +def test_qostodian_nexus_initialization_with_defaults(): + """Test QostodianNexus initializes with default values.""" + import os + from unittest.mock import patch + from litellm.proxy.guardrails.guardrail_hooks.qohash import QostodianNexus + + # Unset env var so the hardcoded default is used + env = {k: v for k, v in os.environ.items() if k != "QOSTODIAN_NEXUS_API_BASE"} + with patch.dict(os.environ, env, clear=True): + guardrail = QostodianNexus() + + # Should use default api_base + assert guardrail.api_base is not None + assert "nexus:8800" in guardrail.api_base + + +def test_qostodian_nexus_initialization_with_custom_api_base(): + """Test QostodianNexus initializes with custom api_base.""" + from litellm.proxy.guardrails.guardrail_hooks.qohash import QostodianNexus + + custom_api_base = "http://custom-nexus:9000" + guardrail = QostodianNexus(api_base=custom_api_base) + + assert custom_api_base in guardrail.api_base + + +def test_qostodian_nexus_in_supported_guardrail_integrations(): + """Test that Qostodian Nexus is registered in SupportedGuardrailIntegrations enum.""" + from litellm.types.guardrails import SupportedGuardrailIntegrations + + # Check enum contains QOSTODIAN_NEXUS + assert hasattr(SupportedGuardrailIntegrations, "QOSTODIAN_NEXUS") + assert SupportedGuardrailIntegrations.QOSTODIAN_NEXUS.value == "qostodian_nexus" + + # Check it's in the list of all values + all_values = [e.value for e in SupportedGuardrailIntegrations] + assert "qostodian_nexus" in all_values + + +def test_qostodian_nexus_in_guardrail_initializer_registry(): + """Test that Qostodian Nexus is registered in guardrail_initializer_registry.""" + from litellm.proxy.guardrails.guardrail_hooks.qohash import ( + guardrail_initializer_registry, + ) + + assert "qostodian_nexus" in guardrail_initializer_registry + assert callable(guardrail_initializer_registry["qostodian_nexus"]) + + +def test_qostodian_nexus_in_guardrail_class_registry(): + """Test that Qostodian Nexus is registered in guardrail_class_registry.""" + from litellm.proxy.guardrails.guardrail_hooks.qohash import ( + guardrail_class_registry, + QostodianNexus, + ) + + assert "qostodian_nexus" in guardrail_class_registry + assert guardrail_class_registry["qostodian_nexus"] == QostodianNexus + + +def test_qostodian_nexus_config_model_initialization(): + """Test QostodianNexusConfigModel can be instantiated.""" + from litellm.types.proxy.guardrails.guardrail_hooks.qohash import ( + QostodianNexusConfigModel, + ) + + config = QostodianNexusConfigModel( + api_base="http://test:8800", + ) + + assert config.api_base == "http://test:8800" + + +def test_qostodian_nexus_config_model_defaults(): + """Test QostodianNexusConfigModel uses correct defaults.""" + from litellm.types.proxy.guardrails.guardrail_hooks.qohash import ( + QostodianNexusConfigModel, + ) + + config = QostodianNexusConfigModel() + + assert config.api_base is None + + +def test_qostodian_nexus_config_model_ui_friendly_name(): + """Test QostodianNexusConfigModel returns correct UI friendly name.""" + from litellm.types.proxy.guardrails.guardrail_hooks.qohash import ( + QostodianNexusConfigModel, + ) + + ui_name = QostodianNexusConfigModel.ui_friendly_name() + assert ui_name == "Qostodian Nexus" + + +def test_qostodian_nexus_initializer_function(): + """Test the initialize_guardrail function.""" + from litellm.proxy.guardrails.guardrail_hooks.qohash import initialize_guardrail + from litellm.types.guardrails import LitellmParams, Guardrail + from unittest.mock import patch + + # Mock litellm.logging_callback_manager + with patch("litellm.logging_callback_manager") as mock_manager: + mock_manager.add_litellm_callback = MagicMock() + + # Create test params + litellm_params = LitellmParams( + guardrail="qostodian_nexus", + mode="pre_call", + api_base="http://test:8800", + default_on=True, + ) + + guardrail_config: Guardrail = {"guardrail_name": "test-qostodian-nexus"} + + # Call initializer + result = initialize_guardrail(litellm_params, guardrail_config) + + # Verify callback was added + mock_manager.add_litellm_callback.assert_called_once() + + # Verify returned instance has correct properties + assert result is not None + assert "test:8800" in result.api_base + + +def test_qostodian_nexus_inherits_from_generic_guardrail_api(): + """Test that QostodianNexus inherits from GenericGuardrailAPI.""" + from litellm.proxy.guardrails.guardrail_hooks.qohash import QostodianNexus + from litellm.proxy.guardrails.guardrail_hooks.generic_guardrail_api.generic_guardrail_api import ( + GenericGuardrailAPI, + ) + + assert issubclass(QostodianNexus, GenericGuardrailAPI) + + +def test_qostodian_nexus_guardrail_name_constant(): + """Test that GUARDRAIL_NAME constant is defined correctly.""" + from litellm.proxy.guardrails.guardrail_hooks.qohash.qohash import GUARDRAIL_NAME + + assert GUARDRAIL_NAME == "qostodian_nexus" + + +def test_qostodian_nexus_get_config_model(): + """Test that QostodianNexus returns the correct config model.""" + from litellm.proxy.guardrails.guardrail_hooks.qohash import QostodianNexus + from litellm.types.proxy.guardrails.guardrail_hooks.qohash import ( + QostodianNexusConfigModel, + ) + + config_model = QostodianNexus.get_config_model() + + assert config_model is not None + assert config_model == QostodianNexusConfigModel + + +def test_qostodian_nexus_env_vars(): + """Test that QOSTODIAN_NEXUS_API_BASE env var is picked up correctly.""" + import os + from unittest.mock import patch + from litellm.proxy.guardrails.guardrail_hooks.qohash import QostodianNexus + + with patch.dict(os.environ, {"QOSTODIAN_NEXUS_API_BASE": "http://new-api:8800"}): + guardrail = QostodianNexus() + assert "new-api:8800" in guardrail.api_base + + +def test_qostodian_nexus_config_model_field_descriptions(): + """Test that QostodianNexusConfigModel has correct field descriptions.""" + from litellm.types.proxy.guardrails.guardrail_hooks.qohash import ( + QostodianNexusConfigModel, + ) + + # Check that field descriptions mention the correct env vars + api_base_field = QostodianNexusConfigModel.model_fields["api_base"] + assert "QOSTODIAN_NEXUS_API_BASE" in api_base_field.description + + +def test_qostodian_nexus_unified_detection(): + """ + Test that QostodianNexus is properly detected by LiteLLM's unified guardrail system. + + This verifies the fix for the detection bug where QostodianNexus wasn't being + recognized because apply_guardrail was only inherited, not in the class's own __dict__. + """ + from litellm.proxy.guardrails.guardrail_hooks.qohash import QostodianNexus + + # Create an instance (this is how LiteLLM uses it) + instance = QostodianNexus(api_base="http://test:8800") + + # Test the exact detection logic used in litellm/proxy/utils.py:868 + # use_unified = "apply_guardrail" in type(callback).__dict__ + use_unified = "apply_guardrail" in type(instance).__dict__ + + # Should be detected as using unified guardrail system + assert use_unified is True, ( + "QostodianNexus should be detected by unified guardrail system. " + "The apply_guardrail method must be present in QostodianNexus.__dict__" + ) + + # Also verify the method is callable + assert hasattr(instance, "apply_guardrail") + assert callable(instance.apply_guardrail) + + +def test_qostodian_nexus_builtin_extra_headers(): + """Test that QostodianNexus includes built-in x-qostodian-nexus-identifiers-* headers.""" + from litellm.proxy.guardrails.guardrail_hooks.qohash import QostodianNexus + + instance = QostodianNexus() + + expected_headers = [ + "x-qostodian-nexus-identifiers-trace", + "x-qostodian-nexus-identifiers-source", + "x-qostodian-nexus-identifiers-container", + "x-qostodian-nexus-identifiers-identity", + ] + + for header in expected_headers: + assert header in instance.extra_headers, ( + f"Expected built-in header '{header}' to be in extra_headers" + ) + + +def test_qostodian_nexus_extra_headers_merged(): + """Test that caller-supplied extra_headers are merged with built-in headers.""" + from litellm.proxy.guardrails.guardrail_hooks.qohash import QostodianNexus + + custom_header = "x-custom-correlation-id" + instance = QostodianNexus(extra_headers=[custom_header]) + + # Built-in headers should be present + assert "x-qostodian-nexus-identifiers-trace" in instance.extra_headers + # Custom header should also be present + assert custom_header in instance.extra_headers + # No duplicates + assert len(instance.extra_headers) == len(set(instance.extra_headers)) diff --git a/tests/test_litellm/proxy/hooks/test_parallel_request_limiter_v3.py b/tests/test_litellm/proxy/hooks/test_parallel_request_limiter_v3.py index d4fb5b7271..370477c360 100644 --- a/tests/test_litellm/proxy/hooks/test_parallel_request_limiter_v3.py +++ b/tests/test_litellm/proxy/hooks/test_parallel_request_limiter_v3.py @@ -1029,6 +1029,83 @@ async def test_team_member_rate_limits_v3(): ), "Team member TPM limit should be set" +@pytest.mark.asyncio +async def test_team_member_rate_limits_v3_raises_429_when_over_limit(): + """ + When should_rate_limit reports OVER_LIMIT for the team_member descriptor, the + pre-call hook raises HTTP 429 with rate_limit headers — same contract as + test_rpm_api_key_rate_limits_v3 / test_tpm_api_key_rate_limits_v3. + """ + _api_key = hash_token("sk-12345") + _team_id = "team_123" + _user_id = "user_456" + + user_api_key_dict = UserAPIKeyAuth( + api_key=_api_key, + team_id=_team_id, + user_id=_user_id, + team_member_rpm_limit=10, + team_member_tpm_limit=1000, + ) + + local_cache = DualCache() + parallel_request_handler = _PROXY_MaxParallelRequestsHandler( + internal_usage_cache=InternalUsageCache(local_cache) + ) + + captured_descriptors = None + + async def mock_should_rate_limit(descriptors, **kwargs): + nonlocal captured_descriptors + captured_descriptors = descriptors + return { + "overall_code": "OVER_LIMIT", + "statuses": [ + { + "code": "OVER_LIMIT", + "current_limit": 10, + "limit_remaining": -1, + "rate_limit_type": "requests", + "descriptor_key": "team_member", + }, + { + "code": "OK", + "current_limit": 1000, + "limit_remaining": 500, + "rate_limit_type": "tokens", + "descriptor_key": "team_member", + }, + ], + } + + parallel_request_handler.should_rate_limit = mock_should_rate_limit + + error = None + try: + await parallel_request_handler.async_pre_call_hook( + user_api_key_dict=user_api_key_dict, + cache=local_cache, + data={"model": "gpt-3.5-turbo"}, + call_type="", + ) + except HTTPException as e: + error = e + assert e.status_code == 429 + assert "rate_limit_type" in e.headers + assert e.headers.get("rate_limit_type") == "requests" + assert "retry-after" in e.headers + + assert error is not None, "An Exception must be thrown" + assert captured_descriptors is not None, "Rate limit descriptors should be captured" + team_member_descriptor = None + for descriptor in captured_descriptors: + if descriptor["key"] == "team_member": + team_member_descriptor = descriptor + break + assert team_member_descriptor is not None + assert team_member_descriptor["value"] == f"{_team_id}:{_user_id}" + + @pytest.mark.asyncio async def test_dynamic_rate_limiting_v3(): """ diff --git a/tests/test_litellm/test_secret_redaction.py b/tests/test_litellm/test_secret_redaction.py index 6cbb2fd7b8..8a0a2221c1 100644 --- a/tests/test_litellm/test_secret_redaction.py +++ b/tests/test_litellm/test_secret_redaction.py @@ -9,11 +9,11 @@ from litellm._logging import ( JsonFormatter, _redact_string, _secret_filter, - _setup_json_exception_handlers, verbose_logger, verbose_proxy_logger, verbose_router_logger, ) +from litellm.litellm_core_utils.secret_redaction import redact_string SECRET = "sk-proj-abc123def456ghi789jklmnopqrst" @@ -57,12 +57,12 @@ def test_redact_string_catches_secret_patterns(): SECRET, ] for secret in cases: - result = _redact_string("msg: " + secret) + result = redact_string("msg: " + secret) assert secret not in result, f"{secret!r} was not redacted" assert "REDACTED" in result normal = "Loaded model gpt-4 with 3 replicas on us-east-1" - assert _redact_string(normal) == normal + assert redact_string(normal) == normal def test_filter_redacts_secrets_in_logger_output(): @@ -155,7 +155,7 @@ def test_x_api_key_regex_does_not_consume_json_delimiters(): """x-api-key pattern must stop before closing quotes/braces so JSON stays valid.""" # Simulates a JSON log line containing an x-api-key header value json_line = '{"headers": {"x-api-key": "secret123"}, "status": 200}' - result = _redact_string(json_line) + result = redact_string(json_line) # The secret value should be redacted assert "secret123" not in result assert "REDACTED" in result @@ -234,12 +234,12 @@ def test_key_name_redaction_catches_secrets_in_dict_repr(): "'slack_webhook_url': 'https://hooks.slack.com/services/T00/B00/xxx'", ] for secret_line in cases: - result = _redact_string(secret_line) + result = redact_string(secret_line) assert "REDACTED" in result, f"Key-name redaction missed: {secret_line!r}" # Non-sensitive keys should NOT be redacted safe = "'enable_jwt_auth': True, 'store_model_in_db': True" - assert _redact_string(safe) == safe + assert redact_string(safe) == safe def test_key_name_redaction_in_general_settings_dict(): @@ -277,7 +277,7 @@ _SAMPLE_SA_JSON = ( def test_pem_private_key_redacted_in_json(): - result = _redact_string(_SAMPLE_SA_JSON) + result = redact_string(_SAMPLE_SA_JSON) assert "MIIEvQIBADA" not in result assert "-----BEGIN" not in result @@ -286,12 +286,12 @@ def test_pem_private_key_redacted_in_dict_repr(): import json sa = json.loads(_SAMPLE_SA_JSON) - result = _redact_string(str(sa)) + result = redact_string(str(sa)) assert "MIIEvQIBADA" not in result def test_service_account_blob_fully_redacted(): - result = _redact_string(f"Got={_SAMPLE_SA_JSON}") + result = redact_string(f"Got={_SAMPLE_SA_JSON}") assert "my-proj-123" not in result assert "sa@my-proj.iam.gserviceaccount.com" not in result assert "abc123def" not in result @@ -320,22 +320,22 @@ def test_vertex_traceback_redacts_pem(): "Unable to load vertex credentials from environment. " f"Got={_SAMPLE_SA_JSON}" ) - result = _redact_string(traceback_text) + result = redact_string(traceback_text) assert "MIIEvQIBADA" not in result assert "-----BEGIN" not in result def test_gcp_oauth_token_redacted(): - result = _redact_string("access token ya29.c.c0ASRK0GZvXlongtokenhere") + result = redact_string("access token ya29.c.c0ASRK0GZvXlongtokenhere") assert "ya29." not in result assert "REDACTED" in result def test_non_pem_private_key_value_redacted(): - result = _redact_string("'private_key': 'some-non-pem-secret-value'") + result = redact_string("'private_key': 'some-non-pem-secret-value'") assert "some-non-pem-secret" not in result def test_normal_vertex_log_not_redacted(): msg = "Vertex: Loading vertex credentials, is_file_path=True, current dir /app" - assert _redact_string(msg) == msg + assert redact_string(msg) == msg diff --git a/ui/litellm-dashboard/public/assets/logos/qohash.jpg b/ui/litellm-dashboard/public/assets/logos/qohash.jpg new file mode 100644 index 0000000000..50227ab391 Binary files /dev/null and b/ui/litellm-dashboard/public/assets/logos/qohash.jpg differ diff --git a/ui/litellm-dashboard/src/components/guardrails/guardrail_info_helpers.tsx b/ui/litellm-dashboard/src/components/guardrails/guardrail_info_helpers.tsx index 2286eba776..ac4b787e96 100644 --- a/ui/litellm-dashboard/src/components/guardrails/guardrail_info_helpers.tsx +++ b/ui/litellm-dashboard/src/components/guardrails/guardrail_info_helpers.tsx @@ -52,6 +52,7 @@ export const guardrail_provider_map: Record = { Promptguard: "promptguard", LlmAsAJudge: "llm_as_a_judge", Xecguard: "xecguard", + QostodianNexus: "qostodian_nexus", }; // Function to populate provider map from API response - updates the original map @@ -138,6 +139,7 @@ export const guardrailLogoMap: Record = { "LiteLLM Content Filter": `${asset_logos_folder}litellm_logo.jpg`, "LiteLLM LLM as a Judge": `${asset_logos_folder}litellm_logo.jpg`, "Akto": `${asset_logos_folder}akto.svg`, + "Qostodian Nexus": `${asset_logos_folder}qohash.jpg`, }; export const getGuardrailLogoAndName = (guardrailValue: string): { logo: string; displayName: string } => {