Added dashscope (alibaba's cloud - qwen) as a provider (#12361)

* Added dashscope as a provider

* Fix some leftover references on nebius

* Porting the dashscope api endpoit international version

* explicit tool_choice = True in config
This commit is contained in:
minghao
2025-07-10 18:09:26 -07:00
committed by GitHub
parent 6ff06bb96f
commit bee63a1ae7
11 changed files with 365 additions and 2 deletions
@@ -0,0 +1,67 @@
# Dashscope
https://dashscope.console.aliyun.com/
**We support ALL Qwen models, just set `dashscope/` as a prefix when sending completion requests**
## API Key
```python
# env variable
os.environ['DASHSCOPE_API_KEY']
```
## Sample Usage
```python
from litellm import completion
import os
os.environ['DASHSCOPE_API_KEY'] = ""
response = completion(
model="dashscope/qwen-turbo",
messages=[
{"role": "user", "content": "hello from litellm"}
],
)
print(response)
```
## Sample Usage - Streaming
```python
from litellm import completion
import os
os.environ['DASHSCOPE_API_KEY'] = ""
response = completion(
model="dashscope/qwen-turbo",
messages=[
{"role": "user", "content": "hello from litellm"}
],
stream=True
)
for chunk in response:
print(chunk)
```
## Supported Models - ALL Qwen Models Supported!
We support ALL Qwen models, just set `dashscope/` as a prefix when sending completion requests
[DashScope Model List](https://help.aliyun.com/zh/model-studio/compatibility-of-openai-with-dashscope?spm=a2c4g.11186623.help-menu-2400256.d_2_8_0.1efd516e2tTXBn&scm=20140722.H_2833609._.OR_help-T_cn~zh-V_1#7f9c78ae99pwz)
| Model Name | Function Call |
|--------------------------|------------------------------------------------------------------------------------------------------------------------------------------------------------------|
| qwen-turbo | `completion(model="dashscope/qwen-turbo", messages)` |
| qwen-plus | `completion(model="dashscope/qwen-plus", messages)` |
| qwen-max | `completion(model="dashscope/qwen-max", messages)` |
| qwen-turbo-latest | `completion(model="dashscope/qwen-turbo-latest", messages)` |
| qwen-plus-latest | `completion(model="dashscope/qwen-plus-latest", messages)` |
| qwen-max-latest | `completion(model="dashscope/qwen-max-latest", messages)` |
| qwen-vl-plus | `completion(model="dashscope/qwen-vl-plus", messages)` |
| qwen-vl-max | `completion(model="dashscope/qwen-vl-max", messages)` |
| qwq-32b | `completion(model="dashscope/qwq-32b", messages)` |
| qwq-32b-preview | `completion(model="dashscope/qwq-32b-preview", messages)` |
| qwen3-235b-a22b | `completion(model="dashscope/qwen3-235b-a22b", messages)` |
| qwen3-32b | `completion(model="dashscope/qwen3-32b", messages)` |
| qwen3-30b-a3b | `completion(model="dashscope/qwen3-30b-a3b", messages)` |
```
+2 -1
View File
@@ -445,7 +445,8 @@ const sidebars = {
"providers/petals",
"providers/snowflake",
"providers/featherless_ai",
"providers/nebius"
"providers/nebius",
"providers/dashscope"
],
},
{
+6 -1
View File
@@ -494,6 +494,7 @@ nebius_models: List = []
nebius_embedding_models: List = []
deepgram_models: List = []
elevenlabs_models: List = []
dashscope_models: List = []
def is_bedrock_pricing_only_model(key: str) -> bool:
@@ -669,7 +670,8 @@ def add_known_models():
deepgram_models.append(key)
elif value.get("litellm_provider") == "elevenlabs":
elevenlabs_models.append(key)
elif value.get("litellm_provider") == "dashscope":
dashscope_models.append(key)
add_known_models()
# known openai compatible endpoints - we'll eventually move this list to the model_prices_and_context_window.json dictionary
@@ -752,6 +754,7 @@ model_list = (
+ nscale_models
+ deepgram_models
+ elevenlabs_models
+ dashscope_models
)
model_list_set = set(model_list)
@@ -817,6 +820,7 @@ models_by_provider: dict = {
"featherless_ai": featherless_ai_models,
"deepgram": deepgram_models,
"elevenlabs": elevenlabs_models,
"dashscope": dashscope_models,
}
# mapping for those models which have larger equivalents
@@ -1133,6 +1137,7 @@ from .llms.watsonx.chat.transformation import IBMWatsonXChatConfig
from .llms.watsonx.embed.transformation import IBMWatsonXEmbeddingConfig
from .llms.github_copilot.chat.transformation import GithubCopilotConfig
from .llms.nebius.chat.transformation import NebiusConfig
from .llms.dashscope.chat.transformation import DashScopeChatConfig
from .main import * # type: ignore
from .integrations import *
from .llms.custom_httpx.async_client_cleanup import close_litellm_async_clients
+17
View File
@@ -267,6 +267,7 @@ LITELLM_CHAT_PROVIDERS = [
"featherless_ai",
"nscale",
"nebius",
"dashscope",
]
LITELLM_EMBEDDING_PROVIDERS_SUPPORTING_INPUT_ARRAY_OF_TOKENS = [
@@ -393,6 +394,7 @@ openai_compatible_endpoints: List = [
"api.featherless.ai/v1",
"inference.api.nscale.com/v1",
"api.studio.nebius.ai/v1",
"https://dashscope-intl.aliyuncs.com/compatible-mode/v1"
]
@@ -428,6 +430,7 @@ openai_compatible_providers: List = [
"featherless_ai",
"nscale",
"nebius",
"dashscope"
]
openai_text_completion_compatible_providers: List = (
[ # providers that support `/v1/completions`
@@ -438,6 +441,7 @@ openai_text_completion_compatible_providers: List = (
"llamafile",
"featherless_ai",
"nebius",
"dashscope"
]
)
_openai_like_providers: List = [
@@ -611,6 +615,19 @@ nebius_models: List = [
"Qwen/Qwen2.5-Coder-32B-Instruct-fast",
]
dashscope_models: List = [
"qwen-turbo",
"qwen-plus",
"qwen-max",
"qwen-turbo-latest",
"qwen-plus-latest",
"qwen-max-latest",
"qwq-32b",
"qwen3-235b-a22b",
"qwen3-32b",
"qwen3-30b-a3b"
]
nebius_embedding_models: List = [
"BAAI/bge-en-icl",
"BAAI/bge-multilingual-gemma2",
@@ -231,6 +231,9 @@ def get_llm_provider( # noqa: PLR0915
elif endpoint == litellm.NscaleConfig.API_BASE_URL:
custom_llm_provider = "nscale"
dynamic_api_key = litellm.NscaleConfig.get_api_key()
elif endpoint == "dashscope-intl.aliyuncs.com/compatible-mode/v1":
custom_llm_provider = "dashscope"
dynamic_api_key = get_secret_str("DASHSCOPE_API_KEY")
if api_base is not None and not isinstance(api_base, str):
raise Exception(
@@ -658,6 +661,13 @@ def _get_openai_compatible_provider_info( # noqa: PLR0915
) = litellm.NscaleConfig()._get_openai_compatible_provider_info(
api_base=api_base, api_key=api_key
)
elif custom_llm_provider == "dashscope":
(
api_base,
dynamic_api_key,
) = litellm.DashScopeChatConfig()._get_openai_compatible_provider_info(
api_base, api_key
)
if api_base is not None and not isinstance(api_base, str):
raise Exception("api base needs to be a string. api_base={}".format(api_base))
@@ -0,0 +1,77 @@
"""
Translates from OpenAI's `/v1/chat/completions` to DashScope's `/v1/chat/completions`
"""
from typing import Any, Coroutine, List, Literal, Optional, Tuple, Union, overload
from litellm.litellm_core_utils.prompt_templates.common_utils import (
handle_messages_with_content_list_to_str_conversion,
)
from litellm.secret_managers.main import get_secret_str
from litellm.types.llms.openai import AllMessageValues
from ...openai.chat.gpt_transformation import OpenAIGPTConfig
class DashScopeChatConfig(OpenAIGPTConfig):
@overload
def _transform_messages(
self, messages: List[AllMessageValues], model: str, is_async: Literal[True]
) -> Coroutine[Any, Any, List[AllMessageValues]]:
...
@overload
def _transform_messages(
self,
messages: List[AllMessageValues],
model: str,
is_async: Literal[False] = False,
) -> List[AllMessageValues]:
...
def _transform_messages(
self, messages: List[AllMessageValues], model: str, is_async: bool = False
) -> Union[List[AllMessageValues], Coroutine[Any, Any, List[AllMessageValues]]]:
"""
DashScope does not support content in list format.
"""
messages = handle_messages_with_content_list_to_str_conversion(messages)
if is_async:
return super()._transform_messages(
messages=messages, model=model, is_async=True
)
else:
return super()._transform_messages(
messages=messages, model=model, is_async=False
)
def _get_openai_compatible_provider_info(
self, api_base: Optional[str], api_key: Optional[str]
) -> Tuple[Optional[str], Optional[str]]:
api_base = (
api_base
or get_secret_str("DASHSCOPE_API_BASE")
or "https://dashscope-intl.aliyuncs.com/compatible-mode/v1"
) # type: ignore
dynamic_api_key = api_key or get_secret_str("DASHSCOPE_API_KEY")
return api_base, dynamic_api_key
def get_complete_url(
self,
api_base: Optional[str],
api_key: Optional[str],
model: str,
optional_params: dict,
litellm_params: dict,
stream: Optional[bool] = None,
) -> str:
"""
If api_base is not provided, use the default DashScope /chat/completions endpoint.
"""
if not api_base:
api_base = "https://dashscope.aliyuncs.com/compatible-mode/v1"
if not api_base.endswith("/chat/completions"):
api_base = f"{api_base}/chat/completions"
return api_base
+21
View File
@@ -0,0 +1,21 @@
"""
Cost calculator for DeepSeek Chat models.
Handles prompt caching scenario.
"""
from typing import Tuple
from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token
from litellm.types.utils import Usage
def cost_per_token(model: str, usage: Usage) -> Tuple[float, float]:
"""
Calculates the cost per token for a given model, prompt tokens, and completion tokens.
Follows the same logic as Anthropic's cost per token calculation.
"""
return generic_cost_per_token(
model=model, usage=usage, custom_llm_provider="deepseek"
)
+1
View File
@@ -2263,6 +2263,7 @@ class LlmProviders(str, Enum):
VOLCENGINE = "volcengine"
CODESTRAL = "codestral"
TEXT_COMPLETION_CODESTRAL = "text-completion-codestral"
DASHSCOPE = "dashscope"
DEEPSEEK = "deepseek"
SAMBANOVA = "sambanova"
MARITALK = "maritalk"
+7
View File
@@ -5347,6 +5347,11 @@ def validate_environment( # noqa: PLR0915
keys_in_environment = True
else:
missing_keys.append("NEBIUS_API_KEY")
elif custom_llm_provider == "dashscope":
if "DASHSCOPE_API_KEY" in os.environ:
keys_in_environment = True
else:
missing_keys.append("DASHSCOPE_API_KEY")
else:
## openai - chatcompletion + text completion
if (
@@ -6804,6 +6809,8 @@ class ProviderConfigManager:
return litellm.NovitaConfig()
elif litellm.LlmProviders.NEBIUS == provider:
return litellm.NebiusConfig()
elif litellm.LlmProviders.DASHSCOPE == provider:
return litellm.DashScopeChatConfig()
elif litellm.LlmProviders.BEDROCK == provider:
bedrock_route = BedrockModelInfo.get_bedrock_route(model)
bedrock_invoke_provider = litellm.BedrockLLM.get_bedrock_invoke_provider(
+44
View File
@@ -15897,5 +15897,49 @@
"calculation": "$0.22/hour = $0.00366/minute = $0.0000611 per second (enterprise pricing)",
"notes": "ElevenLabs Scribe v1 experimental - enhanced version of the main Scribe model"
}
},
"dashscope/qwen-max": {
"max_tokens": 32768,
"max_input_tokens": 30720,
"max_output_tokens": 8192,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://bailian.console.alibabacloud.com/?spm=a2c63.p38356.0.0.4a615d7bjSUCb4&tab=doc#/doc/?type=model&url=https%3A%2F%2Fwww.alibabacloud.com%2Fhelp%2Fen%2Fdoc-detail%2F2840914.html"
},
"dashscope/qwen-plus-latest": {
"max_tokens": 131072,
"max_input_tokens": 129024,
"max_output_tokens": 16384,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://bailian.console.alibabacloud.com/?spm=a2c63.p38356.0.0.4a615d7bjSUCb4&tab=doc#/doc/?type=model&url=https%3A%2F%2Fwww.alibabacloud.com%2Fhelp%2Fen%2Fdoc-detail%2F2840914.html"
},
"dashscope/qwen-turbo-latest": {
"max_tokens": 131072,
"max_input_tokens": 129024,
"max_output_tokens": 16384,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://bailian.console.alibabacloud.com/?spm=a2c63.p38356.0.0.4a615d7bjSUCb4&tab=doc#/doc/?type=model&url=https%3A%2F%2Fwww.alibabacloud.com%2Fhelp%2Fen%2Fdoc-detail%2F2840914.html"
},
"dashscope/qwen3-30b-a3b": {
"max_tokens": 131072,
"max_input_tokens": 129024,
"max_output_tokens": 16384,
"litellm_provider": "dashscope",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"mode": "chat",
"source": "https://bailian.console.alibabacloud.com/?spm=a2c63.p38356.0.0.4a615d7bjSUCb4&tab=doc#/doc/?type=model&url=https%3A%2F%2Fwww.alibabacloud.com%2Fhelp%2Fen%2Fdoc-detail%2F2840914.html"
}
}
@@ -0,0 +1,113 @@
"""
Unit tests for DashScope configuration.
These tests validate the DashScopeConfig class which extends OpenAIGPTConfig.
DashScope is an OpenAI-compatible provider with minor customizations.
"""
import os
import sys
sys.path.insert(
0, os.path.abspath("../../../../..")
) # Adds the parent directory to the system path
import pytest
import litellm
from litellm import completion
from litellm.llms.dashscope.chat.transformation import DashScopeChatConfig
class TestDashScopeConfig:
"""Test class for DashScope functionality"""
def test_default_api_base(self):
"""Test that default API base is used when none is provided"""
config = DashScopeChatConfig()
headers = {}
api_key = "fake-dashscope-key"
# Call validate_environment without specifying api_base
result = config.validate_environment(
headers=headers,
model="qwen-turbo",
messages=[{"role": "user", "content": "Hey"}],
optional_params={},
litellm_params={},
api_key=api_key,
api_base=None, # Not providing api_base
)
# Verify headers are still set correctly
assert result["Authorization"] == f"Bearer {api_key}"
assert result["Content-Type"] == "application/json"
# We can't directly test the api_base value here since validate_environment
# only returns the headers, but we can verify it doesn't raise an exception
# which would happen if api_base handling was incorrect
@pytest.mark.respx()
def test_dashscope_completion_mock(self, respx_mock):
"""
Mock test for Dashscope completion using the model format from docs.
This test mocks the actual HTTP request to test the integration properly.
"""
litellm.disable_aiohttp_transport = (
True # since this uses respx, we need to set use_aiohttp_transport to False
)
# Set up environment variables for the test
api_key = "fake-dashscope-key"
api_base = "https://dashscope-intl.aliyuncs.com/compatible-mode/v1"
model = "dashscope/qwen-turbo"
model_name = "qwen-turbo" # The actual model name without provider prefix
# Mock the HTTP request to the dashscope API
respx_mock.post(f"{api_base}/chat/completions").respond(
json={
"id": "chatcmpl-123",
"object": "chat.completion",
"created": 1677652288,
"model": model_name,
"choices": [
{
"index": 0,
"message": {
"role": "assistant",
"content": '```python\nprint("Hey from LiteLLM!")\n```\n\nThis simple Python code prints a greeting message from LiteLLM.',
},
"finish_reason": "stop",
}
],
"usage": {
"prompt_tokens": 9,
"completion_tokens": 12,
"total_tokens": 21,
},
},
status_code=200,
)
# Make the actual API call through LiteLLM
response = completion(
model=model,
messages=[
{"role": "user", "content": "write code for saying hey from LiteLLM"}
],
api_key=api_key,
api_base=api_base,
)
# Verify response structure
assert response is not None
assert hasattr(response, "choices")
assert len(response.choices) > 0
assert hasattr(response.choices[0], "message")
assert hasattr(response.choices[0].message, "content")
assert response.choices[0].message.content is not None
# Check for specific content in the response
assert "```python" in response.choices[0].message.content
assert "Hey from LiteLLM" in response.choices[0].message.content