mirror of
https://github.com/tiennm99/goclaw.git
synced 2026-10-04 12:13:15 +00:00
* feat(providers): explicit prompt cache for DashScope/Qwen Extend OpenAI-compat path with Anthropic-style cache_control:ephemeral inline blocks for Alibaba DashScope endpoints (provider_type=bailian or URL contains "dashscope"). Verified live: qwen3.6-plus / qwen3.5-plus / qwen3-coder-plus all return 99.7% cache hit rate on 6K-token prefix, yielding ~89% token cost reduction on cached prefix (90% Alibaba discount). - isDashScope(): 3-source detection (URL + providerType + name) handles reverse-proxied endpoints; covers both "dashscope" and "bailian" - buildRequestBody wraps system content with SplitSystemPromptForCache (exported from anthropic_request.go) using <!-- GOCLAW_CACHE_BOUNDARY --> - Tool prefix cache: cache_control on last tool definition, with 4-marker budget guard - Parse cache_creation_input_tokens from prompt_tokens_details into Usage.CacheCreationTokens; propagates via existing span metadata writers - Runtime escape hatch: GOCLAW_DISABLE_DASHSCOPE_CACHE=true - Live integration smoke test (build tag integration, env-gated) * chore(providers): polish per PR #1127 review - Use strings.Repeat instead of custom repeat() helper in smoke test - Clarify BuildRequestBodyForTest is test-only, not public API * feat(providers): enable thinking for Qwen 3.7/3.6, observe cache in smoke test Qwen3.7-plus and qwen3.6-plus support deep thinking but were missing from dashscopeThinkingModels, so enable_thinking/thinking_budget was silently skipped. Add both to the whitelist and test. Add qwen3.7-plus to the cache smoke test as cache-optional: Alibaba's context-cache doc does not yet list 3.7/3.6 for explicit cache, and the cache wrap is a safe no-op when unsupported, so a no-cache result is logged rather than failed to avoid a flaky live assertion. Claude-Session: https://claude.ai/code/session_01X3jkrc7N3ar8ZGzUJWyNS5 * fix(providers): preserve DashScope detection for proxy routes
147 lines
4.7 KiB
Go
147 lines
4.7 KiB
Go
package providers
|
|
|
|
import (
|
|
"context"
|
|
"log/slog"
|
|
"maps"
|
|
)
|
|
|
|
const (
|
|
dashscopeDefaultBase = "https://dashscope-intl.aliyuncs.com/compatible-mode/v1"
|
|
dashscopeDefaultModel = "qwen3-max"
|
|
)
|
|
|
|
// dashscopeThinkingModels lists DashScope models that accept the
|
|
// enable_thinking / thinking_budget parameters (Qwen3 open-weight and Qwen3.5+ series).
|
|
// Models NOT in this set (e.g. qwen3-plus, qwen3-turbo) will silently
|
|
// skip thinking injection to avoid API "model not supported" errors.
|
|
var dashscopeThinkingModels = map[string]bool{
|
|
// Qwen3.7 / 3.6 series — deep thinking + vision
|
|
"qwen3.7-plus": true,
|
|
"qwen3.6-plus": true,
|
|
// Qwen3.5 series — thinking + vision
|
|
"qwen3.5-plus": true,
|
|
"qwen3.5-turbo": true,
|
|
// Qwen3 hosted
|
|
"qwen3-max": true,
|
|
// Qwen3 open-weight (available as hosted inference)
|
|
"qwen3-235b-a22b": true,
|
|
"qwen3-32b": true,
|
|
"qwen3-14b": true,
|
|
"qwen3-8b": true,
|
|
}
|
|
|
|
// DashScopeProvider wraps OpenAIProvider to handle DashScope-specific behaviors.
|
|
// Critical: DashScope does NOT support tools + streaming simultaneously.
|
|
// When tools are present, ChatStream falls back to non-streaming Chat().
|
|
type DashScopeProvider struct {
|
|
*OpenAIProvider
|
|
}
|
|
|
|
func NewDashScopeProvider(name, apiKey, apiBase, defaultModel string) *DashScopeProvider {
|
|
if apiBase == "" {
|
|
apiBase = dashscopeDefaultBase
|
|
}
|
|
if defaultModel == "" {
|
|
defaultModel = dashscopeDefaultModel
|
|
}
|
|
return &DashScopeProvider{
|
|
OpenAIProvider: NewOpenAIProvider(name, apiKey, apiBase, defaultModel).WithProviderType("dashscope"),
|
|
}
|
|
}
|
|
|
|
// Name is inherited from the embedded OpenAIProvider (returns the user-specified name).
|
|
func (p *DashScopeProvider) SupportsThinking() bool { return true }
|
|
|
|
// Capabilities implements CapabilitiesAware for pipeline code-path selection.
|
|
// StreamWithTools=false is THE critical difference: DashScope falls back to
|
|
// non-streaming when tools are present.
|
|
func (p *DashScopeProvider) Capabilities() ProviderCapabilities {
|
|
return ProviderCapabilities{
|
|
Streaming: true,
|
|
ToolCalling: true,
|
|
StreamWithTools: false,
|
|
Thinking: true,
|
|
Vision: true,
|
|
CacheControl: false,
|
|
MaxContextWindow: 128_000,
|
|
TokenizerID: "cl100k_base",
|
|
}
|
|
}
|
|
|
|
// ModelSupportsThinking implements ModelThinkingCapable.
|
|
// Returns true only for models that accept enable_thinking / thinking_budget.
|
|
func (p *DashScopeProvider) ModelSupportsThinking(model string) bool {
|
|
return dashscopeThinkingModels[p.resolveModel(model)]
|
|
}
|
|
|
|
// applyThinkingGuard maps thinking_level to DashScope-specific params
|
|
// (enable_thinking / thinking_budget) only when the model supports it.
|
|
// Returns the (possibly mutated) request. Shared by Chat and ChatStream.
|
|
func (p *DashScopeProvider) applyThinkingGuard(req ChatRequest) ChatRequest {
|
|
level, ok := req.Options[OptThinkingLevel].(string)
|
|
if !ok || level == "" || level == "off" {
|
|
return req
|
|
}
|
|
|
|
if p.ModelSupportsThinking(req.Model) {
|
|
// Clone Options to avoid mutating caller's map
|
|
opts := make(map[string]any, len(req.Options)+2)
|
|
maps.Copy(opts, req.Options)
|
|
opts[OptEnableThinking] = true
|
|
opts[OptThinkingBudget] = dashscopeThinkingBudget(level)
|
|
delete(opts, OptThinkingLevel) // don't pass generic key to OpenAI buildRequestBody
|
|
req.Options = opts
|
|
} else {
|
|
slog.Debug("dashscope: model does not support thinking, skipping enable_thinking",
|
|
"model", p.resolveModel(req.Model), "requested_level", level)
|
|
}
|
|
|
|
return req
|
|
}
|
|
|
|
// Chat overrides OpenAIProvider.Chat to apply the per-model thinking guard.
|
|
func (p *DashScopeProvider) Chat(ctx context.Context, req ChatRequest) (*ChatResponse, error) {
|
|
return p.OpenAIProvider.Chat(ctx, p.applyThinkingGuard(req))
|
|
}
|
|
|
|
// ChatStream handles DashScope's limitation: tools + streaming cannot coexist.
|
|
// When tools are present, falls back to non-streaming Chat() and synthesizes
|
|
// chunk callbacks for the caller.
|
|
func (p *DashScopeProvider) ChatStream(ctx context.Context, req ChatRequest, onChunk func(StreamChunk)) (*ChatResponse, error) {
|
|
req = p.applyThinkingGuard(req)
|
|
|
|
if len(req.Tools) > 0 {
|
|
slog.Debug("dashscope: tools present, falling back to non-streaming Chat")
|
|
resp, err := p.OpenAIProvider.Chat(ctx, req)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
if onChunk != nil {
|
|
if resp.Thinking != "" {
|
|
onChunk(StreamChunk{Thinking: resp.Thinking})
|
|
}
|
|
if resp.Content != "" {
|
|
onChunk(StreamChunk{Content: resp.Content})
|
|
}
|
|
onChunk(StreamChunk{Done: true})
|
|
}
|
|
return resp, nil
|
|
}
|
|
return p.OpenAIProvider.ChatStream(ctx, req, onChunk)
|
|
}
|
|
|
|
// dashscopeThinkingBudget maps a thinking level to a DashScope thinking_budget value.
|
|
func dashscopeThinkingBudget(level string) int {
|
|
switch level {
|
|
case "low":
|
|
return 4096
|
|
case "medium":
|
|
return 16384
|
|
case "high":
|
|
return 32768
|
|
default:
|
|
return 16384
|
|
}
|
|
}
|