Files
goclaw/internal/providers/dashscope.go
T
Plateau Nguyen 0e994f959c feat(providers): explicit prompt cache for DashScope/Qwen (#1127)
* feat(providers): explicit prompt cache for DashScope/Qwen

Extend OpenAI-compat path with Anthropic-style cache_control:ephemeral
inline blocks for Alibaba DashScope endpoints (provider_type=bailian or
URL contains "dashscope"). Verified live: qwen3.6-plus / qwen3.5-plus /
qwen3-coder-plus all return 99.7% cache hit rate on 6K-token prefix,
yielding ~89% token cost reduction on cached prefix (90% Alibaba discount).

- isDashScope(): 3-source detection (URL + providerType + name) handles
  reverse-proxied endpoints; covers both "dashscope" and "bailian"
- buildRequestBody wraps system content with SplitSystemPromptForCache
  (exported from anthropic_request.go) using <!-- GOCLAW_CACHE_BOUNDARY -->
- Tool prefix cache: cache_control on last tool definition, with 4-marker
  budget guard
- Parse cache_creation_input_tokens from prompt_tokens_details into
  Usage.CacheCreationTokens; propagates via existing span metadata writers
- Runtime escape hatch: GOCLAW_DISABLE_DASHSCOPE_CACHE=true
- Live integration smoke test (build tag integration, env-gated)

* chore(providers): polish per PR #1127 review

- Use strings.Repeat instead of custom repeat() helper in smoke test
- Clarify BuildRequestBodyForTest is test-only, not public API

* feat(providers): enable thinking for Qwen 3.7/3.6, observe cache in smoke test

Qwen3.7-plus and qwen3.6-plus support deep thinking but were missing from
dashscopeThinkingModels, so enable_thinking/thinking_budget was silently
skipped. Add both to the whitelist and test.

Add qwen3.7-plus to the cache smoke test as cache-optional: Alibaba's
context-cache doc does not yet list 3.7/3.6 for explicit cache, and the
cache wrap is a safe no-op when unsupported, so a no-cache result is logged
rather than failed to avoid a flaky live assertion.

Claude-Session: https://claude.ai/code/session_01X3jkrc7N3ar8ZGzUJWyNS5

* fix(providers): preserve DashScope detection for proxy routes
2026-06-19 10:59:02 +07:00

147 lines
4.7 KiB
Go

package providers
import (
"context"
"log/slog"
"maps"
)
const (
dashscopeDefaultBase = "https://dashscope-intl.aliyuncs.com/compatible-mode/v1"
dashscopeDefaultModel = "qwen3-max"
)
// dashscopeThinkingModels lists DashScope models that accept the
// enable_thinking / thinking_budget parameters (Qwen3 open-weight and Qwen3.5+ series).
// Models NOT in this set (e.g. qwen3-plus, qwen3-turbo) will silently
// skip thinking injection to avoid API "model not supported" errors.
var dashscopeThinkingModels = map[string]bool{
// Qwen3.7 / 3.6 series — deep thinking + vision
"qwen3.7-plus": true,
"qwen3.6-plus": true,
// Qwen3.5 series — thinking + vision
"qwen3.5-plus": true,
"qwen3.5-turbo": true,
// Qwen3 hosted
"qwen3-max": true,
// Qwen3 open-weight (available as hosted inference)
"qwen3-235b-a22b": true,
"qwen3-32b": true,
"qwen3-14b": true,
"qwen3-8b": true,
}
// DashScopeProvider wraps OpenAIProvider to handle DashScope-specific behaviors.
// Critical: DashScope does NOT support tools + streaming simultaneously.
// When tools are present, ChatStream falls back to non-streaming Chat().
type DashScopeProvider struct {
*OpenAIProvider
}
func NewDashScopeProvider(name, apiKey, apiBase, defaultModel string) *DashScopeProvider {
if apiBase == "" {
apiBase = dashscopeDefaultBase
}
if defaultModel == "" {
defaultModel = dashscopeDefaultModel
}
return &DashScopeProvider{
OpenAIProvider: NewOpenAIProvider(name, apiKey, apiBase, defaultModel).WithProviderType("dashscope"),
}
}
// Name is inherited from the embedded OpenAIProvider (returns the user-specified name).
func (p *DashScopeProvider) SupportsThinking() bool { return true }
// Capabilities implements CapabilitiesAware for pipeline code-path selection.
// StreamWithTools=false is THE critical difference: DashScope falls back to
// non-streaming when tools are present.
func (p *DashScopeProvider) Capabilities() ProviderCapabilities {
return ProviderCapabilities{
Streaming: true,
ToolCalling: true,
StreamWithTools: false,
Thinking: true,
Vision: true,
CacheControl: false,
MaxContextWindow: 128_000,
TokenizerID: "cl100k_base",
}
}
// ModelSupportsThinking implements ModelThinkingCapable.
// Returns true only for models that accept enable_thinking / thinking_budget.
func (p *DashScopeProvider) ModelSupportsThinking(model string) bool {
return dashscopeThinkingModels[p.resolveModel(model)]
}
// applyThinkingGuard maps thinking_level to DashScope-specific params
// (enable_thinking / thinking_budget) only when the model supports it.
// Returns the (possibly mutated) request. Shared by Chat and ChatStream.
func (p *DashScopeProvider) applyThinkingGuard(req ChatRequest) ChatRequest {
level, ok := req.Options[OptThinkingLevel].(string)
if !ok || level == "" || level == "off" {
return req
}
if p.ModelSupportsThinking(req.Model) {
// Clone Options to avoid mutating caller's map
opts := make(map[string]any, len(req.Options)+2)
maps.Copy(opts, req.Options)
opts[OptEnableThinking] = true
opts[OptThinkingBudget] = dashscopeThinkingBudget(level)
delete(opts, OptThinkingLevel) // don't pass generic key to OpenAI buildRequestBody
req.Options = opts
} else {
slog.Debug("dashscope: model does not support thinking, skipping enable_thinking",
"model", p.resolveModel(req.Model), "requested_level", level)
}
return req
}
// Chat overrides OpenAIProvider.Chat to apply the per-model thinking guard.
func (p *DashScopeProvider) Chat(ctx context.Context, req ChatRequest) (*ChatResponse, error) {
return p.OpenAIProvider.Chat(ctx, p.applyThinkingGuard(req))
}
// ChatStream handles DashScope's limitation: tools + streaming cannot coexist.
// When tools are present, falls back to non-streaming Chat() and synthesizes
// chunk callbacks for the caller.
func (p *DashScopeProvider) ChatStream(ctx context.Context, req ChatRequest, onChunk func(StreamChunk)) (*ChatResponse, error) {
req = p.applyThinkingGuard(req)
if len(req.Tools) > 0 {
slog.Debug("dashscope: tools present, falling back to non-streaming Chat")
resp, err := p.OpenAIProvider.Chat(ctx, req)
if err != nil {
return nil, err
}
if onChunk != nil {
if resp.Thinking != "" {
onChunk(StreamChunk{Thinking: resp.Thinking})
}
if resp.Content != "" {
onChunk(StreamChunk{Content: resp.Content})
}
onChunk(StreamChunk{Done: true})
}
return resp, nil
}
return p.OpenAIProvider.ChatStream(ctx, req, onChunk)
}
// dashscopeThinkingBudget maps a thinking level to a DashScope thinking_budget value.
func dashscopeThinkingBudget(level string) int {
switch level {
case "low":
return 4096
case "medium":
return 16384
case "high":
return 32768
default:
return 16384
}
}