Files
goclaw/internal/providers/dashscope_usage_test.go
T
Plateau Nguyen 0e994f959c feat(providers): explicit prompt cache for DashScope/Qwen (#1127)
* feat(providers): explicit prompt cache for DashScope/Qwen

Extend OpenAI-compat path with Anthropic-style cache_control:ephemeral
inline blocks for Alibaba DashScope endpoints (provider_type=bailian or
URL contains "dashscope"). Verified live: qwen3.6-plus / qwen3.5-plus /
qwen3-coder-plus all return 99.7% cache hit rate on 6K-token prefix,
yielding ~89% token cost reduction on cached prefix (90% Alibaba discount).

- isDashScope(): 3-source detection (URL + providerType + name) handles
  reverse-proxied endpoints; covers both "dashscope" and "bailian"
- buildRequestBody wraps system content with SplitSystemPromptForCache
  (exported from anthropic_request.go) using <!-- GOCLAW_CACHE_BOUNDARY -->
- Tool prefix cache: cache_control on last tool definition, with 4-marker
  budget guard
- Parse cache_creation_input_tokens from prompt_tokens_details into
  Usage.CacheCreationTokens; propagates via existing span metadata writers
- Runtime escape hatch: GOCLAW_DISABLE_DASHSCOPE_CACHE=true
- Live integration smoke test (build tag integration, env-gated)

* chore(providers): polish per PR #1127 review

- Use strings.Repeat instead of custom repeat() helper in smoke test
- Clarify BuildRequestBodyForTest is test-only, not public API

* feat(providers): enable thinking for Qwen 3.7/3.6, observe cache in smoke test

Qwen3.7-plus and qwen3.6-plus support deep thinking but were missing from
dashscopeThinkingModels, so enable_thinking/thinking_budget was silently
skipped. Add both to the whitelist and test.

Add qwen3.7-plus to the cache smoke test as cache-optional: Alibaba's
context-cache doc does not yet list 3.7/3.6 for explicit cache, and the
cache wrap is a safe no-op when unsupported, so a no-cache result is logged
rather than failed to avoid a flaky live assertion.

Claude-Session: https://claude.ai/code/session_01X3jkrc7N3ar8ZGzUJWyNS5

* fix(providers): preserve DashScope detection for proxy routes
2026-06-19 10:59:02 +07:00

58 lines
1.5 KiB
Go

package providers
import (
"encoding/json"
"testing"
)
func TestOpenAIUsage_DashScopeCacheHit_Unmarshal(t *testing.T) {
raw := `{
"prompt_tokens": 2318,
"completion_tokens": 195,
"total_tokens": 2513,
"prompt_tokens_details": {
"text_tokens": 2318,
"cache_creation_input_tokens": 0,
"cached_tokens": 2304
}
}`
var u openAIUsage
if err := json.Unmarshal([]byte(raw), &u); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if u.PromptTokensDetails.CachedTokens != 2304 {
t.Errorf("cached_tokens: got %d want 2304", u.PromptTokensDetails.CachedTokens)
}
if u.PromptTokensDetails.CacheCreationInputTokens != 0 {
t.Errorf("cache_creation: got %d want 0", u.PromptTokensDetails.CacheCreationInputTokens)
}
}
func TestOpenAIUsage_DashScopeCacheCreate_Unmarshal(t *testing.T) {
raw := `{
"prompt_tokens": 2318,
"prompt_tokens_details": {
"cache_creation_input_tokens": 2304,
"cached_tokens": 0
}
}`
var u openAIUsage
if err := json.Unmarshal([]byte(raw), &u); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if u.PromptTokensDetails.CacheCreationInputTokens != 2304 {
t.Errorf("got %d want 2304", u.PromptTokensDetails.CacheCreationInputTokens)
}
}
func TestOpenAIUsage_NoDetails_OK(t *testing.T) {
raw := `{"prompt_tokens": 100, "completion_tokens": 20, "total_tokens": 120}`
var u openAIUsage
if err := json.Unmarshal([]byte(raw), &u); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if u.PromptTokensDetails != nil {
t.Errorf("expected nil PromptTokensDetails, got %+v", u.PromptTokensDetails)
}
}