mirror of
https://github.com/tiennm99/goclaw.git
synced 2026-10-04 14:13:15 +00:00
* feat(providers): explicit prompt cache for DashScope/Qwen Extend OpenAI-compat path with Anthropic-style cache_control:ephemeral inline blocks for Alibaba DashScope endpoints (provider_type=bailian or URL contains "dashscope"). Verified live: qwen3.6-plus / qwen3.5-plus / qwen3-coder-plus all return 99.7% cache hit rate on 6K-token prefix, yielding ~89% token cost reduction on cached prefix (90% Alibaba discount). - isDashScope(): 3-source detection (URL + providerType + name) handles reverse-proxied endpoints; covers both "dashscope" and "bailian" - buildRequestBody wraps system content with SplitSystemPromptForCache (exported from anthropic_request.go) using <!-- GOCLAW_CACHE_BOUNDARY --> - Tool prefix cache: cache_control on last tool definition, with 4-marker budget guard - Parse cache_creation_input_tokens from prompt_tokens_details into Usage.CacheCreationTokens; propagates via existing span metadata writers - Runtime escape hatch: GOCLAW_DISABLE_DASHSCOPE_CACHE=true - Live integration smoke test (build tag integration, env-gated) * chore(providers): polish per PR #1127 review - Use strings.Repeat instead of custom repeat() helper in smoke test - Clarify BuildRequestBodyForTest is test-only, not public API * feat(providers): enable thinking for Qwen 3.7/3.6, observe cache in smoke test Qwen3.7-plus and qwen3.6-plus support deep thinking but were missing from dashscopeThinkingModels, so enable_thinking/thinking_budget was silently skipped. Add both to the whitelist and test. Add qwen3.7-plus to the cache smoke test as cache-optional: Alibaba's context-cache doc does not yet list 3.7/3.6 for explicit cache, and the cache wrap is a safe no-op when unsupported, so a no-cache result is logged rather than failed to avoid a flaky live assertion. Claude-Session: https://claude.ai/code/session_01X3jkrc7N3ar8ZGzUJWyNS5 * fix(providers): preserve DashScope detection for proxy routes
58 lines
1.5 KiB
Go
58 lines
1.5 KiB
Go
package providers
|
|
|
|
import (
|
|
"encoding/json"
|
|
"testing"
|
|
)
|
|
|
|
func TestOpenAIUsage_DashScopeCacheHit_Unmarshal(t *testing.T) {
|
|
raw := `{
|
|
"prompt_tokens": 2318,
|
|
"completion_tokens": 195,
|
|
"total_tokens": 2513,
|
|
"prompt_tokens_details": {
|
|
"text_tokens": 2318,
|
|
"cache_creation_input_tokens": 0,
|
|
"cached_tokens": 2304
|
|
}
|
|
}`
|
|
var u openAIUsage
|
|
if err := json.Unmarshal([]byte(raw), &u); err != nil {
|
|
t.Fatalf("unmarshal: %v", err)
|
|
}
|
|
if u.PromptTokensDetails.CachedTokens != 2304 {
|
|
t.Errorf("cached_tokens: got %d want 2304", u.PromptTokensDetails.CachedTokens)
|
|
}
|
|
if u.PromptTokensDetails.CacheCreationInputTokens != 0 {
|
|
t.Errorf("cache_creation: got %d want 0", u.PromptTokensDetails.CacheCreationInputTokens)
|
|
}
|
|
}
|
|
|
|
func TestOpenAIUsage_DashScopeCacheCreate_Unmarshal(t *testing.T) {
|
|
raw := `{
|
|
"prompt_tokens": 2318,
|
|
"prompt_tokens_details": {
|
|
"cache_creation_input_tokens": 2304,
|
|
"cached_tokens": 0
|
|
}
|
|
}`
|
|
var u openAIUsage
|
|
if err := json.Unmarshal([]byte(raw), &u); err != nil {
|
|
t.Fatalf("unmarshal: %v", err)
|
|
}
|
|
if u.PromptTokensDetails.CacheCreationInputTokens != 2304 {
|
|
t.Errorf("got %d want 2304", u.PromptTokensDetails.CacheCreationInputTokens)
|
|
}
|
|
}
|
|
|
|
func TestOpenAIUsage_NoDetails_OK(t *testing.T) {
|
|
raw := `{"prompt_tokens": 100, "completion_tokens": 20, "total_tokens": 120}`
|
|
var u openAIUsage
|
|
if err := json.Unmarshal([]byte(raw), &u); err != nil {
|
|
t.Fatalf("unmarshal: %v", err)
|
|
}
|
|
if u.PromptTokensDetails != nil {
|
|
t.Errorf("expected nil PromptTokensDetails, got %+v", u.PromptTokensDetails)
|
|
}
|
|
}
|