mirror of
https://github.com/tiennm99/goclaw.git
synced 2026-09-20 04:23:29 +00:00
fix(agent/title): disable thinking and raise max_tokens for Gemini
Gemini 2.5/3 default to high thinking via OpenAI-compat, consuming the entire max_tokens budget and truncating titles to a single word (and adding latency). Title generation is a trivial task that does not benefit from reasoning, so set OptThinkingLevel="off" and bump max_tokens from 50 to 256 for headroom across reasoning-capable providers. Also update mapGeminiReasoningEffort to forward "off" as "low" (the minimum effort all Gemini models accept via OpenAI-compat), since not forwarding causes the server to fall back to "high".
This commit is contained in:
1 parent
5590b94994
commit
86d564967f
3 files changed
+16
-5
No files matched your search
@@ -26,8 +26,15 @@ func GenerateTitle(ctx context.Context, provider providers.Provider, model, user
|
||||
},
|
||||
Model: model,
|
||||
Options: map[string]any{
|
||||
providers.OptMaxTokens: 50,
|
||||
providers.OptTemperature: 0.3,
|
||||
// Larger budget: thinking-capable models (Gemini 2.5/3, GPT-5 reasoning)
|
||||
// can consume output tokens on reasoning traces. 256 leaves room for a
|
||||
// 15-word title even when the provider allocates some budget to thinking.
|
||||
providers.OptMaxTokens: 256,
|
||||
providers.OptTemperature: 0.3,
|
||||
// Disable extended thinking for title generation — it's a trivial task
|
||||
// that doesn't benefit from reasoning and defaults (esp. Gemini's "high")
|
||||
// otherwise eat the entire max_tokens budget, truncating the title to 1 word.
|
||||
providers.OptThinkingLevel: "off",
|
||||
},
|
||||
})
|
||||
if err != nil {
|
||||
|
||||
@@ -20,7 +20,7 @@ func TestBuildRequestBody_GeminiForwardsReasoningEffort(t *testing.T) {
|
||||
{"minimal_verbatim", "minimal", "minimal", true},
|
||||
{"high_verbatim", "high", "high", true},
|
||||
{"medium_maps_to_high", "medium", "high", true},
|
||||
{"off_omitted", "off", "", false},
|
||||
{"off_maps_to_low", "off", "low", true},
|
||||
{"empty_omitted", "", "", false},
|
||||
{"unknown_omitted", "garbage", "", false},
|
||||
}
|
||||
|
||||
@@ -251,14 +251,18 @@ func (p *OpenAIProvider) isGeminiRoute(model string) bool {
|
||||
|
||||
// mapGeminiReasoningEffort returns (value, shouldForward). Gemini 3 Preview
|
||||
// rejects "medium" with HTTP 400, so we map it to the nearest valid option.
|
||||
// "off" and unknown values do not forward — respect user intent (off=disable)
|
||||
// and avoid injecting defaults the model might otherwise apply conservatively.
|
||||
// "off" maps to "low" (the minimum effort accepted by all Gemini models via
|
||||
// OpenAI-compat). Forwarding is required because Gemini's default is "high",
|
||||
// which consumes the entire max_tokens budget on reasoning traces and leaves
|
||||
// no room for the response. Unknown values do not forward.
|
||||
func mapGeminiReasoningEffort(level string) (string, bool) {
|
||||
switch level {
|
||||
case "low", "minimal", "high":
|
||||
return level, true
|
||||
case "medium":
|
||||
return "high", true
|
||||
case "off":
|
||||
return "low", true
|
||||
default:
|
||||
return "", false
|
||||
}
|
||||
|
||||
Reference in new issue
Block a user