fix(agent/title): disable thinking and raise max_tokens for Gemini

Gemini 2.5/3 default to high thinking via OpenAI-compat, consuming the
entire max_tokens budget and truncating titles to a single word (and
adding latency). Title generation is a trivial task that does not
benefit from reasoning, so set OptThinkingLevel="off" and bump
max_tokens from 50 to 256 for headroom across reasoning-capable
providers.

Also update mapGeminiReasoningEffort to forward "off" as "low" (the
minimum effort all Gemini models accept via OpenAI-compat), since not
forwarding causes the server to fall back to "high".
This commit is contained in:
viettranx committed 2026-04-15 13:37:37 +07:00
1 parent 5590b94994
commit 86d564967f
3 files changed
+16 -5

No files matched your search

+9 -2
View File
@@ -26,8 +26,15 @@ func GenerateTitle(ctx context.Context, provider providers.Provider, model, user
},
Model: model,
Options: map[string]any{
providers.OptMaxTokens: 50,
providers.OptTemperature: 0.3,
// Larger budget: thinking-capable models (Gemini 2.5/3, GPT-5 reasoning)
// can consume output tokens on reasoning traces. 256 leaves room for a
// 15-word title even when the provider allocates some budget to thinking.
providers.OptMaxTokens: 256,
providers.OptTemperature: 0.3,
// Disable extended thinking for title generation — it's a trivial task
// that doesn't benefit from reasoning and defaults (esp. Gemini's "high")
// otherwise eat the entire max_tokens budget, truncating the title to 1 word.
providers.OptThinkingLevel: "off",
},
})
if err != nil {
@@ -20,7 +20,7 @@ func TestBuildRequestBody_GeminiForwardsReasoningEffort(t *testing.T) {
{"minimal_verbatim", "minimal", "minimal", true},
{"high_verbatim", "high", "high", true},
{"medium_maps_to_high", "medium", "high", true},
{"off_omitted", "off", "", false},
{"off_maps_to_low", "off", "low", true},
{"empty_omitted", "", "", false},
{"unknown_omitted", "garbage", "", false},
}
+6 -2
View File
@@ -251,14 +251,18 @@ func (p *OpenAIProvider) isGeminiRoute(model string) bool {
// mapGeminiReasoningEffort returns (value, shouldForward). Gemini 3 Preview
// rejects "medium" with HTTP 400, so we map it to the nearest valid option.
// "off" and unknown values do not forward — respect user intent (off=disable)
// and avoid injecting defaults the model might otherwise apply conservatively.
// "off" maps to "low" (the minimum effort accepted by all Gemini models via
// OpenAI-compat). Forwarding is required because Gemini's default is "high",
// which consumes the entire max_tokens budget on reasoning traces and leaves
// no room for the response. Unknown values do not forward.
func mapGeminiReasoningEffort(level string) (string, bool) {
switch level {
case "low", "minimal", "high":
return level, true
case "medium":
return "high", true
case "off":
return "low", true
default:
return "", false
}