From 39fde658d67c8f0fc6eda82244da3d2af402a445 Mon Sep 17 00:00:00 2001 From: Sergey Kozyrenko Date: Mon, 13 Jul 2026 12:04:03 +0700 Subject: [PATCH] fix(llms): default adaptive reasoning with no effort to high in one place WithAdaptiveReasoning with an empty effort was handled inconsistently: the Anthropic and Bedrock paths coerced it to high inline, while OpenAI and googleai read GetEffort/GetTokens, which returned none/zero and silently disabled reasoning. The same option produced opposite behavior per provider. Move the default into GetEffort and GetTokens so every provider resolves an empty adaptive effort to high, and drop the now-redundant inline coercions in the three Anthropic paths. Covered by GetEffort/GetTokens table cases and an OpenAI wire test asserting reasoning_effort=high is sent. Co-Authored-By: Claude Fable 5 --- llms/anthropic/anthropicllm.go | 8 ++--- .../bedrockclient/bedrockclient_converse.go | 5 +-- .../bedrockclient/provider_anthropic.go | 6 +--- .../reasoning_effort_passthrough_test.go | 34 +++++++++++++++++++ llms/options.go | 12 ++++++- llms/options_test.go | 18 ++++++++++ 6 files changed, 68 insertions(+), 15 deletions(-) diff --git a/llms/anthropic/anthropicllm.go b/llms/anthropic/anthropicllm.go index 06363c8a..cf159970 100644 --- a/llms/anthropic/anthropicllm.go +++ b/llms/anthropic/anthropicllm.go @@ -177,15 +177,13 @@ func generateMessagesContent(ctx context.Context, o *LLM, messages []llms.Messag // flag: adaptive-only generations reject budget thinking and vice versa, // so honor the caller's preference only where the model accepts it. if reasoning.ResolveClaudeAdaptive(model, opts.Reasoning.Adaptive) { - effort := opts.Reasoning.Effort - if effort == llms.ReasoningNone { - effort = llms.ReasoningHigh // match the server default instead of sending an empty output_config - } thinking = &anthropicclient.ThinkingPayload{ Type: "adaptive", Display: "summarized", } - outputConfig = &anthropicclient.OutputConfig{Effort: string(effort)} + outputConfig = &anthropicclient.OutputConfig{ + Effort: string(opts.Reasoning.GetEffort(opts.GetMaxTokens())), + } } else { thinking = &anthropicclient.ThinkingPayload{ Type: "enabled", diff --git a/llms/bedrock/internal/bedrockclient/bedrockclient_converse.go b/llms/bedrock/internal/bedrockclient/bedrockclient_converse.go index ba9a31f5..aea12fed 100644 --- a/llms/bedrock/internal/bedrockclient/bedrockclient_converse.go +++ b/llms/bedrock/internal/bedrockclient/bedrockclient_converse.go @@ -143,10 +143,7 @@ func (c *ConverseClient) buildConverseInput(input *ConverseInput) (*bedrockrunti if input.ReasoningConfig != nil { additionalModelFields := converseAdditionalModelRequestFields{} setAdaptive := func() { - effort := input.ReasoningConfig.Effort - if effort == llms.ReasoningNone { - effort = llms.ReasoningHigh // match the server default instead of sending an empty output_config - } + effort := input.ReasoningConfig.GetEffort(0) additionalModelFields.Thinking = &converseThinkingPayload{Type: "adaptive", Display: "summarized"} additionalModelFields.OutputConfig = &converseOutputConfig{Effort: string(effort)} // Adaptive models reject sampling params. diff --git a/llms/bedrock/internal/bedrockclient/provider_anthropic.go b/llms/bedrock/internal/bedrockclient/provider_anthropic.go index 75fee1bb..8955c5a5 100644 --- a/llms/bedrock/internal/bedrockclient/provider_anthropic.go +++ b/llms/bedrock/internal/bedrockclient/provider_anthropic.go @@ -670,12 +670,8 @@ func applyAnthropicReasoning( } setAdaptive := func() { - effort := cfg.Effort - if effort == llms.ReasoningNone { - effort = llms.ReasoningHigh // match the server default instead of sending an empty output_config - } input.Thinking = &anthropicThinkingPayload{Type: "adaptive", Display: "summarized"} - input.OutputConfig = &anthropicOutputConfig{Effort: string(effort)} + input.OutputConfig = &anthropicOutputConfig{Effort: string(cfg.GetEffort(maxTokens))} input.Temperature = 0 input.TopP = 0 input.TopK = 0 diff --git a/llms/openai/reasoning_effort_passthrough_test.go b/llms/openai/reasoning_effort_passthrough_test.go index 42ad3fff..403a8604 100644 --- a/llms/openai/reasoning_effort_passthrough_test.go +++ b/llms/openai/reasoning_effort_passthrough_test.go @@ -66,3 +66,37 @@ func TestReasoningEffortPassthroughToWire(t *testing.T) { }) } } + +// WithAdaptiveReasoning with no explicit effort must not silently disable +// reasoning here: it defaults to high, matching the Anthropic/Bedrock paths. +func TestAdaptiveReasoningNoEffortDefaultsToHighOnWire(t *testing.T) { + t.Parallel() + + const completion = `{"id":"x","object":"chat.completion","created":1,"model":"m",` + + `"choices":[{"index":0,"message":{"role":"assistant","content":"ok"},"finish_reason":"stop"}],` + + `"usage":{"prompt_tokens":1,"completion_tokens":1,"total_tokens":2}}` + + var body string + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + b, _ := io.ReadAll(r.Body) + body = string(b) + w.Header().Set("Content-Type", "application/json") + _, _ = io.WriteString(w, completion) + })) + defer srv.Close() + + llm, err := New(WithBaseURL(srv.URL), WithToken("test"), WithModel("reasoning-model")) + if err != nil { + t.Fatalf("New() error: %v", err) + } + if _, err := llm.GenerateContent( + context.Background(), + []llms.MessageContent{llms.TextParts(llms.ChatMessageTypeHuman, "hi")}, + llms.WithAdaptiveReasoning(llms.ReasoningNone), + ); err != nil { + t.Fatalf("GenerateContent() error: %v", err) + } + if !strings.Contains(body, `"reasoning_effort":"high"`) { + t.Fatalf("adaptive-no-effort must send reasoning_effort=high, got body: %s", body) + } +} diff --git a/llms/options.go b/llms/options.go index 58125b23..ecfcf763 100644 --- a/llms/options.go +++ b/llms/options.go @@ -91,6 +91,10 @@ func (r *ReasoningConfig) GetEffort(maxTokens int) ReasoningEffort { return r.Effort } + if r.Adaptive { + return ReasoningHigh // adaptive with no explicit effort defaults to high + } + if maxTokens <= 0 { maxTokens = 8192 } @@ -141,7 +145,13 @@ func (r *ReasoningConfig) GetTokens(maxTokens int) int { // no distinct token budget for xhigh/max; map them to the high budget. tokens = max(maxTokens/2, 4096) case ReasoningNone: - return 0 // disabled + if r.Adaptive { + // adaptive with no explicit effort defaults to the high budget, + // so providers that map reasoning to a token budget don't disable it. + tokens = max(maxTokens/2, 4096) + } else { + return 0 // disabled + } default: return -1 // error value to be handled on the server side } diff --git a/llms/options_test.go b/llms/options_test.go index ef482bd8..d405e50d 100644 --- a/llms/options_test.go +++ b/llms/options_test.go @@ -596,6 +596,18 @@ func TestReasoningConfig_GetEffort(t *testing.T) { maxTokens: maxTokens, expected: llms.ReasoningHigh, }, + { + name: "adaptive with no effort defaults to high", + config: &llms.ReasoningConfig{Adaptive: true}, + maxTokens: maxTokens, + expected: llms.ReasoningHigh, + }, + { + name: "adaptive with explicit effort keeps it", + config: &llms.ReasoningConfig{Adaptive: true, Effort: llms.ReasoningLow}, + maxTokens: maxTokens, + expected: llms.ReasoningLow, + }, { name: "xhigh effort passes through on the effort path", config: &llms.ReasoningConfig{Effort: llms.ReasoningXHigh}, @@ -680,6 +692,12 @@ func TestReasoningConfig_GetTokens(t *testing.T) { maxTokens: maxTokens, expected: 0, }, + { + name: "adaptive with no effort defaults to high budget", + config: &llms.ReasoningConfig{Adaptive: true}, + maxTokens: maxTokens, + expected: max(maxTokens/2, 4096), + }, { name: "config with explicit tokens", config: &llms.ReasoningConfig{Tokens: 3000},