backend: make reasoningEffort the primary knob with documented thresholds
reasoningEffort (low|medium|high) is now the single human-facing control for model deliberation. Backends that speak a discrete level (OpenAI, OpenRouter, Copilot, Gemini, Kiro) send it verbatim; Anthropic, which needs a numeric budget, translates via documented thresholds (low=4096, medium=8192, high=16384). thinkingBudget remains an advanced explicit override that wins when set. - Add EffortLow/Medium/High, threshold consts, ValidReasoningEffort, EffortThinkingBudget, and GenerationParams.ResolvedThinkingBudget. - Anthropic: derive budget from effort, grow max_tokens above the budget (Anthropic requires budget < max_tokens), and force temperature=1 when thinking is enabled (both are Anthropic requirements). - Validate reasoningEffort at config load so typos fail loudly instead of silently no-opping. - theo: drop explicit thinkingBudget/inflated maxTokens; reasoningEffort high now implies the 16384 budget. - Document thresholds and per-backend behavior in data/agents/README.md; fix stale autoLoad/maxSteps references. - Add unit tests for the mapping and Anthropic thinking behavior.
This commit is contained in:
parent
a9fe03eb71
commit
f3933e21e7
|
|
@ -7,6 +7,7 @@ package agent
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
|
"fmt"
|
||||||
"io"
|
"io"
|
||||||
|
|
||||||
"ollie/cmd/olliesrv/internal/backend"
|
"ollie/cmd/olliesrv/internal/backend"
|
||||||
|
|
@ -47,5 +48,8 @@ func Load(r io.Reader) (*AgentConfig, error) {
|
||||||
if err := json.NewDecoder(r).Decode(&cfg); err != nil {
|
if err := json.NewDecoder(r).Decode(&cfg); err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
if !backend.ValidReasoningEffort(cfg.ReasoningEffort) {
|
||||||
|
return nil, fmt.Errorf("invalid reasoningEffort %q: want one of low, medium, high", cfg.ReasoningEffort)
|
||||||
|
}
|
||||||
return &cfg, nil
|
return &cfg, nil
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -136,19 +136,35 @@ func (b *AnthropicBackend) ChatStream(ctx context.Context, messages []Message, t
|
||||||
maxTokens = anthropicDefaultMaxTokens
|
maxTokens = anthropicDefaultMaxTokens
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Resolve extended-thinking budget from the explicit budget or the
|
||||||
|
// reasoning-effort level. Anthropic requires budget_tokens < max_tokens
|
||||||
|
// (thinking tokens are drawn from the output budget), so ensure headroom
|
||||||
|
// for the actual reply by growing max_tokens when necessary.
|
||||||
|
thinkingBudget := params.ResolvedThinkingBudget()
|
||||||
|
if thinkingBudget > 0 && maxTokens <= thinkingBudget {
|
||||||
|
maxTokens = thinkingBudget + anthropicDefaultMaxTokens
|
||||||
|
}
|
||||||
|
|
||||||
|
// Extended thinking requires temperature=1; any other value is rejected.
|
||||||
|
temperature := params.Temperature
|
||||||
|
if thinkingBudget > 0 {
|
||||||
|
one := 1.0
|
||||||
|
temperature = &one
|
||||||
|
}
|
||||||
|
|
||||||
areq := anthropicRequest{
|
areq := anthropicRequest{
|
||||||
Model: model,
|
Model: model,
|
||||||
MaxTokens: maxTokens,
|
MaxTokens: maxTokens,
|
||||||
System: systemBlocks,
|
System: systemBlocks,
|
||||||
Messages: wireMessages,
|
Messages: wireMessages,
|
||||||
Stream: true,
|
Stream: true,
|
||||||
Temperature: params.Temperature,
|
Temperature: temperature,
|
||||||
TopP: params.TopP,
|
TopP: params.TopP,
|
||||||
TopK: params.TopK,
|
TopK: params.TopK,
|
||||||
StopSeqs: params.Stop,
|
StopSeqs: params.Stop,
|
||||||
}
|
}
|
||||||
if params.ThinkingBudget > 0 {
|
if thinkingBudget > 0 {
|
||||||
areq.Thinking = &anthropicThinking{Type: "enabled", BudgetTokens: params.ThinkingBudget}
|
areq.Thinking = &anthropicThinking{Type: "enabled", BudgetTokens: thinkingBudget}
|
||||||
}
|
}
|
||||||
for _, t := range tools {
|
for _, t := range tools {
|
||||||
schema := t.Parameters
|
schema := t.Parameters
|
||||||
|
|
@ -171,7 +187,7 @@ func (b *AnthropicBackend) ChatStream(ctx context.Context, messages []Message, t
|
||||||
httpReq.Header.Set("Content-Type", "application/json")
|
httpReq.Header.Set("Content-Type", "application/json")
|
||||||
httpReq.Header.Set("X-Api-Key", b.apiKey)
|
httpReq.Header.Set("X-Api-Key", b.apiKey)
|
||||||
httpReq.Header.Set("Anthropic-Version", "2023-06-01")
|
httpReq.Header.Set("Anthropic-Version", "2023-06-01")
|
||||||
if params.ThinkingBudget > 0 {
|
if thinkingBudget > 0 {
|
||||||
httpReq.Header.Set("Anthropic-Beta", "interleaved-thinking-2025-05-14")
|
httpReq.Header.Set("Anthropic-Beta", "interleaved-thinking-2025-05-14")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -165,6 +165,63 @@ type GenerationParams struct {
|
||||||
CWD string `json:"-"` // Current working directory
|
CWD string `json:"-"` // Current working directory
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Reasoning effort levels. ReasoningEffort is the primary, human-facing knob
|
||||||
|
// for controlling how much a reasoning-capable model deliberates before
|
||||||
|
// answering. Backends that speak a discrete effort level (OpenAI-family
|
||||||
|
// reasoning_effort) send these strings verbatim. Backends that require a
|
||||||
|
// numeric thinking-token budget (Anthropic extended thinking) translate the
|
||||||
|
// level via EffortThinkingBudget.
|
||||||
|
const (
|
||||||
|
EffortLow = "low"
|
||||||
|
EffortMedium = "medium"
|
||||||
|
EffortHigh = "high"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Thinking-token budgets that each reasoning effort level maps to, for backends
|
||||||
|
// that require a numeric budget rather than a discrete level. These are the
|
||||||
|
// single documented source of truth for the thresholds; keep data/agents/README.md
|
||||||
|
// in sync.
|
||||||
|
const (
|
||||||
|
thinkingBudgetLow = 4096
|
||||||
|
thinkingBudgetMedium = 8192
|
||||||
|
thinkingBudgetHigh = 16384
|
||||||
|
)
|
||||||
|
|
||||||
|
// ValidReasoningEffort reports whether s is a recognised effort level. The
|
||||||
|
// empty string is valid and means "unset" (use the API default).
|
||||||
|
func ValidReasoningEffort(s string) bool {
|
||||||
|
switch s {
|
||||||
|
case "", EffortLow, EffortMedium, EffortHigh:
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// EffortThinkingBudget maps a reasoning effort level to a thinking-token budget.
|
||||||
|
// Returns 0 for an unset or unrecognised level, meaning "no explicit budget".
|
||||||
|
func EffortThinkingBudget(effort string) int {
|
||||||
|
switch effort {
|
||||||
|
case EffortLow:
|
||||||
|
return thinkingBudgetLow
|
||||||
|
case EffortMedium:
|
||||||
|
return thinkingBudgetMedium
|
||||||
|
case EffortHigh:
|
||||||
|
return thinkingBudgetHigh
|
||||||
|
}
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
// ResolvedThinkingBudget returns the thinking-token budget to use for backends
|
||||||
|
// that require a numeric budget. An explicit ThinkingBudget always wins; when
|
||||||
|
// it is unset (0), the budget is derived from ReasoningEffort. Returns 0 when
|
||||||
|
// neither is set, meaning the backend should not enable extended thinking.
|
||||||
|
func (p GenerationParams) ResolvedThinkingBudget() int {
|
||||||
|
if p.ThinkingBudget > 0 {
|
||||||
|
return p.ThinkingBudget
|
||||||
|
}
|
||||||
|
return EffortThinkingBudget(p.ReasoningEffort)
|
||||||
|
}
|
||||||
|
|
||||||
// Backend is the interface all LLM providers must implement.
|
// Backend is the interface all LLM providers must implement.
|
||||||
// Streaming is the only supported mode; backends that wrap blocking APIs
|
// Streaming is the only supported mode; backends that wrap blocking APIs
|
||||||
// should implement ChatStream as a single-event stream.
|
// should implement ChatStream as a single-event stream.
|
||||||
|
|
|
||||||
|
|
@ -0,0 +1,145 @@
|
||||||
|
package backend
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"io"
|
||||||
|
"net/http"
|
||||||
|
"net/http/httptest"
|
||||||
|
"net/url"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestValidReasoningEffort(t *testing.T) {
|
||||||
|
cases := map[string]bool{
|
||||||
|
"": true, // unset means "use default"
|
||||||
|
"low": true,
|
||||||
|
"medium": true,
|
||||||
|
"high": true,
|
||||||
|
"LOW": false, // case-sensitive; profiles use lowercase
|
||||||
|
"max": false,
|
||||||
|
"hihg": false, // typo must be rejected, not silently ignored
|
||||||
|
}
|
||||||
|
for in, want := range cases {
|
||||||
|
if got := ValidReasoningEffort(in); got != want {
|
||||||
|
t.Errorf("ValidReasoningEffort(%q) = %v, want %v", in, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEffortThinkingBudget(t *testing.T) {
|
||||||
|
cases := map[string]int{
|
||||||
|
EffortLow: 4096,
|
||||||
|
EffortMedium: 8192,
|
||||||
|
EffortHigh: 16384,
|
||||||
|
"": 0,
|
||||||
|
"bogus": 0,
|
||||||
|
}
|
||||||
|
for in, want := range cases {
|
||||||
|
if got := EffortThinkingBudget(in); got != want {
|
||||||
|
t.Errorf("EffortThinkingBudget(%q) = %d, want %d", in, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestResolvedThinkingBudget(t *testing.T) {
|
||||||
|
// Explicit budget wins over effort.
|
||||||
|
p := GenerationParams{ThinkingBudget: 5000, ReasoningEffort: EffortHigh}
|
||||||
|
if got := p.ResolvedThinkingBudget(); got != 5000 {
|
||||||
|
t.Errorf("explicit budget: got %d, want 5000", got)
|
||||||
|
}
|
||||||
|
// Derived from effort when budget unset.
|
||||||
|
p = GenerationParams{ReasoningEffort: EffortHigh}
|
||||||
|
if got := p.ResolvedThinkingBudget(); got != 16384 {
|
||||||
|
t.Errorf("derived budget: got %d, want 16384", got)
|
||||||
|
}
|
||||||
|
// Neither set → 0 (no thinking).
|
||||||
|
p = GenerationParams{}
|
||||||
|
if got := p.ResolvedThinkingBudget(); got != 0 {
|
||||||
|
t.Errorf("unset: got %d, want 0", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// captureAnthropicRequest drives AnthropicBackend.ChatStream against a test
|
||||||
|
// server and returns the decoded request body.
|
||||||
|
func captureAnthropicRequest(t *testing.T, params GenerationParams) anthropicRequest {
|
||||||
|
t.Helper()
|
||||||
|
var raw []byte
|
||||||
|
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
raw, _ = io.ReadAll(r.Body)
|
||||||
|
w.Header().Set("Content-Type", "text/event-stream")
|
||||||
|
// Minimal well-formed stream so the reader terminates cleanly.
|
||||||
|
w.Write([]byte("event: message_stop\ndata: {\"type\":\"message_stop\"}\n\n"))
|
||||||
|
}))
|
||||||
|
defer srv.Close()
|
||||||
|
|
||||||
|
b, err := NewAnthropic("key")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
u, _ := url.Parse(srv.URL)
|
||||||
|
b.baseURL = u
|
||||||
|
|
||||||
|
ch, err := b.ChatStream(context.Background(), []Message{{Role: "user", Content: "audit this"}}, nil, params)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
for range ch { //nolint:revive
|
||||||
|
}
|
||||||
|
|
||||||
|
var areq anthropicRequest
|
||||||
|
if err := json.Unmarshal(raw, &areq); err != nil {
|
||||||
|
t.Fatalf("decode request: %v (body=%s)", err, raw)
|
||||||
|
}
|
||||||
|
return areq
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAnthropic_EffortEnablesThinking(t *testing.T) {
|
||||||
|
temp := 0.3
|
||||||
|
areq := captureAnthropicRequest(t, GenerationParams{
|
||||||
|
ReasoningEffort: EffortHigh,
|
||||||
|
MaxTokens: 8192,
|
||||||
|
Temperature: &temp,
|
||||||
|
})
|
||||||
|
|
||||||
|
if areq.Thinking == nil {
|
||||||
|
t.Fatal("thinking not enabled for reasoningEffort=high")
|
||||||
|
}
|
||||||
|
if areq.Thinking.BudgetTokens != 16384 {
|
||||||
|
t.Errorf("budget_tokens = %d, want 16384", areq.Thinking.BudgetTokens)
|
||||||
|
}
|
||||||
|
// max_tokens must be grown above the thinking budget (budget < max_tokens).
|
||||||
|
if areq.MaxTokens <= areq.Thinking.BudgetTokens {
|
||||||
|
t.Errorf("max_tokens (%d) must exceed thinking budget (%d)", areq.MaxTokens, areq.Thinking.BudgetTokens)
|
||||||
|
}
|
||||||
|
// Extended thinking requires temperature=1.
|
||||||
|
if areq.Temperature == nil || *areq.Temperature != 1.0 {
|
||||||
|
t.Errorf("temperature = %v, want 1.0 when thinking enabled", areq.Temperature)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAnthropic_NoEffortNoThinking(t *testing.T) {
|
||||||
|
temp := 0.3
|
||||||
|
areq := captureAnthropicRequest(t, GenerationParams{MaxTokens: 8192, Temperature: &temp})
|
||||||
|
if areq.Thinking != nil {
|
||||||
|
t.Errorf("thinking should be disabled when no effort/budget set")
|
||||||
|
}
|
||||||
|
// Temperature passes through untouched when thinking is off.
|
||||||
|
if areq.Temperature == nil || *areq.Temperature != 0.3 {
|
||||||
|
t.Errorf("temperature = %v, want 0.3", areq.Temperature)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAnthropic_ExplicitBudgetOverridesEffort(t *testing.T) {
|
||||||
|
areq := captureAnthropicRequest(t, GenerationParams{
|
||||||
|
ReasoningEffort: EffortLow,
|
||||||
|
ThinkingBudget: 20000,
|
||||||
|
MaxTokens: 8192,
|
||||||
|
})
|
||||||
|
if areq.Thinking == nil || areq.Thinking.BudgetTokens != 20000 {
|
||||||
|
t.Fatalf("budget_tokens = %v, want 20000 (explicit overrides effort)", areq.Thinking)
|
||||||
|
}
|
||||||
|
if areq.MaxTokens <= 20000 {
|
||||||
|
t.Errorf("max_tokens (%d) must exceed explicit budget (20000)", areq.MaxTokens)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
@ -8,10 +8,10 @@ Each `.json` file defines an agent configuration. Agent configs are installed to
|
||||||
{
|
{
|
||||||
"prompt": ["$XDG_CONFIG_HOME/ollie/prompts/agent-coding.md"],
|
"prompt": ["$XDG_CONFIG_HOME/ollie/prompts/agent-coding.md"],
|
||||||
"userPrompts": ["$XDG_CONFIG_HOME/ollie/prompts/rules.md"],
|
"userPrompts": ["$XDG_CONFIG_HOME/ollie/prompts/rules.md"],
|
||||||
"autoLoad": ["shell", "file_read", "file_edit", "file_glob", "file_grep"],
|
"tools": ["shell", "file_read", "file_edit", "file_glob", "file_grep"],
|
||||||
"maxSteps": 25,
|
|
||||||
"temperature": 0.5,
|
"temperature": 0.5,
|
||||||
"maxTokens": 16384,
|
"maxTokens": 16384,
|
||||||
|
"reasoningEffort": "medium",
|
||||||
"backend": "openrouter",
|
"backend": "openrouter",
|
||||||
"model": "deepseek/deepseek-v4-flash",
|
"model": "deepseek/deepseek-v4-flash",
|
||||||
"systemPrompt": "/path/to/override.md",
|
"systemPrompt": "/path/to/override.md",
|
||||||
|
|
@ -23,15 +23,51 @@ Each `.json` file defines an agent configuration. Agent configs are installed to
|
||||||
|-------|---------|
|
|-------|---------|
|
||||||
| `prompt` | File paths to concatenate as the agent prompt (system preamble). Env vars expanded. |
|
| `prompt` | File paths to concatenate as the agent prompt (system preamble). Env vars expanded. |
|
||||||
| `userPrompts` | File paths to concatenate as active global rules. Prepended to every user message at turn start. Env vars expanded. |
|
| `userPrompts` | File paths to concatenate as active global rules. Prepended to every user message at turn start. Env vars expanded. |
|
||||||
| `autoLoad` | Tools loaded at session start. |
|
| `tools` | Tools loaded at session start. Add more at runtime with the `/tool_load` ctl command. |
|
||||||
| `maxSteps` | Cap on tool-call rounds per turn. 0 = unlimited. |
|
|
||||||
| `temperature` | Sampling temperature. |
|
| `temperature` | Sampling temperature. |
|
||||||
| `maxTokens` | Max output tokens per response. |
|
| `maxTokens` | Max output tokens per response. |
|
||||||
|
| `reasoningEffort` | How hard a reasoning-capable model thinks: `low`, `medium`, or `high`. See below. |
|
||||||
|
| `thinkingBudget` | Advanced: explicit thinking-token budget. Overrides `reasoningEffort`. See below. |
|
||||||
| `backend` | Override backend for this agent. |
|
| `backend` | Override backend for this agent. |
|
||||||
| `model` | Override model for this agent. |
|
| `model` | Override model for this agent. |
|
||||||
| `systemPrompt` | Override the embedded system prompt with a file path. |
|
| `systemPrompt` | Override the embedded system prompt with a file path. |
|
||||||
| `compactionModel` | Use a cheaper model for context compaction. |
|
| `compactionModel` | Use a cheaper model for context compaction. |
|
||||||
|
|
||||||
|
## Reasoning effort
|
||||||
|
|
||||||
|
`reasoningEffort` is the primary, human-facing control for how much a
|
||||||
|
reasoning-capable model deliberates before answering. It takes one of three
|
||||||
|
levels — set it and forget it; you do not need to reason about token counts.
|
||||||
|
|
||||||
|
| Level | Use for | Thinking budget |
|
||||||
|
|-------|---------|-----------------|
|
||||||
|
| `low` | Quick lookups, inline assistance, read-only monitoring | 4096 tokens |
|
||||||
|
| `medium` | General coding, exploration, writing | 8192 tokens |
|
||||||
|
| `high` | Planning, orchestration, security/code review | 16384 tokens |
|
||||||
|
|
||||||
|
How each backend applies it:
|
||||||
|
|
||||||
|
- **OpenAI, OpenRouter, Copilot, Gemini** — sent as the discrete `reasoning_effort`
|
||||||
|
level (`low`/`medium`/`high`).
|
||||||
|
- **Kiro** — enables summarized adaptive thinking and maps the level to its
|
||||||
|
`output_config.effort`.
|
||||||
|
- **Anthropic** — has no discrete level, so the level is translated to the
|
||||||
|
thinking-token budget in the table above (extended thinking). Because thinking
|
||||||
|
tokens are drawn from the output budget, the runtime automatically grows
|
||||||
|
`maxTokens` above the thinking budget and forces `temperature` to 1 (both are
|
||||||
|
Anthropic requirements when thinking is enabled).
|
||||||
|
|
||||||
|
An invalid `reasoningEffort` (anything other than `low`/`medium`/`high`) is a
|
||||||
|
config error and the agent will fail to load, rather than silently doing nothing.
|
||||||
|
|
||||||
|
### thinkingBudget (advanced override)
|
||||||
|
|
||||||
|
`thinkingBudget` sets an explicit thinking-token budget and takes precedence over
|
||||||
|
`reasoningEffort`. Use it only when you need a specific budget that the three
|
||||||
|
levels don't provide; most agents should use `reasoningEffort` instead. It only
|
||||||
|
affects backends that accept a numeric budget (Anthropic).
|
||||||
|
|
||||||
|
|
||||||
## Agents
|
## Agents
|
||||||
|
|
||||||
| Name | Role |
|
| Name | Role |
|
||||||
|
|
@ -64,6 +100,6 @@ Files listed in `userPrompts` are resolved the same way as `prompt` (file paths
|
||||||
|
|
||||||
1. Copy `default.json`, rename it.
|
1. Copy `default.json`, rename it.
|
||||||
2. Adjust the `prompt` array to reference your prompt files in `data/prompts/`.
|
2. Adjust the `prompt` array to reference your prompt files in `data/prompts/`.
|
||||||
3. Set `autoLoad` to the tools this agent needs.
|
3. Set `tools` to the tools this agent needs.
|
||||||
4. Set generation parameters as needed.
|
4. Set generation parameters as needed.
|
||||||
5. Run `make install-data` to install.
|
5. Run `make install-data` to install.
|
||||||
|
|
|
||||||
|
|
@ -26,7 +26,6 @@
|
||||||
"lsp_diagnostics"
|
"lsp_diagnostics"
|
||||||
],
|
],
|
||||||
"temperature": 0.3,
|
"temperature": 0.3,
|
||||||
"maxTokens": 24576,
|
"maxTokens": 8192,
|
||||||
"reasoningEffort": "high",
|
"reasoningEffort": "high"
|
||||||
"thinkingBudget": 16384
|
|
||||||
}
|
}
|
||||||
|
|
|
||||||
Loading…
Reference in New Issue