backend: make reasoningEffort the primary knob with documented thresholds
reasoningEffort (low|medium|high) is now the single human-facing control for model deliberation. Backends that speak a discrete level (OpenAI, OpenRouter, Copilot, Gemini, Kiro) send it verbatim; Anthropic, which needs a numeric budget, translates via documented thresholds (low=4096, medium=8192, high=16384). thinkingBudget remains an advanced explicit override that wins when set. - Add EffortLow/Medium/High, threshold consts, ValidReasoningEffort, EffortThinkingBudget, and GenerationParams.ResolvedThinkingBudget. - Anthropic: derive budget from effort, grow max_tokens above the budget (Anthropic requires budget < max_tokens), and force temperature=1 when thinking is enabled (both are Anthropic requirements). - Validate reasoningEffort at config load so typos fail loudly instead of silently no-opping. - theo: drop explicit thinkingBudget/inflated maxTokens; reasoningEffort high now implies the 16384 budget. - Document thresholds and per-backend behavior in data/agents/README.md; fix stale autoLoad/maxSteps references. - Add unit tests for the mapping and Anthropic thinking behavior.
This commit is contained in:
parent
a9fe03eb71
commit
f3933e21e7
|
|
@ -7,6 +7,7 @@ package agent
|
|||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
|
||||
"ollie/cmd/olliesrv/internal/backend"
|
||||
|
|
@ -47,5 +48,8 @@ func Load(r io.Reader) (*AgentConfig, error) {
|
|||
if err := json.NewDecoder(r).Decode(&cfg); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if !backend.ValidReasoningEffort(cfg.ReasoningEffort) {
|
||||
return nil, fmt.Errorf("invalid reasoningEffort %q: want one of low, medium, high", cfg.ReasoningEffort)
|
||||
}
|
||||
return &cfg, nil
|
||||
}
|
||||
|
|
|
|||
|
|
@ -136,19 +136,35 @@ func (b *AnthropicBackend) ChatStream(ctx context.Context, messages []Message, t
|
|||
maxTokens = anthropicDefaultMaxTokens
|
||||
}
|
||||
|
||||
// Resolve extended-thinking budget from the explicit budget or the
|
||||
// reasoning-effort level. Anthropic requires budget_tokens < max_tokens
|
||||
// (thinking tokens are drawn from the output budget), so ensure headroom
|
||||
// for the actual reply by growing max_tokens when necessary.
|
||||
thinkingBudget := params.ResolvedThinkingBudget()
|
||||
if thinkingBudget > 0 && maxTokens <= thinkingBudget {
|
||||
maxTokens = thinkingBudget + anthropicDefaultMaxTokens
|
||||
}
|
||||
|
||||
// Extended thinking requires temperature=1; any other value is rejected.
|
||||
temperature := params.Temperature
|
||||
if thinkingBudget > 0 {
|
||||
one := 1.0
|
||||
temperature = &one
|
||||
}
|
||||
|
||||
areq := anthropicRequest{
|
||||
Model: model,
|
||||
MaxTokens: maxTokens,
|
||||
System: systemBlocks,
|
||||
Messages: wireMessages,
|
||||
Stream: true,
|
||||
Temperature: params.Temperature,
|
||||
Temperature: temperature,
|
||||
TopP: params.TopP,
|
||||
TopK: params.TopK,
|
||||
StopSeqs: params.Stop,
|
||||
}
|
||||
if params.ThinkingBudget > 0 {
|
||||
areq.Thinking = &anthropicThinking{Type: "enabled", BudgetTokens: params.ThinkingBudget}
|
||||
if thinkingBudget > 0 {
|
||||
areq.Thinking = &anthropicThinking{Type: "enabled", BudgetTokens: thinkingBudget}
|
||||
}
|
||||
for _, t := range tools {
|
||||
schema := t.Parameters
|
||||
|
|
@ -171,7 +187,7 @@ func (b *AnthropicBackend) ChatStream(ctx context.Context, messages []Message, t
|
|||
httpReq.Header.Set("Content-Type", "application/json")
|
||||
httpReq.Header.Set("X-Api-Key", b.apiKey)
|
||||
httpReq.Header.Set("Anthropic-Version", "2023-06-01")
|
||||
if params.ThinkingBudget > 0 {
|
||||
if thinkingBudget > 0 {
|
||||
httpReq.Header.Set("Anthropic-Beta", "interleaved-thinking-2025-05-14")
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -165,6 +165,63 @@ type GenerationParams struct {
|
|||
CWD string `json:"-"` // Current working directory
|
||||
}
|
||||
|
||||
// Reasoning effort levels. ReasoningEffort is the primary, human-facing knob
|
||||
// for controlling how much a reasoning-capable model deliberates before
|
||||
// answering. Backends that speak a discrete effort level (OpenAI-family
|
||||
// reasoning_effort) send these strings verbatim. Backends that require a
|
||||
// numeric thinking-token budget (Anthropic extended thinking) translate the
|
||||
// level via EffortThinkingBudget.
|
||||
const (
|
||||
EffortLow = "low"
|
||||
EffortMedium = "medium"
|
||||
EffortHigh = "high"
|
||||
)
|
||||
|
||||
// Thinking-token budgets that each reasoning effort level maps to, for backends
|
||||
// that require a numeric budget rather than a discrete level. These are the
|
||||
// single documented source of truth for the thresholds; keep data/agents/README.md
|
||||
// in sync.
|
||||
const (
|
||||
thinkingBudgetLow = 4096
|
||||
thinkingBudgetMedium = 8192
|
||||
thinkingBudgetHigh = 16384
|
||||
)
|
||||
|
||||
// ValidReasoningEffort reports whether s is a recognised effort level. The
|
||||
// empty string is valid and means "unset" (use the API default).
|
||||
func ValidReasoningEffort(s string) bool {
|
||||
switch s {
|
||||
case "", EffortLow, EffortMedium, EffortHigh:
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// EffortThinkingBudget maps a reasoning effort level to a thinking-token budget.
|
||||
// Returns 0 for an unset or unrecognised level, meaning "no explicit budget".
|
||||
func EffortThinkingBudget(effort string) int {
|
||||
switch effort {
|
||||
case EffortLow:
|
||||
return thinkingBudgetLow
|
||||
case EffortMedium:
|
||||
return thinkingBudgetMedium
|
||||
case EffortHigh:
|
||||
return thinkingBudgetHigh
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// ResolvedThinkingBudget returns the thinking-token budget to use for backends
|
||||
// that require a numeric budget. An explicit ThinkingBudget always wins; when
|
||||
// it is unset (0), the budget is derived from ReasoningEffort. Returns 0 when
|
||||
// neither is set, meaning the backend should not enable extended thinking.
|
||||
func (p GenerationParams) ResolvedThinkingBudget() int {
|
||||
if p.ThinkingBudget > 0 {
|
||||
return p.ThinkingBudget
|
||||
}
|
||||
return EffortThinkingBudget(p.ReasoningEffort)
|
||||
}
|
||||
|
||||
// Backend is the interface all LLM providers must implement.
|
||||
// Streaming is the only supported mode; backends that wrap blocking APIs
|
||||
// should implement ChatStream as a single-event stream.
|
||||
|
|
|
|||
|
|
@ -0,0 +1,145 @@
|
|||
package backend
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"net/url"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestValidReasoningEffort(t *testing.T) {
|
||||
cases := map[string]bool{
|
||||
"": true, // unset means "use default"
|
||||
"low": true,
|
||||
"medium": true,
|
||||
"high": true,
|
||||
"LOW": false, // case-sensitive; profiles use lowercase
|
||||
"max": false,
|
||||
"hihg": false, // typo must be rejected, not silently ignored
|
||||
}
|
||||
for in, want := range cases {
|
||||
if got := ValidReasoningEffort(in); got != want {
|
||||
t.Errorf("ValidReasoningEffort(%q) = %v, want %v", in, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestEffortThinkingBudget(t *testing.T) {
|
||||
cases := map[string]int{
|
||||
EffortLow: 4096,
|
||||
EffortMedium: 8192,
|
||||
EffortHigh: 16384,
|
||||
"": 0,
|
||||
"bogus": 0,
|
||||
}
|
||||
for in, want := range cases {
|
||||
if got := EffortThinkingBudget(in); got != want {
|
||||
t.Errorf("EffortThinkingBudget(%q) = %d, want %d", in, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestResolvedThinkingBudget(t *testing.T) {
|
||||
// Explicit budget wins over effort.
|
||||
p := GenerationParams{ThinkingBudget: 5000, ReasoningEffort: EffortHigh}
|
||||
if got := p.ResolvedThinkingBudget(); got != 5000 {
|
||||
t.Errorf("explicit budget: got %d, want 5000", got)
|
||||
}
|
||||
// Derived from effort when budget unset.
|
||||
p = GenerationParams{ReasoningEffort: EffortHigh}
|
||||
if got := p.ResolvedThinkingBudget(); got != 16384 {
|
||||
t.Errorf("derived budget: got %d, want 16384", got)
|
||||
}
|
||||
// Neither set → 0 (no thinking).
|
||||
p = GenerationParams{}
|
||||
if got := p.ResolvedThinkingBudget(); got != 0 {
|
||||
t.Errorf("unset: got %d, want 0", got)
|
||||
}
|
||||
}
|
||||
|
||||
// captureAnthropicRequest drives AnthropicBackend.ChatStream against a test
|
||||
// server and returns the decoded request body.
|
||||
func captureAnthropicRequest(t *testing.T, params GenerationParams) anthropicRequest {
|
||||
t.Helper()
|
||||
var raw []byte
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
raw, _ = io.ReadAll(r.Body)
|
||||
w.Header().Set("Content-Type", "text/event-stream")
|
||||
// Minimal well-formed stream so the reader terminates cleanly.
|
||||
w.Write([]byte("event: message_stop\ndata: {\"type\":\"message_stop\"}\n\n"))
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
b, err := NewAnthropic("key")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
u, _ := url.Parse(srv.URL)
|
||||
b.baseURL = u
|
||||
|
||||
ch, err := b.ChatStream(context.Background(), []Message{{Role: "user", Content: "audit this"}}, nil, params)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for range ch { //nolint:revive
|
||||
}
|
||||
|
||||
var areq anthropicRequest
|
||||
if err := json.Unmarshal(raw, &areq); err != nil {
|
||||
t.Fatalf("decode request: %v (body=%s)", err, raw)
|
||||
}
|
||||
return areq
|
||||
}
|
||||
|
||||
func TestAnthropic_EffortEnablesThinking(t *testing.T) {
|
||||
temp := 0.3
|
||||
areq := captureAnthropicRequest(t, GenerationParams{
|
||||
ReasoningEffort: EffortHigh,
|
||||
MaxTokens: 8192,
|
||||
Temperature: &temp,
|
||||
})
|
||||
|
||||
if areq.Thinking == nil {
|
||||
t.Fatal("thinking not enabled for reasoningEffort=high")
|
||||
}
|
||||
if areq.Thinking.BudgetTokens != 16384 {
|
||||
t.Errorf("budget_tokens = %d, want 16384", areq.Thinking.BudgetTokens)
|
||||
}
|
||||
// max_tokens must be grown above the thinking budget (budget < max_tokens).
|
||||
if areq.MaxTokens <= areq.Thinking.BudgetTokens {
|
||||
t.Errorf("max_tokens (%d) must exceed thinking budget (%d)", areq.MaxTokens, areq.Thinking.BudgetTokens)
|
||||
}
|
||||
// Extended thinking requires temperature=1.
|
||||
if areq.Temperature == nil || *areq.Temperature != 1.0 {
|
||||
t.Errorf("temperature = %v, want 1.0 when thinking enabled", areq.Temperature)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnthropic_NoEffortNoThinking(t *testing.T) {
|
||||
temp := 0.3
|
||||
areq := captureAnthropicRequest(t, GenerationParams{MaxTokens: 8192, Temperature: &temp})
|
||||
if areq.Thinking != nil {
|
||||
t.Errorf("thinking should be disabled when no effort/budget set")
|
||||
}
|
||||
// Temperature passes through untouched when thinking is off.
|
||||
if areq.Temperature == nil || *areq.Temperature != 0.3 {
|
||||
t.Errorf("temperature = %v, want 0.3", areq.Temperature)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnthropic_ExplicitBudgetOverridesEffort(t *testing.T) {
|
||||
areq := captureAnthropicRequest(t, GenerationParams{
|
||||
ReasoningEffort: EffortLow,
|
||||
ThinkingBudget: 20000,
|
||||
MaxTokens: 8192,
|
||||
})
|
||||
if areq.Thinking == nil || areq.Thinking.BudgetTokens != 20000 {
|
||||
t.Fatalf("budget_tokens = %v, want 20000 (explicit overrides effort)", areq.Thinking)
|
||||
}
|
||||
if areq.MaxTokens <= 20000 {
|
||||
t.Errorf("max_tokens (%d) must exceed explicit budget (20000)", areq.MaxTokens)
|
||||
}
|
||||
}
|
||||
|
|
@ -8,10 +8,10 @@ Each `.json` file defines an agent configuration. Agent configs are installed to
|
|||
{
|
||||
"prompt": ["$XDG_CONFIG_HOME/ollie/prompts/agent-coding.md"],
|
||||
"userPrompts": ["$XDG_CONFIG_HOME/ollie/prompts/rules.md"],
|
||||
"autoLoad": ["shell", "file_read", "file_edit", "file_glob", "file_grep"],
|
||||
"maxSteps": 25,
|
||||
"tools": ["shell", "file_read", "file_edit", "file_glob", "file_grep"],
|
||||
"temperature": 0.5,
|
||||
"maxTokens": 16384,
|
||||
"reasoningEffort": "medium",
|
||||
"backend": "openrouter",
|
||||
"model": "deepseek/deepseek-v4-flash",
|
||||
"systemPrompt": "/path/to/override.md",
|
||||
|
|
@ -23,15 +23,51 @@ Each `.json` file defines an agent configuration. Agent configs are installed to
|
|||
|-------|---------|
|
||||
| `prompt` | File paths to concatenate as the agent prompt (system preamble). Env vars expanded. |
|
||||
| `userPrompts` | File paths to concatenate as active global rules. Prepended to every user message at turn start. Env vars expanded. |
|
||||
| `autoLoad` | Tools loaded at session start. |
|
||||
| `maxSteps` | Cap on tool-call rounds per turn. 0 = unlimited. |
|
||||
| `tools` | Tools loaded at session start. Add more at runtime with the `/tool_load` ctl command. |
|
||||
| `temperature` | Sampling temperature. |
|
||||
| `maxTokens` | Max output tokens per response. |
|
||||
| `reasoningEffort` | How hard a reasoning-capable model thinks: `low`, `medium`, or `high`. See below. |
|
||||
| `thinkingBudget` | Advanced: explicit thinking-token budget. Overrides `reasoningEffort`. See below. |
|
||||
| `backend` | Override backend for this agent. |
|
||||
| `model` | Override model for this agent. |
|
||||
| `systemPrompt` | Override the embedded system prompt with a file path. |
|
||||
| `compactionModel` | Use a cheaper model for context compaction. |
|
||||
|
||||
## Reasoning effort
|
||||
|
||||
`reasoningEffort` is the primary, human-facing control for how much a
|
||||
reasoning-capable model deliberates before answering. It takes one of three
|
||||
levels — set it and forget it; you do not need to reason about token counts.
|
||||
|
||||
| Level | Use for | Thinking budget |
|
||||
|-------|---------|-----------------|
|
||||
| `low` | Quick lookups, inline assistance, read-only monitoring | 4096 tokens |
|
||||
| `medium` | General coding, exploration, writing | 8192 tokens |
|
||||
| `high` | Planning, orchestration, security/code review | 16384 tokens |
|
||||
|
||||
How each backend applies it:
|
||||
|
||||
- **OpenAI, OpenRouter, Copilot, Gemini** — sent as the discrete `reasoning_effort`
|
||||
level (`low`/`medium`/`high`).
|
||||
- **Kiro** — enables summarized adaptive thinking and maps the level to its
|
||||
`output_config.effort`.
|
||||
- **Anthropic** — has no discrete level, so the level is translated to the
|
||||
thinking-token budget in the table above (extended thinking). Because thinking
|
||||
tokens are drawn from the output budget, the runtime automatically grows
|
||||
`maxTokens` above the thinking budget and forces `temperature` to 1 (both are
|
||||
Anthropic requirements when thinking is enabled).
|
||||
|
||||
An invalid `reasoningEffort` (anything other than `low`/`medium`/`high`) is a
|
||||
config error and the agent will fail to load, rather than silently doing nothing.
|
||||
|
||||
### thinkingBudget (advanced override)
|
||||
|
||||
`thinkingBudget` sets an explicit thinking-token budget and takes precedence over
|
||||
`reasoningEffort`. Use it only when you need a specific budget that the three
|
||||
levels don't provide; most agents should use `reasoningEffort` instead. It only
|
||||
affects backends that accept a numeric budget (Anthropic).
|
||||
|
||||
|
||||
## Agents
|
||||
|
||||
| Name | Role |
|
||||
|
|
@ -64,6 +100,6 @@ Files listed in `userPrompts` are resolved the same way as `prompt` (file paths
|
|||
|
||||
1. Copy `default.json`, rename it.
|
||||
2. Adjust the `prompt` array to reference your prompt files in `data/prompts/`.
|
||||
3. Set `autoLoad` to the tools this agent needs.
|
||||
3. Set `tools` to the tools this agent needs.
|
||||
4. Set generation parameters as needed.
|
||||
5. Run `make install-data` to install.
|
||||
|
|
|
|||
|
|
@ -26,7 +26,6 @@
|
|||
"lsp_diagnostics"
|
||||
],
|
||||
"temperature": 0.3,
|
||||
"maxTokens": 24576,
|
||||
"reasoningEffort": "high",
|
||||
"thinkingBudget": 16384
|
||||
"maxTokens": 8192,
|
||||
"reasoningEffort": "high"
|
||||
}
|
||||
|
|
|
|||
Loading…
Reference in New Issue