diff --git a/cmd/olliesrv/internal/agent/agent_config.go b/cmd/olliesrv/internal/agent/agent_config.go index bc3e9e4..2c54e99 100644 --- a/cmd/olliesrv/internal/agent/agent_config.go +++ b/cmd/olliesrv/internal/agent/agent_config.go @@ -7,6 +7,7 @@ package agent import ( "encoding/json" + "fmt" "io" "ollie/cmd/olliesrv/internal/backend" @@ -47,5 +48,8 @@ func Load(r io.Reader) (*AgentConfig, error) { if err := json.NewDecoder(r).Decode(&cfg); err != nil { return nil, err } + if !backend.ValidReasoningEffort(cfg.ReasoningEffort) { + return nil, fmt.Errorf("invalid reasoningEffort %q: want one of low, medium, high", cfg.ReasoningEffort) + } return &cfg, nil } diff --git a/cmd/olliesrv/internal/backend/anthropic.go b/cmd/olliesrv/internal/backend/anthropic.go index 7abc527..54b67cb 100644 --- a/cmd/olliesrv/internal/backend/anthropic.go +++ b/cmd/olliesrv/internal/backend/anthropic.go @@ -136,19 +136,35 @@ func (b *AnthropicBackend) ChatStream(ctx context.Context, messages []Message, t maxTokens = anthropicDefaultMaxTokens } + // Resolve extended-thinking budget from the explicit budget or the + // reasoning-effort level. Anthropic requires budget_tokens < max_tokens + // (thinking tokens are drawn from the output budget), so ensure headroom + // for the actual reply by growing max_tokens when necessary. + thinkingBudget := params.ResolvedThinkingBudget() + if thinkingBudget > 0 && maxTokens <= thinkingBudget { + maxTokens = thinkingBudget + anthropicDefaultMaxTokens + } + + // Extended thinking requires temperature=1; any other value is rejected. + temperature := params.Temperature + if thinkingBudget > 0 { + one := 1.0 + temperature = &one + } + areq := anthropicRequest{ Model: model, MaxTokens: maxTokens, System: systemBlocks, Messages: wireMessages, Stream: true, - Temperature: params.Temperature, + Temperature: temperature, TopP: params.TopP, TopK: params.TopK, StopSeqs: params.Stop, } - if params.ThinkingBudget > 0 { - areq.Thinking = &anthropicThinking{Type: "enabled", BudgetTokens: params.ThinkingBudget} + if thinkingBudget > 0 { + areq.Thinking = &anthropicThinking{Type: "enabled", BudgetTokens: thinkingBudget} } for _, t := range tools { schema := t.Parameters @@ -171,7 +187,7 @@ func (b *AnthropicBackend) ChatStream(ctx context.Context, messages []Message, t httpReq.Header.Set("Content-Type", "application/json") httpReq.Header.Set("X-Api-Key", b.apiKey) httpReq.Header.Set("Anthropic-Version", "2023-06-01") - if params.ThinkingBudget > 0 { + if thinkingBudget > 0 { httpReq.Header.Set("Anthropic-Beta", "interleaved-thinking-2025-05-14") } diff --git a/cmd/olliesrv/internal/backend/backend.go b/cmd/olliesrv/internal/backend/backend.go index ea2aa76..2101fbb 100644 --- a/cmd/olliesrv/internal/backend/backend.go +++ b/cmd/olliesrv/internal/backend/backend.go @@ -165,6 +165,63 @@ type GenerationParams struct { CWD string `json:"-"` // Current working directory } +// Reasoning effort levels. ReasoningEffort is the primary, human-facing knob +// for controlling how much a reasoning-capable model deliberates before +// answering. Backends that speak a discrete effort level (OpenAI-family +// reasoning_effort) send these strings verbatim. Backends that require a +// numeric thinking-token budget (Anthropic extended thinking) translate the +// level via EffortThinkingBudget. +const ( + EffortLow = "low" + EffortMedium = "medium" + EffortHigh = "high" +) + +// Thinking-token budgets that each reasoning effort level maps to, for backends +// that require a numeric budget rather than a discrete level. These are the +// single documented source of truth for the thresholds; keep data/agents/README.md +// in sync. +const ( + thinkingBudgetLow = 4096 + thinkingBudgetMedium = 8192 + thinkingBudgetHigh = 16384 +) + +// ValidReasoningEffort reports whether s is a recognised effort level. The +// empty string is valid and means "unset" (use the API default). +func ValidReasoningEffort(s string) bool { + switch s { + case "", EffortLow, EffortMedium, EffortHigh: + return true + } + return false +} + +// EffortThinkingBudget maps a reasoning effort level to a thinking-token budget. +// Returns 0 for an unset or unrecognised level, meaning "no explicit budget". +func EffortThinkingBudget(effort string) int { + switch effort { + case EffortLow: + return thinkingBudgetLow + case EffortMedium: + return thinkingBudgetMedium + case EffortHigh: + return thinkingBudgetHigh + } + return 0 +} + +// ResolvedThinkingBudget returns the thinking-token budget to use for backends +// that require a numeric budget. An explicit ThinkingBudget always wins; when +// it is unset (0), the budget is derived from ReasoningEffort. Returns 0 when +// neither is set, meaning the backend should not enable extended thinking. +func (p GenerationParams) ResolvedThinkingBudget() int { + if p.ThinkingBudget > 0 { + return p.ThinkingBudget + } + return EffortThinkingBudget(p.ReasoningEffort) +} + // Backend is the interface all LLM providers must implement. // Streaming is the only supported mode; backends that wrap blocking APIs // should implement ChatStream as a single-event stream. diff --git a/cmd/olliesrv/internal/backend/reasoning_test.go b/cmd/olliesrv/internal/backend/reasoning_test.go new file mode 100644 index 0000000..97ff475 --- /dev/null +++ b/cmd/olliesrv/internal/backend/reasoning_test.go @@ -0,0 +1,145 @@ +package backend + +import ( + "context" + "encoding/json" + "io" + "net/http" + "net/http/httptest" + "net/url" + "testing" +) + +func TestValidReasoningEffort(t *testing.T) { + cases := map[string]bool{ + "": true, // unset means "use default" + "low": true, + "medium": true, + "high": true, + "LOW": false, // case-sensitive; profiles use lowercase + "max": false, + "hihg": false, // typo must be rejected, not silently ignored + } + for in, want := range cases { + if got := ValidReasoningEffort(in); got != want { + t.Errorf("ValidReasoningEffort(%q) = %v, want %v", in, got, want) + } + } +} + +func TestEffortThinkingBudget(t *testing.T) { + cases := map[string]int{ + EffortLow: 4096, + EffortMedium: 8192, + EffortHigh: 16384, + "": 0, + "bogus": 0, + } + for in, want := range cases { + if got := EffortThinkingBudget(in); got != want { + t.Errorf("EffortThinkingBudget(%q) = %d, want %d", in, got, want) + } + } +} + +func TestResolvedThinkingBudget(t *testing.T) { + // Explicit budget wins over effort. + p := GenerationParams{ThinkingBudget: 5000, ReasoningEffort: EffortHigh} + if got := p.ResolvedThinkingBudget(); got != 5000 { + t.Errorf("explicit budget: got %d, want 5000", got) + } + // Derived from effort when budget unset. + p = GenerationParams{ReasoningEffort: EffortHigh} + if got := p.ResolvedThinkingBudget(); got != 16384 { + t.Errorf("derived budget: got %d, want 16384", got) + } + // Neither set → 0 (no thinking). + p = GenerationParams{} + if got := p.ResolvedThinkingBudget(); got != 0 { + t.Errorf("unset: got %d, want 0", got) + } +} + +// captureAnthropicRequest drives AnthropicBackend.ChatStream against a test +// server and returns the decoded request body. +func captureAnthropicRequest(t *testing.T, params GenerationParams) anthropicRequest { + t.Helper() + var raw []byte + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + raw, _ = io.ReadAll(r.Body) + w.Header().Set("Content-Type", "text/event-stream") + // Minimal well-formed stream so the reader terminates cleanly. + w.Write([]byte("event: message_stop\ndata: {\"type\":\"message_stop\"}\n\n")) + })) + defer srv.Close() + + b, err := NewAnthropic("key") + if err != nil { + t.Fatal(err) + } + u, _ := url.Parse(srv.URL) + b.baseURL = u + + ch, err := b.ChatStream(context.Background(), []Message{{Role: "user", Content: "audit this"}}, nil, params) + if err != nil { + t.Fatal(err) + } + for range ch { //nolint:revive + } + + var areq anthropicRequest + if err := json.Unmarshal(raw, &areq); err != nil { + t.Fatalf("decode request: %v (body=%s)", err, raw) + } + return areq +} + +func TestAnthropic_EffortEnablesThinking(t *testing.T) { + temp := 0.3 + areq := captureAnthropicRequest(t, GenerationParams{ + ReasoningEffort: EffortHigh, + MaxTokens: 8192, + Temperature: &temp, + }) + + if areq.Thinking == nil { + t.Fatal("thinking not enabled for reasoningEffort=high") + } + if areq.Thinking.BudgetTokens != 16384 { + t.Errorf("budget_tokens = %d, want 16384", areq.Thinking.BudgetTokens) + } + // max_tokens must be grown above the thinking budget (budget < max_tokens). + if areq.MaxTokens <= areq.Thinking.BudgetTokens { + t.Errorf("max_tokens (%d) must exceed thinking budget (%d)", areq.MaxTokens, areq.Thinking.BudgetTokens) + } + // Extended thinking requires temperature=1. + if areq.Temperature == nil || *areq.Temperature != 1.0 { + t.Errorf("temperature = %v, want 1.0 when thinking enabled", areq.Temperature) + } +} + +func TestAnthropic_NoEffortNoThinking(t *testing.T) { + temp := 0.3 + areq := captureAnthropicRequest(t, GenerationParams{MaxTokens: 8192, Temperature: &temp}) + if areq.Thinking != nil { + t.Errorf("thinking should be disabled when no effort/budget set") + } + // Temperature passes through untouched when thinking is off. + if areq.Temperature == nil || *areq.Temperature != 0.3 { + t.Errorf("temperature = %v, want 0.3", areq.Temperature) + } +} + +func TestAnthropic_ExplicitBudgetOverridesEffort(t *testing.T) { + areq := captureAnthropicRequest(t, GenerationParams{ + ReasoningEffort: EffortLow, + ThinkingBudget: 20000, + MaxTokens: 8192, + }) + if areq.Thinking == nil || areq.Thinking.BudgetTokens != 20000 { + t.Fatalf("budget_tokens = %v, want 20000 (explicit overrides effort)", areq.Thinking) + } + if areq.MaxTokens <= 20000 { + t.Errorf("max_tokens (%d) must exceed explicit budget (20000)", areq.MaxTokens) + } +} diff --git a/data/agents/README.md b/data/agents/README.md index 03c69df..a9c62ad 100644 --- a/data/agents/README.md +++ b/data/agents/README.md @@ -8,10 +8,10 @@ Each `.json` file defines an agent configuration. Agent configs are installed to { "prompt": ["$XDG_CONFIG_HOME/ollie/prompts/agent-coding.md"], "userPrompts": ["$XDG_CONFIG_HOME/ollie/prompts/rules.md"], - "autoLoad": ["shell", "file_read", "file_edit", "file_glob", "file_grep"], - "maxSteps": 25, + "tools": ["shell", "file_read", "file_edit", "file_glob", "file_grep"], "temperature": 0.5, "maxTokens": 16384, + "reasoningEffort": "medium", "backend": "openrouter", "model": "deepseek/deepseek-v4-flash", "systemPrompt": "/path/to/override.md", @@ -23,15 +23,51 @@ Each `.json` file defines an agent configuration. Agent configs are installed to |-------|---------| | `prompt` | File paths to concatenate as the agent prompt (system preamble). Env vars expanded. | | `userPrompts` | File paths to concatenate as active global rules. Prepended to every user message at turn start. Env vars expanded. | -| `autoLoad` | Tools loaded at session start. | -| `maxSteps` | Cap on tool-call rounds per turn. 0 = unlimited. | +| `tools` | Tools loaded at session start. Add more at runtime with the `/tool_load` ctl command. | | `temperature` | Sampling temperature. | | `maxTokens` | Max output tokens per response. | +| `reasoningEffort` | How hard a reasoning-capable model thinks: `low`, `medium`, or `high`. See below. | +| `thinkingBudget` | Advanced: explicit thinking-token budget. Overrides `reasoningEffort`. See below. | | `backend` | Override backend for this agent. | | `model` | Override model for this agent. | | `systemPrompt` | Override the embedded system prompt with a file path. | | `compactionModel` | Use a cheaper model for context compaction. | +## Reasoning effort + +`reasoningEffort` is the primary, human-facing control for how much a +reasoning-capable model deliberates before answering. It takes one of three +levels — set it and forget it; you do not need to reason about token counts. + +| Level | Use for | Thinking budget | +|-------|---------|-----------------| +| `low` | Quick lookups, inline assistance, read-only monitoring | 4096 tokens | +| `medium` | General coding, exploration, writing | 8192 tokens | +| `high` | Planning, orchestration, security/code review | 16384 tokens | + +How each backend applies it: + +- **OpenAI, OpenRouter, Copilot, Gemini** — sent as the discrete `reasoning_effort` + level (`low`/`medium`/`high`). +- **Kiro** — enables summarized adaptive thinking and maps the level to its + `output_config.effort`. +- **Anthropic** — has no discrete level, so the level is translated to the + thinking-token budget in the table above (extended thinking). Because thinking + tokens are drawn from the output budget, the runtime automatically grows + `maxTokens` above the thinking budget and forces `temperature` to 1 (both are + Anthropic requirements when thinking is enabled). + +An invalid `reasoningEffort` (anything other than `low`/`medium`/`high`) is a +config error and the agent will fail to load, rather than silently doing nothing. + +### thinkingBudget (advanced override) + +`thinkingBudget` sets an explicit thinking-token budget and takes precedence over +`reasoningEffort`. Use it only when you need a specific budget that the three +levels don't provide; most agents should use `reasoningEffort` instead. It only +affects backends that accept a numeric budget (Anthropic). + + ## Agents | Name | Role | @@ -64,6 +100,6 @@ Files listed in `userPrompts` are resolved the same way as `prompt` (file paths 1. Copy `default.json`, rename it. 2. Adjust the `prompt` array to reference your prompt files in `data/prompts/`. -3. Set `autoLoad` to the tools this agent needs. +3. Set `tools` to the tools this agent needs. 4. Set generation parameters as needed. 5. Run `make install-data` to install. diff --git a/data/agents/theo.json b/data/agents/theo.json index 2277a0d..c567b96 100644 --- a/data/agents/theo.json +++ b/data/agents/theo.json @@ -26,7 +26,6 @@ "lsp_diagnostics" ], "temperature": 0.3, - "maxTokens": 24576, - "reasoningEffort": "high", - "thinkingBudget": 16384 + "maxTokens": 8192, + "reasoningEffort": "high" }