context: fix budget accounting, tail counting, and output limits

Five improvements for cost/focus efficiency:

1. Budget now accounts for fixed per-request overhead (system prompt +
   tool schemas), which were previously invisible to ContextBuilder and
   could silently exceed intended limits by 30-50%.

2. Tail counting now only counts user and plain-assistant (no tool_calls)
   messages toward TailMessages quota. Processed tool exchanges between
   conversational turns are included but do not crowd out genuine context.
   Trailing in-progress exchanges are always preserved and free.

3. BoundedHistory / BoundedHistoryWithNotice deduplicated into a single
   buildBounded(injectNotice bool) private helper.

4. O(n²) prepend in the greedy inclusion loop replaced with
   append+slices.Reverse.

5. exec.Executor.MaxOutputChars replaces the hardcoded 8000-char const.
   Initialised from OLLIE_TOOL_OUTPUT_CHARS env var, defaults to 8000.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
Levi Neely 2026-04-03 20:27:16 +02:00
parent b7d46015d7
commit 5aea341255
3 changed files with 131 additions and 100 deletions

View File

@ -2,6 +2,7 @@ package agent
import (
"fmt"
"slices"
"strings"
"ollie/backend"
@ -22,9 +23,18 @@ type ContextConfig struct {
// before being added to history. Defaults to 2000.
MaxToolOutputChars int
// TailMessages: always preserve the most recent N messages verbatim,
// even under eviction pressure. Defaults to 6.
// TailMessages: always preserve the most recent N conversational turns
// (user + plain-assistant messages) verbatim, even under eviction pressure.
// Tool-call exchanges between those turns are included, but processed
// tool exchanges do not count toward this limit.
// Defaults to 6.
TailMessages int
// FixedOverheadChars is the estimated character count of fixed per-request
// overhead (system prompt, tool schemas) sent outside the ContextBuilder.
// Subtracted from the budget before greedy inclusion.
// Defaults to 0.
FixedOverheadChars int
}
func defaultContextConfig() ContextConfig {
@ -80,14 +90,28 @@ func (cb *ContextBuilder) Messages() []backend.Message {
}
// BoundedHistory returns a context-window-safe slice of messages.
func (cb *ContextBuilder) BoundedHistory() []backend.Message {
return cb.buildBounded(false)
}
// BoundedHistoryWithNotice is like BoundedHistory but injects a compaction
// notice message when older messages were dropped, so the model is aware.
func (cb *ContextBuilder) BoundedHistoryWithNotice() []backend.Message {
return cb.buildBounded(true)
}
// buildBounded is the shared implementation for BoundedHistory and
// BoundedHistoryWithNotice.
//
// Strategy:
// 1. System messages are always included at the front.
// 2. The most recent TailMessages non-system messages are always kept.
// 3. Older messages are included newest-first until SoftLimit is reached.
// 2. The most recent TailMessages conversational turns (user + plain-assistant)
// are always kept, along with any tool exchanges between or trailing them.
// 3. Older messages are included newest-first until SoftLimit is reached,
// accounting for FixedOverheadChars (system prompt, tool schemas).
// 4. If total still exceeds HardLimit, oldest non-system/non-tail messages
// are dropped entirely.
func (cb *ContextBuilder) BoundedHistory() []backend.Message {
// are dropped atomically (assistant[tool_calls]+tool pairs together).
func (cb *ContextBuilder) buildBounded(injectNotice bool) []backend.Message {
var system []backend.Message
var rest []backend.Message
@ -103,16 +127,12 @@ func (cb *ContextBuilder) BoundedHistory() []backend.Message {
return system
}
// Always keep the tail.
tailStart := len(rest) - cb.cfg.TailMessages
if tailStart < 0 {
tailStart = 0
}
tail := rest[tailStart:]
older := rest[:tailStart]
ts := computeTailStart(rest, cb.cfg.TailMessages)
tail := rest[ts:]
older := rest[:ts]
// Budget remaining after system + tail.
used := msgSliceChars(system) + msgSliceChars(tail)
// Budget: fixed overhead + system + tail chars subtracted up front.
used := cb.cfg.FixedOverheadChars + msgSliceChars(system) + msgSliceChars(tail)
budget := cb.cfg.SoftLimit - used
// Greedily include older messages newest-first until budget exhausted.
@ -123,21 +143,25 @@ func (cb *ContextBuilder) BoundedHistory() []backend.Message {
break
}
budget -= size
included = append([]backend.Message{older[i]}, included...)
included = append(included, older[i])
}
slices.Reverse(included)
result := make([]backend.Message, 0, len(system)+len(included)+len(tail))
evicted := len(older) - len(included)
result := make([]backend.Message, 0, len(system)+len(included)+len(tail)+1)
result = append(result, system...)
if injectNotice && evicted > 0 {
result = append(result, contextSummaryLine(evicted))
}
result = append(result, included...)
result = append(result, tail...)
// Hard-limit safety: truncate from the front (after system) if still over.
// Drop assistant[tool_calls]+tool[result] pairs atomically so we never
// leave an orphaned tool message without its preceding assistant.
for totalChars(result) > cb.cfg.HardLimit && len(result) > len(system)+1 {
drop := len(system) // index of oldest non-system message
// If this message has tool calls, also drop the consecutive tool
// result messages that follow it.
// Hard-limit safety: drop from front (after system) atomically.
// Account for fixed overhead in the ceiling check.
ceiling := cb.cfg.HardLimit - cb.cfg.FixedOverheadChars
for totalChars(result) > ceiling && len(result) > len(system)+1 {
drop := len(system)
if result[drop].Role == "assistant" && len(result[drop].ToolCalls) > 0 {
end := drop + 1
for end < len(result) && result[end].Role == "tool" {
@ -152,6 +176,48 @@ func (cb *ContextBuilder) BoundedHistory() []backend.Message {
return sanitizeHistory(result)
}
// computeTailStart returns the index in rest where the tail begins.
//
// Only user and plain-assistant (no tool calls) messages count toward
// tailCount. Any trailing in-progress exchange (assistant[tool_calls] +
// consecutive tool results with nothing after) is always included and does
// not consume quota. This prevents processed tool exchanges from crowding
// out genuine conversational context.
func computeTailStart(rest []backend.Message, tailCount int) int {
n := len(rest)
if n == 0 || tailCount <= 0 {
return 0
}
// Identify any trailing in-progress exchange: ends with tool result(s)
// that have not yet been followed by an assistant reply.
inProgStart := n
if rest[n-1].Role == "tool" {
j := n - 1
for j > 0 && rest[j].Role == "tool" {
j--
}
if rest[j].Role == "assistant" && len(rest[j].ToolCalls) > 0 {
inProgStart = j
}
}
// Count tailCount conversational messages (user or plain assistant)
// from inProgStart-1 backward.
counted := 0
for i := inProgStart - 1; i >= 0; i-- {
m := rest[i]
if m.Role == "user" || (m.Role == "assistant" && len(m.ToolCalls) == 0) {
counted++
if counted >= tailCount {
return i
}
}
}
// Fewer than tailCount conversational messages: include everything.
return 0
}
// Len returns the number of stored messages.
func (cb *ContextBuilder) Len() int { return len(cb.messages) }
@ -192,70 +258,6 @@ func contextSummaryLine(evicted int) backend.Message {
}
}
// BoundedHistoryWithNotice is like BoundedHistory but injects a compaction
// notice message when older messages were dropped, so the model is aware.
func (cb *ContextBuilder) BoundedHistoryWithNotice() []backend.Message {
var system []backend.Message
var rest []backend.Message
for _, m := range cb.messages {
if m.Role == "system" {
system = append(system, m)
} else {
rest = append(rest, m)
}
}
if len(rest) == 0 {
return system
}
tailStart := len(rest) - cb.cfg.TailMessages
if tailStart < 0 {
tailStart = 0
}
tail := rest[tailStart:]
older := rest[:tailStart]
used := msgSliceChars(system) + msgSliceChars(tail)
budget := cb.cfg.SoftLimit - used
var included []backend.Message
for i := len(older) - 1; i >= 0; i-- {
size := msgChars(older[i])
if budget-size < 0 {
break
}
budget -= size
included = append([]backend.Message{older[i]}, included...)
}
evicted := len(older) - len(included)
result := make([]backend.Message, 0, len(system)+len(included)+len(tail)+1)
result = append(result, system...)
if evicted > 0 {
result = append(result, contextSummaryLine(evicted))
}
result = append(result, included...)
result = append(result, tail...)
for totalChars(result) > cb.cfg.HardLimit && len(result) > len(system)+1 {
drop := len(system)
if result[drop].Role == "assistant" && len(result[drop].ToolCalls) > 0 {
end := drop + 1
for end < len(result) && result[end].Role == "tool" {
end++
}
result = append(result[:drop], result[end:]...)
} else {
result = append(result[:drop], result[drop+1:]...)
}
}
return sanitizeHistory(result)
}
// sanitizeHistory removes tool messages that are not preceded by an assistant
// message with tool calls. This prevents 400 errors from backends that
// reject tool messages not immediately following an assistant[tool_calls]
@ -297,12 +299,6 @@ type ContextStats struct {
func (cb *ContextBuilder) Stats() ContextStats {
bounded := cb.BoundedHistory()
var system []backend.Message
for _, m := range cb.messages {
if m.Role == "system" {
system = append(system, m)
}
}
evicted := len(cb.messages) - len(bounded)
if evicted < 0 {
evicted = 0
@ -310,7 +306,7 @@ func (cb *ContextBuilder) Stats() ContextStats {
return ContextStats{
StoredMessages: len(cb.messages),
BoundedMessages: len(bounded),
ApproxTokens: totalChars(bounded) / 4,
ApproxTokens: (totalChars(bounded) + cb.cfg.FixedOverheadChars) / 4,
Evicted: evicted,
SoftLimit: cb.cfg.SoftLimit,
HardLimit: cb.cfg.HardLimit,

View File

@ -9,6 +9,7 @@ import (
"os/exec"
"path/filepath"
"regexp"
"strconv"
"strings"
"sync"
"time"
@ -96,6 +97,8 @@ func BuildPipeline(steps []PipeStep) (string, bool, error) {
return strings.Join(parts, " |\n"), true, nil
}
const defaultMaxOutputChars = 8000
// Executor runs code in a sandboxed environment.
type Executor struct {
// LogDir is the directory for execution and security event logs.
@ -107,6 +110,11 @@ type Executor struct {
// If empty, the current working directory is used directly.
WorkspaceBase string
// MaxOutputChars is the maximum number of characters returned from Execute.
// Set via OLLIE_TOOL_OUTPUT_CHARS env var or directly on the struct.
// Defaults to 8000.
MaxOutputChars int
// rate limiting state (per-Executor)
rateLimitMu sync.Mutex
validationFailures int
@ -115,8 +123,20 @@ type Executor struct {
}
// New creates a new Executor with the given log directory and workspace base.
// MaxOutputChars is initialised from OLLIE_TOOL_OUTPUT_CHARS if set,
// otherwise defaults to 8000.
func New(logDir, workspaceBase string) *Executor {
return &Executor{LogDir: logDir, WorkspaceBase: workspaceBase}
maxOutput := defaultMaxOutputChars
if s := os.Getenv("OLLIE_TOOL_OUTPUT_CHARS"); s != "" {
if n, err := strconv.Atoi(s); err == nil && n > 0 {
maxOutput = n
}
}
return &Executor{
LogDir: logDir,
WorkspaceBase: workspaceBase,
MaxOutputChars: maxOutput,
}
}
func (e *Executor) logDir() string {
@ -316,7 +336,6 @@ func (e *Executor) Execute(code, language string, timeout int, sandboxName strin
return "", fmt.Errorf("unsupported language: %s (supported: bash)", language)
}
const maxToolOutputSize = 8000
var outputBuf bytes.Buffer
lw := &limitedWriter{w: &outputBuf, limit: 10 * 1024 * 1024}
cmd.Stdout = lw
@ -359,8 +378,12 @@ func (e *Executor) Execute(code, language string, timeout int, sandboxName strin
logExecution(e.logDir(), execLog)
result := string(output)
if len(result) > maxToolOutputSize {
result = result[:maxToolOutputSize] + "\n... (output truncated)"
limit := e.MaxOutputChars
if limit <= 0 {
limit = defaultMaxOutputChars
}
if len(result) > limit {
result = result[:limit] + "\n... (output truncated)"
}
return result, nil
}

14
main.go
View File

@ -174,6 +174,7 @@ type model struct {
doneCh chan struct{}
modelName string
backendName string // e.g. "ollama", "openrouter", "openai"
ctxOverhead int // fixed per-request char overhead (system prompt + tool schemas)
// status bar state
state agentState
@ -267,6 +268,14 @@ func main() {
)
allTools := append(mcpToolsToBackend(mcpTools), executeCodeTool)
// Compute fixed per-request overhead: system prompt + all tool schemas.
// This is subtracted from the context budget so limits mean what they say.
ctxOverhead := len(systemPrompt)
for _, t := range allTools {
ctxOverhead += len(t.Name) + len(t.Description) + len(t.Parameters)
}
serverOf := make(map[string]string, len(mcpTools))
for _, t := range mcpTools {
serverOf[t.Name] = t.Server
@ -320,6 +329,7 @@ func main() {
display: startup,
modelName: modelName,
backendName: backendName,
ctxOverhead: ctxOverhead,
state: agentIdle,
})
@ -521,7 +531,9 @@ func (m model) Update(msg tea.Msg) (tea.Model, tea.Cmd) {
}
if m.session == nil {
m.session = agent.NewSession(input)
m.session = agent.NewSessionWithConfig(input, agent.ContextConfig{
FixedOverheadChars: m.ctxOverhead,
})
} else {
m.session.AppendUserMessage(input)
}