context: fix budget accounting, tail counting, and output limits
Five improvements for cost/focus efficiency: 1. Budget now accounts for fixed per-request overhead (system prompt + tool schemas), which were previously invisible to ContextBuilder and could silently exceed intended limits by 30-50%. 2. Tail counting now only counts user and plain-assistant (no tool_calls) messages toward TailMessages quota. Processed tool exchanges between conversational turns are included but do not crowd out genuine context. Trailing in-progress exchanges are always preserved and free. 3. BoundedHistory / BoundedHistoryWithNotice deduplicated into a single buildBounded(injectNotice bool) private helper. 4. O(n²) prepend in the greedy inclusion loop replaced with append+slices.Reverse. 5. exec.Executor.MaxOutputChars replaces the hardcoded 8000-char const. Initialised from OLLIE_TOOL_OUTPUT_CHARS env var, defaults to 8000. Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
parent
b7d46015d7
commit
5aea341255
186
agent/context.go
186
agent/context.go
|
|
@ -2,6 +2,7 @@ package agent
|
|||
|
||||
import (
|
||||
"fmt"
|
||||
"slices"
|
||||
"strings"
|
||||
|
||||
"ollie/backend"
|
||||
|
|
@ -22,9 +23,18 @@ type ContextConfig struct {
|
|||
// before being added to history. Defaults to 2000.
|
||||
MaxToolOutputChars int
|
||||
|
||||
// TailMessages: always preserve the most recent N messages verbatim,
|
||||
// even under eviction pressure. Defaults to 6.
|
||||
// TailMessages: always preserve the most recent N conversational turns
|
||||
// (user + plain-assistant messages) verbatim, even under eviction pressure.
|
||||
// Tool-call exchanges between those turns are included, but processed
|
||||
// tool exchanges do not count toward this limit.
|
||||
// Defaults to 6.
|
||||
TailMessages int
|
||||
|
||||
// FixedOverheadChars is the estimated character count of fixed per-request
|
||||
// overhead (system prompt, tool schemas) sent outside the ContextBuilder.
|
||||
// Subtracted from the budget before greedy inclusion.
|
||||
// Defaults to 0.
|
||||
FixedOverheadChars int
|
||||
}
|
||||
|
||||
func defaultContextConfig() ContextConfig {
|
||||
|
|
@ -80,14 +90,28 @@ func (cb *ContextBuilder) Messages() []backend.Message {
|
|||
}
|
||||
|
||||
// BoundedHistory returns a context-window-safe slice of messages.
|
||||
func (cb *ContextBuilder) BoundedHistory() []backend.Message {
|
||||
return cb.buildBounded(false)
|
||||
}
|
||||
|
||||
// BoundedHistoryWithNotice is like BoundedHistory but injects a compaction
|
||||
// notice message when older messages were dropped, so the model is aware.
|
||||
func (cb *ContextBuilder) BoundedHistoryWithNotice() []backend.Message {
|
||||
return cb.buildBounded(true)
|
||||
}
|
||||
|
||||
// buildBounded is the shared implementation for BoundedHistory and
|
||||
// BoundedHistoryWithNotice.
|
||||
//
|
||||
// Strategy:
|
||||
// 1. System messages are always included at the front.
|
||||
// 2. The most recent TailMessages non-system messages are always kept.
|
||||
// 3. Older messages are included newest-first until SoftLimit is reached.
|
||||
// 2. The most recent TailMessages conversational turns (user + plain-assistant)
|
||||
// are always kept, along with any tool exchanges between or trailing them.
|
||||
// 3. Older messages are included newest-first until SoftLimit is reached,
|
||||
// accounting for FixedOverheadChars (system prompt, tool schemas).
|
||||
// 4. If total still exceeds HardLimit, oldest non-system/non-tail messages
|
||||
// are dropped entirely.
|
||||
func (cb *ContextBuilder) BoundedHistory() []backend.Message {
|
||||
// are dropped atomically (assistant[tool_calls]+tool pairs together).
|
||||
func (cb *ContextBuilder) buildBounded(injectNotice bool) []backend.Message {
|
||||
var system []backend.Message
|
||||
var rest []backend.Message
|
||||
|
||||
|
|
@ -103,16 +127,12 @@ func (cb *ContextBuilder) BoundedHistory() []backend.Message {
|
|||
return system
|
||||
}
|
||||
|
||||
// Always keep the tail.
|
||||
tailStart := len(rest) - cb.cfg.TailMessages
|
||||
if tailStart < 0 {
|
||||
tailStart = 0
|
||||
}
|
||||
tail := rest[tailStart:]
|
||||
older := rest[:tailStart]
|
||||
ts := computeTailStart(rest, cb.cfg.TailMessages)
|
||||
tail := rest[ts:]
|
||||
older := rest[:ts]
|
||||
|
||||
// Budget remaining after system + tail.
|
||||
used := msgSliceChars(system) + msgSliceChars(tail)
|
||||
// Budget: fixed overhead + system + tail chars subtracted up front.
|
||||
used := cb.cfg.FixedOverheadChars + msgSliceChars(system) + msgSliceChars(tail)
|
||||
budget := cb.cfg.SoftLimit - used
|
||||
|
||||
// Greedily include older messages newest-first until budget exhausted.
|
||||
|
|
@ -123,21 +143,25 @@ func (cb *ContextBuilder) BoundedHistory() []backend.Message {
|
|||
break
|
||||
}
|
||||
budget -= size
|
||||
included = append([]backend.Message{older[i]}, included...)
|
||||
included = append(included, older[i])
|
||||
}
|
||||
slices.Reverse(included)
|
||||
|
||||
result := make([]backend.Message, 0, len(system)+len(included)+len(tail))
|
||||
evicted := len(older) - len(included)
|
||||
|
||||
result := make([]backend.Message, 0, len(system)+len(included)+len(tail)+1)
|
||||
result = append(result, system...)
|
||||
if injectNotice && evicted > 0 {
|
||||
result = append(result, contextSummaryLine(evicted))
|
||||
}
|
||||
result = append(result, included...)
|
||||
result = append(result, tail...)
|
||||
|
||||
// Hard-limit safety: truncate from the front (after system) if still over.
|
||||
// Drop assistant[tool_calls]+tool[result] pairs atomically so we never
|
||||
// leave an orphaned tool message without its preceding assistant.
|
||||
for totalChars(result) > cb.cfg.HardLimit && len(result) > len(system)+1 {
|
||||
drop := len(system) // index of oldest non-system message
|
||||
// If this message has tool calls, also drop the consecutive tool
|
||||
// result messages that follow it.
|
||||
// Hard-limit safety: drop from front (after system) atomically.
|
||||
// Account for fixed overhead in the ceiling check.
|
||||
ceiling := cb.cfg.HardLimit - cb.cfg.FixedOverheadChars
|
||||
for totalChars(result) > ceiling && len(result) > len(system)+1 {
|
||||
drop := len(system)
|
||||
if result[drop].Role == "assistant" && len(result[drop].ToolCalls) > 0 {
|
||||
end := drop + 1
|
||||
for end < len(result) && result[end].Role == "tool" {
|
||||
|
|
@ -152,6 +176,48 @@ func (cb *ContextBuilder) BoundedHistory() []backend.Message {
|
|||
return sanitizeHistory(result)
|
||||
}
|
||||
|
||||
// computeTailStart returns the index in rest where the tail begins.
|
||||
//
|
||||
// Only user and plain-assistant (no tool calls) messages count toward
|
||||
// tailCount. Any trailing in-progress exchange (assistant[tool_calls] +
|
||||
// consecutive tool results with nothing after) is always included and does
|
||||
// not consume quota. This prevents processed tool exchanges from crowding
|
||||
// out genuine conversational context.
|
||||
func computeTailStart(rest []backend.Message, tailCount int) int {
|
||||
n := len(rest)
|
||||
if n == 0 || tailCount <= 0 {
|
||||
return 0
|
||||
}
|
||||
|
||||
// Identify any trailing in-progress exchange: ends with tool result(s)
|
||||
// that have not yet been followed by an assistant reply.
|
||||
inProgStart := n
|
||||
if rest[n-1].Role == "tool" {
|
||||
j := n - 1
|
||||
for j > 0 && rest[j].Role == "tool" {
|
||||
j--
|
||||
}
|
||||
if rest[j].Role == "assistant" && len(rest[j].ToolCalls) > 0 {
|
||||
inProgStart = j
|
||||
}
|
||||
}
|
||||
|
||||
// Count tailCount conversational messages (user or plain assistant)
|
||||
// from inProgStart-1 backward.
|
||||
counted := 0
|
||||
for i := inProgStart - 1; i >= 0; i-- {
|
||||
m := rest[i]
|
||||
if m.Role == "user" || (m.Role == "assistant" && len(m.ToolCalls) == 0) {
|
||||
counted++
|
||||
if counted >= tailCount {
|
||||
return i
|
||||
}
|
||||
}
|
||||
}
|
||||
// Fewer than tailCount conversational messages: include everything.
|
||||
return 0
|
||||
}
|
||||
|
||||
// Len returns the number of stored messages.
|
||||
func (cb *ContextBuilder) Len() int { return len(cb.messages) }
|
||||
|
||||
|
|
@ -192,70 +258,6 @@ func contextSummaryLine(evicted int) backend.Message {
|
|||
}
|
||||
}
|
||||
|
||||
// BoundedHistoryWithNotice is like BoundedHistory but injects a compaction
|
||||
// notice message when older messages were dropped, so the model is aware.
|
||||
func (cb *ContextBuilder) BoundedHistoryWithNotice() []backend.Message {
|
||||
var system []backend.Message
|
||||
var rest []backend.Message
|
||||
|
||||
for _, m := range cb.messages {
|
||||
if m.Role == "system" {
|
||||
system = append(system, m)
|
||||
} else {
|
||||
rest = append(rest, m)
|
||||
}
|
||||
}
|
||||
|
||||
if len(rest) == 0 {
|
||||
return system
|
||||
}
|
||||
|
||||
tailStart := len(rest) - cb.cfg.TailMessages
|
||||
if tailStart < 0 {
|
||||
tailStart = 0
|
||||
}
|
||||
tail := rest[tailStart:]
|
||||
older := rest[:tailStart]
|
||||
|
||||
used := msgSliceChars(system) + msgSliceChars(tail)
|
||||
budget := cb.cfg.SoftLimit - used
|
||||
|
||||
var included []backend.Message
|
||||
for i := len(older) - 1; i >= 0; i-- {
|
||||
size := msgChars(older[i])
|
||||
if budget-size < 0 {
|
||||
break
|
||||
}
|
||||
budget -= size
|
||||
included = append([]backend.Message{older[i]}, included...)
|
||||
}
|
||||
|
||||
evicted := len(older) - len(included)
|
||||
|
||||
result := make([]backend.Message, 0, len(system)+len(included)+len(tail)+1)
|
||||
result = append(result, system...)
|
||||
if evicted > 0 {
|
||||
result = append(result, contextSummaryLine(evicted))
|
||||
}
|
||||
result = append(result, included...)
|
||||
result = append(result, tail...)
|
||||
|
||||
for totalChars(result) > cb.cfg.HardLimit && len(result) > len(system)+1 {
|
||||
drop := len(system)
|
||||
if result[drop].Role == "assistant" && len(result[drop].ToolCalls) > 0 {
|
||||
end := drop + 1
|
||||
for end < len(result) && result[end].Role == "tool" {
|
||||
end++
|
||||
}
|
||||
result = append(result[:drop], result[end:]...)
|
||||
} else {
|
||||
result = append(result[:drop], result[drop+1:]...)
|
||||
}
|
||||
}
|
||||
|
||||
return sanitizeHistory(result)
|
||||
}
|
||||
|
||||
// sanitizeHistory removes tool messages that are not preceded by an assistant
|
||||
// message with tool calls. This prevents 400 errors from backends that
|
||||
// reject tool messages not immediately following an assistant[tool_calls]
|
||||
|
|
@ -297,12 +299,6 @@ type ContextStats struct {
|
|||
|
||||
func (cb *ContextBuilder) Stats() ContextStats {
|
||||
bounded := cb.BoundedHistory()
|
||||
var system []backend.Message
|
||||
for _, m := range cb.messages {
|
||||
if m.Role == "system" {
|
||||
system = append(system, m)
|
||||
}
|
||||
}
|
||||
evicted := len(cb.messages) - len(bounded)
|
||||
if evicted < 0 {
|
||||
evicted = 0
|
||||
|
|
@ -310,7 +306,7 @@ func (cb *ContextBuilder) Stats() ContextStats {
|
|||
return ContextStats{
|
||||
StoredMessages: len(cb.messages),
|
||||
BoundedMessages: len(bounded),
|
||||
ApproxTokens: totalChars(bounded) / 4,
|
||||
ApproxTokens: (totalChars(bounded) + cb.cfg.FixedOverheadChars) / 4,
|
||||
Evicted: evicted,
|
||||
SoftLimit: cb.cfg.SoftLimit,
|
||||
HardLimit: cb.cfg.HardLimit,
|
||||
|
|
|
|||
31
exec/exec.go
31
exec/exec.go
|
|
@ -9,6 +9,7 @@ import (
|
|||
"os/exec"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
|
@ -96,6 +97,8 @@ func BuildPipeline(steps []PipeStep) (string, bool, error) {
|
|||
return strings.Join(parts, " |\n"), true, nil
|
||||
}
|
||||
|
||||
const defaultMaxOutputChars = 8000
|
||||
|
||||
// Executor runs code in a sandboxed environment.
|
||||
type Executor struct {
|
||||
// LogDir is the directory for execution and security event logs.
|
||||
|
|
@ -107,6 +110,11 @@ type Executor struct {
|
|||
// If empty, the current working directory is used directly.
|
||||
WorkspaceBase string
|
||||
|
||||
// MaxOutputChars is the maximum number of characters returned from Execute.
|
||||
// Set via OLLIE_TOOL_OUTPUT_CHARS env var or directly on the struct.
|
||||
// Defaults to 8000.
|
||||
MaxOutputChars int
|
||||
|
||||
// rate limiting state (per-Executor)
|
||||
rateLimitMu sync.Mutex
|
||||
validationFailures int
|
||||
|
|
@ -115,8 +123,20 @@ type Executor struct {
|
|||
}
|
||||
|
||||
// New creates a new Executor with the given log directory and workspace base.
|
||||
// MaxOutputChars is initialised from OLLIE_TOOL_OUTPUT_CHARS if set,
|
||||
// otherwise defaults to 8000.
|
||||
func New(logDir, workspaceBase string) *Executor {
|
||||
return &Executor{LogDir: logDir, WorkspaceBase: workspaceBase}
|
||||
maxOutput := defaultMaxOutputChars
|
||||
if s := os.Getenv("OLLIE_TOOL_OUTPUT_CHARS"); s != "" {
|
||||
if n, err := strconv.Atoi(s); err == nil && n > 0 {
|
||||
maxOutput = n
|
||||
}
|
||||
}
|
||||
return &Executor{
|
||||
LogDir: logDir,
|
||||
WorkspaceBase: workspaceBase,
|
||||
MaxOutputChars: maxOutput,
|
||||
}
|
||||
}
|
||||
|
||||
func (e *Executor) logDir() string {
|
||||
|
|
@ -316,7 +336,6 @@ func (e *Executor) Execute(code, language string, timeout int, sandboxName strin
|
|||
return "", fmt.Errorf("unsupported language: %s (supported: bash)", language)
|
||||
}
|
||||
|
||||
const maxToolOutputSize = 8000
|
||||
var outputBuf bytes.Buffer
|
||||
lw := &limitedWriter{w: &outputBuf, limit: 10 * 1024 * 1024}
|
||||
cmd.Stdout = lw
|
||||
|
|
@ -359,8 +378,12 @@ func (e *Executor) Execute(code, language string, timeout int, sandboxName strin
|
|||
|
||||
logExecution(e.logDir(), execLog)
|
||||
result := string(output)
|
||||
if len(result) > maxToolOutputSize {
|
||||
result = result[:maxToolOutputSize] + "\n... (output truncated)"
|
||||
limit := e.MaxOutputChars
|
||||
if limit <= 0 {
|
||||
limit = defaultMaxOutputChars
|
||||
}
|
||||
if len(result) > limit {
|
||||
result = result[:limit] + "\n... (output truncated)"
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
|
|
|||
14
main.go
14
main.go
|
|
@ -174,6 +174,7 @@ type model struct {
|
|||
doneCh chan struct{}
|
||||
modelName string
|
||||
backendName string // e.g. "ollama", "openrouter", "openai"
|
||||
ctxOverhead int // fixed per-request char overhead (system prompt + tool schemas)
|
||||
|
||||
// status bar state
|
||||
state agentState
|
||||
|
|
@ -267,6 +268,14 @@ func main() {
|
|||
)
|
||||
|
||||
allTools := append(mcpToolsToBackend(mcpTools), executeCodeTool)
|
||||
|
||||
// Compute fixed per-request overhead: system prompt + all tool schemas.
|
||||
// This is subtracted from the context budget so limits mean what they say.
|
||||
ctxOverhead := len(systemPrompt)
|
||||
for _, t := range allTools {
|
||||
ctxOverhead += len(t.Name) + len(t.Description) + len(t.Parameters)
|
||||
}
|
||||
|
||||
serverOf := make(map[string]string, len(mcpTools))
|
||||
for _, t := range mcpTools {
|
||||
serverOf[t.Name] = t.Server
|
||||
|
|
@ -320,6 +329,7 @@ func main() {
|
|||
display: startup,
|
||||
modelName: modelName,
|
||||
backendName: backendName,
|
||||
ctxOverhead: ctxOverhead,
|
||||
state: agentIdle,
|
||||
})
|
||||
|
||||
|
|
@ -521,7 +531,9 @@ func (m model) Update(msg tea.Msg) (tea.Model, tea.Cmd) {
|
|||
}
|
||||
|
||||
if m.session == nil {
|
||||
m.session = agent.NewSession(input)
|
||||
m.session = agent.NewSessionWithConfig(input, agent.ContextConfig{
|
||||
FixedOverheadChars: m.ctxOverhead,
|
||||
})
|
||||
} else {
|
||||
m.session.AppendUserMessage(input)
|
||||
}
|
||||
|
|
|
|||
Reference in New Issue