defer cold summaries until needed

This commit is contained in:
Ollie Agent 2026-08-15 13:27:16 +02:00
parent f167168df3
commit 32180bee4d
3 changed files with 83 additions and 5 deletions

View File

@ -413,7 +413,32 @@ func (s *History) estimateTokens() int {
return chars / 4
}
const maxColdSummariesPerCall = 8
const (
maxColdSummariesPerCall = 8
minColdSummaries = 3
minColdSummaryTokens = 32_000
)
func (s *History) pendingColdSummaryStats() (count, tokens int) {
hot := len(s.messages) - hotTailSize
if hot <= 0 {
return 0, 0
}
for i := 0; i < hot; i++ {
m := &s.messages[i]
if m.Role != "tool" || len(m.Content) <= 200 || m.SummaryApplied {
continue
}
if s.summaryCache != nil {
if _, ok := s.summaryCache[toolSummaryHash(*m)]; ok {
continue
}
}
count++
tokens += len(m.Content) / 4
}
return count, tokens
}
// stripCold summarizes large tool-result messages outside the hot tail
// using a single batched LLM call. Messages in the last hotTailSize slots

View File

@ -0,0 +1,44 @@
package agent
import (
"strings"
"testing"
"ollie/cmd/olliesrv/internal/backend"
)
func TestPendingColdSummaryStatsRequiresCountAndTokens(t *testing.T) {
h := newHistory("test")
for i := 0; i < minColdSummaries; i++ {
h.messages = append(h.messages, backend.Message{Role: "tool", ToolCallID: "tc", Content: strings.Repeat("result ", minColdSummaryTokens/minColdSummaries/2+1000)})
}
for i := 0; i < hotTailSize; i++ {
h.messages = append(h.messages, backend.Message{Role: "user", Content: "hot"})
}
count, tokens := h.pendingColdSummaryStats()
if count != minColdSummaries {
t.Fatalf("pending count = %d, want %d", count, minColdSummaries)
}
if tokens < minColdSummaryTokens {
t.Fatalf("pending tokens = %d, want at least %d", tokens, minColdSummaryTokens)
}
}
func TestPendingColdSummaryStatsSkipsCacheAndApplied(t *testing.T) {
h := newHistory("test")
for i := 0; i < hotTailSize; i++ {
h.messages = append(h.messages, backend.Message{Role: "user", Content: "hot"})
}
cached := backend.Message{Role: "tool", ToolCallID: "cached", Content: strings.Repeat("cached ", 100)}
applied := backend.Message{Role: "tool", ToolCallID: "applied", Content: strings.Repeat("applied ", 100), SummaryApplied: true}
pending := backend.Message{Role: "tool", ToolCallID: "pending", Content: strings.Repeat("pending ", 100)}
h.messages = append(h.messages, cached, applied, pending)
for i := 0; i < hotTailSize; i++ {
h.messages = append(h.messages, backend.Message{Role: "user", Content: "hot"})
}
h.summaryCache[toolSummaryHash(cached)] = "cached summary"
count, _ := h.pendingColdSummaryStats()
if count != 1 {
t.Fatalf("pending count = %d, want 1", count)
}
}

View File

@ -204,10 +204,19 @@ func (ag *Agent) executeTurn(ctx context.Context, input string) string {
// Strip cold-zone tool results (one batched LLM call per turn).
if ag.history != nil {
usage, estimated := ag.history.stripCold(actCtx, ag.runtime.Backend)
if usage.InputTokens > 0 || usage.OutputTokens > 0 {
ag.history.addUsage(usage, estimated)
ag.notifyChange()
pendingCount, pendingTokens := ag.history.pendingColdSummaryStats()
ctxLen := ag.runtime.Backend.ContextLength(actCtx)
if ctxLen <= 0 {
ctxLen = defaultContextLength
}
ctxTokens := ag.history.estimateTokens()
forceStripCold := ctxTokens*100 >= ctxLen*40
if (pendingCount >= minColdSummaries && pendingTokens >= minColdSummaryTokens) || forceStripCold {
usage, estimated := ag.history.stripCold(actCtx, ag.runtime.Backend)
if usage.InputTokens > 0 || usage.OutputTokens > 0 {
ag.history.addUsage(usage, estimated)
ag.notifyChange()
}
}
}