backend,agent,main: retry on HTTP 429 with live countdown

Return RateLimitError from openai backend on HTTP 429, parsing the
Retry-After header (integer seconds or HTTP-date). The agent loop retries
up to 3 times with exponential backoff (5s/10s/20s) when no header is
given, emitting per-second countdown ticks. The status bar renders
"retry {N}s" during the wait.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
Levi Neely 2026-04-04 11:25:05 +02:00
parent b3c4ea0ad4
commit ab575ee385
4 changed files with 103 additions and 9 deletions

View File

@ -3,12 +3,16 @@ package agent
import (
"context"
"encoding/json"
"errors"
"fmt"
"strings"
"time"
"ollie/backend"
)
const maxRateLimitRetries = 3
type ToolExecutor func(name string, args json.RawMessage) (string, error)
type OutputFn func(msg OutputMsg)
@ -41,10 +45,26 @@ func Run(ctx context.Context, cfg Config, state State) error {
history = append([]backend.Message{{Role: "system", Content: cfg.SystemPrompt}}, history...)
}
// Stream the assistant's response.
ch, err := cfg.Backend.ChatStream(ctx, cfg.Model, history, cfg.Tools)
if err != nil {
return fmt.Errorf("step %d: %w", step, err)
// Stream the assistant's response, retrying on HTTP 429.
var ch <-chan backend.StreamEvent
for attempt := range maxRateLimitRetries + 1 {
var err error
ch, err = cfg.Backend.ChatStream(ctx, cfg.Model, history, cfg.Tools)
if err == nil {
break
}
var rlErr *backend.RateLimitError
if !errors.As(err, &rlErr) || attempt >= maxRateLimitRetries {
return fmt.Errorf("step %d: %w", step, err)
}
// Exponential backoff: 5s, 10s, 20s — unless the server told us exactly.
wait := rlErr.RetryAfter
if wait == 0 {
wait = time.Duration(5<<attempt) * time.Second
}
if err := retryCountdown(ctx, cfg, wait); err != nil {
return fmt.Errorf("step %d: %w", step, err)
}
}
var content strings.Builder
@ -128,3 +148,23 @@ func emit(cfg Config, msg OutputMsg) {
cfg.Output(msg)
}
}
// retryCountdown emits one "retry" OutputMsg per second, counting down from
// wait, so the UI can display a live countdown. Returns ctx.Err() if the
// context is cancelled before the wait elapses.
func retryCountdown(ctx context.Context, cfg Config, wait time.Duration) error {
deadline := time.Now().Add(wait)
for {
remaining := time.Until(deadline)
if remaining <= 0 {
return nil
}
secs := int(remaining.Seconds()) + 1
emit(cfg, OutputMsg{Role: "retry", Content: fmt.Sprintf("%d", secs)})
select {
case <-ctx.Done():
return ctx.Err()
case <-time.After(min(remaining, time.Second)):
}
}
}

View File

@ -6,6 +6,8 @@ package backend
import (
"context"
"encoding/json"
"fmt"
"time"
)
// Message is a single conversation turn.
@ -48,6 +50,20 @@ type StreamEvent struct {
Usage Usage // meaningful only when Done==true
}
// RateLimitError is returned when the backend responds with HTTP 429.
// RetryAfter is the suggested wait duration; zero means no hint was given.
type RateLimitError struct {
RetryAfter time.Duration
Message string
}
func (e *RateLimitError) Error() string {
if e.RetryAfter > 0 {
return fmt.Sprintf("rate limited (retry after %v): %s", e.RetryAfter, e.Message)
}
return fmt.Sprintf("rate limited: %s", e.Message)
}
// Backend is the interface all LLM providers must implement.
// Streaming is the only supported mode; backends that wrap blocking APIs
// should implement ChatStream as a single-event stream.

View File

@ -8,7 +8,9 @@ import (
"fmt"
"io"
"net/http"
"strconv"
"strings"
"time"
)
// OpenAIBackend speaks the OpenAI /v1/chat/completions wire format.
@ -152,6 +154,12 @@ func (b *OpenAIBackend) ChatStream(ctx context.Context, model string, messages [
if err != nil {
return nil, err
}
if resp.StatusCode == http.StatusTooManyRequests {
body, _ := io.ReadAll(resp.Body)
resp.Body.Close()
retryAfter := parseRetryAfter(resp.Header.Get("Retry-After"))
return nil, &RateLimitError{RetryAfter: retryAfter, Message: string(body)}
}
if resp.StatusCode != http.StatusOK {
body, _ := io.ReadAll(resp.Body)
resp.Body.Close()
@ -249,3 +257,22 @@ func (b *OpenAIBackend) ChatStream(ctx context.Context, model string, messages [
return ch, nil
}
// parseRetryAfter parses the Retry-After header value, which may be an integer
// number of seconds or an HTTP-date. Returns zero if the header is absent or
// unparseable.
func parseRetryAfter(header string) time.Duration {
if header == "" {
return 0
}
header = strings.TrimSpace(header)
if secs, err := strconv.Atoi(header); err == nil {
return time.Duration(secs) * time.Second
}
if t, err := http.ParseTime(header); err == nil {
if d := time.Until(t); d > 0 {
return d
}
}
return 0
}

21
main.go
View File

@ -8,8 +8,9 @@ import (
"log"
"os"
"os/exec"
"time"
"strconv"
"strings"
"time"
"unicode"
"unicode/utf8"
@ -122,6 +123,7 @@ const (
agentIdle agentState = iota
agentThinking
agentRunningTool
agentRetrying
)
// resolveBackendName returns a short human-readable backend label derived
@ -180,10 +182,11 @@ type model struct {
quitPending bool // whether a second Ctrl+C should quit
lastCtrlC time.Time // timestamp of last Ctrl+C press
// status bar state
state agentState
currentTool string
lastUsage backend.Usage
ctxStats agent.ContextStats
state agentState
currentTool string
retrySecsLeft int
lastUsage backend.Usage
ctxStats agent.ContextStats
}
type agentMsg struct {
@ -387,6 +390,8 @@ func (m model) renderStatusBar() string {
stateStr = "thinking\u2026"
case agentRunningTool:
stateStr = "tool: " + m.currentTool
case agentRetrying:
stateStr = fmt.Sprintf("retry %ds", m.retrySecsLeft)
}
// Token usage segment.
@ -463,6 +468,12 @@ func (m *model) apply(am agentMsg) {
}
m.display = append(m.display, "= "+s)
case "retry":
m.state = agentRetrying
if secs, err := strconv.Atoi(am.content); err == nil {
m.retrySecsLeft = secs
}
case "error":
m.finalizeBuf()
m.display = append(m.display, "Error: "+am.content)