split execute_code into three tools; improve ollama backend and system prompt

- Split execute_code into execute_code, execute_tool, execute_pipe for
  clearer model comprehension, especially on smaller models
- Ollama: accumulate tool calls across all stream chunks (fixes llama3.1
  which delivers tool calls only on the done event)
- Ollama: add HTTP 429 / RateLimitError handling
- Ollama: map ToolCallID on outbound messages
- System prompt: inject cwd, current time, available tools
- System prompt: clarify tool usage rules
- Fix spurious spaces in streamed output (remove addStreamingContent heuristic)
- Fix tool output display: preserve newlines, add blank line separation
This commit is contained in:
Levi Neely 2026-04-04 17:57:55 +02:00
parent b00d970c74
commit c2cbc3e806
5 changed files with 151 additions and 68 deletions

View File

@ -1,5 +1,5 @@
{
"prompt": "",
"prompt": "To discover skills, use discover_skill.sh <keyword> to search skill descriptions. Once a skill is identified, load it using load_skill.sh <skill_name>.",
"mcpServers": {
"superpowers-mcp-server": {
"command": "superpowers-mcp-server",

View File

@ -1,5 +1,5 @@
{
"prompt": "",
"prompt": "To discover skills, use discover_skill.sh <keyword> to search skill descriptions. Once a skill is identified, load it using load_skill.sh <skill_name>.",
"mcpServers": {
"superpowers-mcp-server": {
"command": "superpowers-mcp-server",

View File

@ -1,5 +1,5 @@
{
"prompt": "",
"prompt": "To discover skills, use discover_skill.sh <keyword> to search skill descriptions. Once a skill is identified, load it using load_skill.sh <skill_name>.",
"mcpServers": {},
"hooks": {
"agentSpawn": "true",

View File

@ -29,6 +29,7 @@ type ollamaMessage struct {
Role string `json:"role"`
Content string `json:"content"`
ToolCalls []ollamaToolCall `json:"tool_calls,omitempty"`
ToolCallID string `json:"tool_call_id,omitempty"`
}
type ollamaToolCall struct {
@ -71,7 +72,7 @@ type ollamaChatResponse struct {
func (b *OllamaBackend) ChatStream(ctx context.Context, model string, messages []Message, tools []Tool) (<-chan StreamEvent, error) {
wireMessages := make([]ollamaMessage, len(messages))
for i, m := range messages {
wireMessages[i] = ollamaMessage{Role: m.Role, Content: m.Content}
wireMessages[i] = ollamaMessage{Role: m.Role, Content: m.Content, ToolCallID: m.ToolCallID}
for _, tc := range m.ToolCalls {
wireMessages[i].ToolCalls = append(wireMessages[i].ToolCalls, ollamaToolCall{
Function: ollamaFunction{Name: tc.Name, Arguments: tc.Arguments},
@ -111,6 +112,12 @@ func (b *OllamaBackend) ChatStream(ctx context.Context, model string, messages [
if err != nil {
return nil, err
}
if resp.StatusCode == http.StatusTooManyRequests {
body, _ := io.ReadAll(resp.Body)
resp.Body.Close()
retryAfter := parseRetryAfter(resp.Header.Get("Retry-After"))
return nil, &RateLimitError{RetryAfter: retryAfter, Message: string(body)}
}
if resp.StatusCode != http.StatusOK {
body, _ := io.ReadAll(resp.Body)
resp.Body.Close()
@ -126,6 +133,8 @@ func (b *OllamaBackend) ChatStream(ctx context.Context, model string, messages [
scanner := bufio.NewScanner(resp.Body)
scanner.Buffer(make([]byte, 0, 64*1024), 4*1024*1024)
var accumulated []ToolCall
for scanner.Scan() {
line := scanner.Bytes()
if len(line) == 0 {
@ -138,6 +147,13 @@ func (b *OllamaBackend) ChatStream(ctx context.Context, model string, messages [
return
}
for _, tc := range wire.Message.ToolCalls {
accumulated = append(accumulated, ToolCall{
Name: tc.Function.Name,
Arguments: tc.Function.Arguments,
})
}
if wire.Done {
stopReason := "stop"
if wire.DoneReason != "" {
@ -147,6 +163,7 @@ func (b *OllamaBackend) ChatStream(ctx context.Context, model string, messages [
Done: true,
StopReason: stopReason,
Usage: Usage{InputTokens: wire.PromptEvalCount, OutputTokens: wire.EvalCount},
ToolCalls: accumulated,
}
return
}
@ -155,13 +172,7 @@ func (b *OllamaBackend) ChatStream(ctx context.Context, model string, messages [
if wire.Message.Content != "" {
ev.Content = wire.Message.Content
}
for _, tc := range wire.Message.ToolCalls {
ev.ToolCalls = append(ev.ToolCalls, ToolCall{
Name: tc.Function.Name,
Arguments: tc.Function.Arguments,
})
}
if ev.Content != "" || len(ev.ToolCalls) > 0 {
if ev.Content != "" {
ch <- ev
}
}

164
main.go
View File

@ -11,8 +11,6 @@ import (
"strconv"
"strings"
"time"
"unicode"
"unicode/utf8"
"github.com/charmbracelet/bubbles/key"
"github.com/charmbracelet/bubbles/textarea"
@ -28,25 +26,78 @@ import (
"ollie/tools"
)
const systemPrompt = `Be terse. No preamble, narration, or filler ("Let me...", "I'll now...", "Great!").
const systemPromptBase = `Be terse. No preamble, narration, or filler ("Let me...", "I'll now...", "Great!").
Output only errors, ambiguities requiring clarification, and deliverables.
Do not restate tasks, hedge, or self-congratulate.`
Do not restate tasks, hedge, or self-congratulate.
Always use tools to perform actions; never simulate or guess outputs.
Do not attempt tasks outside your tools.
Use execute_code for all shell commands and scripts. Use execute_tool only for named scripts in ~/mnt/anvillm/tools. Use execute_pipe to chain steps: use {code: "cmd --flags"} for shell commands, {tool, args} only for named scripts in ~/mnt/anvillm/tools.`
var executeCodeTool = backend.Tool{
func systemPrompt(allTools []backend.Tool) string {
cwd, _ := os.Getwd()
now := time.Now().Format("2006-01-02 15:04:05 MST")
names := make([]string, len(allTools))
for i, t := range allTools {
names[i] = t.Name
}
s := systemPromptBase + "\n\nWorking directory: " + cwd +
"\nCurrent time: " + now +
"\nAvailable tools: " + strings.Join(names, ", ")
return s
}
var builtinTools = []backend.Tool{
{
Name: "execute_code",
Description: "Execute shell code or a named tool script in a sandboxed environment. Use 'code' for inline bash, 'tool'+'args' for a named script, or 'pipe' for a sequence of {tool, args} steps.",
Description: "Run inline code in a sandboxed environment.",
Parameters: json.RawMessage(`{
"type": "object",
"required": ["code"],
"properties": {
"code": {"type": "string", "description": "Inline shell code to run (bash)."},
"code": {"type": "string", "description": "Code to execute."},
"language": {"type": "string", "description": "Language interpreter (default: bash)."},
"timeout": {"type": "integer", "description": "Timeout in seconds (default: 30)."},
"sandbox": {"type": "string", "description": "Sandbox name (default: default)."},
"tool": {"type": "string", "description": "Named tool script to run instead of inline code."},
"args": {"type": "array", "items": {"type": "string"}, "description": "Arguments for the tool script."},
"pipe": {"type": "array", "description": "Pipeline: array of {tool, args} objects run in sequence."}
"sandbox": {"type": "string", "description": "Sandbox name (default: default)."}
}
}`),
},
{
Name: "execute_tool",
Description: "Run a named tool script from ~/mnt/anvillm/tools in a sandboxed environment. Use this only for named scripts, not for inline shell commands.",
Parameters: json.RawMessage(`{
"type": "object",
"required": ["tool"],
"properties": {
"tool": {"type": "string", "description": "Name of the tool script (e.g. discover_skill.sh)."},
"args": {"type": "array", "items": {"type": "string"}, "description": "Arguments for the tool script."},
"timeout": {"type": "integer", "description": "Timeout in seconds (default: 30)."},
"sandbox": {"type": "string", "description": "Sandbox name (default: default)."}
}
}`),
},
{
Name: "execute_pipe",
Description: "Run a pipeline of steps, piping stdout of each into stdin of the next. Use {code: \"cmd --flags\"} for shell commands; use {tool, args} only for named scripts in ~/mnt/anvillm/tools.",
Parameters: json.RawMessage(`{
"type": "object",
"required": ["pipe"],
"properties": {
"pipe": {
"type": "array",
"items": {
"type": "object",
"properties": {
"tool": {"type": "string"},
"args": {"type": "array", "items": {"type": "string"}},
"code": {"type": "string"}
}
}
},
"timeout": {"type": "integer", "description": "Timeout in seconds (default: 30)."},
"sandbox": {"type": "string", "description": "Sandbox name (default: default)."}
}
}`),
},
}
// agentState represents what the agent is currently doing.
@ -187,7 +238,7 @@ func buildAgentEnv(cfg *config.Config, builtinExec *execpkg.Executor) agentEnv {
serverOf[t.Name] = t.Server
}
allTools := append(mcpToolsToBackend(mcpTools), executeCodeTool)
allTools := append(mcpToolsToBackend(mcpTools), builtinTools...)
hooks := map[string]string{}
agentPrompt := ""
@ -198,9 +249,9 @@ func buildAgentEnv(cfg *config.Config, builtinExec *execpkg.Executor) agentEnv {
agentPrompt = cfg.Prompt
}
sp := systemPrompt
sp := systemPrompt(allTools)
if agentPrompt != "" {
sp = systemPrompt + "\n\n" + agentPrompt
sp = systemPrompt(allTools) + "\n\n" + agentPrompt
}
overhead := len(sp)
@ -208,6 +259,11 @@ func buildAgentEnv(cfg *config.Config, builtinExec *execpkg.Executor) agentEnv {
overhead += len(t.Name) + len(t.Description) + len(t.Parameters)
}
builtinNames := make(map[string]struct{}, len(builtinTools))
for _, t := range builtinTools {
builtinNames[t.Name] = struct{}{}
}
execFn := func(name string, args json.RawMessage) (string, error) {
if server, ok := serverOf[name]; ok {
raw, err := mcpExec.Execute(server, name, args)
@ -216,8 +272,8 @@ func buildAgentEnv(cfg *config.Config, builtinExec *execpkg.Executor) agentEnv {
}
return extractMCPText(raw), nil
}
if name == "execute_code" {
return dispatchBuiltinExec(builtinExec, args)
if _, ok := builtinNames[name]; ok {
return dispatchBuiltinExec(name, builtinExec, args)
}
return "", fmt.Errorf("unknown tool: %s", name)
}
@ -358,7 +414,7 @@ func (m model) renderDisplay() string {
}
if m.buf != "" {
if len(m.display) > 0 {
b.WriteByte('\n')
b.WriteString("\n\n")
}
b.WriteString("Bot: " + m.buf)
}
@ -416,20 +472,19 @@ func (m model) renderStatusBar() string {
// LLM APIs always deliver complete BPE tokens — which never split mid-word —
// inserting a space whenever two non-space runes meet is safe in practice.
func addStreamingContent(buf, chunk string) string {
if buf == "" || chunk == "" {
return buf + chunk
}
last, _ := utf8.DecodeLastRuneInString(buf)
first, _ := utf8.DecodeRuneInString(chunk)
if !unicode.IsSpace(last) && !unicode.IsSpace(first) {
return buf + " " + chunk
func (m *model) appendDisplay(line string) {
if len(m.display) > 0 {
m.display = append(m.display, "")
}
return buf + chunk
m.display = append(m.display, line)
}
func (m *model) finalizeBuf() {
if m.buf != "" {
m.display = append(m.display, "Bot: "+m.buf)
m.appendDisplay("Bot: " + m.buf)
m.buf = ""
}
}
@ -449,15 +504,20 @@ func (m *model) apply(am agentMsg) {
if len(args) > 500 {
args = args[:500] + "..."
}
m.display = append(m.display, fmt.Sprintf("-> %s(%s)", am.name, args))
m.appendDisplay(fmt.Sprintf("-> %s(%s)", am.name, args))
case "tool":
m.state = agentThinking
s := squashWhitespace(am.content)
lines := strings.Split(strings.TrimRight(am.content, "\n"), "\n")
var squashed []string
for _, l := range lines {
squashed = append(squashed, squashWhitespace(l))
}
s := strings.Join(squashed, "\n")
if len(s) > 500 {
s = s[:500] + "..."
}
m.display = append(m.display, "= "+s)
m.appendDisplay("= " + s)
case "retry":
m.state = agentRetrying
@ -467,7 +527,7 @@ func (m *model) apply(am agentMsg) {
case "error":
m.finalizeBuf()
m.display = append(m.display, "Error: "+am.content)
m.appendDisplay("Error: " + am.content)
case "usage":
// Update status bar only; no display line appended.
@ -581,7 +641,7 @@ func (m model) Update(msg tea.Msg) (tea.Model, tea.Cmd) {
}
m.drainAgent()
m.finalizeBuf()
m.display = append(m.display, "You: "+input)
m.appendDisplay("You: " + input)
m.refreshView()
m.textarea.Reset()
@ -833,7 +893,7 @@ func extractMCPText(raw json.RawMessage) string {
return strings.Join(parts, "\n")
}
func dispatchBuiltinExec(e *execpkg.Executor, args json.RawMessage) (string, error) {
func dispatchBuiltinExec(name string, e *execpkg.Executor, args json.RawMessage) (string, error) {
var a struct {
Code string `json:"code"`
Language string `json:"language"`
@ -844,18 +904,36 @@ func dispatchBuiltinExec(e *execpkg.Executor, args json.RawMessage) (string, err
Pipe []execpkg.PipeStep `json:"pipe"`
}
if err := json.Unmarshal(args, &a); err != nil {
return "", fmt.Errorf("execute_code: bad args: %w", err)
return "", fmt.Errorf("%s: bad args: %w", name, err)
}
code := a.Code
if a.Language == "" {
a.Language = "bash"
}
if a.Timeout <= 0 {
a.Timeout = 30
}
if a.Sandbox == "" {
a.Sandbox = "default"
}
var code string
trusted := false
switch {
case len(a.Pipe) > 0:
switch name {
case "execute_pipe":
if len(a.Pipe) == 0 {
return "", fmt.Errorf("execute_pipe: 'pipe' is required")
}
var err error
code, trusted, err = execpkg.BuildPipeline(a.Pipe)
if err != nil {
return "", err
}
case a.Tool != "":
case "execute_tool":
if a.Tool == "" {
return "", fmt.Errorf("execute_tool: 'tool' is required")
}
toolCode, err := execpkg.ReadTool(a.Tool)
if err != nil {
return "", err
@ -869,18 +947,12 @@ func dispatchBuiltinExec(e *execpkg.Executor, args json.RawMessage) (string, err
}
code = fmt.Sprintf("set -- %s\n%s", strings.Join(escaped, " "), code)
}
default: // execute_code
if a.Code == "" {
return "", fmt.Errorf("execute_code: 'code' is required")
}
if code == "" {
return "", fmt.Errorf("execute_code: one of 'code', 'tool', or 'pipe' is required")
}
if a.Language == "" {
a.Language = "bash"
}
if a.Timeout <= 0 {
a.Timeout = 30
}
if a.Sandbox == "" {
a.Sandbox = "default"
code = a.Code
}
return e.Execute(code, a.Language, a.Timeout, a.Sandbox, trusted)
}