feat: cap tool output at source + improve context pruning pipeline

Problem: Agent sessions accumulated 71K+ input tokens (83% history)
because read_file and exec had no output limits. SOUL personality
drowned by massive context.

Changes:
- read_file: add offset/limit params + 50K char output cap with
  pagination hints (model can re-read with offset)
- exec/shell: cap output at 30K chars with smart head+tail truncation
  (preserves errors/summaries at tail)
- pruning: add per-result 30% context guard, tune softTrimRatio
  0.3→0.25 and softTrimMaxChars 4K→3K, add tail-aware soft trim
- mid-loop: allow pruning to re-trigger each iteration (was one-shot)

Design: cap at source, preserve full data in session, prune at
consumption time. read_file offset/limit enables recovery of
truncated content.
This commit is contained in:
viettranx committed 2026-03-31 08:35:06 +07:00
1 parent fe163eca30
commit ab5bad11ff
8 files changed
+259 -19

No files matched your search

+2 -2
View File
@@ -375,8 +375,8 @@ func (l *Loop) runLoop(ctx context.Context, req RunRequest) (result *RunResult,
}
// Phase 1: Prune old tool results before resorting to full compaction (at 70% of budget).
if historyTokens >= int(float64(historyBudget)*0.7) && !rs.midLoopPruned {
rs.midLoopPruned = true
// Re-triggers each iteration — new tool results may have grown context since last prune.
if historyTokens >= int(float64(historyBudget)*0.7) {
pruned := pruneContextMessages(messages, l.contextWindow, l.contextPruningCfg)
if len(pruned) > 0 {
messages = pruned
-1
View File
@@ -519,7 +519,6 @@ type runState struct {
// Mid-loop compaction and overhead calibration
midLoopCompacted bool
midLoopPruned bool
overheadTokens int // non-history token overhead (system prompt + tools + context files)
overheadCalibrated bool
+60 -6
View File
@@ -11,10 +11,10 @@ import (
// Context pruning defaults matching TS DEFAULT_CONTEXT_PRUNING_SETTINGS.
const (
defaultKeepLastAssistants = 3
defaultSoftTrimRatio = 0.3
defaultSoftTrimRatio = 0.25
defaultHardClearRatio = 0.5
defaultMinPrunableToolChars = 50000
defaultSoftTrimMaxChars = 4000
defaultSoftTrimMaxChars = 3000
defaultSoftTrimHeadChars = 1500
defaultSoftTrimTailChars = 1500
defaultHardClearPlaceholder = "[Old tool result content cleared]"
@@ -148,8 +148,42 @@ func pruneContextMessages(msgs []providers.Message, contextWindowTokens int, cfg
return msgs
}
// Pass 1: Soft trim long tool results.
// Pass 0: Per-result context guard — force-trim any single tool result
// exceeding 30% of the context window. Catches outlier outputs even
// when overall context ratio is low.
maxSingleResultChars := charWindow * 3 / 10
var result []providers.Message
for _, idx := range prunableIndexes {
msgChars := estimateMessageChars(msgs[idx])
if msgChars > maxSingleResultChars {
if result == nil {
result = make([]providers.Message, len(msgs))
copy(result, msgs)
}
msg := msgs[idx]
head := takeHead(msg.Content, maxSingleResultChars*7/10)
tail := takeTail(msg.Content, maxSingleResultChars*3/10)
trimmed := fmt.Sprintf("%s\n\n⚠️ [... middle content omitted ...]\n\n%s\n\n[Single tool result trimmed: %d chars exceeded per-result limit of %d chars.]",
head, tail, msgChars, maxSingleResultChars)
result[idx] = providers.Message{
Role: msg.Role,
Content: trimmed,
ToolCallID: msg.ToolCallID,
}
totalChars += len(trimmed) - msgChars
}
}
if result != nil {
msgs = result
result = nil
// Re-check ratio after per-result guard.
ratio = float64(totalChars) / float64(charWindow)
if ratio < settings.softTrimRatio {
return msgs
}
}
// Pass 1: Soft trim long tool results.
for i := range prunableIndexes {
idx := prunableIndexes[i]
msg := msgs[idx]
@@ -165,10 +199,19 @@ func pruneContextMessages(msgs []providers.Message, contextWindowTokens int, cfg
copy(result, msgs)
}
head := takeHead(msg.Content, settings.softTrimHeadChars)
tail := takeTail(msg.Content, settings.softTrimTailChars)
// Tail-aware split: if tail has important content (errors, summaries),
// use dynamic 70/30 split. Otherwise use configured head/tail sizes.
headChars := settings.softTrimHeadChars
tailChars := settings.softTrimTailChars
if hasImportantTail(msg.Content) {
totalBudget := headChars + tailChars
headChars = totalBudget * 7 / 10
tailChars = totalBudget - headChars
}
head := takeHead(msg.Content, headChars)
tail := takeTail(msg.Content, tailChars)
trimmed := fmt.Sprintf("%s\n...\n%s\n\n[Tool result trimmed: kept first %d chars and last %d chars of %d chars.]",
head, tail, settings.softTrimHeadChars, settings.softTrimTailChars, msgChars)
head, tail, headChars, tailChars, msgChars)
result[idx] = providers.Message{
Role: msg.Role,
@@ -250,6 +293,17 @@ func estimateMessageChars(m providers.Message) int {
return utf8.RuneCountInString(m.Content)
}
// hasImportantTail checks if the last ~500 chars of content contain error/summary keywords.
func hasImportantTail(content string) bool {
runes := []rune(content)
checkLen := 500
if checkLen > len(runes) {
checkLen = len(runes)
}
tail := string(runes[len(runes)-checkLen:])
return importantTailRe.MatchString(tail)
}
// takeHead returns the first n runes of s.
func takeHead(s string, n int) string {
if n <= 0 {
+8
View File
@@ -0,0 +1,8 @@
package agent
import "regexp"
// importantTailRe matches keywords in the tail of output that indicate
// the tail contains important information (errors, summaries, results).
// Used by pruning.go's hasImportantTail() for tail-aware soft trim.
var importantTailRe = regexp.MustCompile(`(?i)(error|exception|failed|fatal|traceback|panic|stack trace|exit code|total|summary|result|complete|finished|done)\b`)
+6 -2
View File
@@ -223,7 +223,9 @@ func (t *ExecTool) executeCredentialedSandbox(ctx context.Context, absPath strin
if output == "" {
output = "(command completed with no output)"
}
return SilentResult(ScrubCredentials(output))
output = ScrubCredentials(output)
output = capExecOutput(output, execMaxOutputChars)
return SilentResult(output)
}
// buildCredentialedEnv creates a minimal environment with injected credentials.
@@ -270,7 +272,9 @@ func formatCredentialedResult(binary string, args []string,
if output == "" {
output = "(command completed with no output)"
}
return SilentResult(ScrubCredentials(output))
output = ScrubCredentials(output)
output = capExecOutput(output, execMaxOutputChars)
return SilentResult(output)
}
// lookupCredentialedBinary checks if a command's binary has credential config.
+84
View File
@@ -0,0 +1,84 @@
package tools
import (
"fmt"
"regexp"
"strings"
"unicode/utf8"
)
// cutAtLastNewline trims text to the nearest newline in the last 20% of the string.
func cutAtLastNewline(text string) string {
threshold := len(text) * 80 / 100
if idx := strings.LastIndex(text[threshold:], "\n"); idx >= 0 {
return text[:threshold+idx]
}
return text
}
// cutAtFirstNewline trims text from the start at the nearest newline in the first 20%.
func cutAtFirstNewline(text string) string {
searchEnd := len(text) * 20 / 100
if searchEnd == 0 {
return text
}
if idx := strings.Index(text[:searchEnd], "\n"); idx >= 0 {
return text[idx+1:]
}
return text
}
// execMaxOutputChars is the maximum characters kept from exec/shell output.
// Larger outputs are truncated with head+tail strategy.
const execMaxOutputChars = 30000
// execImportantTailRe matches keywords indicating the tail contains important info.
var execImportantTailRe = regexp.MustCompile(`(?i)(error|exception|failed|fatal|traceback|panic|stack trace|exit code|total|summary|result|complete|finished|done|}\s*$)`)
// capExecOutput truncates exec output to maxChars using smart head+tail strategy.
// If the tail contains important content (errors, summaries), keeps 70% head + 30% tail.
// Otherwise keeps head only. Returns the original text if it fits within maxChars.
func capExecOutput(output string, maxChars int) string {
if utf8.RuneCountInString(output) <= maxChars {
return output
}
runes := []rune(output)
totalRunes := len(runes)
suffix := fmt.Sprintf("\n\n[Output truncated: %d chars total. Redirect to file for full output: command > output.txt]", totalRunes)
budget := maxChars - utf8.RuneCountInString(suffix)
if budget < 2000 {
budget = 2000
}
// Check if tail has important content.
tailCheckLen := 2000
if tailCheckLen > totalRunes {
tailCheckLen = totalRunes
}
tailSample := string(runes[totalRunes-tailCheckLen:])
if execImportantTailRe.MatchString(tailSample) && budget > 4000 {
// Smart split: 70% head + 30% tail.
headBudget := budget * 7 / 10
tailBudget := budget - headBudget
if tailBudget > 4000 {
tailBudget = 4000
headBudget = budget - tailBudget
}
head := string(runes[:headBudget])
tail := string(runes[totalRunes-tailBudget:])
// Cut at newline boundaries for cleaner output.
head = cutAtLastNewline(head)
tail = cutAtFirstNewline(tail)
return head + "\n\n⚠️ [... middle content omitted ...]\n\n" + tail + suffix
}
// Head-only truncation.
head := string(runes[:budget])
head = cutAtLastNewline(head)
return head + suffix
}
+97 -6
View File
@@ -69,8 +69,10 @@ func NewSandboxedReadFileTool(workspace string, restrict bool, mgr sandbox.Manag
// SetSandboxKey is a no-op; sandbox key is now read from ctx (thread-safe).
func (t *ReadFileTool) SetSandboxKey(key string) {}
func (t *ReadFileTool) Name() string { return "read_file" }
func (t *ReadFileTool) Description() string { return "Read the contents of a file" }
func (t *ReadFileTool) Name() string { return "read_file" }
func (t *ReadFileTool) Description() string {
return "Read the contents of a file. For large files, use offset and limit to read specific line ranges."
}
func (t *ReadFileTool) Parameters() map[string]any {
return map[string]any{
"type": "object",
@@ -79,6 +81,14 @@ func (t *ReadFileTool) Parameters() map[string]any {
"type": "string",
"description": "File path (relative to workspace, or absolute)",
},
"offset": map[string]any{
"type": "integer",
"description": "Start reading from this line number (0-indexed). Defaults to 0.",
},
"limit": map[string]any{
"type": "integer",
"description": "Maximum number of lines to return. Omit to read until output cap.",
},
},
"required": []string{"path"},
}
@@ -136,7 +146,7 @@ func (t *ReadFileTool) Execute(ctx context.Context, args map[string]any) *Result
// Sandbox routing (sandboxKey from ctx — thread-safe)
sandboxKey := ToolSandboxKeyFromCtx(ctx)
if t.sandboxMgr != nil && sandboxKey != "" {
return t.executeInSandbox(ctx, path, sandboxKey)
return t.executeInSandbox(ctx, path, sandboxKey, args)
}
// Host execution — use per-user workspace from context if available
@@ -170,10 +180,10 @@ func (t *ReadFileTool) Execute(ctx context.Context, args map[string]any) *Result
return ErrorResult(msg)
}
return SilentResult(string(data))
return t.paginateOutput(string(data), args)
}
func (t *ReadFileTool) executeInSandbox(ctx context.Context, path, sandboxKey string) *Result {
func (t *ReadFileTool) executeInSandbox(ctx context.Context, path, sandboxKey string, args map[string]any) *Result {
bridge, err := t.getFsBridge(ctx, sandboxKey)
if err != nil {
return ErrorResult(fmt.Sprintf("sandbox error: %v", err))
@@ -190,7 +200,7 @@ func (t *ReadFileTool) executeInSandbox(ctx context.Context, path, sandboxKey st
return ErrorResult(fmt.Sprintf("failed to read file: %v", err) + MaybeFsBridgeHint(err))
}
return SilentResult(data)
return t.paginateOutput(data, args)
}
func (t *ReadFileTool) getFsBridge(ctx context.Context, sandboxKey string) (*sandbox.FsBridge, error) {
@@ -201,6 +211,87 @@ func (t *ReadFileTool) getFsBridge(ctx context.Context, sandboxKey string) (*san
return sandbox.NewFsBridge(sb.ID(), sandbox.DefaultContainerWorkdir), nil
}
// readFileMaxChars is the output cap for read_file. Large files require offset/limit pagination.
const readFileMaxChars = 50000
// paginateOutput applies offset/limit slicing and output capping to file content.
// Returns a SilentResult with pagination metadata when the output is truncated.
func (t *ReadFileTool) paginateOutput(content string, args map[string]any) *Result {
lines := strings.Split(content, "\n")
totalLines := len(lines)
// Parse offset (0-indexed line number).
offset := 0
if v, ok := args["offset"]; ok {
switch n := v.(type) {
case float64:
offset = int(n)
case int:
offset = n
}
}
if offset < 0 {
offset = 0
}
if offset >= totalLines {
return SilentResult(fmt.Sprintf("(offset %d exceeds file length of %d lines)", offset, totalLines))
}
// Parse limit (max lines to return).
limit := 0 // 0 = no explicit limit
if v, ok := args["limit"]; ok {
switch n := v.(type) {
case float64:
limit = int(n)
case int:
limit = n
}
}
// Slice lines by offset and limit.
sliced := lines[offset:]
if limit > 0 && limit < len(sliced) {
sliced = sliced[:limit]
}
output := strings.Join(sliced, "\n")
shownLines := len(sliced)
endLine := offset + shownLines
// Check output char cap.
runeCount := len([]rune(output))
if runeCount <= readFileMaxChars {
// Fits within cap — add line info if offset was used or file was partially read.
if offset > 0 || endLine < totalLines {
output += fmt.Sprintf("\n\n[Showing lines %d-%d of %d total]", offset, endLine-1, totalLines)
}
return SilentResult(output)
}
// Output exceeds cap — truncate at line boundary within budget.
charCount := 0
truncIdx := len(sliced)
for i, line := range sliced {
charCount += len([]rune(line)) + 1 // +1 for newline
if charCount > readFileMaxChars {
truncIdx = i
break
}
}
if truncIdx < 1 {
truncIdx = 1
}
output = strings.Join(sliced[:truncIdx], "\n")
shownLines = truncIdx
nextOffset := offset + shownLines
output += fmt.Sprintf("\n\n[Output capped. File has %d lines, showed %d (lines %d-%d). Use offset=%d to continue reading.]",
totalLines, shownLines, offset, offset+shownLines-1, nextOffset)
return SilentResult(output)
}
// allowedWithTeamWorkspace returns the allowed prefixes with team workspace appended
// if present in context. Thread-safe: creates a new slice per request.
func allowedWithTeamWorkspace(ctx context.Context, base []string) []string {
+2 -2
View File
@@ -290,7 +290,7 @@ func (t *ExecTool) executeOnHost(ctx context.Context, command, cwd string) *Resu
result = "(command completed with no output)"
}
return SilentResult(result)
return SilentResult(capExecOutput(result, execMaxOutputChars))
}
// executeInSandbox routes a command through a Docker sandbox container.
@@ -339,7 +339,7 @@ func (t *ExecTool) executeInSandbox(ctx context.Context, command, cwd, sandboxKe
output = "(command completed with no output)"
}
return SilentResult(output)
return SilentResult(capExecOutput(output, execMaxOutputChars))
}
// limitedBuffer caps output to prevent OOM from runaway commands.