fix(agent): make same-call repeat guard progress-aware
CI / Tidy (pull_request) Successful in 9m25s
Gadfly review (reusable) / review (pull_request) Successful in 9m48s
Adversarial Review (Gadfly) / review (pull_request) Successful in 9m48s
CI / Build & Test (pull_request) Successful in 10m33s

The maxSameCallRepeats guard counted identical (name+arguments) tool calls
across a run and tripped ErrToolLoop past the ceiling — regardless of whether
each call made progress. This killed legitimate polling of long-running
background jobs: code_exec_poll must be called with identical args (same
job_id), so a render/encode that needs more than N polls was guillotined
mid-flight even as each poll returned an advancing result (elapsed/status
moving forward).

Only count an identical call toward the trip when its RESULT is unchanged
from the previous identical call. A call whose result keeps changing is
progress and resets its count; a genuinely stuck call returning the same
output still trips. This can never trip more than before, only less, and
covers every idempotent poller with no per-tool configuration — matching the
progress-over-usage thesis behind the stall-detection work.

Co-Authored-By: Claude Opus 4.8 (1M context) <[email protected]>
Claude-Session: https://claude.ai/code/session_01HgEuVfZJN9mhRhzEsMEVog
This commit is contained in:
2026-07-18 18:56:56 -04:00
co-authored by Claude Opus 4.8
parent 54b295efc3
commit 68bf7157d3
2 changed files with 91 additions and 11 deletions
+43 -11
View File
@@ -125,10 +125,12 @@ func WithCompactor(fn func(ctx context.Context, msgs []llm.Message) ([]llm.Messa
}
// WithToolErrorLimits installs loop guards: maxConsecutiveErrors bounds
// successive steps whose tool results were ALL errors, and
// maxSameCallRepeats bounds identical (name + arguments) tool calls within
// one run. Either guard tripping ends the run with ErrToolLoop and the
// partial result. Zero disables a guard.
// successive steps whose tool results were ALL errors, and maxSameCallRepeats
// bounds identical (name + arguments) tool calls that ALSO return an unchanged
// result within one run — a call whose result keeps advancing (e.g. polling a
// long-running background job) is progress and never trips this guard. Either
// guard tripping ends the run with ErrToolLoop and the partial result. Zero
// disables a guard.
func WithToolErrorLimits(maxConsecutiveErrors, maxSameCallRepeats int) Option {
return func(a *Agent) {
a.maxConsecutiveToolErrors = maxConsecutiveErrors
@@ -276,7 +278,13 @@ func (a *Agent) Run(ctx context.Context, input string, opts ...RunOption) (*Resu
// Loop-guard state (WithToolErrorLimits).
consecutiveErrorSteps := 0
// callCounts tracks a run-length of identical (name+arguments) tool calls
// that ALSO returned the same result. lastResults holds the previous result
// per signature so a call whose result keeps changing (e.g. polling a
// background job whose progress advances) resets the count instead of
// tripping the guard — a changing result is progress, not a stuck loop.
callCounts := make(map[string]int)
lastResults := make(map[string]string)
maxSteps := func() int {
if a.maxStepsFunc != nil {
@@ -330,13 +338,6 @@ func (a *Agent) Run(ctx context.Context, input string, opts ...RunOption) (*Resu
result.Messages = msgs
return result, err
}
if a.maxSameCallRepeats > 0 {
sig := call.Name + "\x00" + string(call.Arguments)
callCounts[sig]++
if callCounts[sig] > a.maxSameCallRepeats {
repeatTripped = call.Name
}
}
tool, ok := byName[call.Name]
if !ok {
results = append(results, llm.ToolResult{
@@ -356,6 +357,37 @@ func (a *Agent) Run(ctx context.Context, input string, opts ...RunOption) (*Resu
a.notify(rc, step)
msgs = append(msgs, llm.ToolResultsMessage(results...))
// Same-call repeat guard (progress-aware). An identical (name+arguments)
// call only counts toward the loop trip when its result is unchanged from
// the previous identical call. A call whose result advances each time —
// canonically polling a long-running background job (elapsed/status keeps
// moving) — resets its count and never trips, while a genuinely stuck
// call returning the same output repeats until it exceeds the ceiling.
// results[i] pairs with resp.ToolCalls[i]: every call appends exactly one
// result (unknown tools append an error result before continue), and the
// only early exit above is a full return on ctx cancellation.
if a.maxSameCallRepeats > 0 {
for i, call := range resp.ToolCalls {
sig := call.Name + "\x00" + string(call.Arguments)
resKey := ""
if i < len(results) {
if results[i].IsError {
resKey = "e\x00"
}
resKey += results[i].Content
}
if prev, seen := lastResults[sig]; seen && prev == resKey {
callCounts[sig]++
} else {
callCounts[sig] = 1
}
lastResults[sig] = resKey
if callCounts[sig] > a.maxSameCallRepeats {
repeatTripped = call.Name
}
}
}
if repeatTripped != "" {
result.Messages = msgs
return result, fmt.Errorf("%w: %q called identically more than %d times",
+48
View File
@@ -173,3 +173,51 @@ func TestSameCallRepeatGuard(t *testing.T) {
t.Errorf("varied calls must not trip the guard: %v", err)
}
}
// TestSameCallRepeatGuardProgressAware: identical (name+args) calls whose
// RESULT keeps changing — canonically polling a long-running background job
// whose progress advances — do not trip the repeat guard even well past the
// limit; but identical calls returning an unchanged result still trip it.
func TestSameCallRepeatGuardProgressAware(t *testing.T) {
// A poll tool that advances every call, so identical args yield a
// different result each time.
calls := 0
polling := llm.NewToolbox("jobs", llm.Tool{
Name: "poll",
Handler: func(context.Context, json.RawMessage) (any, error) {
calls++
return map[string]any{"status": "running", "elapsed": calls}, nil
},
})
n := 0
fp := fake.New("fp", fake.WithDefault(func(string, llm.Request) fake.Step {
n++
if n > 6 { // six identical polls, well past the limit of 3
return fake.Reply("done")
}
return toolCallReply("c", "poll", `{"job":"x"}`)
}))
a := New(newModel(t, fp), "", WithToolbox(polling), WithToolErrorLimits(0, 3), WithMaxSteps(20))
res, err := a.Run(context.Background(), "go")
if err != nil {
t.Fatalf("advancing-result polls must not trip the guard: %v", err)
}
if res.Output != "done" {
t.Errorf("output = %q, want run to complete after polling", res.Output)
}
// A call that returns an UNCHANGED result each time still trips the guard.
frozen := llm.NewToolbox("jobs", llm.Tool{
Name: "poll",
Handler: func(context.Context, json.RawMessage) (any, error) {
return map[string]any{"status": "running"}, nil // never advances
},
})
fp2 := fake.New("fp", fake.WithDefault(func(string, llm.Request) fake.Step {
return toolCallReply("c", "poll", `{"job":"x"}`)
}))
a2 := New(newModel(t, fp2), "", WithToolbox(frozen), WithToolErrorLimits(0, 3), WithMaxSteps(20))
if _, err := a2.Run(context.Background(), "go"); !errors.Is(err, ErrToolLoop) {
t.Fatalf("frozen identical result must still trip the guard: %v", err)
}
}