Skip to content

Commit 7b09b92

Browse files
Coalesce consecutive tool-failure recovery audits into one count (#900)
* Nudge once when assistants print tool-call markup as text Models sometimes emit <tool_call><function=...> wrappers as assistant text instead of real tool_call blocks. Catch that narrow shape, give one corrective nudge per no-real-tool epoch without counting it as incomplete-report narration, then fall through to the existing report policy. Thinking blocks and arbitrary XML stay out of scope; the epoch resets only on genuine tool activity or a parent follow-up. * Let a complete report envelope win over quoted tool-call markup * Coalesce consecutive tool-failure recovery audits into one count Several failed tool.done events before the pending recovery nudge is consumed were each writing a separate intervention line. Keep the recovery nudge text and arming behavior the same, but count the burst in director memory and flush a single record with count when the nudge is applied. Forensics treats missing count as one. * Flush coalesced tool-failure audits on terminal paths
1 parent b491d45 commit 7b09b92

5 files changed

Lines changed: 131 additions & 10 deletions

File tree

‎scripts/intervention-forensics.ts‎

Lines changed: 8 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -151,15 +151,19 @@ for (const file of files) {
151151
bucket = emptyBucket();
152152
buckets.set(key, bucket);
153153
}
154-
bucket.count++;
154+
const occurrence = record.count ?? 1;
155+
bucket.count += occurrence;
155156
const family = record.family ?? record.model ?? "unknown";
156-
bucket.byFamily.set(family, (bucket.byFamily.get(family) ?? 0) + 1);
157+
bucket.byFamily.set(
158+
family,
159+
(bucket.byFamily.get(family) ?? 0) + occurrence,
160+
);
157161
const model = record.model ?? "unknown";
158-
bucket.byModel.set(model, (bucket.byModel.get(model) ?? 0) + 1);
162+
bucket.byModel.set(model, (bucket.byModel.get(model) ?? 0) + occurrence);
159163
if (record.class === "stop" || record.class === "nudge") {
160164
interventionsByModel.set(
161165
model,
162-
(interventionsByModel.get(model) ?? 0) + 1,
166+
(interventionsByModel.get(model) ?? 0) + occurrence,
163167
);
164168
}
165169
if (record.measurement !== undefined) {

‎src/subagent/intervention-log.test.ts‎

Lines changed: 10 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -76,6 +76,16 @@ describe("intervention log", () => {
7676
expect(records.map((r) => r.id)).toEqual(["report-forced", "turn-budget"]);
7777
});
7878

79+
test("preserves an optional coalesced count on the record", async () => {
80+
const dir = await mkdtemp(join(tmpdir(), "intervention-log-"));
81+
const sink = createInterventionLog(dir, { role: "leaf" });
82+
sink({ id: "tool-failure-recovery", class: "nudge", count: 3 });
83+
await flush();
84+
85+
const [record] = await readRecords(dir);
86+
expect(record?.count).toBe(3);
87+
});
88+
7989
test("a write failure never throws into the caller", async () => {
8090
const sink = createInterventionLog(
8191
join(tmpdir(), "intervention-log-missing-dir-xyz"),

‎src/subagent/intervention-log.ts‎

Lines changed: 6 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -90,6 +90,12 @@ export interface InterventionRecord {
9090
};
9191
/** Free-form specifics, kept short (a looped window, a refused fingerprint). */
9292
detail?: string;
93+
/**
94+
* How many consecutive same-trigger audits this record represents. Present when
95+
* the director coalesced a burst (e.g. several failed tool.done events before
96+
* the pending recovery nudge was consumed) into one flush. Absent means one.
97+
*/
98+
count?: number;
9399
}
94100

95101
/** Fields every record from one run shares, supplied once at construction. */

‎src/subagent/nudge-director.test.ts‎

Lines changed: 77 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -179,6 +179,83 @@ describe("SubAgentDirector tool failure recovery", () => {
179179
expect(texts?.[0]).toContain("report the blocker");
180180
});
181181

182+
test("coalesces consecutive failed tool audits into one counted intervention", async () => {
183+
const director = new SubAgentDirector("system", [], undefined, 30);
184+
const caps = capabilities();
185+
const records: { id: string; count?: number }[] = [];
186+
director.observeInterventions((event) => {
187+
records.push(
188+
event.count === undefined
189+
? { id: event.id }
190+
: { id: event.id, count: event.count },
191+
);
192+
});
193+
194+
await director.decide(
195+
inferenceDone(["fail-a", "fail-b", "ok-c"]),
196+
state,
197+
caps,
198+
);
199+
await director.decide(toolDone("fail-a", true), state, caps);
200+
expect(records).toEqual([]);
201+
await director.decide(toolDone("fail-b", true), state, caps);
202+
expect(records).toEqual([]);
203+
204+
const texts = ephemeralTexts(
205+
inferAction(await director.decide(toolDone("ok-c"), state, caps)),
206+
);
207+
expect(texts).toHaveLength(1);
208+
expect(texts?.[0]).toContain("A tool call failed");
209+
expect(records).toEqual([{ id: "tool-failure-recovery", count: 2 }]);
210+
});
211+
212+
test("a single failed tool audit omits the count field", async () => {
213+
const director = new SubAgentDirector("system", [], undefined, 30);
214+
const caps = capabilities();
215+
const records: { id: string; count: number | null }[] = [];
216+
director.observeInterventions((event) => {
217+
records.push({ id: event.id, count: event.count ?? null });
218+
});
219+
220+
await director.decide(inferenceDone(["fail-a"]), state, caps);
221+
await director.decide(toolDone("fail-a", true), state, caps);
222+
await director.decide(inferenceDoneText(REPORT_ENVELOPE), state, caps);
223+
224+
expect(records).toEqual([{ id: "tool-failure-recovery", count: null }]);
225+
});
226+
227+
test("flushes an undelivered recovery burst when the run goes terminal", async () => {
228+
const director = new SubAgentDirector("system", [], undefined, 30);
229+
const caps = capabilities();
230+
const records: { id: string; count?: number }[] = [];
231+
director.observeInterventions((event) => {
232+
records.push(
233+
event.count === undefined
234+
? { id: event.id }
235+
: { id: event.id, count: event.count },
236+
);
237+
});
238+
239+
// ok-c stays pending so the armed recovery nudge never reaches an infer.
240+
await director.decide(
241+
inferenceDone(["fail-a", "fail-b", "ok-c"]),
242+
state,
243+
caps,
244+
);
245+
await director.decide(toolDone("fail-a", true), state, caps);
246+
await director.decide(toolDone("fail-b", true), state, caps);
247+
expect(records).toEqual([]);
248+
249+
const result = actions(
250+
await director.decide(inferenceDoneText(REPORT_ENVELOPE), state, caps),
251+
);
252+
expect(result).toContainEqual({
253+
type: "checkpoint",
254+
message: "subagent-complete",
255+
});
256+
expect(records).toEqual([{ id: "tool-failure-recovery", count: 2 }]);
257+
});
258+
182259
test("successful tool result has no ephemeral recovery turn", async () => {
183260
const director = new SubAgentDirector("system", [], undefined, 30);
184261
const caps = capabilities();

‎src/subagent/nudge-director.ts‎

Lines changed: 30 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -136,6 +136,10 @@ export class SubAgentDirector extends DefaultDirector {
136136
// overflow compact (interceptOverflow re-arms from lastConsumedNudgeText if
137137
// the infer that consumed pending never completed).
138138
private pendingNudgeText: string | null = null;
139+
// How many consecutive failed tool.done audits are waiting to be flushed as
140+
// one tool-failure-recovery intervention when applyPendingNudge consumes the
141+
// pending recovery nudge. Coalesces the audit trail without changing nudge text.
142+
private pendingToolFailureRecoveryCount = 0;
139143
// The text applyPendingNudge last attached to a returned infer. Overflow of
140144
// that infer means the model never saw it, so interceptOverflow re-arms
141145
// pending from this when pending is still null. Cleared on a successful
@@ -325,6 +329,7 @@ export class SubAgentDirector extends DefaultDirector {
325329

326330
if (stop === "complete") {
327331
this.reportReplied = true;
332+
this.flushToolFailureRecoveryAudit();
328333
const terminal: ReactorAction[] = [
329334
capabilities.checkpoint("subagent-complete"),
330335
capabilities.reply(lastText(content)),
@@ -384,6 +389,7 @@ export class SubAgentDirector extends DefaultDirector {
384389
});
385390
this.onForcedStop("incomplete-report");
386391
this.reportReplied = true;
392+
this.flushToolFailureRecoveryAudit();
387393
const terminal: ReactorAction[] = [
388394
capabilities.checkpoint("subagent-incomplete-report"),
389395
capabilities.reply(
@@ -406,13 +412,10 @@ export class SubAgentDirector extends DefaultDirector {
406412
this.lastActivityAt = this.now();
407413
this.consecutiveStalls = 0;
408414
if (event.result.isError === true) {
409-
// Failed-tool recovery guidance.
415+
// Failed-tool recovery guidance. Arm once; coalesce consecutive failure
416+
// audits until applyPendingNudge flushes a single counted record.
410417
this.pendingNudgeText = TOOL_FAILURE_RECOVERY_NUDGE;
411-
this.interventions({
412-
id: "tool-failure-recovery",
413-
class: "nudge",
414-
state: this.interventionState(),
415-
});
418+
this.pendingToolFailureRecoveryCount += 1;
416419
}
417420
}
418421
const base = await super.decide(event, state, capabilities);
@@ -483,6 +486,7 @@ export class SubAgentDirector extends DefaultDirector {
483486
});
484487
this.onForcedStop("stalled");
485488
this.reportReplied = true;
489+
this.flushToolFailureRecoveryAudit();
486490
const terminal: ReactorAction[] = [
487491
capabilities.checkpoint("subagent-stalled"),
488492
capabilities.reply(
@@ -495,6 +499,25 @@ export class SubAgentDirector extends DefaultDirector {
495499
return terminal;
496500
}
497501

502+
/**
503+
* Write the coalesced tool-failure-recovery audit once the burst ends —
504+
* when the armed nudge lands on an infer, or when the run goes terminal
505+
* (complete / forced stop / stalled) with the nudge still undelivered.
506+
* Without the terminal-path flush a burst that is never followed by an
507+
* infer would vanish from the audit trail entirely.
508+
*/
509+
private flushToolFailureRecoveryAudit(): void {
510+
if (this.pendingToolFailureRecoveryCount === 0) return;
511+
const count = this.pendingToolFailureRecoveryCount;
512+
this.pendingToolFailureRecoveryCount = 0;
513+
this.interventions({
514+
id: "tool-failure-recovery",
515+
class: "nudge",
516+
...(count > 1 ? { count } : {}),
517+
state: this.interventionState(),
518+
});
519+
}
520+
498521
/**
499522
* Rewrite the infer action in a fall-through actions batch to carry the
500523
* armed nudge, once — this matches the infer after report-forced or
@@ -510,6 +533,7 @@ export class SubAgentDirector extends DefaultDirector {
510533
const text = this.pendingNudgeText;
511534
this.pendingNudgeText = null;
512535
this.lastConsumedNudgeText = text;
536+
this.flushToolFailureRecoveryAudit();
513537
const existing = actions[inferIndex] as Extract<
514538
ReactorAction,
515539
{ type: "infer" }

0 commit comments

Comments
 (0)