@caupulican/pi-agent-core 0.93.8 → 0.93.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1 +1 @@
1
- {"version":3,"file":"agent-loop.d.ts","sourceRoot":"","sources":["../src/agent-loop.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAEH,OAAO,EAAE,WAAW,EAAE,MAAM,gCAAgC,CAAC;AAiC7D,OAAO,EAEN,uBAAuB,EAGvB,MAAM,iCAAiC,CAAC;AAEzC,OAAO,KAAK,EACX,YAAY,EACZ,UAAU,EACV,eAAe,EACf,YAAY,EAEZ,aAAa,EAEb,QAAQ,EACR,kBAAkB,EAClB,MAAM,YAAY,CAAC;AAIpB,OAAO,EACN,0BAA0B,EAC1B,sBAAsB,EACtB,gCAAgC,GAChC,MAAM,+BAA+B,CAAC;AAEvC,MAAM,MAAM,cAAc,GAAG,CAAC,KAAK,EAAE,UAAU,KAAK,OAAO,CAAC,IAAI,CAAC,GAAG,IAAI,CAAC;AAKzE,gGAAgG;AAChG,MAAM,WAAW,0BAA0B;IAC1C,aAAa,EAAE,MAAM,CAAC;IACtB,WAAW,EAAE,MAAM,EAAE,CAAC;IACtB,uBAAuB,EAAE,uBAAuB,CAAC;CACjD;AAED,wBAAgB,gCAAgC,IAAI,0BAA0B,CAE7E;AAED;;;GAGG;AACH,wBAAgB,SAAS,CACxB,OAAO,EAAE,YAAY,EAAE,EACvB,OAAO,EAAE,YAAY,EACrB,MAAM,EAAE,eAAe,EACvB,MAAM,CAAC,EAAE,WAAW,EACpB,QAAQ,CAAC,EAAE,QAAQ,GACjB,WAAW,CAAC,UAAU,EAAE,YAAY,EAAE,CAAC,CAEzC;AAED;;;;;;;GAOG;AACH,wBAAgB,iBAAiB,CAChC,OAAO,EAAE,YAAY,EACrB,MAAM,EAAE,eAAe,EACvB,MAAM,CAAC,EAAE,WAAW,EACpB,QAAQ,CAAC,EAAE,QAAQ,GACjB,WAAW,CAAC,UAAU,EAAE,YAAY,EAAE,CAAC,CAIzC;AAuBD,wBAAsB,YAAY,CACjC,OAAO,EAAE,YAAY,EAAE,EACvB,OAAO,EAAE,YAAY,EACrB,MAAM,EAAE,eAAe,EACvB,IAAI,EAAE,cAAc,EACpB,MAAM,CAAC,EAAE,WAAW,EACpB,QAAQ,CAAC,EAAE,QAAQ,EACnB,iBAAiB,GAAE,0BAA+D,GAChF,OAAO,CAAC,YAAY,EAAE,CAAC,CAgBzB;AAED,wBAAsB,oBAAoB,CACzC,OAAO,EAAE,YAAY,EACrB,MAAM,EAAE,eAAe,EACvB,IAAI,EAAE,cAAc,EACpB,MAAM,CAAC,EAAE,WAAW,EACpB,QAAQ,CAAC,EAAE,QAAQ,EACnB,iBAAiB,GAAE,0BAA+D,GAChF,OAAO,CAAC,YAAY,EAAE,CAAC,CAWzB;AAoVD;;;;;;GAMG;AACH,wBAAsB,yBAAyB,CAC9C,OAAO,EAAE,YAAY,EACrB,MAAM,EAAE,eAAe,EACvB,MAAM,EAAE,WAAW,GAAG,SAAS,EAC/B,QAAQ,CAAC,EAAE,QAAQ,GACjB,OAAO,CAAC,OAAO,CAAC,UAAU,CAAC,QAAQ,CAAC,CAAC,CAAC,CAExC;AAqlCD,wBAAgB,qBAAqB,CAAC,QAAQ,EAAE,aAAa,GAAG,kBAAkB,GAAG,SAAS,CAO7F"}
1
+ {"version":3,"file":"agent-loop.d.ts","sourceRoot":"","sources":["../src/agent-loop.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAEH,OAAO,EAAE,WAAW,EAAE,MAAM,gCAAgC,CAAC;AAgC7D,OAAO,EAAE,uBAAuB,EAAsC,MAAM,iCAAiC,CAAC;AAE9G,OAAO,KAAK,EACX,YAAY,EACZ,UAAU,EACV,eAAe,EACf,YAAY,EAEZ,aAAa,EAGb,QAAQ,EACR,kBAAkB,EAClB,MAAM,YAAY,CAAC;AAIpB,OAAO,EACN,0BAA0B,EAC1B,sBAAsB,EACtB,gCAAgC,GAChC,MAAM,+BAA+B,CAAC;AAEvC,MAAM,MAAM,cAAc,GAAG,CAAC,KAAK,EAAE,UAAU,KAAK,OAAO,CAAC,IAAI,CAAC,GAAG,IAAI,CAAC;AAKzE,gGAAgG;AAChG,MAAM,WAAW,0BAA0B;IAC1C,aAAa,EAAE,MAAM,CAAC;IACtB,WAAW,EAAE,MAAM,EAAE,CAAC;IACtB,uBAAuB,EAAE,uBAAuB,CAAC;CACjD;AAED,wBAAgB,gCAAgC,IAAI,0BAA0B,CAE7E;AAED;;;GAGG;AACH,wBAAgB,SAAS,CACxB,OAAO,EAAE,YAAY,EAAE,EACvB,OAAO,EAAE,YAAY,EACrB,MAAM,EAAE,eAAe,EACvB,MAAM,CAAC,EAAE,WAAW,EACpB,QAAQ,CAAC,EAAE,QAAQ,GACjB,WAAW,CAAC,UAAU,EAAE,YAAY,EAAE,CAAC,CAEzC;AAED;;;;;;;GAOG;AACH,wBAAgB,iBAAiB,CAChC,OAAO,EAAE,YAAY,EACrB,MAAM,EAAE,eAAe,EACvB,MAAM,CAAC,EAAE,WAAW,EACpB,QAAQ,CAAC,EAAE,QAAQ,GACjB,WAAW,CAAC,UAAU,EAAE,YAAY,EAAE,CAAC,CAIzC;AAuBD,wBAAsB,YAAY,CACjC,OAAO,EAAE,YAAY,EAAE,EACvB,OAAO,EAAE,YAAY,EACrB,MAAM,EAAE,eAAe,EACvB,IAAI,EAAE,cAAc,EACpB,MAAM,CAAC,EAAE,WAAW,EACpB,QAAQ,CAAC,EAAE,QAAQ,EACnB,iBAAiB,GAAE,0BAA+D,GAChF,OAAO,CAAC,YAAY,EAAE,CAAC,CAwBzB;AAED,wBAAsB,oBAAoB,CACzC,OAAO,EAAE,YAAY,EACrB,MAAM,EAAE,eAAe,EACvB,IAAI,EAAE,cAAc,EACpB,MAAM,CAAC,EAAE,WAAW,EACpB,QAAQ,CAAC,EAAE,QAAQ,EACnB,iBAAiB,GAAE,0BAA+D,GAChF,OAAO,CAAC,YAAY,EAAE,CAAC,CAezB;AAmWD;;;;;;GAMG;AACH,wBAAsB,yBAAyB,CAC9C,OAAO,EAAE,YAAY,EACrB,MAAM,EAAE,eAAe,EACvB,MAAM,EAAE,WAAW,GAAG,SAAS,EAC/B,QAAQ,CAAC,EAAE,QAAQ,GACjB,OAAO,CAAC,OAAO,CAAC,UAAU,CAAC,QAAQ,CAAC,CAAC,CAAC,CAExC;AA6iCD,wBAAgB,qBAAqB,CAAC,QAAQ,EAAE,aAAa,GAAG,kBAAkB,GAAG,SAAS,CAO7F"}
@@ -7,9 +7,9 @@ import { formatToolRepairStandingRule, REPEATED_SUCCESSFUL_TOOL_CALL_FAILURE, }
7
7
  import { ToolArgumentValidationError, validateToolArguments, } from "@caupulican/pi-ai/validation";
8
8
  import { assistantMessageText, collapseDegenerateAssistantMessage, shouldAbortDegenerateStream, } from "./degenerate-assistant-text.js";
9
9
  import { startPlannedAgentProviderRequest } from "./provider-request-planner.js";
10
- import { assessToolFailure, clearToolFailure, createRepeatedToolFailureResult, createToolFailureMemoryTracker, createToolFailureOperationExhaustedResult, createToolFailureRecoveryExhaustedResult, createToolFailureResult, getUnresolvedToolFailure, normalizeToolSignature, rememberToolFailure, toolFailureCorrection, } from "./tool-failure-memory.js";
11
- import { ToolFailureRecoveryGate, } from "./tool-failure-recovery-gate.js";
12
- import { rejectNativeToolProtocolResidue } from "./tool-protocol-residue.js";
10
+ import { assessToolFailure, clearToolFailure, createRepeatedToolFailureResult, createToolFailureMemoryTracker, createToolFailureResult, describeOperationOutcome, getUnresolvedToolFailure, normalizeToolSignature, rememberToolFailure, toolFailureCorrection, } from "./tool-failure-memory.js";
11
+ import { ToolFailureRecoveryGate } from "./tool-failure-recovery-gate.js";
12
+ import { rejectNativeToolProtocolResidue, rejectToolCallsFromToolFreeResponse } from "./tool-protocol-residue.js";
13
13
  import { AgentToolExecutionError, DEFAULT_MAX_PROVIDER_TURNS, DEFAULT_MAX_STALL_TURNS } from "./types.js";
14
14
  import { createEmptyUsage } from "./usage.js";
15
15
  export { composeRequestSystemPrompt, narrowRequestMaxTokens, resolveRequestPreflightMaxTokens, } from "./provider-request-planner.js";
@@ -59,6 +59,14 @@ export async function runAgentLoop(prompts, context, config, emit, signal, strea
59
59
  messages: [...context.messages, ...prompts],
60
60
  };
61
61
  await emit({ type: "agent_start" });
62
+ if (providerTurnLimitReached(config, continuationState)) {
63
+ for (const prompt of prompts) {
64
+ await emit({ type: "message_start", message: prompt });
65
+ await emit({ type: "message_end", message: prompt });
66
+ }
67
+ await emitProviderTurnLimitStop(config, continuationState, newMessages, emit);
68
+ return newMessages;
69
+ }
62
70
  await emit({ type: "turn_start" });
63
71
  for (const prompt of prompts) {
64
72
  await emit({ type: "message_start", message: prompt });
@@ -72,6 +80,10 @@ export async function runAgentLoopContinue(context, config, emit, signal, stream
72
80
  const newMessages = [];
73
81
  const currentContext = { ...context, messages: [...context.messages] };
74
82
  await emit({ type: "agent_start" });
83
+ if (providerTurnLimitReached(config, continuationState)) {
84
+ await emitProviderTurnLimitStop(config, continuationState, newMessages, emit);
85
+ return newMessages;
86
+ }
75
87
  await emit({ type: "turn_start" });
76
88
  await runLoop(currentContext, newMessages, config, signal, emit, streamFn, continuationState);
77
89
  return newMessages;
@@ -97,36 +109,21 @@ function createLoopFailureMessage(error, config, aborted) {
97
109
  timestamp: Date.now(),
98
110
  };
99
111
  }
100
- function createMandatoryRecoveryDeliveryFallback(halt, config) {
101
- const diagnostic = halt.record.diagnostic ?? halt.diagnostic;
102
- return createLocalDiagnosticMessage(config, `Tool recovery stopped for ${halt.record.tool}: ${diagnostic} Required recovery: ${halt.record.correction}`);
103
- }
104
- function createProviderTurnLimitMessage(config, providerTurns) {
105
- return createLocalDiagnosticMessage(config, `Configured provider turn limit (${providerTurns}) reached. The harness stopped before another provider request; continue explicitly if more work is needed.`);
112
+ function createAgentStream() {
113
+ return new EventStream((event) => event.type === "agent_end", (event) => (event.type === "agent_end" ? event.messages : []));
106
114
  }
107
- function createLocalDiagnosticMessage(config, text) {
108
- return {
109
- role: "assistant",
110
- content: [{ type: "text", text }],
111
- api: config.model.api,
112
- provider: config.model.provider,
113
- model: config.model.id,
114
- usage: createEmptyUsage(),
115
- stopReason: "stop",
116
- timestamp: Date.now(),
117
- };
115
+ function providerTurnLimitReached(config, continuationState) {
116
+ const limit = config.maxProviderTurns ?? DEFAULT_MAX_PROVIDER_TURNS;
117
+ return limit > 0 && continuationState.providerTurns >= limit;
118
118
  }
119
- async function emitTerminalLocalMessage(message, newMessages, emit, startTurn) {
120
- if (startTurn)
121
- await emit({ type: "turn_start" });
122
- await emit({ type: "message_start", message, origin: "local" });
123
- await emit({ type: "message_end", message, origin: "local" });
124
- await emit({ type: "turn_end", message, toolResults: [] });
119
+ async function emitProviderTurnLimitStop(config, continuationState, newMessages, emit) {
120
+ config.onRunawayStop?.({
121
+ reason: "provider_turn_limit",
122
+ signature: "provider_turn_limit",
123
+ repeats: continuationState.providerTurns,
124
+ });
125
125
  await emit({ type: "agent_end", messages: newMessages });
126
126
  }
127
- function createAgentStream() {
128
- return new EventStream((event) => event.type === "agent_end", (event) => (event.type === "agent_end" ? event.messages : []));
129
- }
130
127
  /**
131
128
  * How many `stallLimit`-length periods the runaway-loop window spans. A window of `stallLimit * P`
132
129
  * turns lets the count-based detector catch oscillating cycles of period up to `P` (each signature in a
@@ -206,40 +203,41 @@ async function runLoop(initialContext, newMessages, initialConfig, signal, emit,
206
203
  let lastSuccessfulTextProtocolBatch;
207
204
  // Check for steering messages at start (user may have typed while waiting)
208
205
  let pendingMessages = (await config.getSteeringMessages?.()) || [];
206
+ const processPendingMessages = async () => {
207
+ if (pendingMessages.length === 0)
208
+ return;
209
+ lastSuccessfulTextProtocolBatch = undefined;
210
+ // A new user turn can change authority, intent, or the workspace itself, so every
211
+ // operation that was blocked as an unproductive replay becomes worth attempting again.
212
+ toolFailureRecoveryGate.noteWorldAdvance();
213
+ for (const message of pendingMessages) {
214
+ await emit({ type: "message_start", message });
215
+ await emit({ type: "message_end", message });
216
+ currentContext.messages.push(message);
217
+ newMessages.push(message);
218
+ }
219
+ pendingMessages = [];
220
+ };
209
221
  // Outer loop: continues when queued follow-up messages arrive after agent would stop
210
222
  while (true) {
211
223
  let hasMoreToolCalls = true;
212
224
  // Inner loop: process tool calls and steering messages
213
225
  while (hasMoreToolCalls || pendingMessages.length > 0) {
226
+ if (providerTurnLimit > 0 && continuationState.providerTurns >= providerTurnLimit) {
227
+ // Preserve already-dequeued steering, but do not announce an assistant turn that will
228
+ // never start. A turn is one provider response plus its tools/results.
229
+ await processPendingMessages();
230
+ await emitProviderTurnLimitStop(config, continuationState, newMessages, emit);
231
+ return;
232
+ }
214
233
  if (!firstTurn) {
215
234
  await emit({ type: "turn_start" });
216
235
  }
217
236
  else {
218
237
  firstTurn = false;
219
238
  }
220
- // Process pending messages (inject before next assistant response)
221
- if (pendingMessages.length > 0) {
222
- lastSuccessfulTextProtocolBatch = undefined;
223
- for (const message of pendingMessages) {
224
- await emit({ type: "message_start", message });
225
- await emit({ type: "message_end", message });
226
- currentContext.messages.push(message);
227
- newMessages.push(message);
228
- }
229
- pendingMessages = [];
230
- }
231
- if (providerTurnLimit > 0 && continuationState.providerTurns >= providerTurnLimit) {
232
- const fallback = createProviderTurnLimitMessage(config, continuationState.providerTurns);
233
- currentContext.messages.push(fallback);
234
- newMessages.push(fallback);
235
- config.onRunawayStop?.({
236
- reason: "provider_turn_limit",
237
- signature: "provider_turn_limit",
238
- repeats: continuationState.providerTurns,
239
- });
240
- await emitTerminalLocalMessage(fallback, newMessages, emit, false);
241
- return;
242
- }
239
+ // Process pending messages (inject before next assistant response).
240
+ await processPendingMessages();
243
241
  continuationState.providerTurns++;
244
242
  const message = await streamAssistantResponse(currentContext, config, signal, emit, streamFn);
245
243
  newMessages.push(message);
@@ -251,7 +249,6 @@ async function runLoop(initialContext, newMessages, initialConfig, signal, emit,
251
249
  // Check for tool calls
252
250
  const toolCalls = message.content.filter((c) => c.type === "toolCall");
253
251
  const toolResults = [];
254
- let recoveryHalt;
255
252
  hasMoreToolCalls = false;
256
253
  if (toolCalls.length > 0) {
257
254
  const textProtocolBatch = toolCalls.every((toolCall) => toolCall.source === "text-protocol");
@@ -262,14 +259,7 @@ async function runLoop(initialContext, newMessages, initialConfig, signal, emit,
262
259
  : undefined;
263
260
  const executedToolBatch = await executeToolCalls(currentContext, message, config, validationFailureTracker, repairTeachTracker, toolFailureMemory, toolFailureRecoveryGate, previousSuccessfulTextProtocolResults, signal, emit);
264
261
  toolResults.push(...executedToolBatch.messages);
265
- if (toolFailureRecoveryGate.isHalted()) {
266
- recoveryHalt = toolFailureRecoveryGate.getHalt();
267
- if (!recoveryHalt)
268
- throw new Error("Tool recovery halted without a failure record");
269
- }
270
- else {
271
- hasMoreToolCalls = !executedToolBatch.terminate;
272
- }
262
+ hasMoreToolCalls = !executedToolBatch.terminate;
273
263
  for (const result of toolResults) {
274
264
  currentContext.messages.push(result);
275
265
  newMessages.push(result);
@@ -291,13 +281,6 @@ async function runLoop(initialContext, newMessages, initialConfig, signal, emit,
291
281
  lastSuccessfulTextProtocolBatch = undefined;
292
282
  }
293
283
  await emit({ type: "turn_end", message, toolResults });
294
- if (recoveryHalt) {
295
- const fallback = createMandatoryRecoveryDeliveryFallback(recoveryHalt, config);
296
- currentContext.messages.push(fallback);
297
- newMessages.push(fallback);
298
- await emitTerminalLocalMessage(fallback, newMessages, emit, true);
299
- return;
300
- }
301
284
  // Runaway-loop backstop (cost guard): detect a model stuck repeating one action.
302
285
  if (stallLimit > 0 && toolCalls.length > 0) {
303
286
  const signature = normalizeToolSignature(toolCalls.map((c) => [c.name, c.arguments ?? null]));
@@ -307,6 +290,7 @@ async function runLoop(initialContext, newMessages, initialConfig, signal, emit,
307
290
  const repeats = stallWindow.reduce((n, s) => (s === signature ? n + 1 : n), 0);
308
291
  if (repeats >= stallLimit) {
309
292
  config.onRunawayStop?.({ reason: "repeated_tool_call", signature, repeats });
293
+ await streamToollessClosingTurn(currentContext, newMessages, config, continuationState, providerTurnLimit, signal, emit, streamFn);
310
294
  await emit({ type: "agent_end", messages: newMessages });
311
295
  return;
312
296
  }
@@ -352,6 +336,41 @@ async function runLoop(initialContext, newMessages, initialConfig, signal, emit,
352
336
  }
353
337
  await emit({ type: "agent_end", messages: newMessages });
354
338
  }
339
+ const RUNAWAY_STOP_CLOSING_SYSTEM_PROMPT = [
340
+ "RUNAWAY STOP CLOSING TURN",
341
+ "The host stopped a repeated tool-call loop. No tools are available in this final request.",
342
+ "Write one concise factual closing message for the user: completed work, the unresolved operation or blocker, and the safest next action.",
343
+ "Do not claim unperformed work. Do not emit a tool call or tool-call markup.",
344
+ ].join("\n");
345
+ /**
346
+ * Spend one final provider request, with no tools, so a stopped run closes in the model's own words.
347
+ *
348
+ * The harness never writes that message itself. Tools are removed from the request, so this request
349
+ * cannot open another tool batch and cannot re-enter the loop; it runs through the same planned
350
+ * provider boundary as every other request. If the run is aborted, or the configured provider-turn
351
+ * limit leaves no budget, the run ends with no closing message rather than a fabricated one — and a
352
+ * provider error keeps its own error message for the same reason.
353
+ */
354
+ async function streamToollessClosingTurn(currentContext, newMessages, config, continuationState, providerTurnLimit, signal, emit, streamFn) {
355
+ if (signal?.aborted)
356
+ return;
357
+ if (providerTurnLimit > 0 && continuationState.providerTurns >= providerTurnLimit)
358
+ return;
359
+ continuationState.providerTurns++;
360
+ await emit({ type: "turn_start" });
361
+ const closingContext = {
362
+ ...currentContext,
363
+ systemPrompt: currentContext.systemPrompt
364
+ ? `${currentContext.systemPrompt}\n\n${RUNAWAY_STOP_CLOSING_SYSTEM_PROMPT}`
365
+ : RUNAWAY_STOP_CLOSING_SYSTEM_PROMPT,
366
+ tools: [],
367
+ };
368
+ const message = await streamAssistantResponse(closingContext, config, signal, emit, streamFn, {
369
+ rejectToolCalls: true,
370
+ });
371
+ newMessages.push(message);
372
+ await emit({ type: "turn_end", message, toolResults: [] });
373
+ }
355
374
  /**
356
375
  * Start one provider request through the canonical agent-loop boundary.
357
376
  *
@@ -366,7 +385,7 @@ export async function startAgentProviderRequest(context, config, signal, streamF
366
385
  * Stream an assistant response from the LLM.
367
386
  * This is where AgentMessage[] gets transformed to Message[] for the LLM.
368
387
  */
369
- async function streamAssistantResponse(context, config, signal, emit, streamFn) {
388
+ async function streamAssistantResponse(context, config, signal, emit, streamFn, policy) {
370
389
  const degenerationAbort = new AbortController();
371
390
  const onOuterAbort = () => degenerationAbort.abort();
372
391
  if (signal?.aborted)
@@ -417,7 +436,10 @@ async function streamAssistantResponse(context, config, signal, emit, streamFn)
417
436
  }
418
437
  }
419
438
  signal?.removeEventListener("abort", onOuterAbort);
420
- let finalMessage = rejectNativeToolProtocolResidue(await response.result(), context.tools ?? [], Boolean(config.textToolCallProtocol));
439
+ const providerMessage = await response.result();
440
+ let finalMessage = policy?.rejectToolCalls
441
+ ? rejectToolCallsFromToolFreeResponse(providerMessage)
442
+ : rejectNativeToolProtocolResidue(providerMessage, context.tools ?? [], Boolean(config.textToolCallProtocol));
421
443
  if (abortedForDegeneration && !signal?.aborted && finalMessage.stopReason === "aborted") {
422
444
  finalMessage = { ...finalMessage, stopReason: "stop" };
423
445
  delete finalMessage.errorMessage;
@@ -465,7 +487,7 @@ async function prepareAndStartToolCall(execCtx, toolCall, index) {
465
487
  emitToolArgumentValidationTelemetry(execCtx.config, preparation.validationEvent, "not_run", "none");
466
488
  return {
467
489
  kind: "finalized",
468
- finalized: finalizeRejectedToolCall(toolCall, execCtx.context.tools?.find((tool) => tool.name === toolCall.name), preparation, execCtx.toolFailureMemory, execCtx.toolFailureRecoveryGate),
490
+ finalized: finalizeRejectedToolCall(toolCall, execCtx.context.tools?.find((tool) => tool.name === toolCall.name), preparation, execCtx.toolFailureMemory),
469
491
  };
470
492
  }
471
493
  return { kind: "prepared", preparation };
@@ -480,7 +502,8 @@ async function executeToolCallsSequential(execCtx, toolCalls) {
480
502
  const messages = [];
481
503
  for (const [index, toolCall] of toolCalls.entries()) {
482
504
  const started = await prepareAndStartToolCall(execCtx, toolCall, index);
483
- const finalized = applyToolFailureRecoveryEffect(execCtx.toolFailureRecoveryGate, await finalizeStartedToolCall(execCtx, started));
505
+ const finalized = await finalizeStartedToolCall(execCtx, started);
506
+ execCtx.toolFailureRecoveryGate.apply(finalized.executionGateEffect);
484
507
  await emitToolExecutionEnd(finalized, execCtx.emit);
485
508
  const toolResultMessage = createToolResultMessage(finalized);
486
509
  await emitToolResultMessage(toolResultMessage, execCtx.emit);
@@ -490,10 +513,7 @@ async function executeToolCallsSequential(execCtx, toolCalls) {
490
513
  break;
491
514
  }
492
515
  }
493
- return {
494
- messages,
495
- terminate: execCtx.toolFailureRecoveryGate.isHalted() || shouldTerminateToolBatch(finalizedCalls),
496
- };
516
+ return { messages, terminate: shouldTerminateToolBatch(finalizedCalls) };
497
517
  }
498
518
  async function executeToolCallsParallel(execCtx, toolCalls) {
499
519
  const orderedFinalizedCalls = [];
@@ -523,15 +543,11 @@ async function executeToolCallsParallel(execCtx, toolCalls) {
523
543
  break;
524
544
  }
525
545
  }
526
- const unappliedWave = await Promise.all(wave.map((entry) => (typeof entry === "function" ? entry() : Promise.resolve(entry))));
527
- const appliedByOriginal = new Map();
528
- const finalizedWave = unappliedWave.map((finalized) => {
529
- const applied = applyToolFailureRecoveryEffect(execCtx.toolFailureRecoveryGate, finalized);
530
- appliedByOriginal.set(finalized, applied);
531
- return applied;
532
- });
546
+ const finalizedWave = await Promise.all(wave.map((entry) => (typeof entry === "function" ? entry() : Promise.resolve(entry))));
547
+ for (const finalized of finalizedWave)
548
+ execCtx.toolFailureRecoveryGate.apply(finalized.executionGateEffect);
533
549
  for (const finalized of completionOrder) {
534
- await emitToolExecutionEnd(appliedByOriginal.get(finalized) ?? finalized, execCtx.emit);
550
+ await emitToolExecutionEnd(finalized, execCtx.emit);
535
551
  }
536
552
  orderedFinalizedCalls.push(...finalizedWave);
537
553
  }
@@ -541,27 +557,7 @@ async function executeToolCallsParallel(execCtx, toolCalls) {
541
557
  await emitToolResultMessage(toolResultMessage, execCtx.emit);
542
558
  messages.push(toolResultMessage);
543
559
  }
544
- return {
545
- messages,
546
- terminate: execCtx.toolFailureRecoveryGate.isHalted() || shouldTerminateToolBatch(orderedFinalizedCalls),
547
- };
548
- }
549
- function applyToolFailureRecoveryEffect(gate, finalized) {
550
- const halt = gate.apply(finalized.executionGateEffect);
551
- if (!halt)
552
- return finalized;
553
- return createRecoveryExhaustedToolCallOutcome(finalized, halt);
554
- }
555
- function createRecoveryExhaustedToolCallOutcome(finalized, halt) {
556
- const exhaustedResult = createToolFailureRecoveryExhaustedResult(halt.record, halt.diagnostic);
557
- return {
558
- ...finalized,
559
- result: {
560
- ...exhaustedResult,
561
- ...(finalized.result.usage ? { usage: finalized.result.usage } : {}),
562
- },
563
- isError: true,
564
- };
560
+ return { messages, terminate: shouldTerminateToolBatch(orderedFinalizedCalls) };
565
561
  }
566
562
  const DEFAULT_TOOL_VALIDATION_ESCALATION_THRESHOLD = 3;
567
563
  const TOOL_REPAIR_TEACH_EVERY = 5;
@@ -586,7 +582,7 @@ function createRepeatedSuccessfulToolCallOutcome(previousResult) {
586
582
  ...(previousResult ? { repeatedSuccessfulCall: { previousToolCallId: previousResult.toolCallId } } : {}),
587
583
  };
588
584
  }
589
- function createAbortedToolCallOutcome(validationEvent, executionGateReservation) {
585
+ function createAbortedToolCallOutcome(validationEvent) {
590
586
  return {
591
587
  kind: "immediate",
592
588
  result: createErrorToolResult("Operation aborted"),
@@ -594,7 +590,6 @@ function createAbortedToolCallOutcome(validationEvent, executionGateReservation)
594
590
  phase: "cancelled",
595
591
  failureCode: "aborted",
596
592
  correction: "Retry only if the operation is still required.",
597
- ...(executionGateReservation ? { executionGateReservation } : {}),
598
593
  validationEvent,
599
594
  };
600
595
  }
@@ -717,7 +712,6 @@ async function prepareToolCall(currentContext, assistantMessage, toolCall, confi
717
712
  };
718
713
  }
719
714
  let validationEvent;
720
- let executionGateReservation;
721
715
  try {
722
716
  const preparedToolCall = prepareToolCallArguments(tool, toolCall);
723
717
  const validatedArgs = validateToolArguments(tool, preparedToolCall, {
@@ -739,16 +733,7 @@ async function prepareToolCall(currentContext, assistantMessage, toolCall, confi
739
733
  const unresolvedRecord = getUnresolvedToolFailure(toolFailureMemory, toolCall.name, validatedArgs);
740
734
  const admission = toolFailureRecoveryGate.admit(tool, validatedArgs, unresolvedRecord, currentContext.messages);
741
735
  if (admission.kind === "blocked") {
742
- const failureCode = admission.exhausted
743
- ? admission.scope === "operation"
744
- ? "operation_recovery_exhausted"
745
- : "recovery_exhausted"
746
- : "repeated_failed_operation";
747
- const result = admission.exhausted
748
- ? admission.scope === "operation"
749
- ? createToolFailureOperationExhaustedResult(admission.record, admission.diagnostic ?? "Tool operation recovery budget exhausted.")
750
- : createToolFailureRecoveryExhaustedResult(admission.record, admission.diagnostic ?? "Tool failure recovery budget exhausted.")
751
- : createRepeatedToolFailureResult(admission.record);
736
+ const result = createRepeatedToolFailureResult(admission.record);
752
737
  const memoryRecord = result.details.piToolFailureMemory;
753
738
  toolFailureMemory.set(admission.record.failureKey, memoryRecord);
754
739
  return {
@@ -756,14 +741,13 @@ async function prepareToolCall(currentContext, assistantMessage, toolCall, confi
756
741
  result,
757
742
  isError: true,
758
743
  phase: admission.record.phase,
759
- failureCode,
744
+ failureCode: "repeated_failed_operation",
760
745
  correction: memoryRecord.correction,
761
746
  diagnostic: memoryRecord.diagnostic,
762
747
  repeatedToolFailure: true,
763
- validationEvent: createValidationBounceTelemetry(config, toolCall, failureCode),
748
+ validationEvent: createValidationBounceTelemetry(config, toolCall, "repeated_failed_operation"),
764
749
  };
765
750
  }
766
- executionGateReservation = admission.reservation;
767
751
  if (config.beforeToolCall) {
768
752
  const beforeResult = await config.beforeToolCall({
769
753
  assistantMessage,
@@ -772,7 +756,7 @@ async function prepareToolCall(currentContext, assistantMessage, toolCall, confi
772
756
  context: currentContext,
773
757
  }, signal);
774
758
  if (signal?.aborted) {
775
- return createAbortedToolCallOutcome(validationEvent, executionGateReservation);
759
+ return createAbortedToolCallOutcome(validationEvent);
776
760
  }
777
761
  if (beforeResult?.block) {
778
762
  const reason = beforeResult.reason || "Tool execution was blocked";
@@ -784,20 +768,18 @@ async function prepareToolCall(currentContext, assistantMessage, toolCall, confi
784
768
  failureCode: "blocked",
785
769
  correction: "Choose an allowed approach or request the required authority before retrying.",
786
770
  diagnostic: reason,
787
- ...(executionGateReservation ? { executionGateReservation } : {}),
788
771
  validationEvent,
789
772
  };
790
773
  }
791
774
  }
792
775
  if (signal?.aborted) {
793
- return createAbortedToolCallOutcome(validationEvent, executionGateReservation);
776
+ return createAbortedToolCallOutcome(validationEvent);
794
777
  }
795
778
  return {
796
779
  kind: "prepared",
797
780
  toolCall,
798
781
  tool,
799
782
  args: validatedArgs,
800
- ...(executionGateReservation ? { executionGateReservation } : {}),
801
783
  validationEvent,
802
784
  };
803
785
  }
@@ -817,12 +799,11 @@ async function prepareToolCall(currentContext, assistantMessage, toolCall, confi
817
799
  ? validationFailureCorrection(validationEvent, toolCall.name)
818
800
  : toolFailureCorrection(message, "rejected", "preflight"),
819
801
  diagnostic: isToolArgumentValidationError(error) ? undefined : message,
820
- ...(executionGateReservation ? { executionGateReservation } : {}),
821
802
  validationEvent,
822
803
  };
823
804
  }
824
805
  }
825
- function finalizeRejectedToolCall(toolCall, tool, outcome, tracker, toolFailureRecoveryGate) {
806
+ function finalizeRejectedToolCall(toolCall, tool, outcome, tracker) {
826
807
  if (outcome.repeatedToolFailure) {
827
808
  return {
828
809
  toolCall,
@@ -836,16 +817,7 @@ function finalizeRejectedToolCall(toolCall, tool, outcome, tracker, toolFailureR
836
817
  .map((block) => block.text)
837
818
  .join("\n"),
838
819
  });
839
- const halt = toolFailureRecoveryGate.apply({
840
- kind: "failure",
841
- ...(tool ? { tool } : {}),
842
- record,
843
- args: toolCall.arguments,
844
- ...(outcome.executionGateReservation ? { reservation: outcome.executionGateReservation } : {}),
845
- });
846
- const failureResult = halt
847
- ? createToolFailureRecoveryExhaustedResult(halt.record, halt.diagnostic)
848
- : createToolFailureResult(record, outcome.result.terminate);
820
+ const failureResult = createToolFailureResult(record, outcome.result.terminate);
849
821
  return {
850
822
  toolCall,
851
823
  result: outcome.repeatedSuccessfulCall
@@ -858,6 +830,12 @@ function finalizeRejectedToolCall(toolCall, tool, outcome, tracker, toolFailureR
858
830
  }
859
831
  : failureResult,
860
832
  isError: true,
833
+ executionGateEffect: {
834
+ kind: "unproductive",
835
+ ...(tool ? { tool } : {}),
836
+ record,
837
+ args: toolCall.arguments,
838
+ },
861
839
  };
862
840
  }
863
841
  function createLinkedToolAbort(foregroundSignal) {
@@ -941,7 +919,13 @@ async function executeAndFinalizePreparedToolCall(currentContext, assistantMessa
941
919
  if (executionAbort.signal.aborted)
942
920
  return completion;
943
921
  let handoffAccepted = false;
944
- const handedOffCompletion = completion.then((finalized) => handoffAccepted ? applyToolFailureRecoveryEffect(toolFailureRecoveryGate, finalized) : finalized);
922
+ // A handed-off call leaves the batch, so the batch never applies its effect: the background
923
+ // completion is the only place left that can tell the governor what this operation did.
924
+ const handedOffCompletion = completion.then((finalized) => {
925
+ if (handoffAccepted)
926
+ toolFailureRecoveryGate.apply(finalized.executionGateEffect);
927
+ return finalized;
928
+ });
945
929
  void handedOffCompletion.catch(() => undefined);
946
930
  let handoff;
947
931
  try {
@@ -988,7 +972,9 @@ async function executePreparedToolCall(prepared, signal, emit) {
988
972
  // throwing. Keep the returned result intact through afterToolCall so
989
973
  // policy hooks can inspect its bounded diagnostics and metadata.
990
974
  isError: result.isError === true,
991
- ...(result.isError === true ? { errorClass: "tool_result_error" } : {}),
975
+ ...(result.isError === true
976
+ ? { errorClass: "tool_result_error", errorKind: result.errorKind ?? "tool_failure" }
977
+ : {}),
992
978
  };
993
979
  }
994
980
  catch (error) {
@@ -1000,6 +986,7 @@ async function executePreparedToolCall(prepared, signal, emit) {
1000
986
  isError: true,
1001
987
  errorClass: error instanceof Error ? error.name : typeof error,
1002
988
  failureMessage: message,
989
+ errorKind: toolFailure?.errorKind ?? "tool_failure",
1003
990
  ...(toolFailure ? { failureCode: toolFailure.failureCode, outputSignature: toolFailure.outputSignature } : {}),
1004
991
  };
1005
992
  }
@@ -1035,6 +1022,7 @@ async function finalizeExecutedToolCall(currentContext, assistantMessage, prepar
1035
1022
  let errorClass = executed.errorClass;
1036
1023
  let failureCode = executed.failureCode;
1037
1024
  let outputSignature = executed.outputSignature;
1025
+ let errorKind = executed.errorKind;
1038
1026
  let executionGateEffect;
1039
1027
  if (config.afterToolCall) {
1040
1028
  try {
@@ -1057,10 +1045,12 @@ async function finalizeExecutedToolCall(currentContext, assistantMessage, prepar
1057
1045
  }
1058
1046
  }
1059
1047
  catch (error) {
1048
+ // The hook itself failed, so nothing about the tool's own completed operation survives.
1060
1049
  failureMessage = error instanceof Error ? error.message : String(error);
1061
1050
  errorClass = error instanceof Error ? error.name : typeof error;
1062
1051
  failureCode = undefined;
1063
1052
  outputSignature = undefined;
1053
+ errorKind = "tool_failure";
1064
1054
  result = { ...createErrorToolResult(failureMessage), usage: result.usage };
1065
1055
  isError = true;
1066
1056
  }
@@ -1076,20 +1066,34 @@ async function finalizeExecutedToolCall(currentContext, assistantMessage, prepar
1076
1066
  const effectiveFailureMessage = failureMessage || result.content.find((block) => block.type === "text")?.text || "Tool execution failed";
1077
1067
  const assessment = assessToolFailure(effectiveFailureMessage, "failed", errorClass);
1078
1068
  const effectiveFailureCode = failureCode ?? assessment.failureCode;
1079
- const recoveryPlan = toolFailureRecoveryGate.planFailure(prepared.tool, prepared.args, { failureCode: effectiveFailureCode, message: effectiveFailureMessage }, currentContext.tools ?? [], prepared.executionGateReservation);
1080
- const correction = assessment.policyGuidance
1081
- ? `${assessment.policyGuidance} ${recoveryPlan.guidance}`
1082
- : recoveryPlan.guidance;
1083
- const record = rememberToolFailure(toolFailureMemory, prepared.toolCall.name, prepared.args, "failed", effectiveFailureCode, correction, assessment.diagnostic, assessment.phase, recoveryPlan.evidence ?? assessment.evidence, { output: failureOutput, outputSignature });
1084
- executionGateEffect = {
1085
- kind: "failure",
1086
- tool: prepared.tool,
1087
- record,
1088
- args: prepared.args,
1089
- targets: recoveryPlan.targets,
1090
- ...(prepared.executionGateReservation ? { reservation: prepared.executionGateReservation } : {}),
1091
- };
1092
- result = { ...createToolFailureResult(record, result.terminate), usage };
1069
+ if (errorKind === "operation_outcome") {
1070
+ // The tool ran the operation to completion; the non-zero status is the observation the
1071
+ // agent asked for. Nothing here is a mistake, so no failure record is remembered and the
1072
+ // tool's own output stands exactly as written. The governor still notes that repeating
1073
+ // this identical operation cannot say anything new until something else changes.
1074
+ clearToolFailure(toolFailureMemory, prepared.toolCall.name, prepared.args);
1075
+ result = { ...result, errorKind: "operation_outcome", usage };
1076
+ executionGateEffect = {
1077
+ kind: "unproductive",
1078
+ tool: prepared.tool,
1079
+ record: describeOperationOutcome(prepared.toolCall.name, prepared.args, effectiveFailureCode, assessment.diagnostic),
1080
+ args: prepared.args,
1081
+ };
1082
+ }
1083
+ else {
1084
+ const recoveryPlan = toolFailureRecoveryGate.planFailure(prepared.tool, prepared.args, { failureCode: effectiveFailureCode, message: effectiveFailureMessage }, currentContext.tools ?? []);
1085
+ const correction = assessment.policyGuidance
1086
+ ? `${assessment.policyGuidance} ${recoveryPlan.guidance}`
1087
+ : recoveryPlan.guidance;
1088
+ const record = rememberToolFailure(toolFailureMemory, prepared.toolCall.name, prepared.args, "failed", effectiveFailureCode, correction, assessment.diagnostic, assessment.phase, recoveryPlan.evidence ?? assessment.evidence, { output: failureOutput, outputSignature });
1089
+ executionGateEffect = {
1090
+ kind: "unproductive",
1091
+ tool: prepared.tool,
1092
+ record,
1093
+ args: prepared.args,
1094
+ };
1095
+ result = { ...createToolFailureResult(record, result.terminate), usage };
1096
+ }
1093
1097
  }
1094
1098
  else {
1095
1099
  clearToolFailure(toolFailureMemory, prepared.toolCall.name, prepared.args);
@@ -1098,7 +1102,6 @@ async function finalizeExecutedToolCall(currentContext, assistantMessage, prepar
1098
1102
  kind: "success",
1099
1103
  tool: prepared.tool,
1100
1104
  args: prepared.args,
1101
- evidenceResult: executed.result,
1102
1105
  };
1103
1106
  }
1104
1107
  }
@@ -1156,6 +1159,9 @@ function createToolResultMessage(finalized) {
1156
1159
  details: finalized.result.details,
1157
1160
  usage: finalized.result.usage,
1158
1161
  isError: finalized.isError,
1162
+ // Persisted so a reloaded transcript still separates a completed operation's own status from a
1163
+ // tool that could not run at all.
1164
+ ...(finalized.isError && finalized.result.errorKind ? { errorKind: finalized.result.errorKind } : {}),
1159
1165
  timestamp: Date.now(),
1160
1166
  };
1161
1167
  }