@caupulican/pi-agent-core 0.93.8 → 0.93.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-loop.d.ts.map +1 -1
- package/dist/agent-loop.js +160 -154
- package/dist/agent-loop.js.map +1 -1
- package/dist/index.d.ts +1 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -0
- package/dist/index.js.map +1 -1
- package/dist/tool-failure-memory.d.ts +8 -4
- package/dist/tool-failure-memory.d.ts.map +1 -1
- package/dist/tool-failure-memory.js +87 -79
- package/dist/tool-failure-memory.js.map +1 -1
- package/dist/tool-failure-recovery-gate.d.ts +32 -42
- package/dist/tool-failure-recovery-gate.d.ts.map +1 -1
- package/dist/tool-failure-recovery-gate.js +161 -287
- package/dist/tool-failure-recovery-gate.js.map +1 -1
- package/dist/tool-failure-recovery-protocol.d.ts +2 -0
- package/dist/tool-failure-recovery-protocol.d.ts.map +1 -1
- package/dist/tool-failure-recovery-protocol.js +7 -6
- package/dist/tool-failure-recovery-protocol.js.map +1 -1
- package/dist/tool-protocol-residue.d.ts +9 -0
- package/dist/tool-protocol-residue.d.ts.map +1 -1
- package/dist/tool-protocol-residue.js +28 -1
- package/dist/tool-protocol-residue.js.map +1 -1
- package/dist/types.d.ts +41 -20
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +5 -2
- package/dist/types.js.map +1 -1
- package/package.json +2 -2
package/dist/agent-loop.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"agent-loop.d.ts","sourceRoot":"","sources":["../src/agent-loop.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAEH,OAAO,EAAE,WAAW,EAAE,MAAM,gCAAgC,CAAC;
|
|
1
|
+
{"version":3,"file":"agent-loop.d.ts","sourceRoot":"","sources":["../src/agent-loop.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAEH,OAAO,EAAE,WAAW,EAAE,MAAM,gCAAgC,CAAC;AAgC7D,OAAO,EAAE,uBAAuB,EAAsC,MAAM,iCAAiC,CAAC;AAE9G,OAAO,KAAK,EACX,YAAY,EACZ,UAAU,EACV,eAAe,EACf,YAAY,EAEZ,aAAa,EAGb,QAAQ,EACR,kBAAkB,EAClB,MAAM,YAAY,CAAC;AAIpB,OAAO,EACN,0BAA0B,EAC1B,sBAAsB,EACtB,gCAAgC,GAChC,MAAM,+BAA+B,CAAC;AAEvC,MAAM,MAAM,cAAc,GAAG,CAAC,KAAK,EAAE,UAAU,KAAK,OAAO,CAAC,IAAI,CAAC,GAAG,IAAI,CAAC;AAKzE,gGAAgG;AAChG,MAAM,WAAW,0BAA0B;IAC1C,aAAa,EAAE,MAAM,CAAC;IACtB,WAAW,EAAE,MAAM,EAAE,CAAC;IACtB,uBAAuB,EAAE,uBAAuB,CAAC;CACjD;AAED,wBAAgB,gCAAgC,IAAI,0BAA0B,CAE7E;AAED;;;GAGG;AACH,wBAAgB,SAAS,CACxB,OAAO,EAAE,YAAY,EAAE,EACvB,OAAO,EAAE,YAAY,EACrB,MAAM,EAAE,eAAe,EACvB,MAAM,CAAC,EAAE,WAAW,EACpB,QAAQ,CAAC,EAAE,QAAQ,GACjB,WAAW,CAAC,UAAU,EAAE,YAAY,EAAE,CAAC,CAEzC;AAED;;;;;;;GAOG;AACH,wBAAgB,iBAAiB,CAChC,OAAO,EAAE,YAAY,EACrB,MAAM,EAAE,eAAe,EACvB,MAAM,CAAC,EAAE,WAAW,EACpB,QAAQ,CAAC,EAAE,QAAQ,GACjB,WAAW,CAAC,UAAU,EAAE,YAAY,EAAE,CAAC,CAIzC;AAuBD,wBAAsB,YAAY,CACjC,OAAO,EAAE,YAAY,EAAE,EACvB,OAAO,EAAE,YAAY,EACrB,MAAM,EAAE,eAAe,EACvB,IAAI,EAAE,cAAc,EACpB,MAAM,CAAC,EAAE,WAAW,EACpB,QAAQ,CAAC,EAAE,QAAQ,EACnB,iBAAiB,GAAE,0BAA+D,GAChF,OAAO,CAAC,YAAY,EAAE,CAAC,CAwBzB;AAED,wBAAsB,oBAAoB,CACzC,OAAO,EAAE,YAAY,EACrB,MAAM,EAAE,eAAe,EACvB,IAAI,EAAE,cAAc,EACpB,MAAM,CAAC,EAAE,WAAW,EACpB,QAAQ,CAAC,EAAE,QAAQ,EACnB,iBAAiB,GAAE,0BAA+D,GAChF,OAAO,CAAC,YAAY,EAAE,CAAC,CAezB;AAmWD;;;;;;GAMG;AACH,wBAAsB,yBAAyB,CAC9C,OAAO,EAAE,YAAY,EACrB,MAAM,EAAE,eAAe,EACvB,MAAM,EAAE,WAAW,GAAG,SAAS,EAC/B,QAAQ,CAAC,EAAE,QAAQ,GACjB,OAAO,CAAC,OAAO,CAAC,UAAU,CAAC,QAAQ,CAAC,CAAC,CAAC,CAExC;AA6iCD,wBAAgB,qBAAqB,CAAC,QAAQ,EAAE,aAAa,GAAG,kBAAkB,GAAG,SAAS,CAO7F"}
|
package/dist/agent-loop.js
CHANGED
|
@@ -7,9 +7,9 @@ import { formatToolRepairStandingRule, REPEATED_SUCCESSFUL_TOOL_CALL_FAILURE, }
|
|
|
7
7
|
import { ToolArgumentValidationError, validateToolArguments, } from "@caupulican/pi-ai/validation";
|
|
8
8
|
import { assistantMessageText, collapseDegenerateAssistantMessage, shouldAbortDegenerateStream, } from "./degenerate-assistant-text.js";
|
|
9
9
|
import { startPlannedAgentProviderRequest } from "./provider-request-planner.js";
|
|
10
|
-
import { assessToolFailure, clearToolFailure, createRepeatedToolFailureResult, createToolFailureMemoryTracker,
|
|
11
|
-
import { ToolFailureRecoveryGate
|
|
12
|
-
import { rejectNativeToolProtocolResidue } from "./tool-protocol-residue.js";
|
|
10
|
+
import { assessToolFailure, clearToolFailure, createRepeatedToolFailureResult, createToolFailureMemoryTracker, createToolFailureResult, describeOperationOutcome, getUnresolvedToolFailure, normalizeToolSignature, rememberToolFailure, toolFailureCorrection, } from "./tool-failure-memory.js";
|
|
11
|
+
import { ToolFailureRecoveryGate } from "./tool-failure-recovery-gate.js";
|
|
12
|
+
import { rejectNativeToolProtocolResidue, rejectToolCallsFromToolFreeResponse } from "./tool-protocol-residue.js";
|
|
13
13
|
import { AgentToolExecutionError, DEFAULT_MAX_PROVIDER_TURNS, DEFAULT_MAX_STALL_TURNS } from "./types.js";
|
|
14
14
|
import { createEmptyUsage } from "./usage.js";
|
|
15
15
|
export { composeRequestSystemPrompt, narrowRequestMaxTokens, resolveRequestPreflightMaxTokens, } from "./provider-request-planner.js";
|
|
@@ -59,6 +59,14 @@ export async function runAgentLoop(prompts, context, config, emit, signal, strea
|
|
|
59
59
|
messages: [...context.messages, ...prompts],
|
|
60
60
|
};
|
|
61
61
|
await emit({ type: "agent_start" });
|
|
62
|
+
if (providerTurnLimitReached(config, continuationState)) {
|
|
63
|
+
for (const prompt of prompts) {
|
|
64
|
+
await emit({ type: "message_start", message: prompt });
|
|
65
|
+
await emit({ type: "message_end", message: prompt });
|
|
66
|
+
}
|
|
67
|
+
await emitProviderTurnLimitStop(config, continuationState, newMessages, emit);
|
|
68
|
+
return newMessages;
|
|
69
|
+
}
|
|
62
70
|
await emit({ type: "turn_start" });
|
|
63
71
|
for (const prompt of prompts) {
|
|
64
72
|
await emit({ type: "message_start", message: prompt });
|
|
@@ -72,6 +80,10 @@ export async function runAgentLoopContinue(context, config, emit, signal, stream
|
|
|
72
80
|
const newMessages = [];
|
|
73
81
|
const currentContext = { ...context, messages: [...context.messages] };
|
|
74
82
|
await emit({ type: "agent_start" });
|
|
83
|
+
if (providerTurnLimitReached(config, continuationState)) {
|
|
84
|
+
await emitProviderTurnLimitStop(config, continuationState, newMessages, emit);
|
|
85
|
+
return newMessages;
|
|
86
|
+
}
|
|
75
87
|
await emit({ type: "turn_start" });
|
|
76
88
|
await runLoop(currentContext, newMessages, config, signal, emit, streamFn, continuationState);
|
|
77
89
|
return newMessages;
|
|
@@ -97,36 +109,21 @@ function createLoopFailureMessage(error, config, aborted) {
|
|
|
97
109
|
timestamp: Date.now(),
|
|
98
110
|
};
|
|
99
111
|
}
|
|
100
|
-
function
|
|
101
|
-
|
|
102
|
-
return createLocalDiagnosticMessage(config, `Tool recovery stopped for ${halt.record.tool}: ${diagnostic} Required recovery: ${halt.record.correction}`);
|
|
103
|
-
}
|
|
104
|
-
function createProviderTurnLimitMessage(config, providerTurns) {
|
|
105
|
-
return createLocalDiagnosticMessage(config, `Configured provider turn limit (${providerTurns}) reached. The harness stopped before another provider request; continue explicitly if more work is needed.`);
|
|
112
|
+
function createAgentStream() {
|
|
113
|
+
return new EventStream((event) => event.type === "agent_end", (event) => (event.type === "agent_end" ? event.messages : []));
|
|
106
114
|
}
|
|
107
|
-
function
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
content: [{ type: "text", text }],
|
|
111
|
-
api: config.model.api,
|
|
112
|
-
provider: config.model.provider,
|
|
113
|
-
model: config.model.id,
|
|
114
|
-
usage: createEmptyUsage(),
|
|
115
|
-
stopReason: "stop",
|
|
116
|
-
timestamp: Date.now(),
|
|
117
|
-
};
|
|
115
|
+
function providerTurnLimitReached(config, continuationState) {
|
|
116
|
+
const limit = config.maxProviderTurns ?? DEFAULT_MAX_PROVIDER_TURNS;
|
|
117
|
+
return limit > 0 && continuationState.providerTurns >= limit;
|
|
118
118
|
}
|
|
119
|
-
async function
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
119
|
+
async function emitProviderTurnLimitStop(config, continuationState, newMessages, emit) {
|
|
120
|
+
config.onRunawayStop?.({
|
|
121
|
+
reason: "provider_turn_limit",
|
|
122
|
+
signature: "provider_turn_limit",
|
|
123
|
+
repeats: continuationState.providerTurns,
|
|
124
|
+
});
|
|
125
125
|
await emit({ type: "agent_end", messages: newMessages });
|
|
126
126
|
}
|
|
127
|
-
function createAgentStream() {
|
|
128
|
-
return new EventStream((event) => event.type === "agent_end", (event) => (event.type === "agent_end" ? event.messages : []));
|
|
129
|
-
}
|
|
130
127
|
/**
|
|
131
128
|
* How many `stallLimit`-length periods the runaway-loop window spans. A window of `stallLimit * P`
|
|
132
129
|
* turns lets the count-based detector catch oscillating cycles of period up to `P` (each signature in a
|
|
@@ -206,40 +203,41 @@ async function runLoop(initialContext, newMessages, initialConfig, signal, emit,
|
|
|
206
203
|
let lastSuccessfulTextProtocolBatch;
|
|
207
204
|
// Check for steering messages at start (user may have typed while waiting)
|
|
208
205
|
let pendingMessages = (await config.getSteeringMessages?.()) || [];
|
|
206
|
+
const processPendingMessages = async () => {
|
|
207
|
+
if (pendingMessages.length === 0)
|
|
208
|
+
return;
|
|
209
|
+
lastSuccessfulTextProtocolBatch = undefined;
|
|
210
|
+
// A new user turn can change authority, intent, or the workspace itself, so every
|
|
211
|
+
// operation that was blocked as an unproductive replay becomes worth attempting again.
|
|
212
|
+
toolFailureRecoveryGate.noteWorldAdvance();
|
|
213
|
+
for (const message of pendingMessages) {
|
|
214
|
+
await emit({ type: "message_start", message });
|
|
215
|
+
await emit({ type: "message_end", message });
|
|
216
|
+
currentContext.messages.push(message);
|
|
217
|
+
newMessages.push(message);
|
|
218
|
+
}
|
|
219
|
+
pendingMessages = [];
|
|
220
|
+
};
|
|
209
221
|
// Outer loop: continues when queued follow-up messages arrive after agent would stop
|
|
210
222
|
while (true) {
|
|
211
223
|
let hasMoreToolCalls = true;
|
|
212
224
|
// Inner loop: process tool calls and steering messages
|
|
213
225
|
while (hasMoreToolCalls || pendingMessages.length > 0) {
|
|
226
|
+
if (providerTurnLimit > 0 && continuationState.providerTurns >= providerTurnLimit) {
|
|
227
|
+
// Preserve already-dequeued steering, but do not announce an assistant turn that will
|
|
228
|
+
// never start. A turn is one provider response plus its tools/results.
|
|
229
|
+
await processPendingMessages();
|
|
230
|
+
await emitProviderTurnLimitStop(config, continuationState, newMessages, emit);
|
|
231
|
+
return;
|
|
232
|
+
}
|
|
214
233
|
if (!firstTurn) {
|
|
215
234
|
await emit({ type: "turn_start" });
|
|
216
235
|
}
|
|
217
236
|
else {
|
|
218
237
|
firstTurn = false;
|
|
219
238
|
}
|
|
220
|
-
// Process pending messages (inject before next assistant response)
|
|
221
|
-
|
|
222
|
-
lastSuccessfulTextProtocolBatch = undefined;
|
|
223
|
-
for (const message of pendingMessages) {
|
|
224
|
-
await emit({ type: "message_start", message });
|
|
225
|
-
await emit({ type: "message_end", message });
|
|
226
|
-
currentContext.messages.push(message);
|
|
227
|
-
newMessages.push(message);
|
|
228
|
-
}
|
|
229
|
-
pendingMessages = [];
|
|
230
|
-
}
|
|
231
|
-
if (providerTurnLimit > 0 && continuationState.providerTurns >= providerTurnLimit) {
|
|
232
|
-
const fallback = createProviderTurnLimitMessage(config, continuationState.providerTurns);
|
|
233
|
-
currentContext.messages.push(fallback);
|
|
234
|
-
newMessages.push(fallback);
|
|
235
|
-
config.onRunawayStop?.({
|
|
236
|
-
reason: "provider_turn_limit",
|
|
237
|
-
signature: "provider_turn_limit",
|
|
238
|
-
repeats: continuationState.providerTurns,
|
|
239
|
-
});
|
|
240
|
-
await emitTerminalLocalMessage(fallback, newMessages, emit, false);
|
|
241
|
-
return;
|
|
242
|
-
}
|
|
239
|
+
// Process pending messages (inject before next assistant response).
|
|
240
|
+
await processPendingMessages();
|
|
243
241
|
continuationState.providerTurns++;
|
|
244
242
|
const message = await streamAssistantResponse(currentContext, config, signal, emit, streamFn);
|
|
245
243
|
newMessages.push(message);
|
|
@@ -251,7 +249,6 @@ async function runLoop(initialContext, newMessages, initialConfig, signal, emit,
|
|
|
251
249
|
// Check for tool calls
|
|
252
250
|
const toolCalls = message.content.filter((c) => c.type === "toolCall");
|
|
253
251
|
const toolResults = [];
|
|
254
|
-
let recoveryHalt;
|
|
255
252
|
hasMoreToolCalls = false;
|
|
256
253
|
if (toolCalls.length > 0) {
|
|
257
254
|
const textProtocolBatch = toolCalls.every((toolCall) => toolCall.source === "text-protocol");
|
|
@@ -262,14 +259,7 @@ async function runLoop(initialContext, newMessages, initialConfig, signal, emit,
|
|
|
262
259
|
: undefined;
|
|
263
260
|
const executedToolBatch = await executeToolCalls(currentContext, message, config, validationFailureTracker, repairTeachTracker, toolFailureMemory, toolFailureRecoveryGate, previousSuccessfulTextProtocolResults, signal, emit);
|
|
264
261
|
toolResults.push(...executedToolBatch.messages);
|
|
265
|
-
|
|
266
|
-
recoveryHalt = toolFailureRecoveryGate.getHalt();
|
|
267
|
-
if (!recoveryHalt)
|
|
268
|
-
throw new Error("Tool recovery halted without a failure record");
|
|
269
|
-
}
|
|
270
|
-
else {
|
|
271
|
-
hasMoreToolCalls = !executedToolBatch.terminate;
|
|
272
|
-
}
|
|
262
|
+
hasMoreToolCalls = !executedToolBatch.terminate;
|
|
273
263
|
for (const result of toolResults) {
|
|
274
264
|
currentContext.messages.push(result);
|
|
275
265
|
newMessages.push(result);
|
|
@@ -291,13 +281,6 @@ async function runLoop(initialContext, newMessages, initialConfig, signal, emit,
|
|
|
291
281
|
lastSuccessfulTextProtocolBatch = undefined;
|
|
292
282
|
}
|
|
293
283
|
await emit({ type: "turn_end", message, toolResults });
|
|
294
|
-
if (recoveryHalt) {
|
|
295
|
-
const fallback = createMandatoryRecoveryDeliveryFallback(recoveryHalt, config);
|
|
296
|
-
currentContext.messages.push(fallback);
|
|
297
|
-
newMessages.push(fallback);
|
|
298
|
-
await emitTerminalLocalMessage(fallback, newMessages, emit, true);
|
|
299
|
-
return;
|
|
300
|
-
}
|
|
301
284
|
// Runaway-loop backstop (cost guard): detect a model stuck repeating one action.
|
|
302
285
|
if (stallLimit > 0 && toolCalls.length > 0) {
|
|
303
286
|
const signature = normalizeToolSignature(toolCalls.map((c) => [c.name, c.arguments ?? null]));
|
|
@@ -307,6 +290,7 @@ async function runLoop(initialContext, newMessages, initialConfig, signal, emit,
|
|
|
307
290
|
const repeats = stallWindow.reduce((n, s) => (s === signature ? n + 1 : n), 0);
|
|
308
291
|
if (repeats >= stallLimit) {
|
|
309
292
|
config.onRunawayStop?.({ reason: "repeated_tool_call", signature, repeats });
|
|
293
|
+
await streamToollessClosingTurn(currentContext, newMessages, config, continuationState, providerTurnLimit, signal, emit, streamFn);
|
|
310
294
|
await emit({ type: "agent_end", messages: newMessages });
|
|
311
295
|
return;
|
|
312
296
|
}
|
|
@@ -352,6 +336,41 @@ async function runLoop(initialContext, newMessages, initialConfig, signal, emit,
|
|
|
352
336
|
}
|
|
353
337
|
await emit({ type: "agent_end", messages: newMessages });
|
|
354
338
|
}
|
|
339
|
+
const RUNAWAY_STOP_CLOSING_SYSTEM_PROMPT = [
|
|
340
|
+
"RUNAWAY STOP CLOSING TURN",
|
|
341
|
+
"The host stopped a repeated tool-call loop. No tools are available in this final request.",
|
|
342
|
+
"Write one concise factual closing message for the user: completed work, the unresolved operation or blocker, and the safest next action.",
|
|
343
|
+
"Do not claim unperformed work. Do not emit a tool call or tool-call markup.",
|
|
344
|
+
].join("\n");
|
|
345
|
+
/**
|
|
346
|
+
* Spend one final provider request, with no tools, so a stopped run closes in the model's own words.
|
|
347
|
+
*
|
|
348
|
+
* The harness never writes that message itself. Tools are removed from the request, so this request
|
|
349
|
+
* cannot open another tool batch and cannot re-enter the loop; it runs through the same planned
|
|
350
|
+
* provider boundary as every other request. If the run is aborted, or the configured provider-turn
|
|
351
|
+
* limit leaves no budget, the run ends with no closing message rather than a fabricated one — and a
|
|
352
|
+
* provider error keeps its own error message for the same reason.
|
|
353
|
+
*/
|
|
354
|
+
async function streamToollessClosingTurn(currentContext, newMessages, config, continuationState, providerTurnLimit, signal, emit, streamFn) {
|
|
355
|
+
if (signal?.aborted)
|
|
356
|
+
return;
|
|
357
|
+
if (providerTurnLimit > 0 && continuationState.providerTurns >= providerTurnLimit)
|
|
358
|
+
return;
|
|
359
|
+
continuationState.providerTurns++;
|
|
360
|
+
await emit({ type: "turn_start" });
|
|
361
|
+
const closingContext = {
|
|
362
|
+
...currentContext,
|
|
363
|
+
systemPrompt: currentContext.systemPrompt
|
|
364
|
+
? `${currentContext.systemPrompt}\n\n${RUNAWAY_STOP_CLOSING_SYSTEM_PROMPT}`
|
|
365
|
+
: RUNAWAY_STOP_CLOSING_SYSTEM_PROMPT,
|
|
366
|
+
tools: [],
|
|
367
|
+
};
|
|
368
|
+
const message = await streamAssistantResponse(closingContext, config, signal, emit, streamFn, {
|
|
369
|
+
rejectToolCalls: true,
|
|
370
|
+
});
|
|
371
|
+
newMessages.push(message);
|
|
372
|
+
await emit({ type: "turn_end", message, toolResults: [] });
|
|
373
|
+
}
|
|
355
374
|
/**
|
|
356
375
|
* Start one provider request through the canonical agent-loop boundary.
|
|
357
376
|
*
|
|
@@ -366,7 +385,7 @@ export async function startAgentProviderRequest(context, config, signal, streamF
|
|
|
366
385
|
* Stream an assistant response from the LLM.
|
|
367
386
|
* This is where AgentMessage[] gets transformed to Message[] for the LLM.
|
|
368
387
|
*/
|
|
369
|
-
async function streamAssistantResponse(context, config, signal, emit, streamFn) {
|
|
388
|
+
async function streamAssistantResponse(context, config, signal, emit, streamFn, policy) {
|
|
370
389
|
const degenerationAbort = new AbortController();
|
|
371
390
|
const onOuterAbort = () => degenerationAbort.abort();
|
|
372
391
|
if (signal?.aborted)
|
|
@@ -417,7 +436,10 @@ async function streamAssistantResponse(context, config, signal, emit, streamFn)
|
|
|
417
436
|
}
|
|
418
437
|
}
|
|
419
438
|
signal?.removeEventListener("abort", onOuterAbort);
|
|
420
|
-
|
|
439
|
+
const providerMessage = await response.result();
|
|
440
|
+
let finalMessage = policy?.rejectToolCalls
|
|
441
|
+
? rejectToolCallsFromToolFreeResponse(providerMessage)
|
|
442
|
+
: rejectNativeToolProtocolResidue(providerMessage, context.tools ?? [], Boolean(config.textToolCallProtocol));
|
|
421
443
|
if (abortedForDegeneration && !signal?.aborted && finalMessage.stopReason === "aborted") {
|
|
422
444
|
finalMessage = { ...finalMessage, stopReason: "stop" };
|
|
423
445
|
delete finalMessage.errorMessage;
|
|
@@ -465,7 +487,7 @@ async function prepareAndStartToolCall(execCtx, toolCall, index) {
|
|
|
465
487
|
emitToolArgumentValidationTelemetry(execCtx.config, preparation.validationEvent, "not_run", "none");
|
|
466
488
|
return {
|
|
467
489
|
kind: "finalized",
|
|
468
|
-
finalized: finalizeRejectedToolCall(toolCall, execCtx.context.tools?.find((tool) => tool.name === toolCall.name), preparation, execCtx.toolFailureMemory
|
|
490
|
+
finalized: finalizeRejectedToolCall(toolCall, execCtx.context.tools?.find((tool) => tool.name === toolCall.name), preparation, execCtx.toolFailureMemory),
|
|
469
491
|
};
|
|
470
492
|
}
|
|
471
493
|
return { kind: "prepared", preparation };
|
|
@@ -480,7 +502,8 @@ async function executeToolCallsSequential(execCtx, toolCalls) {
|
|
|
480
502
|
const messages = [];
|
|
481
503
|
for (const [index, toolCall] of toolCalls.entries()) {
|
|
482
504
|
const started = await prepareAndStartToolCall(execCtx, toolCall, index);
|
|
483
|
-
const finalized =
|
|
505
|
+
const finalized = await finalizeStartedToolCall(execCtx, started);
|
|
506
|
+
execCtx.toolFailureRecoveryGate.apply(finalized.executionGateEffect);
|
|
484
507
|
await emitToolExecutionEnd(finalized, execCtx.emit);
|
|
485
508
|
const toolResultMessage = createToolResultMessage(finalized);
|
|
486
509
|
await emitToolResultMessage(toolResultMessage, execCtx.emit);
|
|
@@ -490,10 +513,7 @@ async function executeToolCallsSequential(execCtx, toolCalls) {
|
|
|
490
513
|
break;
|
|
491
514
|
}
|
|
492
515
|
}
|
|
493
|
-
return {
|
|
494
|
-
messages,
|
|
495
|
-
terminate: execCtx.toolFailureRecoveryGate.isHalted() || shouldTerminateToolBatch(finalizedCalls),
|
|
496
|
-
};
|
|
516
|
+
return { messages, terminate: shouldTerminateToolBatch(finalizedCalls) };
|
|
497
517
|
}
|
|
498
518
|
async function executeToolCallsParallel(execCtx, toolCalls) {
|
|
499
519
|
const orderedFinalizedCalls = [];
|
|
@@ -523,15 +543,11 @@ async function executeToolCallsParallel(execCtx, toolCalls) {
|
|
|
523
543
|
break;
|
|
524
544
|
}
|
|
525
545
|
}
|
|
526
|
-
const
|
|
527
|
-
const
|
|
528
|
-
|
|
529
|
-
const applied = applyToolFailureRecoveryEffect(execCtx.toolFailureRecoveryGate, finalized);
|
|
530
|
-
appliedByOriginal.set(finalized, applied);
|
|
531
|
-
return applied;
|
|
532
|
-
});
|
|
546
|
+
const finalizedWave = await Promise.all(wave.map((entry) => (typeof entry === "function" ? entry() : Promise.resolve(entry))));
|
|
547
|
+
for (const finalized of finalizedWave)
|
|
548
|
+
execCtx.toolFailureRecoveryGate.apply(finalized.executionGateEffect);
|
|
533
549
|
for (const finalized of completionOrder) {
|
|
534
|
-
await emitToolExecutionEnd(
|
|
550
|
+
await emitToolExecutionEnd(finalized, execCtx.emit);
|
|
535
551
|
}
|
|
536
552
|
orderedFinalizedCalls.push(...finalizedWave);
|
|
537
553
|
}
|
|
@@ -541,27 +557,7 @@ async function executeToolCallsParallel(execCtx, toolCalls) {
|
|
|
541
557
|
await emitToolResultMessage(toolResultMessage, execCtx.emit);
|
|
542
558
|
messages.push(toolResultMessage);
|
|
543
559
|
}
|
|
544
|
-
return {
|
|
545
|
-
messages,
|
|
546
|
-
terminate: execCtx.toolFailureRecoveryGate.isHalted() || shouldTerminateToolBatch(orderedFinalizedCalls),
|
|
547
|
-
};
|
|
548
|
-
}
|
|
549
|
-
function applyToolFailureRecoveryEffect(gate, finalized) {
|
|
550
|
-
const halt = gate.apply(finalized.executionGateEffect);
|
|
551
|
-
if (!halt)
|
|
552
|
-
return finalized;
|
|
553
|
-
return createRecoveryExhaustedToolCallOutcome(finalized, halt);
|
|
554
|
-
}
|
|
555
|
-
function createRecoveryExhaustedToolCallOutcome(finalized, halt) {
|
|
556
|
-
const exhaustedResult = createToolFailureRecoveryExhaustedResult(halt.record, halt.diagnostic);
|
|
557
|
-
return {
|
|
558
|
-
...finalized,
|
|
559
|
-
result: {
|
|
560
|
-
...exhaustedResult,
|
|
561
|
-
...(finalized.result.usage ? { usage: finalized.result.usage } : {}),
|
|
562
|
-
},
|
|
563
|
-
isError: true,
|
|
564
|
-
};
|
|
560
|
+
return { messages, terminate: shouldTerminateToolBatch(orderedFinalizedCalls) };
|
|
565
561
|
}
|
|
566
562
|
const DEFAULT_TOOL_VALIDATION_ESCALATION_THRESHOLD = 3;
|
|
567
563
|
const TOOL_REPAIR_TEACH_EVERY = 5;
|
|
@@ -586,7 +582,7 @@ function createRepeatedSuccessfulToolCallOutcome(previousResult) {
|
|
|
586
582
|
...(previousResult ? { repeatedSuccessfulCall: { previousToolCallId: previousResult.toolCallId } } : {}),
|
|
587
583
|
};
|
|
588
584
|
}
|
|
589
|
-
function createAbortedToolCallOutcome(validationEvent
|
|
585
|
+
function createAbortedToolCallOutcome(validationEvent) {
|
|
590
586
|
return {
|
|
591
587
|
kind: "immediate",
|
|
592
588
|
result: createErrorToolResult("Operation aborted"),
|
|
@@ -594,7 +590,6 @@ function createAbortedToolCallOutcome(validationEvent, executionGateReservation)
|
|
|
594
590
|
phase: "cancelled",
|
|
595
591
|
failureCode: "aborted",
|
|
596
592
|
correction: "Retry only if the operation is still required.",
|
|
597
|
-
...(executionGateReservation ? { executionGateReservation } : {}),
|
|
598
593
|
validationEvent,
|
|
599
594
|
};
|
|
600
595
|
}
|
|
@@ -717,7 +712,6 @@ async function prepareToolCall(currentContext, assistantMessage, toolCall, confi
|
|
|
717
712
|
};
|
|
718
713
|
}
|
|
719
714
|
let validationEvent;
|
|
720
|
-
let executionGateReservation;
|
|
721
715
|
try {
|
|
722
716
|
const preparedToolCall = prepareToolCallArguments(tool, toolCall);
|
|
723
717
|
const validatedArgs = validateToolArguments(tool, preparedToolCall, {
|
|
@@ -739,16 +733,7 @@ async function prepareToolCall(currentContext, assistantMessage, toolCall, confi
|
|
|
739
733
|
const unresolvedRecord = getUnresolvedToolFailure(toolFailureMemory, toolCall.name, validatedArgs);
|
|
740
734
|
const admission = toolFailureRecoveryGate.admit(tool, validatedArgs, unresolvedRecord, currentContext.messages);
|
|
741
735
|
if (admission.kind === "blocked") {
|
|
742
|
-
const
|
|
743
|
-
? admission.scope === "operation"
|
|
744
|
-
? "operation_recovery_exhausted"
|
|
745
|
-
: "recovery_exhausted"
|
|
746
|
-
: "repeated_failed_operation";
|
|
747
|
-
const result = admission.exhausted
|
|
748
|
-
? admission.scope === "operation"
|
|
749
|
-
? createToolFailureOperationExhaustedResult(admission.record, admission.diagnostic ?? "Tool operation recovery budget exhausted.")
|
|
750
|
-
: createToolFailureRecoveryExhaustedResult(admission.record, admission.diagnostic ?? "Tool failure recovery budget exhausted.")
|
|
751
|
-
: createRepeatedToolFailureResult(admission.record);
|
|
736
|
+
const result = createRepeatedToolFailureResult(admission.record);
|
|
752
737
|
const memoryRecord = result.details.piToolFailureMemory;
|
|
753
738
|
toolFailureMemory.set(admission.record.failureKey, memoryRecord);
|
|
754
739
|
return {
|
|
@@ -756,14 +741,13 @@ async function prepareToolCall(currentContext, assistantMessage, toolCall, confi
|
|
|
756
741
|
result,
|
|
757
742
|
isError: true,
|
|
758
743
|
phase: admission.record.phase,
|
|
759
|
-
failureCode,
|
|
744
|
+
failureCode: "repeated_failed_operation",
|
|
760
745
|
correction: memoryRecord.correction,
|
|
761
746
|
diagnostic: memoryRecord.diagnostic,
|
|
762
747
|
repeatedToolFailure: true,
|
|
763
|
-
validationEvent: createValidationBounceTelemetry(config, toolCall,
|
|
748
|
+
validationEvent: createValidationBounceTelemetry(config, toolCall, "repeated_failed_operation"),
|
|
764
749
|
};
|
|
765
750
|
}
|
|
766
|
-
executionGateReservation = admission.reservation;
|
|
767
751
|
if (config.beforeToolCall) {
|
|
768
752
|
const beforeResult = await config.beforeToolCall({
|
|
769
753
|
assistantMessage,
|
|
@@ -772,7 +756,7 @@ async function prepareToolCall(currentContext, assistantMessage, toolCall, confi
|
|
|
772
756
|
context: currentContext,
|
|
773
757
|
}, signal);
|
|
774
758
|
if (signal?.aborted) {
|
|
775
|
-
return createAbortedToolCallOutcome(validationEvent
|
|
759
|
+
return createAbortedToolCallOutcome(validationEvent);
|
|
776
760
|
}
|
|
777
761
|
if (beforeResult?.block) {
|
|
778
762
|
const reason = beforeResult.reason || "Tool execution was blocked";
|
|
@@ -784,20 +768,18 @@ async function prepareToolCall(currentContext, assistantMessage, toolCall, confi
|
|
|
784
768
|
failureCode: "blocked",
|
|
785
769
|
correction: "Choose an allowed approach or request the required authority before retrying.",
|
|
786
770
|
diagnostic: reason,
|
|
787
|
-
...(executionGateReservation ? { executionGateReservation } : {}),
|
|
788
771
|
validationEvent,
|
|
789
772
|
};
|
|
790
773
|
}
|
|
791
774
|
}
|
|
792
775
|
if (signal?.aborted) {
|
|
793
|
-
return createAbortedToolCallOutcome(validationEvent
|
|
776
|
+
return createAbortedToolCallOutcome(validationEvent);
|
|
794
777
|
}
|
|
795
778
|
return {
|
|
796
779
|
kind: "prepared",
|
|
797
780
|
toolCall,
|
|
798
781
|
tool,
|
|
799
782
|
args: validatedArgs,
|
|
800
|
-
...(executionGateReservation ? { executionGateReservation } : {}),
|
|
801
783
|
validationEvent,
|
|
802
784
|
};
|
|
803
785
|
}
|
|
@@ -817,12 +799,11 @@ async function prepareToolCall(currentContext, assistantMessage, toolCall, confi
|
|
|
817
799
|
? validationFailureCorrection(validationEvent, toolCall.name)
|
|
818
800
|
: toolFailureCorrection(message, "rejected", "preflight"),
|
|
819
801
|
diagnostic: isToolArgumentValidationError(error) ? undefined : message,
|
|
820
|
-
...(executionGateReservation ? { executionGateReservation } : {}),
|
|
821
802
|
validationEvent,
|
|
822
803
|
};
|
|
823
804
|
}
|
|
824
805
|
}
|
|
825
|
-
function finalizeRejectedToolCall(toolCall, tool, outcome, tracker
|
|
806
|
+
function finalizeRejectedToolCall(toolCall, tool, outcome, tracker) {
|
|
826
807
|
if (outcome.repeatedToolFailure) {
|
|
827
808
|
return {
|
|
828
809
|
toolCall,
|
|
@@ -836,16 +817,7 @@ function finalizeRejectedToolCall(toolCall, tool, outcome, tracker, toolFailureR
|
|
|
836
817
|
.map((block) => block.text)
|
|
837
818
|
.join("\n"),
|
|
838
819
|
});
|
|
839
|
-
const
|
|
840
|
-
kind: "failure",
|
|
841
|
-
...(tool ? { tool } : {}),
|
|
842
|
-
record,
|
|
843
|
-
args: toolCall.arguments,
|
|
844
|
-
...(outcome.executionGateReservation ? { reservation: outcome.executionGateReservation } : {}),
|
|
845
|
-
});
|
|
846
|
-
const failureResult = halt
|
|
847
|
-
? createToolFailureRecoveryExhaustedResult(halt.record, halt.diagnostic)
|
|
848
|
-
: createToolFailureResult(record, outcome.result.terminate);
|
|
820
|
+
const failureResult = createToolFailureResult(record, outcome.result.terminate);
|
|
849
821
|
return {
|
|
850
822
|
toolCall,
|
|
851
823
|
result: outcome.repeatedSuccessfulCall
|
|
@@ -858,6 +830,12 @@ function finalizeRejectedToolCall(toolCall, tool, outcome, tracker, toolFailureR
|
|
|
858
830
|
}
|
|
859
831
|
: failureResult,
|
|
860
832
|
isError: true,
|
|
833
|
+
executionGateEffect: {
|
|
834
|
+
kind: "unproductive",
|
|
835
|
+
...(tool ? { tool } : {}),
|
|
836
|
+
record,
|
|
837
|
+
args: toolCall.arguments,
|
|
838
|
+
},
|
|
861
839
|
};
|
|
862
840
|
}
|
|
863
841
|
function createLinkedToolAbort(foregroundSignal) {
|
|
@@ -941,7 +919,13 @@ async function executeAndFinalizePreparedToolCall(currentContext, assistantMessa
|
|
|
941
919
|
if (executionAbort.signal.aborted)
|
|
942
920
|
return completion;
|
|
943
921
|
let handoffAccepted = false;
|
|
944
|
-
|
|
922
|
+
// A handed-off call leaves the batch, so the batch never applies its effect: the background
|
|
923
|
+
// completion is the only place left that can tell the governor what this operation did.
|
|
924
|
+
const handedOffCompletion = completion.then((finalized) => {
|
|
925
|
+
if (handoffAccepted)
|
|
926
|
+
toolFailureRecoveryGate.apply(finalized.executionGateEffect);
|
|
927
|
+
return finalized;
|
|
928
|
+
});
|
|
945
929
|
void handedOffCompletion.catch(() => undefined);
|
|
946
930
|
let handoff;
|
|
947
931
|
try {
|
|
@@ -988,7 +972,9 @@ async function executePreparedToolCall(prepared, signal, emit) {
|
|
|
988
972
|
// throwing. Keep the returned result intact through afterToolCall so
|
|
989
973
|
// policy hooks can inspect its bounded diagnostics and metadata.
|
|
990
974
|
isError: result.isError === true,
|
|
991
|
-
...(result.isError === true
|
|
975
|
+
...(result.isError === true
|
|
976
|
+
? { errorClass: "tool_result_error", errorKind: result.errorKind ?? "tool_failure" }
|
|
977
|
+
: {}),
|
|
992
978
|
};
|
|
993
979
|
}
|
|
994
980
|
catch (error) {
|
|
@@ -1000,6 +986,7 @@ async function executePreparedToolCall(prepared, signal, emit) {
|
|
|
1000
986
|
isError: true,
|
|
1001
987
|
errorClass: error instanceof Error ? error.name : typeof error,
|
|
1002
988
|
failureMessage: message,
|
|
989
|
+
errorKind: toolFailure?.errorKind ?? "tool_failure",
|
|
1003
990
|
...(toolFailure ? { failureCode: toolFailure.failureCode, outputSignature: toolFailure.outputSignature } : {}),
|
|
1004
991
|
};
|
|
1005
992
|
}
|
|
@@ -1035,6 +1022,7 @@ async function finalizeExecutedToolCall(currentContext, assistantMessage, prepar
|
|
|
1035
1022
|
let errorClass = executed.errorClass;
|
|
1036
1023
|
let failureCode = executed.failureCode;
|
|
1037
1024
|
let outputSignature = executed.outputSignature;
|
|
1025
|
+
let errorKind = executed.errorKind;
|
|
1038
1026
|
let executionGateEffect;
|
|
1039
1027
|
if (config.afterToolCall) {
|
|
1040
1028
|
try {
|
|
@@ -1057,10 +1045,12 @@ async function finalizeExecutedToolCall(currentContext, assistantMessage, prepar
|
|
|
1057
1045
|
}
|
|
1058
1046
|
}
|
|
1059
1047
|
catch (error) {
|
|
1048
|
+
// The hook itself failed, so nothing about the tool's own completed operation survives.
|
|
1060
1049
|
failureMessage = error instanceof Error ? error.message : String(error);
|
|
1061
1050
|
errorClass = error instanceof Error ? error.name : typeof error;
|
|
1062
1051
|
failureCode = undefined;
|
|
1063
1052
|
outputSignature = undefined;
|
|
1053
|
+
errorKind = "tool_failure";
|
|
1064
1054
|
result = { ...createErrorToolResult(failureMessage), usage: result.usage };
|
|
1065
1055
|
isError = true;
|
|
1066
1056
|
}
|
|
@@ -1076,20 +1066,34 @@ async function finalizeExecutedToolCall(currentContext, assistantMessage, prepar
|
|
|
1076
1066
|
const effectiveFailureMessage = failureMessage || result.content.find((block) => block.type === "text")?.text || "Tool execution failed";
|
|
1077
1067
|
const assessment = assessToolFailure(effectiveFailureMessage, "failed", errorClass);
|
|
1078
1068
|
const effectiveFailureCode = failureCode ?? assessment.failureCode;
|
|
1079
|
-
|
|
1080
|
-
|
|
1081
|
-
|
|
1082
|
-
|
|
1083
|
-
|
|
1084
|
-
|
|
1085
|
-
|
|
1086
|
-
|
|
1087
|
-
|
|
1088
|
-
|
|
1089
|
-
|
|
1090
|
-
|
|
1091
|
-
|
|
1092
|
-
|
|
1069
|
+
if (errorKind === "operation_outcome") {
|
|
1070
|
+
// The tool ran the operation to completion; the non-zero status is the observation the
|
|
1071
|
+
// agent asked for. Nothing here is a mistake, so no failure record is remembered and the
|
|
1072
|
+
// tool's own output stands exactly as written. The governor still notes that repeating
|
|
1073
|
+
// this identical operation cannot say anything new until something else changes.
|
|
1074
|
+
clearToolFailure(toolFailureMemory, prepared.toolCall.name, prepared.args);
|
|
1075
|
+
result = { ...result, errorKind: "operation_outcome", usage };
|
|
1076
|
+
executionGateEffect = {
|
|
1077
|
+
kind: "unproductive",
|
|
1078
|
+
tool: prepared.tool,
|
|
1079
|
+
record: describeOperationOutcome(prepared.toolCall.name, prepared.args, effectiveFailureCode, assessment.diagnostic),
|
|
1080
|
+
args: prepared.args,
|
|
1081
|
+
};
|
|
1082
|
+
}
|
|
1083
|
+
else {
|
|
1084
|
+
const recoveryPlan = toolFailureRecoveryGate.planFailure(prepared.tool, prepared.args, { failureCode: effectiveFailureCode, message: effectiveFailureMessage }, currentContext.tools ?? []);
|
|
1085
|
+
const correction = assessment.policyGuidance
|
|
1086
|
+
? `${assessment.policyGuidance} ${recoveryPlan.guidance}`
|
|
1087
|
+
: recoveryPlan.guidance;
|
|
1088
|
+
const record = rememberToolFailure(toolFailureMemory, prepared.toolCall.name, prepared.args, "failed", effectiveFailureCode, correction, assessment.diagnostic, assessment.phase, recoveryPlan.evidence ?? assessment.evidence, { output: failureOutput, outputSignature });
|
|
1089
|
+
executionGateEffect = {
|
|
1090
|
+
kind: "unproductive",
|
|
1091
|
+
tool: prepared.tool,
|
|
1092
|
+
record,
|
|
1093
|
+
args: prepared.args,
|
|
1094
|
+
};
|
|
1095
|
+
result = { ...createToolFailureResult(record, result.terminate), usage };
|
|
1096
|
+
}
|
|
1093
1097
|
}
|
|
1094
1098
|
else {
|
|
1095
1099
|
clearToolFailure(toolFailureMemory, prepared.toolCall.name, prepared.args);
|
|
@@ -1098,7 +1102,6 @@ async function finalizeExecutedToolCall(currentContext, assistantMessage, prepar
|
|
|
1098
1102
|
kind: "success",
|
|
1099
1103
|
tool: prepared.tool,
|
|
1100
1104
|
args: prepared.args,
|
|
1101
|
-
evidenceResult: executed.result,
|
|
1102
1105
|
};
|
|
1103
1106
|
}
|
|
1104
1107
|
}
|
|
@@ -1156,6 +1159,9 @@ function createToolResultMessage(finalized) {
|
|
|
1156
1159
|
details: finalized.result.details,
|
|
1157
1160
|
usage: finalized.result.usage,
|
|
1158
1161
|
isError: finalized.isError,
|
|
1162
|
+
// Persisted so a reloaded transcript still separates a completed operation's own status from a
|
|
1163
|
+
// tool that could not run at all.
|
|
1164
|
+
...(finalized.isError && finalized.result.errorKind ? { errorKind: finalized.result.errorKind } : {}),
|
|
1159
1165
|
timestamp: Date.now(),
|
|
1160
1166
|
};
|
|
1161
1167
|
}
|