@caupulican/pi-agent-core 0.93.7 → 0.93.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-loop.d.ts.map +1 -1
- package/dist/agent-loop.js +179 -155
- package/dist/agent-loop.js.map +1 -1
- package/dist/index.d.ts +1 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -0
- package/dist/index.js.map +1 -1
- package/dist/tool-failure-memory.d.ts +17 -5
- package/dist/tool-failure-memory.d.ts.map +1 -1
- package/dist/tool-failure-memory.js +128 -82
- package/dist/tool-failure-memory.js.map +1 -1
- package/dist/tool-failure-recovery-gate.d.ts +32 -42
- package/dist/tool-failure-recovery-gate.d.ts.map +1 -1
- package/dist/tool-failure-recovery-gate.js +161 -267
- package/dist/tool-failure-recovery-gate.js.map +1 -1
- package/dist/tool-failure-recovery-protocol.d.ts +2 -0
- package/dist/tool-failure-recovery-protocol.d.ts.map +1 -1
- package/dist/tool-failure-recovery-protocol.js +7 -6
- package/dist/tool-failure-recovery-protocol.js.map +1 -1
- package/dist/tool-protocol-residue.d.ts +9 -0
- package/dist/tool-protocol-residue.d.ts.map +1 -1
- package/dist/tool-protocol-residue.js +28 -1
- package/dist/tool-protocol-residue.js.map +1 -1
- package/dist/types.d.ts +50 -18
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +17 -0
- package/dist/types.js.map +1 -1
- package/package.json +2 -2
package/dist/agent-loop.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"agent-loop.d.ts","sourceRoot":"","sources":["../src/agent-loop.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAEH,OAAO,EAAE,WAAW,EAAE,MAAM,gCAAgC,CAAC;
|
|
1
|
+
{"version":3,"file":"agent-loop.d.ts","sourceRoot":"","sources":["../src/agent-loop.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAEH,OAAO,EAAE,WAAW,EAAE,MAAM,gCAAgC,CAAC;AAgC7D,OAAO,EAAE,uBAAuB,EAAsC,MAAM,iCAAiC,CAAC;AAE9G,OAAO,KAAK,EACX,YAAY,EACZ,UAAU,EACV,eAAe,EACf,YAAY,EAEZ,aAAa,EAGb,QAAQ,EACR,kBAAkB,EAClB,MAAM,YAAY,CAAC;AAIpB,OAAO,EACN,0BAA0B,EAC1B,sBAAsB,EACtB,gCAAgC,GAChC,MAAM,+BAA+B,CAAC;AAEvC,MAAM,MAAM,cAAc,GAAG,CAAC,KAAK,EAAE,UAAU,KAAK,OAAO,CAAC,IAAI,CAAC,GAAG,IAAI,CAAC;AAKzE,gGAAgG;AAChG,MAAM,WAAW,0BAA0B;IAC1C,aAAa,EAAE,MAAM,CAAC;IACtB,WAAW,EAAE,MAAM,EAAE,CAAC;IACtB,uBAAuB,EAAE,uBAAuB,CAAC;CACjD;AAED,wBAAgB,gCAAgC,IAAI,0BAA0B,CAE7E;AAED;;;GAGG;AACH,wBAAgB,SAAS,CACxB,OAAO,EAAE,YAAY,EAAE,EACvB,OAAO,EAAE,YAAY,EACrB,MAAM,EAAE,eAAe,EACvB,MAAM,CAAC,EAAE,WAAW,EACpB,QAAQ,CAAC,EAAE,QAAQ,GACjB,WAAW,CAAC,UAAU,EAAE,YAAY,EAAE,CAAC,CAEzC;AAED;;;;;;;GAOG;AACH,wBAAgB,iBAAiB,CAChC,OAAO,EAAE,YAAY,EACrB,MAAM,EAAE,eAAe,EACvB,MAAM,CAAC,EAAE,WAAW,EACpB,QAAQ,CAAC,EAAE,QAAQ,GACjB,WAAW,CAAC,UAAU,EAAE,YAAY,EAAE,CAAC,CAIzC;AAuBD,wBAAsB,YAAY,CACjC,OAAO,EAAE,YAAY,EAAE,EACvB,OAAO,EAAE,YAAY,EACrB,MAAM,EAAE,eAAe,EACvB,IAAI,EAAE,cAAc,EACpB,MAAM,CAAC,EAAE,WAAW,EACpB,QAAQ,CAAC,EAAE,QAAQ,EACnB,iBAAiB,GAAE,0BAA+D,GAChF,OAAO,CAAC,YAAY,EAAE,CAAC,CAwBzB;AAED,wBAAsB,oBAAoB,CACzC,OAAO,EAAE,YAAY,EACrB,MAAM,EAAE,eAAe,EACvB,IAAI,EAAE,cAAc,EACpB,MAAM,CAAC,EAAE,WAAW,EACpB,QAAQ,CAAC,EAAE,QAAQ,EACnB,iBAAiB,GAAE,0BAA+D,GAChF,OAAO,CAAC,YAAY,EAAE,CAAC,CAezB;AAmWD;;;;;;GAMG;AACH,wBAAsB,yBAAyB,CAC9C,OAAO,EAAE,YAAY,EACrB,MAAM,EAAE,eAAe,EACvB,MAAM,EAAE,WAAW,GAAG,SAAS,EAC/B,QAAQ,CAAC,EAAE,QAAQ,GACjB,OAAO,CAAC,OAAO,CAAC,UAAU,CAAC,QAAQ,CAAC,CAAC,CAAC,CAExC;AA6iCD,wBAAgB,qBAAqB,CAAC,QAAQ,EAAE,aAAa,GAAG,kBAAkB,GAAG,SAAS,CAO7F"}
|
package/dist/agent-loop.js
CHANGED
|
@@ -7,10 +7,10 @@ import { formatToolRepairStandingRule, REPEATED_SUCCESSFUL_TOOL_CALL_FAILURE, }
|
|
|
7
7
|
import { ToolArgumentValidationError, validateToolArguments, } from "@caupulican/pi-ai/validation";
|
|
8
8
|
import { assistantMessageText, collapseDegenerateAssistantMessage, shouldAbortDegenerateStream, } from "./degenerate-assistant-text.js";
|
|
9
9
|
import { startPlannedAgentProviderRequest } from "./provider-request-planner.js";
|
|
10
|
-
import { assessToolFailure, clearToolFailure, createRepeatedToolFailureResult, createToolFailureMemoryTracker,
|
|
11
|
-
import { ToolFailureRecoveryGate
|
|
12
|
-
import { rejectNativeToolProtocolResidue } from "./tool-protocol-residue.js";
|
|
13
|
-
import { DEFAULT_MAX_PROVIDER_TURNS, DEFAULT_MAX_STALL_TURNS } from "./types.js";
|
|
10
|
+
import { assessToolFailure, clearToolFailure, createRepeatedToolFailureResult, createToolFailureMemoryTracker, createToolFailureResult, describeOperationOutcome, getUnresolvedToolFailure, normalizeToolSignature, rememberToolFailure, toolFailureCorrection, } from "./tool-failure-memory.js";
|
|
11
|
+
import { ToolFailureRecoveryGate } from "./tool-failure-recovery-gate.js";
|
|
12
|
+
import { rejectNativeToolProtocolResidue, rejectToolCallsFromToolFreeResponse } from "./tool-protocol-residue.js";
|
|
13
|
+
import { AgentToolExecutionError, DEFAULT_MAX_PROVIDER_TURNS, DEFAULT_MAX_STALL_TURNS } from "./types.js";
|
|
14
14
|
import { createEmptyUsage } from "./usage.js";
|
|
15
15
|
export { composeRequestSystemPrompt, narrowRequestMaxTokens, resolveRequestPreflightMaxTokens, } from "./provider-request-planner.js";
|
|
16
16
|
/** Bound simultaneous tool starts without imposing a failure-count or operation-count stop. */
|
|
@@ -59,6 +59,14 @@ export async function runAgentLoop(prompts, context, config, emit, signal, strea
|
|
|
59
59
|
messages: [...context.messages, ...prompts],
|
|
60
60
|
};
|
|
61
61
|
await emit({ type: "agent_start" });
|
|
62
|
+
if (providerTurnLimitReached(config, continuationState)) {
|
|
63
|
+
for (const prompt of prompts) {
|
|
64
|
+
await emit({ type: "message_start", message: prompt });
|
|
65
|
+
await emit({ type: "message_end", message: prompt });
|
|
66
|
+
}
|
|
67
|
+
await emitProviderTurnLimitStop(config, continuationState, newMessages, emit);
|
|
68
|
+
return newMessages;
|
|
69
|
+
}
|
|
62
70
|
await emit({ type: "turn_start" });
|
|
63
71
|
for (const prompt of prompts) {
|
|
64
72
|
await emit({ type: "message_start", message: prompt });
|
|
@@ -72,6 +80,10 @@ export async function runAgentLoopContinue(context, config, emit, signal, stream
|
|
|
72
80
|
const newMessages = [];
|
|
73
81
|
const currentContext = { ...context, messages: [...context.messages] };
|
|
74
82
|
await emit({ type: "agent_start" });
|
|
83
|
+
if (providerTurnLimitReached(config, continuationState)) {
|
|
84
|
+
await emitProviderTurnLimitStop(config, continuationState, newMessages, emit);
|
|
85
|
+
return newMessages;
|
|
86
|
+
}
|
|
75
87
|
await emit({ type: "turn_start" });
|
|
76
88
|
await runLoop(currentContext, newMessages, config, signal, emit, streamFn, continuationState);
|
|
77
89
|
return newMessages;
|
|
@@ -97,36 +109,21 @@ function createLoopFailureMessage(error, config, aborted) {
|
|
|
97
109
|
timestamp: Date.now(),
|
|
98
110
|
};
|
|
99
111
|
}
|
|
100
|
-
function
|
|
101
|
-
|
|
102
|
-
return createLocalDiagnosticMessage(config, `Tool recovery stopped for ${halt.record.tool}: ${diagnostic} Required recovery: ${halt.record.correction}`);
|
|
103
|
-
}
|
|
104
|
-
function createProviderTurnLimitMessage(config, providerTurns) {
|
|
105
|
-
return createLocalDiagnosticMessage(config, `Configured provider turn limit (${providerTurns}) reached. The harness stopped before another provider request; continue explicitly if more work is needed.`);
|
|
112
|
+
function createAgentStream() {
|
|
113
|
+
return new EventStream((event) => event.type === "agent_end", (event) => (event.type === "agent_end" ? event.messages : []));
|
|
106
114
|
}
|
|
107
|
-
function
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
content: [{ type: "text", text }],
|
|
111
|
-
api: config.model.api,
|
|
112
|
-
provider: config.model.provider,
|
|
113
|
-
model: config.model.id,
|
|
114
|
-
usage: createEmptyUsage(),
|
|
115
|
-
stopReason: "stop",
|
|
116
|
-
timestamp: Date.now(),
|
|
117
|
-
};
|
|
115
|
+
function providerTurnLimitReached(config, continuationState) {
|
|
116
|
+
const limit = config.maxProviderTurns ?? DEFAULT_MAX_PROVIDER_TURNS;
|
|
117
|
+
return limit > 0 && continuationState.providerTurns >= limit;
|
|
118
118
|
}
|
|
119
|
-
async function
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
119
|
+
async function emitProviderTurnLimitStop(config, continuationState, newMessages, emit) {
|
|
120
|
+
config.onRunawayStop?.({
|
|
121
|
+
reason: "provider_turn_limit",
|
|
122
|
+
signature: "provider_turn_limit",
|
|
123
|
+
repeats: continuationState.providerTurns,
|
|
124
|
+
});
|
|
125
125
|
await emit({ type: "agent_end", messages: newMessages });
|
|
126
126
|
}
|
|
127
|
-
function createAgentStream() {
|
|
128
|
-
return new EventStream((event) => event.type === "agent_end", (event) => (event.type === "agent_end" ? event.messages : []));
|
|
129
|
-
}
|
|
130
127
|
/**
|
|
131
128
|
* How many `stallLimit`-length periods the runaway-loop window spans. A window of `stallLimit * P`
|
|
132
129
|
* turns lets the count-based detector catch oscillating cycles of period up to `P` (each signature in a
|
|
@@ -206,40 +203,41 @@ async function runLoop(initialContext, newMessages, initialConfig, signal, emit,
|
|
|
206
203
|
let lastSuccessfulTextProtocolBatch;
|
|
207
204
|
// Check for steering messages at start (user may have typed while waiting)
|
|
208
205
|
let pendingMessages = (await config.getSteeringMessages?.()) || [];
|
|
206
|
+
const processPendingMessages = async () => {
|
|
207
|
+
if (pendingMessages.length === 0)
|
|
208
|
+
return;
|
|
209
|
+
lastSuccessfulTextProtocolBatch = undefined;
|
|
210
|
+
// A new user turn can change authority, intent, or the workspace itself, so every
|
|
211
|
+
// operation that was blocked as an unproductive replay becomes worth attempting again.
|
|
212
|
+
toolFailureRecoveryGate.noteWorldAdvance();
|
|
213
|
+
for (const message of pendingMessages) {
|
|
214
|
+
await emit({ type: "message_start", message });
|
|
215
|
+
await emit({ type: "message_end", message });
|
|
216
|
+
currentContext.messages.push(message);
|
|
217
|
+
newMessages.push(message);
|
|
218
|
+
}
|
|
219
|
+
pendingMessages = [];
|
|
220
|
+
};
|
|
209
221
|
// Outer loop: continues when queued follow-up messages arrive after agent would stop
|
|
210
222
|
while (true) {
|
|
211
223
|
let hasMoreToolCalls = true;
|
|
212
224
|
// Inner loop: process tool calls and steering messages
|
|
213
225
|
while (hasMoreToolCalls || pendingMessages.length > 0) {
|
|
226
|
+
if (providerTurnLimit > 0 && continuationState.providerTurns >= providerTurnLimit) {
|
|
227
|
+
// Preserve already-dequeued steering, but do not announce an assistant turn that will
|
|
228
|
+
// never start. A turn is one provider response plus its tools/results.
|
|
229
|
+
await processPendingMessages();
|
|
230
|
+
await emitProviderTurnLimitStop(config, continuationState, newMessages, emit);
|
|
231
|
+
return;
|
|
232
|
+
}
|
|
214
233
|
if (!firstTurn) {
|
|
215
234
|
await emit({ type: "turn_start" });
|
|
216
235
|
}
|
|
217
236
|
else {
|
|
218
237
|
firstTurn = false;
|
|
219
238
|
}
|
|
220
|
-
// Process pending messages (inject before next assistant response)
|
|
221
|
-
|
|
222
|
-
lastSuccessfulTextProtocolBatch = undefined;
|
|
223
|
-
for (const message of pendingMessages) {
|
|
224
|
-
await emit({ type: "message_start", message });
|
|
225
|
-
await emit({ type: "message_end", message });
|
|
226
|
-
currentContext.messages.push(message);
|
|
227
|
-
newMessages.push(message);
|
|
228
|
-
}
|
|
229
|
-
pendingMessages = [];
|
|
230
|
-
}
|
|
231
|
-
if (providerTurnLimit > 0 && continuationState.providerTurns >= providerTurnLimit) {
|
|
232
|
-
const fallback = createProviderTurnLimitMessage(config, continuationState.providerTurns);
|
|
233
|
-
currentContext.messages.push(fallback);
|
|
234
|
-
newMessages.push(fallback);
|
|
235
|
-
config.onRunawayStop?.({
|
|
236
|
-
reason: "provider_turn_limit",
|
|
237
|
-
signature: "provider_turn_limit",
|
|
238
|
-
repeats: continuationState.providerTurns,
|
|
239
|
-
});
|
|
240
|
-
await emitTerminalLocalMessage(fallback, newMessages, emit, false);
|
|
241
|
-
return;
|
|
242
|
-
}
|
|
239
|
+
// Process pending messages (inject before next assistant response).
|
|
240
|
+
await processPendingMessages();
|
|
243
241
|
continuationState.providerTurns++;
|
|
244
242
|
const message = await streamAssistantResponse(currentContext, config, signal, emit, streamFn);
|
|
245
243
|
newMessages.push(message);
|
|
@@ -251,7 +249,6 @@ async function runLoop(initialContext, newMessages, initialConfig, signal, emit,
|
|
|
251
249
|
// Check for tool calls
|
|
252
250
|
const toolCalls = message.content.filter((c) => c.type === "toolCall");
|
|
253
251
|
const toolResults = [];
|
|
254
|
-
let recoveryHalt;
|
|
255
252
|
hasMoreToolCalls = false;
|
|
256
253
|
if (toolCalls.length > 0) {
|
|
257
254
|
const textProtocolBatch = toolCalls.every((toolCall) => toolCall.source === "text-protocol");
|
|
@@ -262,14 +259,7 @@ async function runLoop(initialContext, newMessages, initialConfig, signal, emit,
|
|
|
262
259
|
: undefined;
|
|
263
260
|
const executedToolBatch = await executeToolCalls(currentContext, message, config, validationFailureTracker, repairTeachTracker, toolFailureMemory, toolFailureRecoveryGate, previousSuccessfulTextProtocolResults, signal, emit);
|
|
264
261
|
toolResults.push(...executedToolBatch.messages);
|
|
265
|
-
|
|
266
|
-
recoveryHalt = toolFailureRecoveryGate.getHalt();
|
|
267
|
-
if (!recoveryHalt)
|
|
268
|
-
throw new Error("Tool recovery halted without a failure record");
|
|
269
|
-
}
|
|
270
|
-
else {
|
|
271
|
-
hasMoreToolCalls = !executedToolBatch.terminate;
|
|
272
|
-
}
|
|
262
|
+
hasMoreToolCalls = !executedToolBatch.terminate;
|
|
273
263
|
for (const result of toolResults) {
|
|
274
264
|
currentContext.messages.push(result);
|
|
275
265
|
newMessages.push(result);
|
|
@@ -291,13 +281,6 @@ async function runLoop(initialContext, newMessages, initialConfig, signal, emit,
|
|
|
291
281
|
lastSuccessfulTextProtocolBatch = undefined;
|
|
292
282
|
}
|
|
293
283
|
await emit({ type: "turn_end", message, toolResults });
|
|
294
|
-
if (recoveryHalt) {
|
|
295
|
-
const fallback = createMandatoryRecoveryDeliveryFallback(recoveryHalt, config);
|
|
296
|
-
currentContext.messages.push(fallback);
|
|
297
|
-
newMessages.push(fallback);
|
|
298
|
-
await emitTerminalLocalMessage(fallback, newMessages, emit, true);
|
|
299
|
-
return;
|
|
300
|
-
}
|
|
301
284
|
// Runaway-loop backstop (cost guard): detect a model stuck repeating one action.
|
|
302
285
|
if (stallLimit > 0 && toolCalls.length > 0) {
|
|
303
286
|
const signature = normalizeToolSignature(toolCalls.map((c) => [c.name, c.arguments ?? null]));
|
|
@@ -307,6 +290,7 @@ async function runLoop(initialContext, newMessages, initialConfig, signal, emit,
|
|
|
307
290
|
const repeats = stallWindow.reduce((n, s) => (s === signature ? n + 1 : n), 0);
|
|
308
291
|
if (repeats >= stallLimit) {
|
|
309
292
|
config.onRunawayStop?.({ reason: "repeated_tool_call", signature, repeats });
|
|
293
|
+
await streamToollessClosingTurn(currentContext, newMessages, config, continuationState, providerTurnLimit, signal, emit, streamFn);
|
|
310
294
|
await emit({ type: "agent_end", messages: newMessages });
|
|
311
295
|
return;
|
|
312
296
|
}
|
|
@@ -352,6 +336,41 @@ async function runLoop(initialContext, newMessages, initialConfig, signal, emit,
|
|
|
352
336
|
}
|
|
353
337
|
await emit({ type: "agent_end", messages: newMessages });
|
|
354
338
|
}
|
|
339
|
+
const RUNAWAY_STOP_CLOSING_SYSTEM_PROMPT = [
|
|
340
|
+
"RUNAWAY STOP CLOSING TURN",
|
|
341
|
+
"The host stopped a repeated tool-call loop. No tools are available in this final request.",
|
|
342
|
+
"Write one concise factual closing message for the user: completed work, the unresolved operation or blocker, and the safest next action.",
|
|
343
|
+
"Do not claim unperformed work. Do not emit a tool call or tool-call markup.",
|
|
344
|
+
].join("\n");
|
|
345
|
+
/**
|
|
346
|
+
* Spend one final provider request, with no tools, so a stopped run closes in the model's own words.
|
|
347
|
+
*
|
|
348
|
+
* The harness never writes that message itself. Tools are removed from the request, so this request
|
|
349
|
+
* cannot open another tool batch and cannot re-enter the loop; it runs through the same planned
|
|
350
|
+
* provider boundary as every other request. If the run is aborted, or the configured provider-turn
|
|
351
|
+
* limit leaves no budget, the run ends with no closing message rather than a fabricated one — and a
|
|
352
|
+
* provider error keeps its own error message for the same reason.
|
|
353
|
+
*/
|
|
354
|
+
async function streamToollessClosingTurn(currentContext, newMessages, config, continuationState, providerTurnLimit, signal, emit, streamFn) {
|
|
355
|
+
if (signal?.aborted)
|
|
356
|
+
return;
|
|
357
|
+
if (providerTurnLimit > 0 && continuationState.providerTurns >= providerTurnLimit)
|
|
358
|
+
return;
|
|
359
|
+
continuationState.providerTurns++;
|
|
360
|
+
await emit({ type: "turn_start" });
|
|
361
|
+
const closingContext = {
|
|
362
|
+
...currentContext,
|
|
363
|
+
systemPrompt: currentContext.systemPrompt
|
|
364
|
+
? `${currentContext.systemPrompt}\n\n${RUNAWAY_STOP_CLOSING_SYSTEM_PROMPT}`
|
|
365
|
+
: RUNAWAY_STOP_CLOSING_SYSTEM_PROMPT,
|
|
366
|
+
tools: [],
|
|
367
|
+
};
|
|
368
|
+
const message = await streamAssistantResponse(closingContext, config, signal, emit, streamFn, {
|
|
369
|
+
rejectToolCalls: true,
|
|
370
|
+
});
|
|
371
|
+
newMessages.push(message);
|
|
372
|
+
await emit({ type: "turn_end", message, toolResults: [] });
|
|
373
|
+
}
|
|
355
374
|
/**
|
|
356
375
|
* Start one provider request through the canonical agent-loop boundary.
|
|
357
376
|
*
|
|
@@ -366,7 +385,7 @@ export async function startAgentProviderRequest(context, config, signal, streamF
|
|
|
366
385
|
* Stream an assistant response from the LLM.
|
|
367
386
|
* This is where AgentMessage[] gets transformed to Message[] for the LLM.
|
|
368
387
|
*/
|
|
369
|
-
async function streamAssistantResponse(context, config, signal, emit, streamFn) {
|
|
388
|
+
async function streamAssistantResponse(context, config, signal, emit, streamFn, policy) {
|
|
370
389
|
const degenerationAbort = new AbortController();
|
|
371
390
|
const onOuterAbort = () => degenerationAbort.abort();
|
|
372
391
|
if (signal?.aborted)
|
|
@@ -417,7 +436,10 @@ async function streamAssistantResponse(context, config, signal, emit, streamFn)
|
|
|
417
436
|
}
|
|
418
437
|
}
|
|
419
438
|
signal?.removeEventListener("abort", onOuterAbort);
|
|
420
|
-
|
|
439
|
+
const providerMessage = await response.result();
|
|
440
|
+
let finalMessage = policy?.rejectToolCalls
|
|
441
|
+
? rejectToolCallsFromToolFreeResponse(providerMessage)
|
|
442
|
+
: rejectNativeToolProtocolResidue(providerMessage, context.tools ?? [], Boolean(config.textToolCallProtocol));
|
|
421
443
|
if (abortedForDegeneration && !signal?.aborted && finalMessage.stopReason === "aborted") {
|
|
422
444
|
finalMessage = { ...finalMessage, stopReason: "stop" };
|
|
423
445
|
delete finalMessage.errorMessage;
|
|
@@ -465,7 +487,7 @@ async function prepareAndStartToolCall(execCtx, toolCall, index) {
|
|
|
465
487
|
emitToolArgumentValidationTelemetry(execCtx.config, preparation.validationEvent, "not_run", "none");
|
|
466
488
|
return {
|
|
467
489
|
kind: "finalized",
|
|
468
|
-
finalized: finalizeRejectedToolCall(toolCall, execCtx.context.tools?.find((tool) => tool.name === toolCall.name), preparation, execCtx.toolFailureMemory
|
|
490
|
+
finalized: finalizeRejectedToolCall(toolCall, execCtx.context.tools?.find((tool) => tool.name === toolCall.name), preparation, execCtx.toolFailureMemory),
|
|
469
491
|
};
|
|
470
492
|
}
|
|
471
493
|
return { kind: "prepared", preparation };
|
|
@@ -480,7 +502,8 @@ async function executeToolCallsSequential(execCtx, toolCalls) {
|
|
|
480
502
|
const messages = [];
|
|
481
503
|
for (const [index, toolCall] of toolCalls.entries()) {
|
|
482
504
|
const started = await prepareAndStartToolCall(execCtx, toolCall, index);
|
|
483
|
-
const finalized =
|
|
505
|
+
const finalized = await finalizeStartedToolCall(execCtx, started);
|
|
506
|
+
execCtx.toolFailureRecoveryGate.apply(finalized.executionGateEffect);
|
|
484
507
|
await emitToolExecutionEnd(finalized, execCtx.emit);
|
|
485
508
|
const toolResultMessage = createToolResultMessage(finalized);
|
|
486
509
|
await emitToolResultMessage(toolResultMessage, execCtx.emit);
|
|
@@ -490,10 +513,7 @@ async function executeToolCallsSequential(execCtx, toolCalls) {
|
|
|
490
513
|
break;
|
|
491
514
|
}
|
|
492
515
|
}
|
|
493
|
-
return {
|
|
494
|
-
messages,
|
|
495
|
-
terminate: execCtx.toolFailureRecoveryGate.isHalted() || shouldTerminateToolBatch(finalizedCalls),
|
|
496
|
-
};
|
|
516
|
+
return { messages, terminate: shouldTerminateToolBatch(finalizedCalls) };
|
|
497
517
|
}
|
|
498
518
|
async function executeToolCallsParallel(execCtx, toolCalls) {
|
|
499
519
|
const orderedFinalizedCalls = [];
|
|
@@ -523,15 +543,11 @@ async function executeToolCallsParallel(execCtx, toolCalls) {
|
|
|
523
543
|
break;
|
|
524
544
|
}
|
|
525
545
|
}
|
|
526
|
-
const
|
|
527
|
-
const
|
|
528
|
-
|
|
529
|
-
const applied = applyToolFailureRecoveryEffect(execCtx.toolFailureRecoveryGate, finalized);
|
|
530
|
-
appliedByOriginal.set(finalized, applied);
|
|
531
|
-
return applied;
|
|
532
|
-
});
|
|
546
|
+
const finalizedWave = await Promise.all(wave.map((entry) => (typeof entry === "function" ? entry() : Promise.resolve(entry))));
|
|
547
|
+
for (const finalized of finalizedWave)
|
|
548
|
+
execCtx.toolFailureRecoveryGate.apply(finalized.executionGateEffect);
|
|
533
549
|
for (const finalized of completionOrder) {
|
|
534
|
-
await emitToolExecutionEnd(
|
|
550
|
+
await emitToolExecutionEnd(finalized, execCtx.emit);
|
|
535
551
|
}
|
|
536
552
|
orderedFinalizedCalls.push(...finalizedWave);
|
|
537
553
|
}
|
|
@@ -541,27 +557,7 @@ async function executeToolCallsParallel(execCtx, toolCalls) {
|
|
|
541
557
|
await emitToolResultMessage(toolResultMessage, execCtx.emit);
|
|
542
558
|
messages.push(toolResultMessage);
|
|
543
559
|
}
|
|
544
|
-
return {
|
|
545
|
-
messages,
|
|
546
|
-
terminate: execCtx.toolFailureRecoveryGate.isHalted() || shouldTerminateToolBatch(orderedFinalizedCalls),
|
|
547
|
-
};
|
|
548
|
-
}
|
|
549
|
-
function applyToolFailureRecoveryEffect(gate, finalized) {
|
|
550
|
-
const halt = gate.apply(finalized.executionGateEffect);
|
|
551
|
-
if (!halt)
|
|
552
|
-
return finalized;
|
|
553
|
-
return createRecoveryExhaustedToolCallOutcome(finalized, halt);
|
|
554
|
-
}
|
|
555
|
-
function createRecoveryExhaustedToolCallOutcome(finalized, halt) {
|
|
556
|
-
const exhaustedResult = createToolFailureRecoveryExhaustedResult(halt.record, halt.diagnostic);
|
|
557
|
-
return {
|
|
558
|
-
...finalized,
|
|
559
|
-
result: {
|
|
560
|
-
...exhaustedResult,
|
|
561
|
-
...(finalized.result.usage ? { usage: finalized.result.usage } : {}),
|
|
562
|
-
},
|
|
563
|
-
isError: true,
|
|
564
|
-
};
|
|
560
|
+
return { messages, terminate: shouldTerminateToolBatch(orderedFinalizedCalls) };
|
|
565
561
|
}
|
|
566
562
|
const DEFAULT_TOOL_VALIDATION_ESCALATION_THRESHOLD = 3;
|
|
567
563
|
const TOOL_REPAIR_TEACH_EVERY = 5;
|
|
@@ -586,7 +582,7 @@ function createRepeatedSuccessfulToolCallOutcome(previousResult) {
|
|
|
586
582
|
...(previousResult ? { repeatedSuccessfulCall: { previousToolCallId: previousResult.toolCallId } } : {}),
|
|
587
583
|
};
|
|
588
584
|
}
|
|
589
|
-
function createAbortedToolCallOutcome(validationEvent
|
|
585
|
+
function createAbortedToolCallOutcome(validationEvent) {
|
|
590
586
|
return {
|
|
591
587
|
kind: "immediate",
|
|
592
588
|
result: createErrorToolResult("Operation aborted"),
|
|
@@ -594,7 +590,6 @@ function createAbortedToolCallOutcome(validationEvent, executionGateReservation)
|
|
|
594
590
|
phase: "cancelled",
|
|
595
591
|
failureCode: "aborted",
|
|
596
592
|
correction: "Retry only if the operation is still required.",
|
|
597
|
-
...(executionGateReservation ? { executionGateReservation } : {}),
|
|
598
593
|
validationEvent,
|
|
599
594
|
};
|
|
600
595
|
}
|
|
@@ -717,7 +712,6 @@ async function prepareToolCall(currentContext, assistantMessage, toolCall, confi
|
|
|
717
712
|
};
|
|
718
713
|
}
|
|
719
714
|
let validationEvent;
|
|
720
|
-
let executionGateReservation;
|
|
721
715
|
try {
|
|
722
716
|
const preparedToolCall = prepareToolCallArguments(tool, toolCall);
|
|
723
717
|
const validatedArgs = validateToolArguments(tool, preparedToolCall, {
|
|
@@ -739,16 +733,7 @@ async function prepareToolCall(currentContext, assistantMessage, toolCall, confi
|
|
|
739
733
|
const unresolvedRecord = getUnresolvedToolFailure(toolFailureMemory, toolCall.name, validatedArgs);
|
|
740
734
|
const admission = toolFailureRecoveryGate.admit(tool, validatedArgs, unresolvedRecord, currentContext.messages);
|
|
741
735
|
if (admission.kind === "blocked") {
|
|
742
|
-
const
|
|
743
|
-
? admission.scope === "operation"
|
|
744
|
-
? "operation_recovery_exhausted"
|
|
745
|
-
: "recovery_exhausted"
|
|
746
|
-
: "repeated_failed_operation";
|
|
747
|
-
const result = admission.exhausted
|
|
748
|
-
? admission.scope === "operation"
|
|
749
|
-
? createToolFailureOperationExhaustedResult(admission.record, admission.diagnostic ?? "Tool operation recovery budget exhausted.")
|
|
750
|
-
: createToolFailureRecoveryExhaustedResult(admission.record, admission.diagnostic ?? "Tool failure recovery budget exhausted.")
|
|
751
|
-
: createRepeatedToolFailureResult(admission.record);
|
|
736
|
+
const result = createRepeatedToolFailureResult(admission.record);
|
|
752
737
|
const memoryRecord = result.details.piToolFailureMemory;
|
|
753
738
|
toolFailureMemory.set(admission.record.failureKey, memoryRecord);
|
|
754
739
|
return {
|
|
@@ -756,14 +741,13 @@ async function prepareToolCall(currentContext, assistantMessage, toolCall, confi
|
|
|
756
741
|
result,
|
|
757
742
|
isError: true,
|
|
758
743
|
phase: admission.record.phase,
|
|
759
|
-
failureCode,
|
|
744
|
+
failureCode: "repeated_failed_operation",
|
|
760
745
|
correction: memoryRecord.correction,
|
|
761
746
|
diagnostic: memoryRecord.diagnostic,
|
|
762
747
|
repeatedToolFailure: true,
|
|
763
|
-
validationEvent: createValidationBounceTelemetry(config, toolCall,
|
|
748
|
+
validationEvent: createValidationBounceTelemetry(config, toolCall, "repeated_failed_operation"),
|
|
764
749
|
};
|
|
765
750
|
}
|
|
766
|
-
executionGateReservation = admission.reservation;
|
|
767
751
|
if (config.beforeToolCall) {
|
|
768
752
|
const beforeResult = await config.beforeToolCall({
|
|
769
753
|
assistantMessage,
|
|
@@ -772,7 +756,7 @@ async function prepareToolCall(currentContext, assistantMessage, toolCall, confi
|
|
|
772
756
|
context: currentContext,
|
|
773
757
|
}, signal);
|
|
774
758
|
if (signal?.aborted) {
|
|
775
|
-
return createAbortedToolCallOutcome(validationEvent
|
|
759
|
+
return createAbortedToolCallOutcome(validationEvent);
|
|
776
760
|
}
|
|
777
761
|
if (beforeResult?.block) {
|
|
778
762
|
const reason = beforeResult.reason || "Tool execution was blocked";
|
|
@@ -784,20 +768,18 @@ async function prepareToolCall(currentContext, assistantMessage, toolCall, confi
|
|
|
784
768
|
failureCode: "blocked",
|
|
785
769
|
correction: "Choose an allowed approach or request the required authority before retrying.",
|
|
786
770
|
diagnostic: reason,
|
|
787
|
-
...(executionGateReservation ? { executionGateReservation } : {}),
|
|
788
771
|
validationEvent,
|
|
789
772
|
};
|
|
790
773
|
}
|
|
791
774
|
}
|
|
792
775
|
if (signal?.aborted) {
|
|
793
|
-
return createAbortedToolCallOutcome(validationEvent
|
|
776
|
+
return createAbortedToolCallOutcome(validationEvent);
|
|
794
777
|
}
|
|
795
778
|
return {
|
|
796
779
|
kind: "prepared",
|
|
797
780
|
toolCall,
|
|
798
781
|
tool,
|
|
799
782
|
args: validatedArgs,
|
|
800
|
-
...(executionGateReservation ? { executionGateReservation } : {}),
|
|
801
783
|
validationEvent,
|
|
802
784
|
};
|
|
803
785
|
}
|
|
@@ -817,12 +799,11 @@ async function prepareToolCall(currentContext, assistantMessage, toolCall, confi
|
|
|
817
799
|
? validationFailureCorrection(validationEvent, toolCall.name)
|
|
818
800
|
: toolFailureCorrection(message, "rejected", "preflight"),
|
|
819
801
|
diagnostic: isToolArgumentValidationError(error) ? undefined : message,
|
|
820
|
-
...(executionGateReservation ? { executionGateReservation } : {}),
|
|
821
802
|
validationEvent,
|
|
822
803
|
};
|
|
823
804
|
}
|
|
824
805
|
}
|
|
825
|
-
function finalizeRejectedToolCall(toolCall, tool, outcome, tracker
|
|
806
|
+
function finalizeRejectedToolCall(toolCall, tool, outcome, tracker) {
|
|
826
807
|
if (outcome.repeatedToolFailure) {
|
|
827
808
|
return {
|
|
828
809
|
toolCall,
|
|
@@ -830,17 +811,13 @@ function finalizeRejectedToolCall(toolCall, tool, outcome, tracker, toolFailureR
|
|
|
830
811
|
isError: true,
|
|
831
812
|
};
|
|
832
813
|
}
|
|
833
|
-
const record = rememberToolFailure(tracker, toolCall.name, toolCall.arguments, "rejected", outcome.failureCode, outcome.correction, outcome.diagnostic, outcome.phase
|
|
834
|
-
|
|
835
|
-
|
|
836
|
-
|
|
837
|
-
|
|
838
|
-
args: toolCall.arguments,
|
|
839
|
-
...(outcome.executionGateReservation ? { reservation: outcome.executionGateReservation } : {}),
|
|
814
|
+
const record = rememberToolFailure(tracker, toolCall.name, toolCall.arguments, "rejected", outcome.failureCode, outcome.correction, outcome.diagnostic, outcome.phase, undefined, {
|
|
815
|
+
output: outcome.result.content
|
|
816
|
+
.filter((block) => block.type === "text")
|
|
817
|
+
.map((block) => block.text)
|
|
818
|
+
.join("\n"),
|
|
840
819
|
});
|
|
841
|
-
const failureResult =
|
|
842
|
-
? createToolFailureRecoveryExhaustedResult(halt.record, halt.diagnostic)
|
|
843
|
-
: createToolFailureResult(record, outcome.result.terminate);
|
|
820
|
+
const failureResult = createToolFailureResult(record, outcome.result.terminate);
|
|
844
821
|
return {
|
|
845
822
|
toolCall,
|
|
846
823
|
result: outcome.repeatedSuccessfulCall
|
|
@@ -853,6 +830,12 @@ function finalizeRejectedToolCall(toolCall, tool, outcome, tracker, toolFailureR
|
|
|
853
830
|
}
|
|
854
831
|
: failureResult,
|
|
855
832
|
isError: true,
|
|
833
|
+
executionGateEffect: {
|
|
834
|
+
kind: "unproductive",
|
|
835
|
+
...(tool ? { tool } : {}),
|
|
836
|
+
record,
|
|
837
|
+
args: toolCall.arguments,
|
|
838
|
+
},
|
|
856
839
|
};
|
|
857
840
|
}
|
|
858
841
|
function createLinkedToolAbort(foregroundSignal) {
|
|
@@ -936,7 +919,13 @@ async function executeAndFinalizePreparedToolCall(currentContext, assistantMessa
|
|
|
936
919
|
if (executionAbort.signal.aborted)
|
|
937
920
|
return completion;
|
|
938
921
|
let handoffAccepted = false;
|
|
939
|
-
|
|
922
|
+
// A handed-off call leaves the batch, so the batch never applies its effect: the background
|
|
923
|
+
// completion is the only place left that can tell the governor what this operation did.
|
|
924
|
+
const handedOffCompletion = completion.then((finalized) => {
|
|
925
|
+
if (handoffAccepted)
|
|
926
|
+
toolFailureRecoveryGate.apply(finalized.executionGateEffect);
|
|
927
|
+
return finalized;
|
|
928
|
+
});
|
|
940
929
|
void handedOffCompletion.catch(() => undefined);
|
|
941
930
|
let handoff;
|
|
942
931
|
try {
|
|
@@ -983,17 +972,22 @@ async function executePreparedToolCall(prepared, signal, emit) {
|
|
|
983
972
|
// throwing. Keep the returned result intact through afterToolCall so
|
|
984
973
|
// policy hooks can inspect its bounded diagnostics and metadata.
|
|
985
974
|
isError: result.isError === true,
|
|
986
|
-
...(result.isError === true
|
|
975
|
+
...(result.isError === true
|
|
976
|
+
? { errorClass: "tool_result_error", errorKind: result.errorKind ?? "tool_failure" }
|
|
977
|
+
: {}),
|
|
987
978
|
};
|
|
988
979
|
}
|
|
989
980
|
catch (error) {
|
|
990
981
|
await Promise.all(updateEvents);
|
|
991
982
|
const message = error instanceof Error ? error.message : String(error);
|
|
983
|
+
const toolFailure = error instanceof AgentToolExecutionError ? error : undefined;
|
|
992
984
|
return {
|
|
993
985
|
result: createErrorToolResult(message),
|
|
994
986
|
isError: true,
|
|
995
987
|
errorClass: error instanceof Error ? error.name : typeof error,
|
|
996
988
|
failureMessage: message,
|
|
989
|
+
errorKind: toolFailure?.errorKind ?? "tool_failure",
|
|
990
|
+
...(toolFailure ? { failureCode: toolFailure.failureCode, outputSignature: toolFailure.outputSignature } : {}),
|
|
997
991
|
};
|
|
998
992
|
}
|
|
999
993
|
}
|
|
@@ -1026,6 +1020,9 @@ async function finalizeExecutedToolCall(currentContext, assistantMessage, prepar
|
|
|
1026
1020
|
let isError = executed.isError;
|
|
1027
1021
|
let failureMessage = executed.failureMessage ?? "";
|
|
1028
1022
|
let errorClass = executed.errorClass;
|
|
1023
|
+
let failureCode = executed.failureCode;
|
|
1024
|
+
let outputSignature = executed.outputSignature;
|
|
1025
|
+
let errorKind = executed.errorKind;
|
|
1029
1026
|
let executionGateEffect;
|
|
1030
1027
|
if (config.afterToolCall) {
|
|
1031
1028
|
try {
|
|
@@ -1048,30 +1045,55 @@ async function finalizeExecutedToolCall(currentContext, assistantMessage, prepar
|
|
|
1048
1045
|
}
|
|
1049
1046
|
}
|
|
1050
1047
|
catch (error) {
|
|
1048
|
+
// The hook itself failed, so nothing about the tool's own completed operation survives.
|
|
1051
1049
|
failureMessage = error instanceof Error ? error.message : String(error);
|
|
1052
1050
|
errorClass = error instanceof Error ? error.name : typeof error;
|
|
1051
|
+
failureCode = undefined;
|
|
1052
|
+
outputSignature = undefined;
|
|
1053
|
+
errorKind = "tool_failure";
|
|
1053
1054
|
result = { ...createErrorToolResult(failureMessage), usage: result.usage };
|
|
1054
1055
|
isError = true;
|
|
1055
1056
|
}
|
|
1056
1057
|
}
|
|
1057
1058
|
if (isError) {
|
|
1058
1059
|
const usage = result.usage;
|
|
1060
|
+
const failureOutput = failureMessage ||
|
|
1061
|
+
result.content
|
|
1062
|
+
.filter((block) => block.type === "text")
|
|
1063
|
+
.map((block) => block.text)
|
|
1064
|
+
.join("\n") ||
|
|
1065
|
+
"Tool execution failed";
|
|
1059
1066
|
const effectiveFailureMessage = failureMessage || result.content.find((block) => block.type === "text")?.text || "Tool execution failed";
|
|
1060
1067
|
const assessment = assessToolFailure(effectiveFailureMessage, "failed", errorClass);
|
|
1061
|
-
const
|
|
1062
|
-
|
|
1063
|
-
|
|
1064
|
-
|
|
1065
|
-
|
|
1066
|
-
|
|
1067
|
-
|
|
1068
|
-
|
|
1069
|
-
|
|
1070
|
-
|
|
1071
|
-
|
|
1072
|
-
|
|
1073
|
-
|
|
1074
|
-
|
|
1068
|
+
const effectiveFailureCode = failureCode ?? assessment.failureCode;
|
|
1069
|
+
if (errorKind === "operation_outcome") {
|
|
1070
|
+
// The tool ran the operation to completion; the non-zero status is the observation the
|
|
1071
|
+
// agent asked for. Nothing here is a mistake, so no failure record is remembered and the
|
|
1072
|
+
// tool's own output stands exactly as written. The governor still notes that repeating
|
|
1073
|
+
// this identical operation cannot say anything new until something else changes.
|
|
1074
|
+
clearToolFailure(toolFailureMemory, prepared.toolCall.name, prepared.args);
|
|
1075
|
+
result = { ...result, errorKind: "operation_outcome", usage };
|
|
1076
|
+
executionGateEffect = {
|
|
1077
|
+
kind: "unproductive",
|
|
1078
|
+
tool: prepared.tool,
|
|
1079
|
+
record: describeOperationOutcome(prepared.toolCall.name, prepared.args, effectiveFailureCode, assessment.diagnostic),
|
|
1080
|
+
args: prepared.args,
|
|
1081
|
+
};
|
|
1082
|
+
}
|
|
1083
|
+
else {
|
|
1084
|
+
const recoveryPlan = toolFailureRecoveryGate.planFailure(prepared.tool, prepared.args, { failureCode: effectiveFailureCode, message: effectiveFailureMessage }, currentContext.tools ?? []);
|
|
1085
|
+
const correction = assessment.policyGuidance
|
|
1086
|
+
? `${assessment.policyGuidance} ${recoveryPlan.guidance}`
|
|
1087
|
+
: recoveryPlan.guidance;
|
|
1088
|
+
const record = rememberToolFailure(toolFailureMemory, prepared.toolCall.name, prepared.args, "failed", effectiveFailureCode, correction, assessment.diagnostic, assessment.phase, recoveryPlan.evidence ?? assessment.evidence, { output: failureOutput, outputSignature });
|
|
1089
|
+
executionGateEffect = {
|
|
1090
|
+
kind: "unproductive",
|
|
1091
|
+
tool: prepared.tool,
|
|
1092
|
+
record,
|
|
1093
|
+
args: prepared.args,
|
|
1094
|
+
};
|
|
1095
|
+
result = { ...createToolFailureResult(record, result.terminate), usage };
|
|
1096
|
+
}
|
|
1075
1097
|
}
|
|
1076
1098
|
else {
|
|
1077
1099
|
clearToolFailure(toolFailureMemory, prepared.toolCall.name, prepared.args);
|
|
@@ -1080,7 +1102,6 @@ async function finalizeExecutedToolCall(currentContext, assistantMessage, prepar
|
|
|
1080
1102
|
kind: "success",
|
|
1081
1103
|
tool: prepared.tool,
|
|
1082
1104
|
args: prepared.args,
|
|
1083
|
-
evidenceResult: executed.result,
|
|
1084
1105
|
};
|
|
1085
1106
|
}
|
|
1086
1107
|
}
|
|
@@ -1138,6 +1159,9 @@ function createToolResultMessage(finalized) {
|
|
|
1138
1159
|
details: finalized.result.details,
|
|
1139
1160
|
usage: finalized.result.usage,
|
|
1140
1161
|
isError: finalized.isError,
|
|
1162
|
+
// Persisted so a reloaded transcript still separates a completed operation's own status from a
|
|
1163
|
+
// tool that could not run at all.
|
|
1164
|
+
...(finalized.isError && finalized.result.errorKind ? { errorKind: finalized.result.errorKind } : {}),
|
|
1141
1165
|
timestamp: Date.now(),
|
|
1142
1166
|
};
|
|
1143
1167
|
}
|