@buoy-gg/agent-core 7.0.36 → 7.0.39
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +13 -9
- package/lib/commonjs/catalog/catalog.g.js +93 -6
- package/lib/commonjs/catalog/catalog.g.js.map +1 -1
- package/lib/commonjs/catalog/catalog.source.json +106 -4
- package/lib/commonjs/catalog/catalog.types.g.js +3 -3
- package/lib/commonjs/catalog/catalog.types.g.js.map +1 -1
- package/lib/commonjs/catalog/toProviderTools.js +3 -1
- package/lib/commonjs/catalog/toProviderTools.js.map +1 -1
- package/lib/commonjs/effects/ledger.js +13 -1
- package/lib/commonjs/effects/ledger.js.map +1 -1
- package/lib/commonjs/engine/evidence.js +113 -0
- package/lib/commonjs/engine/evidence.js.map +1 -0
- package/lib/commonjs/engine/historyBudget.js +158 -55
- package/lib/commonjs/engine/historyBudget.js.map +1 -1
- package/lib/commonjs/engine/retrieve.js +214 -0
- package/lib/commonjs/engine/retrieve.js.map +1 -0
- package/lib/commonjs/engine/runAgentTurn.js +364 -94
- package/lib/commonjs/engine/runAgentTurn.js.map +1 -1
- package/lib/commonjs/engine/systemPrompt.js +31 -1
- package/lib/commonjs/engine/systemPrompt.js.map +1 -1
- package/lib/commonjs/engine/tokenCalibration.js +81 -0
- package/lib/commonjs/engine/tokenCalibration.js.map +1 -0
- package/lib/commonjs/engine/verify.js +300 -0
- package/lib/commonjs/engine/verify.js.map +1 -0
- package/lib/commonjs/index.js +16 -0
- package/lib/commonjs/index.js.map +1 -1
- package/lib/commonjs/providers/anthropic.js +12 -3
- package/lib/commonjs/providers/anthropic.js.map +1 -1
- package/lib/commonjs/providers/openai.js +11 -3
- package/lib/commonjs/providers/openai.js.map +1 -1
- package/lib/commonjs/providers/problem.js +98 -0
- package/lib/commonjs/providers/problem.js.map +1 -0
- package/lib/commonjs/providers/sse.js +6 -5
- package/lib/commonjs/providers/sse.js.map +1 -1
- package/lib/commonjs/providers/transport.js +20 -4
- package/lib/commonjs/providers/transport.js.map +1 -1
- package/lib/commonjs/providers/xhrStream.js +3 -0
- package/lib/commonjs/providers/xhrStream.js.map +1 -1
- package/lib/commonjs/session.js +32 -5
- package/lib/commonjs/session.js.map +1 -1
- package/lib/module/catalog/catalog.g.js +93 -6
- package/lib/module/catalog/catalog.g.js.map +1 -1
- package/lib/module/catalog/catalog.source.json +106 -4
- package/lib/module/catalog/catalog.types.g.js +3 -3
- package/lib/module/catalog/catalog.types.g.js.map +1 -1
- package/lib/module/catalog/toProviderTools.js +3 -1
- package/lib/module/catalog/toProviderTools.js.map +1 -1
- package/lib/module/effects/ledger.js +13 -1
- package/lib/module/effects/ledger.js.map +1 -1
- package/lib/module/engine/evidence.js +107 -0
- package/lib/module/engine/evidence.js.map +1 -0
- package/lib/module/engine/historyBudget.js +158 -54
- package/lib/module/engine/historyBudget.js.map +1 -1
- package/lib/module/engine/retrieve.js +210 -0
- package/lib/module/engine/retrieve.js.map +1 -0
- package/lib/module/engine/runAgentTurn.js +365 -96
- package/lib/module/engine/runAgentTurn.js.map +1 -1
- package/lib/module/engine/systemPrompt.js +30 -1
- package/lib/module/engine/systemPrompt.js.map +1 -1
- package/lib/module/engine/tokenCalibration.js +75 -0
- package/lib/module/engine/tokenCalibration.js.map +1 -0
- package/lib/module/engine/verify.js +292 -0
- package/lib/module/engine/verify.js.map +1 -0
- package/lib/module/index.js +2 -0
- package/lib/module/index.js.map +1 -1
- package/lib/module/providers/anthropic.js +12 -3
- package/lib/module/providers/anthropic.js.map +1 -1
- package/lib/module/providers/openai.js +11 -3
- package/lib/module/providers/openai.js.map +1 -1
- package/lib/module/providers/problem.js +92 -0
- package/lib/module/providers/problem.js.map +1 -0
- package/lib/module/providers/sse.js +6 -5
- package/lib/module/providers/sse.js.map +1 -1
- package/lib/module/providers/transport.js +20 -4
- package/lib/module/providers/transport.js.map +1 -1
- package/lib/module/providers/xhrStream.js +3 -0
- package/lib/module/providers/xhrStream.js.map +1 -1
- package/lib/module/session.js +33 -6
- package/lib/module/session.js.map +1 -1
- package/lib/typescript/catalog/catalog.g.d.ts +2 -2
- package/lib/typescript/catalog/catalog.g.d.ts.map +1 -1
- package/lib/typescript/catalog/catalog.types.g.d.ts +3 -3
- package/lib/typescript/catalog/catalog.types.g.d.ts.map +1 -1
- package/lib/typescript/catalog/toProviderTools.d.ts.map +1 -1
- package/lib/typescript/effects/ledger.d.ts +18 -0
- package/lib/typescript/effects/ledger.d.ts.map +1 -1
- package/lib/typescript/engine/evidence.d.ts +67 -0
- package/lib/typescript/engine/evidence.d.ts.map +1 -0
- package/lib/typescript/engine/historyBudget.d.ts +31 -8
- package/lib/typescript/engine/historyBudget.d.ts.map +1 -1
- package/lib/typescript/engine/retrieve.d.ts +32 -0
- package/lib/typescript/engine/retrieve.d.ts.map +1 -0
- package/lib/typescript/engine/runAgentTurn.d.ts +96 -2
- package/lib/typescript/engine/runAgentTurn.d.ts.map +1 -1
- package/lib/typescript/engine/systemPrompt.d.ts +39 -0
- package/lib/typescript/engine/systemPrompt.d.ts.map +1 -1
- package/lib/typescript/engine/tokenCalibration.d.ts +46 -0
- package/lib/typescript/engine/tokenCalibration.d.ts.map +1 -0
- package/lib/typescript/engine/verify.d.ts +67 -0
- package/lib/typescript/engine/verify.d.ts.map +1 -0
- package/lib/typescript/index.d.ts +4 -2
- package/lib/typescript/index.d.ts.map +1 -1
- package/lib/typescript/providers/anthropic.d.ts.map +1 -1
- package/lib/typescript/providers/openai.d.ts.map +1 -1
- package/lib/typescript/providers/problem.d.ts +50 -0
- package/lib/typescript/providers/problem.d.ts.map +1 -0
- package/lib/typescript/providers/sse.d.ts +4 -1
- package/lib/typescript/providers/sse.d.ts.map +1 -1
- package/lib/typescript/providers/transport.d.ts +8 -0
- package/lib/typescript/providers/transport.d.ts.map +1 -1
- package/lib/typescript/providers/types.d.ts +14 -0
- package/lib/typescript/providers/types.d.ts.map +1 -1
- package/lib/typescript/providers/xhrStream.d.ts +2 -0
- package/lib/typescript/providers/xhrStream.d.ts.map +1 -1
- package/lib/typescript/session.d.ts +15 -2
- package/lib/typescript/session.d.ts.map +1 -1
- package/lib/typescript/types.d.ts +6 -0
- package/lib/typescript/types.d.ts.map +1 -1
- package/package.json +1 -1
|
@@ -35,8 +35,11 @@ import { projectReceipt } from "../blocks/receipts";
|
|
|
35
35
|
import { isAskBlock } from "../blocks/types";
|
|
36
36
|
import { ACQUISITION_CALLS, synthesizeAskBlock, synthesizeShortfallBlock } from "./askGate";
|
|
37
37
|
import { couldBeToolCallEnvelope, parseTextToolCalls, SALVAGE_NOTE } from "./textToolCalls";
|
|
38
|
-
import { budgetForRequest } from "./historyBudget";
|
|
39
|
-
|
|
38
|
+
import { budgetForRequest, MAX_HISTORY_TOKENS, messageSize } from "./historyBudget";
|
|
39
|
+
import { promptTokensOf, TokenCalibration } from "./tokenCalibration";
|
|
40
|
+
import { truncationMarker } from "./evidence";
|
|
41
|
+
import { retrieveEvidence } from "./retrieve";
|
|
42
|
+
import { verificationTrailer, verifyOutcome } from "./verify";
|
|
40
43
|
/** Beyond this a tool result costs more in tokens than it can possibly be worth. */
|
|
41
44
|
const MAX_RESULT_CHARS = 24_000;
|
|
42
45
|
/**
|
|
@@ -71,6 +74,73 @@ function callSignature(calls) {
|
|
|
71
74
|
const LOOP_WARNING = "[Buoy] You have now made this exact call three times in a row and gotten the same thing back. Repeating it again will not change the answer. Read the result you already have, then either try a different tool or different arguments, or tell the user plainly what you could not get and what you tried.";
|
|
72
75
|
const DEFAULT_MAX_STEPS = 12;
|
|
73
76
|
const DEFAULT_TURN_MS = 180_000;
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* Retrying a model request — narrowly.
|
|
80
|
+
*
|
|
81
|
+
* A 429 from a shared gateway, a 529 / `overloaded_error` from Anthropic, or
|
|
82
|
+
* a "prompt is too long" 400 used to end the turn with an error card and Try
|
|
83
|
+
* again — and Try again re-sends the QUESTION, restarting an investigation
|
|
84
|
+
* that may already have written things. All three fail before a model has
|
|
85
|
+
* generated anything, so asking again is safe: nothing runs twice, nothing
|
|
86
|
+
* is billed twice. The rule that makes it safe is enforced, not assumed —
|
|
87
|
+
* a request is only retried when its stream produced NO text, tool call or
|
|
88
|
+
* thinking; once anything has come back, the existing no-replay handling
|
|
89
|
+
* stands (see providers/transport.ts on consumed-exactly-once).
|
|
90
|
+
*
|
|
91
|
+
* Throttled: up to two retries, 1 s then 2 s (plus jitter), a `Retry-After`
|
|
92
|
+
* winning when the endpoint sent one. Overflow: one retry, with the history
|
|
93
|
+
* ceiling halved first — the proactive budget is chars ÷ 4 and the provider
|
|
94
|
+
* just said that was optimistic. Both wait inside the turn deadline and stop
|
|
95
|
+
* on the user's Stop. Small numbers on purpose: this is a phone, not a
|
|
96
|
+
* server queue.
|
|
97
|
+
*/
|
|
98
|
+
const THROTTLE_RETRIES = 2;
|
|
99
|
+
const OVERFLOW_RETRIES = 1;
|
|
100
|
+
const RETRY_BASE_MS = 1_000;
|
|
101
|
+
const RETRY_JITTER_MS = 250;
|
|
102
|
+
const RETRY_AFTER_CAP_MS = 30_000;
|
|
103
|
+
function sleep(ms, signal) {
|
|
104
|
+
return new Promise(resolve => {
|
|
105
|
+
if (signal?.aborted) return resolve();
|
|
106
|
+
const t = setTimeout(done, ms);
|
|
107
|
+
function done() {
|
|
108
|
+
clearTimeout(t);
|
|
109
|
+
signal?.removeEventListener("abort", done);
|
|
110
|
+
resolve();
|
|
111
|
+
}
|
|
112
|
+
signal?.addEventListener("abort", done, {
|
|
113
|
+
once: true
|
|
114
|
+
});
|
|
115
|
+
});
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/**
|
|
119
|
+
* Why the turn ended — the machine-readable half of the notices. The text in
|
|
120
|
+
* the notice blocks is unchanged (the eval classifier pins it); this is for a
|
|
121
|
+
* host, the bank and the desktop, which used to have to grep the prose.
|
|
122
|
+
*/
|
|
123
|
+
|
|
124
|
+
/** Normalise an answer to its parts, so both call sites read one shape. */
|
|
125
|
+
function readAnswer(answer) {
|
|
126
|
+
if (typeof answer === "boolean") return {
|
|
127
|
+
approved: answer,
|
|
128
|
+
trust: false
|
|
129
|
+
};
|
|
130
|
+
const reason = answer.reason?.trim();
|
|
131
|
+
return {
|
|
132
|
+
approved: answer.approved,
|
|
133
|
+
...(reason ? {
|
|
134
|
+
reason
|
|
135
|
+
} : {}),
|
|
136
|
+
trust: answer.trust === true && answer.approved
|
|
137
|
+
};
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/** What the model is told when the user says no — with their note, when they left one. */
|
|
141
|
+
function declinedResult(reason) {
|
|
142
|
+
return reason ? `The user declined this change and said: "${reason}". Do not retry it as proposed; take their note into account and, if a different change would fit, propose that instead.` : "The user declined this change. Do not retry it; ask what they would prefer.";
|
|
143
|
+
}
|
|
74
144
|
function findDescriptor(catalog, toolId, action) {
|
|
75
145
|
return catalog.find(t => t.toolId === toolId)?.actions.find(a => a.action === action);
|
|
76
146
|
}
|
|
@@ -85,15 +155,20 @@ function findDescriptor(catalog, toolId, action) {
|
|
|
85
155
|
*/
|
|
86
156
|
const URL_PROVENANCE_REJECTION = "Rejected: every image URL must be one you read from a tool result in THIS conversation (images.list, a storage value, a response body). Never invent or remember URLs — read them first, then show them.";
|
|
87
157
|
const droppedImageNote = n => ` NOTE: ${n} image ${n === 1 ? "URL was" : "URLs were"} dropped — they were not read from a tool result in this conversation, so the card the user is looking at has no picture there. Do not tell them it does; read the URLs and show it again, or say the artwork is missing.`;
|
|
88
|
-
|
|
89
|
-
|
|
158
|
+
|
|
159
|
+
/** The whole result as text — what the evidence store keeps. */
|
|
160
|
+
function fullText(value) {
|
|
90
161
|
try {
|
|
91
|
-
|
|
162
|
+
return JSON.stringify(value ?? null);
|
|
92
163
|
} catch {
|
|
93
|
-
|
|
164
|
+
return String(value);
|
|
94
165
|
}
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/** What the model is sent: the text, cut at the cap with a marker that names where the rest is. */
|
|
169
|
+
function encodeResult(text, ref) {
|
|
95
170
|
if (text.length <= MAX_RESULT_CHARS) return text;
|
|
96
|
-
return `${text.slice(0, MAX_RESULT_CHARS)}
|
|
171
|
+
return `${text.slice(0, MAX_RESULT_CHARS)}${truncationMarker(text.length, ref)}`;
|
|
97
172
|
}
|
|
98
173
|
|
|
99
174
|
/** A short line for the transcript row, so the UI never renders raw payloads. */
|
|
@@ -162,8 +237,14 @@ export async function* runAgentTurn(input) {
|
|
|
162
237
|
policy,
|
|
163
238
|
isRelease,
|
|
164
239
|
requestApproval,
|
|
165
|
-
signal
|
|
240
|
+
signal,
|
|
241
|
+
evidence,
|
|
242
|
+
trusted,
|
|
243
|
+
procedures
|
|
166
244
|
} = input;
|
|
245
|
+
/** Notes the user left with a decline, by call id — see declinedResult. */
|
|
246
|
+
const declineReasons = new Map();
|
|
247
|
+
const calibration = input.calibration ?? new TokenCalibration();
|
|
167
248
|
const messages = [...input.messages];
|
|
168
249
|
const tools = toProviderTools(catalog, {
|
|
169
250
|
availableToolIds: input.availableToolIds,
|
|
@@ -187,7 +268,9 @@ export async function* runAgentTurn(input) {
|
|
|
187
268
|
* much to leave out of a budget. `maxTokens` is multiplied by four because
|
|
188
269
|
* the budget is in characters.
|
|
189
270
|
*/
|
|
190
|
-
const
|
|
271
|
+
const fixedChars = system.length + (systemVolatile?.length ?? 0) + tools.reduce((n, t) => n + t.description.length + JSON.stringify(t.inputSchema).length, 0);
|
|
272
|
+
/** The reserve at the ratio believed RIGHT NOW — it moves as usage reports come in. */
|
|
273
|
+
const requestReserve = () => fixedChars + calibration.chars(maxTokens);
|
|
191
274
|
/**
|
|
192
275
|
* Caps the AGENT's wall-clock, not the user's. Time spent parked on an
|
|
193
276
|
* approval sheet is pushed onto the deadline as it is spent (see the
|
|
@@ -196,6 +279,13 @@ export async function* runAgentTurn(input) {
|
|
|
196
279
|
* for the turn that applied it.
|
|
197
280
|
*/
|
|
198
281
|
let deadline = Date.now() + DEFAULT_TURN_MS;
|
|
282
|
+
/** Retries spent this turn, per kind — the budget is per turn, not per step. */
|
|
283
|
+
const retries = {
|
|
284
|
+
throttled: 0,
|
|
285
|
+
overflow: 0
|
|
286
|
+
};
|
|
287
|
+
/** Tightened after an overflow; see budgetForRequest's `cap`. In tokens, so a calibration change re-scales it. */
|
|
288
|
+
let historyCapTokens = MAX_HISTORY_TOKENS;
|
|
199
289
|
|
|
200
290
|
// Every http(s) URL that appeared in a tool result in this CONVERSATION.
|
|
201
291
|
// The prompt tells the model image URLs must come from here; this Set is
|
|
@@ -241,7 +331,8 @@ export async function* runAgentTurn(input) {
|
|
|
241
331
|
for (let step = 0; step < maxSteps; step++) {
|
|
242
332
|
if (signal?.aborted) {
|
|
243
333
|
yield {
|
|
244
|
-
type: "done"
|
|
334
|
+
type: "done",
|
|
335
|
+
stopReason: "stopped"
|
|
245
336
|
};
|
|
246
337
|
return messages;
|
|
247
338
|
}
|
|
@@ -268,7 +359,8 @@ export async function* runAgentTurn(input) {
|
|
|
268
359
|
}
|
|
269
360
|
};
|
|
270
361
|
yield {
|
|
271
|
-
type: "done"
|
|
362
|
+
type: "done",
|
|
363
|
+
stopReason: "time-cap"
|
|
272
364
|
};
|
|
273
365
|
return messages;
|
|
274
366
|
}
|
|
@@ -286,13 +378,14 @@ export async function* runAgentTurn(input) {
|
|
|
286
378
|
*
|
|
287
379
|
* The current round is never touched: it is the question being answered.
|
|
288
380
|
*/
|
|
289
|
-
const budgeted = budgetForRequest(messages, requestReserve);
|
|
381
|
+
const budgeted = budgetForRequest(messages, requestReserve(), calibration.chars(historyCapTokens), calibration.charsPerToken, ledger.liveCallIds());
|
|
290
382
|
if (budgeted.droppedRounds > 0) {
|
|
291
383
|
messages.length = 0;
|
|
292
384
|
messages.push(...budgeted.messages);
|
|
293
385
|
yield {
|
|
294
386
|
type: "history-trimmed",
|
|
295
|
-
droppedRounds: budgeted.droppedRounds
|
|
387
|
+
droppedRounds: budgeted.droppedRounds,
|
|
388
|
+
droppedPinned: budgeted.droppedPinned
|
|
296
389
|
};
|
|
297
390
|
} else if (budgeted.messages !== messages) {
|
|
298
391
|
messages.length = 0;
|
|
@@ -318,7 +411,7 @@ export async function* runAgentTurn(input) {
|
|
|
318
411
|
delta
|
|
319
412
|
};
|
|
320
413
|
};
|
|
321
|
-
|
|
414
|
+
let calls = [];
|
|
322
415
|
/**
|
|
323
416
|
* Recovered from text rather than emitted as calls — see textToolCalls.ts.
|
|
324
417
|
* Tracked by id so each one's result can tell the model to stop doing that;
|
|
@@ -335,7 +428,7 @@ export async function* runAgentTurn(input) {
|
|
|
335
428
|
let holding = true;
|
|
336
429
|
// Carried, never read: Anthropic requires the turn's thinking blocks back
|
|
337
430
|
// verbatim with its tool results. See providers/anthropic.ts note 5.
|
|
338
|
-
|
|
431
|
+
let thinking = [];
|
|
339
432
|
let failed = false;
|
|
340
433
|
/**
|
|
341
434
|
* What the provider said about how the stream ended.
|
|
@@ -348,80 +441,166 @@ export async function* runAgentTurn(input) {
|
|
|
348
441
|
let outcome;
|
|
349
442
|
/** Set for the two kinds that are recoverable rather than a hard failure. */
|
|
350
443
|
let incomplete;
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
}
|
|
444
|
+
|
|
445
|
+
/**
|
|
446
|
+
* THE REQUEST, with its retries. One request per pass; a pass that fails
|
|
447
|
+
* before the model generated anything may be sent again (see
|
|
448
|
+
* THROTTLE_RETRIES). Everything the pass accumulated is reset first —
|
|
449
|
+
* there is nothing to keep, by the rule that made the retry safe.
|
|
450
|
+
*/
|
|
451
|
+
for (;;) {
|
|
452
|
+
text = "";
|
|
453
|
+
calls = [];
|
|
454
|
+
held = "";
|
|
455
|
+
holding = true;
|
|
456
|
+
thinking = [];
|
|
457
|
+
failed = false;
|
|
458
|
+
outcome = undefined;
|
|
459
|
+
incomplete = undefined;
|
|
460
|
+
/** The failure that ended this pass, when it is one the engine may retry. */
|
|
461
|
+
let retryable;
|
|
462
|
+
/** What this pass sends, in characters — the numerator of the calibration sample. */
|
|
463
|
+
const sentChars = fixedChars + messages.reduce((n, m) => n + messageSize(m), 0);
|
|
464
|
+
for await (const ev of provider.send({
|
|
465
|
+
messages,
|
|
466
|
+
system,
|
|
467
|
+
systemVolatile,
|
|
468
|
+
tools,
|
|
469
|
+
model,
|
|
470
|
+
maxTokens,
|
|
471
|
+
signal
|
|
472
|
+
})) {
|
|
473
|
+
if (ev.type === "text") {
|
|
474
|
+
text += ev.delta;
|
|
475
|
+
if (!holding) {
|
|
476
|
+
yield textEvent(ev.delta);
|
|
477
|
+
} else if (couldBeToolCallEnvelope(text)) {
|
|
478
|
+
held += ev.delta;
|
|
479
|
+
} else {
|
|
480
|
+
// Not an envelope after all. Release everything at once and stream
|
|
481
|
+
// the rest as usual — the reader loses nothing but a few characters
|
|
482
|
+
// of latency at the very start of the answer.
|
|
483
|
+
holding = false;
|
|
484
|
+
held = "";
|
|
485
|
+
yield textEvent(text);
|
|
486
|
+
}
|
|
487
|
+
} else if (ev.type === "tool-call") {
|
|
488
|
+
calls.push(ev.call);
|
|
489
|
+
} else if (ev.type === "thinking") {
|
|
490
|
+
thinking.push(ev.block);
|
|
491
|
+
// Surfaced as well as carried: the block goes back to the provider
|
|
492
|
+
// verbatim (note above), and the readable half goes to the UI so a
|
|
493
|
+
// tester can see WHY a turn did what it did. Redacted blocks have no
|
|
494
|
+
// readable half and are carried only.
|
|
495
|
+
if (ev.block.type === "thinking" && ev.block.thinking.trim()) {
|
|
496
|
+
yield {
|
|
497
|
+
type: "reasoning",
|
|
498
|
+
text: ev.block.thinking
|
|
499
|
+
};
|
|
500
|
+
}
|
|
501
|
+
} else if (ev.type === "error") {
|
|
502
|
+
if (ev.kind === "truncated" || ev.kind === "timeout") {
|
|
503
|
+
// Not a failure the user did anything about, and not one Try again
|
|
504
|
+
// can fix — re-sending the QUESTION restarts an investigation that
|
|
505
|
+
// may already have written things. Held back from the error card
|
|
506
|
+
// (which is what draws Try again) and handled below as a recoverable
|
|
507
|
+
// stop with a Continue button.
|
|
508
|
+
incomplete = {
|
|
509
|
+
message: ev.message
|
|
510
|
+
};
|
|
511
|
+
} else if (ev.problem && (ev.problem.kind === "throttled" || ev.problem.kind === "overflow") && text === "" && calls.length === 0 && thinking.length === 0) {
|
|
512
|
+
// Nothing generated, and a kind that a wait or a trim can fix.
|
|
513
|
+
// Decided after the stream closes; the adapter returns on error.
|
|
514
|
+
retryable = {
|
|
515
|
+
problem: ev.problem,
|
|
516
|
+
message: ev.message
|
|
517
|
+
};
|
|
518
|
+
} else {
|
|
519
|
+
// A mid-stream error can arrive on an already-committed 200.
|
|
520
|
+
yield {
|
|
521
|
+
type: "error",
|
|
522
|
+
message: ev.message
|
|
523
|
+
};
|
|
524
|
+
failed = true;
|
|
525
|
+
}
|
|
526
|
+
} else if (ev.type === "done") {
|
|
527
|
+
outcome = ev.outcome;
|
|
528
|
+
if (ev.usage) {
|
|
529
|
+
// The provider just counted this prompt. One sample per request
|
|
530
|
+
// keeps the chars↔tokens ratio honest for the NEXT budget.
|
|
531
|
+
calibration.observe(sentChars, promptTokensOf(ev.usage, provider.protocol));
|
|
532
|
+
// Surfaced so a host can meter cost per seat. Parsed for a long time;
|
|
533
|
+
// went nowhere anyone could see.
|
|
534
|
+
yield {
|
|
535
|
+
type: "usage",
|
|
536
|
+
...ev.usage,
|
|
537
|
+
model: ev.model
|
|
538
|
+
};
|
|
539
|
+
}
|
|
387
540
|
}
|
|
388
|
-
}
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
541
|
+
}
|
|
542
|
+
if (!retryable) break;
|
|
543
|
+
const kind = retryable.problem.kind;
|
|
544
|
+
const maxAttempts = 1 + (kind === "throttled" ? THROTTLE_RETRIES : OVERFLOW_RETRIES);
|
|
545
|
+
const spent = retries[kind];
|
|
546
|
+
let waitMs = kind === "throttled" ? Math.min(retryable.problem.retryAfterMs ?? RETRY_BASE_MS * 2 ** spent, RETRY_AFTER_CAP_MS) + Math.floor(Math.random() * RETRY_JITTER_MS) : 0;
|
|
547
|
+
let giveUp = spent >= maxAttempts - 1 || signal?.aborted === true || Date.now() + waitMs > deadline;
|
|
548
|
+
if (!giveUp && kind === "overflow") {
|
|
549
|
+
// Halve the ceiling and trim again. If that changes nothing — the
|
|
550
|
+
// current round alone is over the limit — a retry would only repeat
|
|
551
|
+
// the refusal, so report it instead.
|
|
552
|
+
historyCapTokens = Math.floor(historyCapTokens / 2);
|
|
553
|
+
const before = messages.reduce((n, m) => n + messageSize(m), 0);
|
|
554
|
+
const again = budgetForRequest(messages, requestReserve(), calibration.chars(historyCapTokens), calibration.charsPerToken, ledger.liveCallIds());
|
|
555
|
+
const after = again.messages.reduce((n, m) => n + messageSize(m), 0);
|
|
556
|
+
if (after >= before) {
|
|
557
|
+
giveUp = true;
|
|
398
558
|
} else {
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
}
|
|
406
|
-
} else if (ev.type === "done") {
|
|
407
|
-
outcome = ev.outcome;
|
|
408
|
-
if (ev.usage) {
|
|
409
|
-
// Surfaced so a host can meter cost per seat. Parsed for a long time;
|
|
410
|
-
// went nowhere anyone could see.
|
|
411
|
-
yield {
|
|
412
|
-
type: "usage",
|
|
413
|
-
...ev.usage,
|
|
414
|
-
model: ev.model
|
|
559
|
+
messages.length = 0;
|
|
560
|
+
messages.push(...again.messages);
|
|
561
|
+
if (again.droppedRounds > 0) yield {
|
|
562
|
+
type: "history-trimmed",
|
|
563
|
+
droppedRounds: again.droppedRounds,
|
|
564
|
+
droppedPinned: again.droppedPinned
|
|
415
565
|
};
|
|
416
566
|
}
|
|
417
567
|
}
|
|
568
|
+
if (giveUp) {
|
|
569
|
+
// Plain words first; the endpoint's own text after, for whoever files
|
|
570
|
+
// the bug. A user reading "prompt is too long: 213000 tokens" has no
|
|
571
|
+
// move; "start a new conversation" is one.
|
|
572
|
+
const plain = kind === "throttled" ? spent > 0 ? `The AI endpoint is rate-limiting requests — tried ${spent + 1} times.` : "The AI endpoint is rate-limiting requests." : "This conversation is too large for the AI endpoint and there was nothing older to trim. Start a new conversation, or ask a shorter question.";
|
|
573
|
+
yield {
|
|
574
|
+
type: "error",
|
|
575
|
+
message: `${plain} (${retryable.message})`
|
|
576
|
+
};
|
|
577
|
+
failed = true;
|
|
578
|
+
break;
|
|
579
|
+
}
|
|
580
|
+
retries[kind] += 1;
|
|
581
|
+
yield {
|
|
582
|
+
type: "retrying",
|
|
583
|
+
reason: kind,
|
|
584
|
+
attempt: retries[kind] + 1,
|
|
585
|
+
maxAttempts,
|
|
586
|
+
inMs: waitMs
|
|
587
|
+
};
|
|
588
|
+
if (waitMs > 0) await sleep(waitMs, signal);
|
|
589
|
+
if (signal?.aborted) {
|
|
590
|
+
yield {
|
|
591
|
+
type: "done",
|
|
592
|
+
stopReason: "stopped"
|
|
593
|
+
};
|
|
594
|
+
return messages;
|
|
595
|
+
}
|
|
418
596
|
}
|
|
419
597
|
if (failed) {
|
|
420
598
|
// Whatever was withheld is the model's own words on the way to an error.
|
|
421
599
|
// Show it rather than swallowing it.
|
|
422
600
|
if (held) yield textEvent(held);
|
|
423
601
|
yield {
|
|
424
|
-
type: "done"
|
|
602
|
+
type: "done",
|
|
603
|
+
stopReason: "error"
|
|
425
604
|
};
|
|
426
605
|
return messages;
|
|
427
606
|
}
|
|
@@ -477,7 +656,8 @@ export async function* runAgentTurn(input) {
|
|
|
477
656
|
}
|
|
478
657
|
};
|
|
479
658
|
yield {
|
|
480
|
-
type: "done"
|
|
659
|
+
type: "done",
|
|
660
|
+
stopReason: "incomplete"
|
|
481
661
|
};
|
|
482
662
|
return messages;
|
|
483
663
|
}
|
|
@@ -535,7 +715,8 @@ export async function* runAgentTurn(input) {
|
|
|
535
715
|
};
|
|
536
716
|
}
|
|
537
717
|
yield {
|
|
538
|
-
type: "done"
|
|
718
|
+
type: "done",
|
|
719
|
+
stopReason: asked ? "awaiting-user" : "answered"
|
|
539
720
|
};
|
|
540
721
|
return messages;
|
|
541
722
|
}
|
|
@@ -688,6 +869,11 @@ export async function* runAgentTurn(input) {
|
|
|
688
869
|
barrierRun = true;
|
|
689
870
|
for (const p of planned) {
|
|
690
871
|
if (!p?.gated || decided.has(p.call.id)) continue;
|
|
872
|
+
if (trusted?.has(`${p.toolId}.${p.action}`)) {
|
|
873
|
+
// Waived for this conversation: runs like any allowed write, no card.
|
|
874
|
+
decided.set(p.call.id, true);
|
|
875
|
+
continue;
|
|
876
|
+
}
|
|
691
877
|
yield {
|
|
692
878
|
type: "approval-required",
|
|
693
879
|
id: p.call.id,
|
|
@@ -705,7 +891,7 @@ export async function* runAgentTurn(input) {
|
|
|
705
891
|
dispatch,
|
|
706
892
|
catalog
|
|
707
893
|
});
|
|
708
|
-
const
|
|
894
|
+
const answer = readAnswer(trusted?.has(`${p.toolId}.${p.action}`) ? true : requestApproval ? await requestApproval({
|
|
709
895
|
id: p.call.id,
|
|
710
896
|
toolId: p.toolId,
|
|
711
897
|
action: p.action,
|
|
@@ -714,10 +900,13 @@ export async function* runAgentTurn(input) {
|
|
|
714
900
|
reason: p.reason,
|
|
715
901
|
params: p.params,
|
|
716
902
|
targetDigest
|
|
717
|
-
}) : false;
|
|
903
|
+
}) : false);
|
|
718
904
|
// The user's deliberation is not the agent's runtime.
|
|
719
905
|
deadline += Date.now() - askedAt;
|
|
906
|
+
const approved = answer.approved;
|
|
720
907
|
decided.set(p.call.id, approved);
|
|
908
|
+
if (answer.trust) trusted?.add(`${p.toolId}.${p.action}`);
|
|
909
|
+
if (answer.reason) declineReasons.set(p.call.id, answer.reason);
|
|
721
910
|
if (!approved) {
|
|
722
911
|
// One refusal ends the PLAN, not just the call. The remaining
|
|
723
912
|
// changes were proposed together and nothing has run yet, so
|
|
@@ -969,7 +1158,7 @@ export async function* runAgentTurn(input) {
|
|
|
969
1158
|
for (const ev of unrunStep("declined")) yield ev;
|
|
970
1159
|
results.push({
|
|
971
1160
|
toolCallId: call.id,
|
|
972
|
-
content: decided.get(call.id) === false ?
|
|
1161
|
+
content: decided.get(call.id) === false ? declinedResult(declineReasons.get(call.id)) : "Not run — the user declined another change in this same batch, so none of it was applied. Ask what they would prefer before proposing it again.",
|
|
973
1162
|
isError: true
|
|
974
1163
|
});
|
|
975
1164
|
continue;
|
|
@@ -980,6 +1169,10 @@ export async function* runAgentTurn(input) {
|
|
|
980
1169
|
// The barrier already put this card up, before the batch's first
|
|
981
1170
|
// mutation. Asking again would be the same question twice.
|
|
982
1171
|
approved = decided.get(call.id);
|
|
1172
|
+
} else if (trusted?.has(`${toolId}.${action}`)) {
|
|
1173
|
+
// Waived for this conversation — see RunTurnInput.trusted.
|
|
1174
|
+
approved = true;
|
|
1175
|
+
decided.set(call.id, true);
|
|
983
1176
|
} else {
|
|
984
1177
|
// The barrier could not resolve this call (see `planned`), so it was
|
|
985
1178
|
// never offered. Ask here, exactly as this always did — a gate that
|
|
@@ -1004,7 +1197,7 @@ export async function* runAgentTurn(input) {
|
|
|
1004
1197
|
dispatch,
|
|
1005
1198
|
catalog
|
|
1006
1199
|
});
|
|
1007
|
-
|
|
1200
|
+
const answer = readAnswer(requestApproval ? await requestApproval({
|
|
1008
1201
|
id: call.id,
|
|
1009
1202
|
toolId,
|
|
1010
1203
|
action,
|
|
@@ -1013,11 +1206,14 @@ export async function* runAgentTurn(input) {
|
|
|
1013
1206
|
reason: verdict.reason,
|
|
1014
1207
|
params,
|
|
1015
1208
|
targetDigest
|
|
1016
|
-
}) : false;
|
|
1209
|
+
}) : false);
|
|
1017
1210
|
// The user's deliberation is not the agent's runtime. Give the clock
|
|
1018
1211
|
// back before anything else can trip the deadline check.
|
|
1019
1212
|
deadline += Date.now() - askedAt;
|
|
1213
|
+
approved = answer.approved;
|
|
1020
1214
|
decided.set(call.id, approved);
|
|
1215
|
+
if (answer.trust) trusted?.add(`${toolId}.${action}`);
|
|
1216
|
+
if (answer.reason) declineReasons.set(call.id, answer.reason);
|
|
1021
1217
|
}
|
|
1022
1218
|
if (!approved) {
|
|
1023
1219
|
// The row is the whole point here. Without it the bubble read as a
|
|
@@ -1027,7 +1223,7 @@ export async function* runAgentTurn(input) {
|
|
|
1027
1223
|
for (const ev of unrunStep("declined")) yield ev;
|
|
1028
1224
|
results.push({
|
|
1029
1225
|
toolCallId: call.id,
|
|
1030
|
-
content:
|
|
1226
|
+
content: declinedResult(declineReasons.get(call.id)),
|
|
1031
1227
|
isError: true
|
|
1032
1228
|
});
|
|
1033
1229
|
continue;
|
|
@@ -1101,17 +1297,30 @@ export async function* runAgentTurn(input) {
|
|
|
1101
1297
|
// ledger the model asks about lives HERE, not behind an adapter, so
|
|
1102
1298
|
// routing it through dispatch would only work on a device and would
|
|
1103
1299
|
// record the undo as a fresh effect.
|
|
1104
|
-
const result = action === SNAPSHOT_ACTION ? projectSnapshot(toolId, await requireSnapshot(readSnapshot, toolId), params) : toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch) : await (prefetch.get(call.id) ?? dispatch(toolId, action, params));
|
|
1300
|
+
const result = action === SNAPSHOT_ACTION ? projectSnapshot(toolId, await requireSnapshot(readSnapshot, toolId), params) : toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch, evidence, params, procedures) : await (prefetch.get(call.id) ?? dispatch(toolId, action, params));
|
|
1105
1301
|
const cleaned = redact(stripSelfTraffic(result));
|
|
1106
1302
|
const refused = reportsFailure(result);
|
|
1107
|
-
|
|
1303
|
+
|
|
1304
|
+
// Kept in full — after redaction, never before — so the model can go
|
|
1305
|
+
// back to it. Not a retrieve's own result (a retrieve of a retrieve is
|
|
1306
|
+
// a loop), and not a refusal (nothing to go back to).
|
|
1307
|
+
const text = fullText(cleaned);
|
|
1308
|
+
const ref = evidence && !refused && !(toolId === "ask-buoy" && action === "retrieve") ? evidence.stash({
|
|
1309
|
+
toolId,
|
|
1310
|
+
action,
|
|
1311
|
+
params,
|
|
1312
|
+
capturedAt: Date.now(),
|
|
1313
|
+
text
|
|
1314
|
+
}) : undefined;
|
|
1315
|
+
const encodedResult = encodeResult(text, ref);
|
|
1316
|
+
const traceResult = encodedResult.length > MAX_TRACE_RESULT_CHARS ? `${encodedResult.slice(0, MAX_TRACE_RESULT_CHARS)}…` : encodedResult;
|
|
1108
1317
|
yield {
|
|
1109
1318
|
type: "tool-end",
|
|
1110
1319
|
id: call.id,
|
|
1111
1320
|
ok: !refused,
|
|
1112
1321
|
summary: summarise(result),
|
|
1113
1322
|
durationMs: Date.now() - startedAt,
|
|
1114
|
-
result:
|
|
1323
|
+
result: traceResult
|
|
1115
1324
|
};
|
|
1116
1325
|
|
|
1117
1326
|
// The receipt: what the user sees of this result, built from the
|
|
@@ -1135,7 +1344,12 @@ export async function* runAgentTurn(input) {
|
|
|
1135
1344
|
before: stateBefore,
|
|
1136
1345
|
after: stateAfter
|
|
1137
1346
|
});
|
|
1138
|
-
if (fx)
|
|
1347
|
+
if (fx) {
|
|
1348
|
+
const entry = ledger.record(toolId, action, fx, ledgerGeneration);
|
|
1349
|
+
// Ties the change to its round, so the budget keeps that round
|
|
1350
|
+
// while the change is live. See LedgerEntry.callId.
|
|
1351
|
+
if (entry) entry.callId = call.id;
|
|
1352
|
+
}
|
|
1139
1353
|
}
|
|
1140
1354
|
// No receipt for a call that changed nothing: a "what changed" card
|
|
1141
1355
|
// under a refusal is the same lie as the ledger entry, drawn bigger.
|
|
@@ -1153,10 +1367,36 @@ export async function* runAgentTurn(input) {
|
|
|
1153
1367
|
block: receipt,
|
|
1154
1368
|
forToolCallId: call.id
|
|
1155
1369
|
};
|
|
1370
|
+
/**
|
|
1371
|
+
* THE OUTCOME CHECK. A write's own `{ok:true}` says the adapter
|
|
1372
|
+
* accepted it; this reads the app back and says whether the requested
|
|
1373
|
+
* state is actually there. Only for actions that have a verifier, only
|
|
1374
|
+
* for writes that were not refused, and never itself a write. Its
|
|
1375
|
+
* verdict rides in the tool result as a trailer the model reads, and
|
|
1376
|
+
* on a second tool-end so the row says "verified" / "unverified" /
|
|
1377
|
+
* "check failed". See engine/verify.ts for the three meanings.
|
|
1378
|
+
*/
|
|
1379
|
+
const verification = descriptor.effect !== "read" && !refused ? await verifyOutcome({
|
|
1380
|
+
toolId,
|
|
1381
|
+
action,
|
|
1382
|
+
params,
|
|
1383
|
+
result,
|
|
1384
|
+
dispatch,
|
|
1385
|
+
signal,
|
|
1386
|
+
after: stateAfter
|
|
1387
|
+
}) : undefined;
|
|
1388
|
+
if (verification) yield {
|
|
1389
|
+
type: "tool-verified",
|
|
1390
|
+
id: call.id,
|
|
1391
|
+
verification
|
|
1392
|
+
};
|
|
1156
1393
|
for (const url of encodedResult.match(URL_RE) ?? []) seenUrls.add(url);
|
|
1157
1394
|
results.push({
|
|
1158
1395
|
toolCallId: call.id,
|
|
1159
|
-
content: encodedResult + ownedKeyNote(toolId, descriptor, params, storeKeyOwners) + aliasTrailer + (salvagedIds.has(call.id) ? SALVAGE_NOTE : "")
|
|
1396
|
+
content: encodedResult + ownedKeyNote(toolId, descriptor, params, storeKeyOwners) + aliasTrailer + (verification ? verificationTrailer(verification) : "") + (salvagedIds.has(call.id) ? SALVAGE_NOTE : ""),
|
|
1397
|
+
...(ref ? {
|
|
1398
|
+
ref
|
|
1399
|
+
} : {})
|
|
1160
1400
|
});
|
|
1161
1401
|
if (toolId === "app" && (action === "reloadApp" || action === "reload")) {
|
|
1162
1402
|
realmWillDie = true;
|
|
@@ -1201,7 +1441,8 @@ export async function* runAgentTurn(input) {
|
|
|
1201
1441
|
blockId: awaiting
|
|
1202
1442
|
};
|
|
1203
1443
|
yield {
|
|
1204
|
-
type: "done"
|
|
1444
|
+
type: "done",
|
|
1445
|
+
stopReason: "awaiting-user"
|
|
1205
1446
|
};
|
|
1206
1447
|
return messages;
|
|
1207
1448
|
}
|
|
@@ -1209,7 +1450,8 @@ export async function* runAgentTurn(input) {
|
|
|
1209
1450
|
// Anything after this dies with the JS realm. Stop cleanly instead of
|
|
1210
1451
|
// sending a request whose answer can never arrive.
|
|
1211
1452
|
yield {
|
|
1212
|
-
type: "done"
|
|
1453
|
+
type: "done",
|
|
1454
|
+
stopReason: "realm-died"
|
|
1213
1455
|
};
|
|
1214
1456
|
return messages;
|
|
1215
1457
|
}
|
|
@@ -1229,7 +1471,8 @@ export async function* runAgentTurn(input) {
|
|
|
1229
1471
|
}
|
|
1230
1472
|
};
|
|
1231
1473
|
yield {
|
|
1232
|
-
type: "done"
|
|
1474
|
+
type: "done",
|
|
1475
|
+
stopReason: "step-cap"
|
|
1233
1476
|
};
|
|
1234
1477
|
return messages;
|
|
1235
1478
|
}
|
|
@@ -1356,7 +1599,7 @@ export async function runGatedAction(input) {
|
|
|
1356
1599
|
};
|
|
1357
1600
|
}
|
|
1358
1601
|
try {
|
|
1359
|
-
const result = toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch) : await dispatch(toolId, action, params);
|
|
1602
|
+
const result = toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch, undefined, params) : await dispatch(toolId, action, params);
|
|
1360
1603
|
|
|
1361
1604
|
// Same rule as the model's path: a resolved `{ok:false}` is a refusal, not
|
|
1362
1605
|
// a change. Without this a block button (or an approval answered after a
|
|
@@ -1398,7 +1641,33 @@ export async function runGatedAction(input) {
|
|
|
1398
1641
|
* adapter's same-named actions remain for desktop/MCP, where the ledger is
|
|
1399
1642
|
* reached over the broker instead.
|
|
1400
1643
|
*/
|
|
1401
|
-
async function runAskBuoyAction(action, ledger, dispatch) {
|
|
1644
|
+
async function runAskBuoyAction(action, ledger, dispatch, evidence, params, procedures) {
|
|
1645
|
+
if (action === "retrieve") {
|
|
1646
|
+
return retrieveEvidence(evidence, params ?? {});
|
|
1647
|
+
}
|
|
1648
|
+
if (action === "openProcedure") {
|
|
1649
|
+
const id = typeof params?.id === "string" ? params.id : "";
|
|
1650
|
+
const found = procedures?.find(p => p.id === id);
|
|
1651
|
+
if (!found) {
|
|
1652
|
+
const ids = (procedures ?? []).map(p => p.id);
|
|
1653
|
+
return {
|
|
1654
|
+
ok: false,
|
|
1655
|
+
error: ids.length ? `No procedure "${id}". This app's procedures: ${ids.join(", ")}.` : "This app has no procedures written for it."
|
|
1656
|
+
};
|
|
1657
|
+
}
|
|
1658
|
+
return {
|
|
1659
|
+
id: found.id,
|
|
1660
|
+
title: found.title,
|
|
1661
|
+
...(found.version ? {
|
|
1662
|
+
version: found.version
|
|
1663
|
+
} : {}),
|
|
1664
|
+
...(found.requires?.length ? {
|
|
1665
|
+
requires: found.requires
|
|
1666
|
+
} : {}),
|
|
1667
|
+
body: found.body,
|
|
1668
|
+
note: "Follow these steps with the ordinary tools. Every write still goes through the same policy, approval and undo as any other call; a step that is refused stays refused."
|
|
1669
|
+
};
|
|
1670
|
+
}
|
|
1402
1671
|
if (action === "listChanges") {
|
|
1403
1672
|
const changes = ledger.list().filter(e => !e.undoneAt).map(e => ({
|
|
1404
1673
|
toolId: e.toolId,
|