@buoy-gg/agent-core 7.0.36 → 7.0.39

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/README.md +13 -9
  2. package/lib/commonjs/catalog/catalog.g.js +93 -6
  3. package/lib/commonjs/catalog/catalog.g.js.map +1 -1
  4. package/lib/commonjs/catalog/catalog.source.json +106 -4
  5. package/lib/commonjs/catalog/catalog.types.g.js +3 -3
  6. package/lib/commonjs/catalog/catalog.types.g.js.map +1 -1
  7. package/lib/commonjs/catalog/toProviderTools.js +3 -1
  8. package/lib/commonjs/catalog/toProviderTools.js.map +1 -1
  9. package/lib/commonjs/effects/ledger.js +13 -1
  10. package/lib/commonjs/effects/ledger.js.map +1 -1
  11. package/lib/commonjs/engine/evidence.js +113 -0
  12. package/lib/commonjs/engine/evidence.js.map +1 -0
  13. package/lib/commonjs/engine/historyBudget.js +158 -55
  14. package/lib/commonjs/engine/historyBudget.js.map +1 -1
  15. package/lib/commonjs/engine/retrieve.js +214 -0
  16. package/lib/commonjs/engine/retrieve.js.map +1 -0
  17. package/lib/commonjs/engine/runAgentTurn.js +364 -94
  18. package/lib/commonjs/engine/runAgentTurn.js.map +1 -1
  19. package/lib/commonjs/engine/systemPrompt.js +31 -1
  20. package/lib/commonjs/engine/systemPrompt.js.map +1 -1
  21. package/lib/commonjs/engine/tokenCalibration.js +81 -0
  22. package/lib/commonjs/engine/tokenCalibration.js.map +1 -0
  23. package/lib/commonjs/engine/verify.js +300 -0
  24. package/lib/commonjs/engine/verify.js.map +1 -0
  25. package/lib/commonjs/index.js +16 -0
  26. package/lib/commonjs/index.js.map +1 -1
  27. package/lib/commonjs/providers/anthropic.js +12 -3
  28. package/lib/commonjs/providers/anthropic.js.map +1 -1
  29. package/lib/commonjs/providers/openai.js +11 -3
  30. package/lib/commonjs/providers/openai.js.map +1 -1
  31. package/lib/commonjs/providers/problem.js +98 -0
  32. package/lib/commonjs/providers/problem.js.map +1 -0
  33. package/lib/commonjs/providers/sse.js +6 -5
  34. package/lib/commonjs/providers/sse.js.map +1 -1
  35. package/lib/commonjs/providers/transport.js +20 -4
  36. package/lib/commonjs/providers/transport.js.map +1 -1
  37. package/lib/commonjs/providers/xhrStream.js +3 -0
  38. package/lib/commonjs/providers/xhrStream.js.map +1 -1
  39. package/lib/commonjs/session.js +32 -5
  40. package/lib/commonjs/session.js.map +1 -1
  41. package/lib/module/catalog/catalog.g.js +93 -6
  42. package/lib/module/catalog/catalog.g.js.map +1 -1
  43. package/lib/module/catalog/catalog.source.json +106 -4
  44. package/lib/module/catalog/catalog.types.g.js +3 -3
  45. package/lib/module/catalog/catalog.types.g.js.map +1 -1
  46. package/lib/module/catalog/toProviderTools.js +3 -1
  47. package/lib/module/catalog/toProviderTools.js.map +1 -1
  48. package/lib/module/effects/ledger.js +13 -1
  49. package/lib/module/effects/ledger.js.map +1 -1
  50. package/lib/module/engine/evidence.js +107 -0
  51. package/lib/module/engine/evidence.js.map +1 -0
  52. package/lib/module/engine/historyBudget.js +158 -54
  53. package/lib/module/engine/historyBudget.js.map +1 -1
  54. package/lib/module/engine/retrieve.js +210 -0
  55. package/lib/module/engine/retrieve.js.map +1 -0
  56. package/lib/module/engine/runAgentTurn.js +365 -96
  57. package/lib/module/engine/runAgentTurn.js.map +1 -1
  58. package/lib/module/engine/systemPrompt.js +30 -1
  59. package/lib/module/engine/systemPrompt.js.map +1 -1
  60. package/lib/module/engine/tokenCalibration.js +75 -0
  61. package/lib/module/engine/tokenCalibration.js.map +1 -0
  62. package/lib/module/engine/verify.js +292 -0
  63. package/lib/module/engine/verify.js.map +1 -0
  64. package/lib/module/index.js +2 -0
  65. package/lib/module/index.js.map +1 -1
  66. package/lib/module/providers/anthropic.js +12 -3
  67. package/lib/module/providers/anthropic.js.map +1 -1
  68. package/lib/module/providers/openai.js +11 -3
  69. package/lib/module/providers/openai.js.map +1 -1
  70. package/lib/module/providers/problem.js +92 -0
  71. package/lib/module/providers/problem.js.map +1 -0
  72. package/lib/module/providers/sse.js +6 -5
  73. package/lib/module/providers/sse.js.map +1 -1
  74. package/lib/module/providers/transport.js +20 -4
  75. package/lib/module/providers/transport.js.map +1 -1
  76. package/lib/module/providers/xhrStream.js +3 -0
  77. package/lib/module/providers/xhrStream.js.map +1 -1
  78. package/lib/module/session.js +33 -6
  79. package/lib/module/session.js.map +1 -1
  80. package/lib/typescript/catalog/catalog.g.d.ts +2 -2
  81. package/lib/typescript/catalog/catalog.g.d.ts.map +1 -1
  82. package/lib/typescript/catalog/catalog.types.g.d.ts +3 -3
  83. package/lib/typescript/catalog/catalog.types.g.d.ts.map +1 -1
  84. package/lib/typescript/catalog/toProviderTools.d.ts.map +1 -1
  85. package/lib/typescript/effects/ledger.d.ts +18 -0
  86. package/lib/typescript/effects/ledger.d.ts.map +1 -1
  87. package/lib/typescript/engine/evidence.d.ts +67 -0
  88. package/lib/typescript/engine/evidence.d.ts.map +1 -0
  89. package/lib/typescript/engine/historyBudget.d.ts +31 -8
  90. package/lib/typescript/engine/historyBudget.d.ts.map +1 -1
  91. package/lib/typescript/engine/retrieve.d.ts +32 -0
  92. package/lib/typescript/engine/retrieve.d.ts.map +1 -0
  93. package/lib/typescript/engine/runAgentTurn.d.ts +96 -2
  94. package/lib/typescript/engine/runAgentTurn.d.ts.map +1 -1
  95. package/lib/typescript/engine/systemPrompt.d.ts +39 -0
  96. package/lib/typescript/engine/systemPrompt.d.ts.map +1 -1
  97. package/lib/typescript/engine/tokenCalibration.d.ts +46 -0
  98. package/lib/typescript/engine/tokenCalibration.d.ts.map +1 -0
  99. package/lib/typescript/engine/verify.d.ts +67 -0
  100. package/lib/typescript/engine/verify.d.ts.map +1 -0
  101. package/lib/typescript/index.d.ts +4 -2
  102. package/lib/typescript/index.d.ts.map +1 -1
  103. package/lib/typescript/providers/anthropic.d.ts.map +1 -1
  104. package/lib/typescript/providers/openai.d.ts.map +1 -1
  105. package/lib/typescript/providers/problem.d.ts +50 -0
  106. package/lib/typescript/providers/problem.d.ts.map +1 -0
  107. package/lib/typescript/providers/sse.d.ts +4 -1
  108. package/lib/typescript/providers/sse.d.ts.map +1 -1
  109. package/lib/typescript/providers/transport.d.ts +8 -0
  110. package/lib/typescript/providers/transport.d.ts.map +1 -1
  111. package/lib/typescript/providers/types.d.ts +14 -0
  112. package/lib/typescript/providers/types.d.ts.map +1 -1
  113. package/lib/typescript/providers/xhrStream.d.ts +2 -0
  114. package/lib/typescript/providers/xhrStream.d.ts.map +1 -1
  115. package/lib/typescript/session.d.ts +15 -2
  116. package/lib/typescript/session.d.ts.map +1 -1
  117. package/lib/typescript/types.d.ts +6 -0
  118. package/lib/typescript/types.d.ts.map +1 -1
  119. package/package.json +1 -1
@@ -35,8 +35,11 @@ import { projectReceipt } from "../blocks/receipts";
35
35
  import { isAskBlock } from "../blocks/types";
36
36
  import { ACQUISITION_CALLS, synthesizeAskBlock, synthesizeShortfallBlock } from "./askGate";
37
37
  import { couldBeToolCallEnvelope, parseTextToolCalls, SALVAGE_NOTE } from "./textToolCalls";
38
- import { budgetForRequest } from "./historyBudget";
39
-
38
+ import { budgetForRequest, MAX_HISTORY_TOKENS, messageSize } from "./historyBudget";
39
+ import { promptTokensOf, TokenCalibration } from "./tokenCalibration";
40
+ import { truncationMarker } from "./evidence";
41
+ import { retrieveEvidence } from "./retrieve";
42
+ import { verificationTrailer, verifyOutcome } from "./verify";
40
43
  /** Beyond this a tool result costs more in tokens than it can possibly be worth. */
41
44
  const MAX_RESULT_CHARS = 24_000;
42
45
  /**
@@ -71,6 +74,73 @@ function callSignature(calls) {
71
74
  const LOOP_WARNING = "[Buoy] You have now made this exact call three times in a row and gotten the same thing back. Repeating it again will not change the answer. Read the result you already have, then either try a different tool or different arguments, or tell the user plainly what you could not get and what you tried.";
72
75
  const DEFAULT_MAX_STEPS = 12;
73
76
  const DEFAULT_TURN_MS = 180_000;
77
+
78
+ /**
79
+ * Retrying a model request — narrowly.
80
+ *
81
+ * A 429 from a shared gateway, a 529 / `overloaded_error` from Anthropic, or
82
+ * a "prompt is too long" 400 used to end the turn with an error card and Try
83
+ * again — and Try again re-sends the QUESTION, restarting an investigation
84
+ * that may already have written things. All three fail before a model has
85
+ * generated anything, so asking again is safe: nothing runs twice, nothing
86
+ * is billed twice. The rule that makes it safe is enforced, not assumed —
87
+ * a request is only retried when its stream produced NO text, tool call or
88
+ * thinking; once anything has come back, the existing no-replay handling
89
+ * stands (see providers/transport.ts on consumed-exactly-once).
90
+ *
91
+ * Throttled: up to two retries, 1 s then 2 s (plus jitter), a `Retry-After`
92
+ * winning when the endpoint sent one. Overflow: one retry, with the history
93
+ * ceiling halved first — the proactive budget is chars ÷ 4 and the provider
94
+ * just said that was optimistic. Both wait inside the turn deadline and stop
95
+ * on the user's Stop. Small numbers on purpose: this is a phone, not a
96
+ * server queue.
97
+ */
98
+ const THROTTLE_RETRIES = 2;
99
+ const OVERFLOW_RETRIES = 1;
100
+ const RETRY_BASE_MS = 1_000;
101
+ const RETRY_JITTER_MS = 250;
102
+ const RETRY_AFTER_CAP_MS = 30_000;
103
+ function sleep(ms, signal) {
104
+ return new Promise(resolve => {
105
+ if (signal?.aborted) return resolve();
106
+ const t = setTimeout(done, ms);
107
+ function done() {
108
+ clearTimeout(t);
109
+ signal?.removeEventListener("abort", done);
110
+ resolve();
111
+ }
112
+ signal?.addEventListener("abort", done, {
113
+ once: true
114
+ });
115
+ });
116
+ }
117
+
118
+ /**
119
+ * Why the turn ended — the machine-readable half of the notices. The text in
120
+ * the notice blocks is unchanged (the eval classifier pins it); this is for a
121
+ * host, the bank and the desktop, which used to have to grep the prose.
122
+ */
123
+
124
+ /** Normalise an answer to its parts, so both call sites read one shape. */
125
+ function readAnswer(answer) {
126
+ if (typeof answer === "boolean") return {
127
+ approved: answer,
128
+ trust: false
129
+ };
130
+ const reason = answer.reason?.trim();
131
+ return {
132
+ approved: answer.approved,
133
+ ...(reason ? {
134
+ reason
135
+ } : {}),
136
+ trust: answer.trust === true && answer.approved
137
+ };
138
+ }
139
+
140
+ /** What the model is told when the user says no — with their note, when they left one. */
141
+ function declinedResult(reason) {
142
+ return reason ? `The user declined this change and said: "${reason}". Do not retry it as proposed; take their note into account and, if a different change would fit, propose that instead.` : "The user declined this change. Do not retry it; ask what they would prefer.";
143
+ }
74
144
  function findDescriptor(catalog, toolId, action) {
75
145
  return catalog.find(t => t.toolId === toolId)?.actions.find(a => a.action === action);
76
146
  }
@@ -85,15 +155,20 @@ function findDescriptor(catalog, toolId, action) {
85
155
  */
86
156
  const URL_PROVENANCE_REJECTION = "Rejected: every image URL must be one you read from a tool result in THIS conversation (images.list, a storage value, a response body). Never invent or remember URLs — read them first, then show them.";
87
157
  const droppedImageNote = n => ` NOTE: ${n} image ${n === 1 ? "URL was" : "URLs were"} dropped — they were not read from a tool result in this conversation, so the card the user is looking at has no picture there. Do not tell them it does; read the URLs and show it again, or say the artwork is missing.`;
88
- function encodeResult(value) {
89
- let text;
158
+
159
+ /** The whole result as text — what the evidence store keeps. */
160
+ function fullText(value) {
90
161
  try {
91
- text = JSON.stringify(value ?? null);
162
+ return JSON.stringify(value ?? null);
92
163
  } catch {
93
- text = String(value);
164
+ return String(value);
94
165
  }
166
+ }
167
+
168
+ /** What the model is sent: the text, cut at the cap with a marker that names where the rest is. */
169
+ function encodeResult(text, ref) {
95
170
  if (text.length <= MAX_RESULT_CHARS) return text;
96
- return `${text.slice(0, MAX_RESULT_CHARS)}\n\n[truncated — ${text.length} characters total. Ask for a narrower slice if you need more.]`;
171
+ return `${text.slice(0, MAX_RESULT_CHARS)}${truncationMarker(text.length, ref)}`;
97
172
  }
98
173
 
99
174
  /** A short line for the transcript row, so the UI never renders raw payloads. */
@@ -162,8 +237,14 @@ export async function* runAgentTurn(input) {
162
237
  policy,
163
238
  isRelease,
164
239
  requestApproval,
165
- signal
240
+ signal,
241
+ evidence,
242
+ trusted,
243
+ procedures
166
244
  } = input;
245
+ /** Notes the user left with a decline, by call id — see declinedResult. */
246
+ const declineReasons = new Map();
247
+ const calibration = input.calibration ?? new TokenCalibration();
167
248
  const messages = [...input.messages];
168
249
  const tools = toProviderTools(catalog, {
169
250
  availableToolIds: input.availableToolIds,
@@ -187,7 +268,9 @@ export async function* runAgentTurn(input) {
187
268
  * much to leave out of a budget. `maxTokens` is multiplied by four because
188
269
  * the budget is in characters.
189
270
  */
190
- const requestReserve = system.length + (systemVolatile?.length ?? 0) + tools.reduce((n, t) => n + t.description.length + JSON.stringify(t.inputSchema).length, 0) + maxTokens * 4;
271
+ const fixedChars = system.length + (systemVolatile?.length ?? 0) + tools.reduce((n, t) => n + t.description.length + JSON.stringify(t.inputSchema).length, 0);
272
+ /** The reserve at the ratio believed RIGHT NOW — it moves as usage reports come in. */
273
+ const requestReserve = () => fixedChars + calibration.chars(maxTokens);
191
274
  /**
192
275
  * Caps the AGENT's wall-clock, not the user's. Time spent parked on an
193
276
  * approval sheet is pushed onto the deadline as it is spent (see the
@@ -196,6 +279,13 @@ export async function* runAgentTurn(input) {
196
279
  * for the turn that applied it.
197
280
  */
198
281
  let deadline = Date.now() + DEFAULT_TURN_MS;
282
+ /** Retries spent this turn, per kind — the budget is per turn, not per step. */
283
+ const retries = {
284
+ throttled: 0,
285
+ overflow: 0
286
+ };
287
+ /** Tightened after an overflow; see budgetForRequest's `cap`. In tokens, so a calibration change re-scales it. */
288
+ let historyCapTokens = MAX_HISTORY_TOKENS;
199
289
 
200
290
  // Every http(s) URL that appeared in a tool result in this CONVERSATION.
201
291
  // The prompt tells the model image URLs must come from here; this Set is
@@ -241,7 +331,8 @@ export async function* runAgentTurn(input) {
241
331
  for (let step = 0; step < maxSteps; step++) {
242
332
  if (signal?.aborted) {
243
333
  yield {
244
- type: "done"
334
+ type: "done",
335
+ stopReason: "stopped"
245
336
  };
246
337
  return messages;
247
338
  }
@@ -268,7 +359,8 @@ export async function* runAgentTurn(input) {
268
359
  }
269
360
  };
270
361
  yield {
271
- type: "done"
362
+ type: "done",
363
+ stopReason: "time-cap"
272
364
  };
273
365
  return messages;
274
366
  }
@@ -286,13 +378,14 @@ export async function* runAgentTurn(input) {
286
378
  *
287
379
  * The current round is never touched: it is the question being answered.
288
380
  */
289
- const budgeted = budgetForRequest(messages, requestReserve);
381
+ const budgeted = budgetForRequest(messages, requestReserve(), calibration.chars(historyCapTokens), calibration.charsPerToken, ledger.liveCallIds());
290
382
  if (budgeted.droppedRounds > 0) {
291
383
  messages.length = 0;
292
384
  messages.push(...budgeted.messages);
293
385
  yield {
294
386
  type: "history-trimmed",
295
- droppedRounds: budgeted.droppedRounds
387
+ droppedRounds: budgeted.droppedRounds,
388
+ droppedPinned: budgeted.droppedPinned
296
389
  };
297
390
  } else if (budgeted.messages !== messages) {
298
391
  messages.length = 0;
@@ -318,7 +411,7 @@ export async function* runAgentTurn(input) {
318
411
  delta
319
412
  };
320
413
  };
321
- const calls = [];
414
+ let calls = [];
322
415
  /**
323
416
  * Recovered from text rather than emitted as calls — see textToolCalls.ts.
324
417
  * Tracked by id so each one's result can tell the model to stop doing that;
@@ -335,7 +428,7 @@ export async function* runAgentTurn(input) {
335
428
  let holding = true;
336
429
  // Carried, never read: Anthropic requires the turn's thinking blocks back
337
430
  // verbatim with its tool results. See providers/anthropic.ts note 5.
338
- const thinking = [];
431
+ let thinking = [];
339
432
  let failed = false;
340
433
  /**
341
434
  * What the provider said about how the stream ended.
@@ -348,80 +441,166 @@ export async function* runAgentTurn(input) {
348
441
  let outcome;
349
442
  /** Set for the two kinds that are recoverable rather than a hard failure. */
350
443
  let incomplete;
351
- for await (const ev of provider.send({
352
- messages,
353
- system,
354
- systemVolatile,
355
- tools,
356
- model,
357
- maxTokens,
358
- signal
359
- })) {
360
- if (ev.type === "text") {
361
- text += ev.delta;
362
- if (!holding) {
363
- yield textEvent(ev.delta);
364
- } else if (couldBeToolCallEnvelope(text)) {
365
- held += ev.delta;
366
- } else {
367
- // Not an envelope after all. Release everything at once and stream
368
- // the rest as usual — the reader loses nothing but a few characters
369
- // of latency at the very start of the answer.
370
- holding = false;
371
- held = "";
372
- yield textEvent(text);
373
- }
374
- } else if (ev.type === "tool-call") {
375
- calls.push(ev.call);
376
- } else if (ev.type === "thinking") {
377
- thinking.push(ev.block);
378
- // Surfaced as well as carried: the block goes back to the provider
379
- // verbatim (note above), and the readable half goes to the UI so a
380
- // tester can see WHY a turn did what it did. Redacted blocks have no
381
- // readable half and are carried only.
382
- if (ev.block.type === "thinking" && ev.block.thinking.trim()) {
383
- yield {
384
- type: "reasoning",
385
- text: ev.block.thinking
386
- };
444
+
445
+ /**
446
+ * THE REQUEST, with its retries. One request per pass; a pass that fails
447
+ * before the model generated anything may be sent again (see
448
+ * THROTTLE_RETRIES). Everything the pass accumulated is reset first —
449
+ * there is nothing to keep, by the rule that made the retry safe.
450
+ */
451
+ for (;;) {
452
+ text = "";
453
+ calls = [];
454
+ held = "";
455
+ holding = true;
456
+ thinking = [];
457
+ failed = false;
458
+ outcome = undefined;
459
+ incomplete = undefined;
460
+ /** The failure that ended this pass, when it is one the engine may retry. */
461
+ let retryable;
462
+ /** What this pass sends, in characters — the numerator of the calibration sample. */
463
+ const sentChars = fixedChars + messages.reduce((n, m) => n + messageSize(m), 0);
464
+ for await (const ev of provider.send({
465
+ messages,
466
+ system,
467
+ systemVolatile,
468
+ tools,
469
+ model,
470
+ maxTokens,
471
+ signal
472
+ })) {
473
+ if (ev.type === "text") {
474
+ text += ev.delta;
475
+ if (!holding) {
476
+ yield textEvent(ev.delta);
477
+ } else if (couldBeToolCallEnvelope(text)) {
478
+ held += ev.delta;
479
+ } else {
480
+ // Not an envelope after all. Release everything at once and stream
481
+ // the rest as usual — the reader loses nothing but a few characters
482
+ // of latency at the very start of the answer.
483
+ holding = false;
484
+ held = "";
485
+ yield textEvent(text);
486
+ }
487
+ } else if (ev.type === "tool-call") {
488
+ calls.push(ev.call);
489
+ } else if (ev.type === "thinking") {
490
+ thinking.push(ev.block);
491
+ // Surfaced as well as carried: the block goes back to the provider
492
+ // verbatim (note above), and the readable half goes to the UI so a
493
+ // tester can see WHY a turn did what it did. Redacted blocks have no
494
+ // readable half and are carried only.
495
+ if (ev.block.type === "thinking" && ev.block.thinking.trim()) {
496
+ yield {
497
+ type: "reasoning",
498
+ text: ev.block.thinking
499
+ };
500
+ }
501
+ } else if (ev.type === "error") {
502
+ if (ev.kind === "truncated" || ev.kind === "timeout") {
503
+ // Not a failure the user did anything about, and not one Try again
504
+ // can fix — re-sending the QUESTION restarts an investigation that
505
+ // may already have written things. Held back from the error card
506
+ // (which is what draws Try again) and handled below as a recoverable
507
+ // stop with a Continue button.
508
+ incomplete = {
509
+ message: ev.message
510
+ };
511
+ } else if (ev.problem && (ev.problem.kind === "throttled" || ev.problem.kind === "overflow") && text === "" && calls.length === 0 && thinking.length === 0) {
512
+ // Nothing generated, and a kind that a wait or a trim can fix.
513
+ // Decided after the stream closes; the adapter returns on error.
514
+ retryable = {
515
+ problem: ev.problem,
516
+ message: ev.message
517
+ };
518
+ } else {
519
+ // A mid-stream error can arrive on an already-committed 200.
520
+ yield {
521
+ type: "error",
522
+ message: ev.message
523
+ };
524
+ failed = true;
525
+ }
526
+ } else if (ev.type === "done") {
527
+ outcome = ev.outcome;
528
+ if (ev.usage) {
529
+ // The provider just counted this prompt. One sample per request
530
+ // keeps the chars↔tokens ratio honest for the NEXT budget.
531
+ calibration.observe(sentChars, promptTokensOf(ev.usage, provider.protocol));
532
+ // Surfaced so a host can meter cost per seat. Parsed for a long time;
533
+ // went nowhere anyone could see.
534
+ yield {
535
+ type: "usage",
536
+ ...ev.usage,
537
+ model: ev.model
538
+ };
539
+ }
387
540
  }
388
- } else if (ev.type === "error") {
389
- if (ev.kind === "truncated" || ev.kind === "timeout") {
390
- // Not a failure the user did anything about, and not one Try again
391
- // can fix — re-sending the QUESTION restarts an investigation that
392
- // may already have written things. Held back from the error card
393
- // (which is what draws Try again) and handled below as a recoverable
394
- // stop with a Continue button.
395
- incomplete = {
396
- message: ev.message
397
- };
541
+ }
542
+ if (!retryable) break;
543
+ const kind = retryable.problem.kind;
544
+ const maxAttempts = 1 + (kind === "throttled" ? THROTTLE_RETRIES : OVERFLOW_RETRIES);
545
+ const spent = retries[kind];
546
+ let waitMs = kind === "throttled" ? Math.min(retryable.problem.retryAfterMs ?? RETRY_BASE_MS * 2 ** spent, RETRY_AFTER_CAP_MS) + Math.floor(Math.random() * RETRY_JITTER_MS) : 0;
547
+ let giveUp = spent >= maxAttempts - 1 || signal?.aborted === true || Date.now() + waitMs > deadline;
548
+ if (!giveUp && kind === "overflow") {
549
+ // Halve the ceiling and trim again. If that changes nothing — the
550
+ // current round alone is over the limit — a retry would only repeat
551
+ // the refusal, so report it instead.
552
+ historyCapTokens = Math.floor(historyCapTokens / 2);
553
+ const before = messages.reduce((n, m) => n + messageSize(m), 0);
554
+ const again = budgetForRequest(messages, requestReserve(), calibration.chars(historyCapTokens), calibration.charsPerToken, ledger.liveCallIds());
555
+ const after = again.messages.reduce((n, m) => n + messageSize(m), 0);
556
+ if (after >= before) {
557
+ giveUp = true;
398
558
  } else {
399
- // A mid-stream error can arrive on an already-committed 200.
400
- yield {
401
- type: "error",
402
- message: ev.message
403
- };
404
- failed = true;
405
- }
406
- } else if (ev.type === "done") {
407
- outcome = ev.outcome;
408
- if (ev.usage) {
409
- // Surfaced so a host can meter cost per seat. Parsed for a long time;
410
- // went nowhere anyone could see.
411
- yield {
412
- type: "usage",
413
- ...ev.usage,
414
- model: ev.model
559
+ messages.length = 0;
560
+ messages.push(...again.messages);
561
+ if (again.droppedRounds > 0) yield {
562
+ type: "history-trimmed",
563
+ droppedRounds: again.droppedRounds,
564
+ droppedPinned: again.droppedPinned
415
565
  };
416
566
  }
417
567
  }
568
+ if (giveUp) {
569
+ // Plain words first; the endpoint's own text after, for whoever files
570
+ // the bug. A user reading "prompt is too long: 213000 tokens" has no
571
+ // move; "start a new conversation" is one.
572
+ const plain = kind === "throttled" ? spent > 0 ? `The AI endpoint is rate-limiting requests — tried ${spent + 1} times.` : "The AI endpoint is rate-limiting requests." : "This conversation is too large for the AI endpoint and there was nothing older to trim. Start a new conversation, or ask a shorter question.";
573
+ yield {
574
+ type: "error",
575
+ message: `${plain} (${retryable.message})`
576
+ };
577
+ failed = true;
578
+ break;
579
+ }
580
+ retries[kind] += 1;
581
+ yield {
582
+ type: "retrying",
583
+ reason: kind,
584
+ attempt: retries[kind] + 1,
585
+ maxAttempts,
586
+ inMs: waitMs
587
+ };
588
+ if (waitMs > 0) await sleep(waitMs, signal);
589
+ if (signal?.aborted) {
590
+ yield {
591
+ type: "done",
592
+ stopReason: "stopped"
593
+ };
594
+ return messages;
595
+ }
418
596
  }
419
597
  if (failed) {
420
598
  // Whatever was withheld is the model's own words on the way to an error.
421
599
  // Show it rather than swallowing it.
422
600
  if (held) yield textEvent(held);
423
601
  yield {
424
- type: "done"
602
+ type: "done",
603
+ stopReason: "error"
425
604
  };
426
605
  return messages;
427
606
  }
@@ -477,7 +656,8 @@ export async function* runAgentTurn(input) {
477
656
  }
478
657
  };
479
658
  yield {
480
- type: "done"
659
+ type: "done",
660
+ stopReason: "incomplete"
481
661
  };
482
662
  return messages;
483
663
  }
@@ -535,7 +715,8 @@ export async function* runAgentTurn(input) {
535
715
  };
536
716
  }
537
717
  yield {
538
- type: "done"
718
+ type: "done",
719
+ stopReason: asked ? "awaiting-user" : "answered"
539
720
  };
540
721
  return messages;
541
722
  }
@@ -688,6 +869,11 @@ export async function* runAgentTurn(input) {
688
869
  barrierRun = true;
689
870
  for (const p of planned) {
690
871
  if (!p?.gated || decided.has(p.call.id)) continue;
872
+ if (trusted?.has(`${p.toolId}.${p.action}`)) {
873
+ // Waived for this conversation: runs like any allowed write, no card.
874
+ decided.set(p.call.id, true);
875
+ continue;
876
+ }
691
877
  yield {
692
878
  type: "approval-required",
693
879
  id: p.call.id,
@@ -705,7 +891,7 @@ export async function* runAgentTurn(input) {
705
891
  dispatch,
706
892
  catalog
707
893
  });
708
- const approved = requestApproval ? await requestApproval({
894
+ const answer = readAnswer(trusted?.has(`${p.toolId}.${p.action}`) ? true : requestApproval ? await requestApproval({
709
895
  id: p.call.id,
710
896
  toolId: p.toolId,
711
897
  action: p.action,
@@ -714,10 +900,13 @@ export async function* runAgentTurn(input) {
714
900
  reason: p.reason,
715
901
  params: p.params,
716
902
  targetDigest
717
- }) : false;
903
+ }) : false);
718
904
  // The user's deliberation is not the agent's runtime.
719
905
  deadline += Date.now() - askedAt;
906
+ const approved = answer.approved;
720
907
  decided.set(p.call.id, approved);
908
+ if (answer.trust) trusted?.add(`${p.toolId}.${p.action}`);
909
+ if (answer.reason) declineReasons.set(p.call.id, answer.reason);
721
910
  if (!approved) {
722
911
  // One refusal ends the PLAN, not just the call. The remaining
723
912
  // changes were proposed together and nothing has run yet, so
@@ -969,7 +1158,7 @@ export async function* runAgentTurn(input) {
969
1158
  for (const ev of unrunStep("declined")) yield ev;
970
1159
  results.push({
971
1160
  toolCallId: call.id,
972
- content: decided.get(call.id) === false ? "The user declined this change. Do not retry it; ask what they would prefer." : "Not run — the user declined another change in this same batch, so none of it was applied. Ask what they would prefer before proposing it again.",
1161
+ content: decided.get(call.id) === false ? declinedResult(declineReasons.get(call.id)) : "Not run — the user declined another change in this same batch, so none of it was applied. Ask what they would prefer before proposing it again.",
973
1162
  isError: true
974
1163
  });
975
1164
  continue;
@@ -980,6 +1169,10 @@ export async function* runAgentTurn(input) {
980
1169
  // The barrier already put this card up, before the batch's first
981
1170
  // mutation. Asking again would be the same question twice.
982
1171
  approved = decided.get(call.id);
1172
+ } else if (trusted?.has(`${toolId}.${action}`)) {
1173
+ // Waived for this conversation — see RunTurnInput.trusted.
1174
+ approved = true;
1175
+ decided.set(call.id, true);
983
1176
  } else {
984
1177
  // The barrier could not resolve this call (see `planned`), so it was
985
1178
  // never offered. Ask here, exactly as this always did — a gate that
@@ -1004,7 +1197,7 @@ export async function* runAgentTurn(input) {
1004
1197
  dispatch,
1005
1198
  catalog
1006
1199
  });
1007
- approved = requestApproval ? await requestApproval({
1200
+ const answer = readAnswer(requestApproval ? await requestApproval({
1008
1201
  id: call.id,
1009
1202
  toolId,
1010
1203
  action,
@@ -1013,11 +1206,14 @@ export async function* runAgentTurn(input) {
1013
1206
  reason: verdict.reason,
1014
1207
  params,
1015
1208
  targetDigest
1016
- }) : false;
1209
+ }) : false);
1017
1210
  // The user's deliberation is not the agent's runtime. Give the clock
1018
1211
  // back before anything else can trip the deadline check.
1019
1212
  deadline += Date.now() - askedAt;
1213
+ approved = answer.approved;
1020
1214
  decided.set(call.id, approved);
1215
+ if (answer.trust) trusted?.add(`${toolId}.${action}`);
1216
+ if (answer.reason) declineReasons.set(call.id, answer.reason);
1021
1217
  }
1022
1218
  if (!approved) {
1023
1219
  // The row is the whole point here. Without it the bubble read as a
@@ -1027,7 +1223,7 @@ export async function* runAgentTurn(input) {
1027
1223
  for (const ev of unrunStep("declined")) yield ev;
1028
1224
  results.push({
1029
1225
  toolCallId: call.id,
1030
- content: "The user declined this change. Do not retry it; ask what they would prefer.",
1226
+ content: declinedResult(declineReasons.get(call.id)),
1031
1227
  isError: true
1032
1228
  });
1033
1229
  continue;
@@ -1101,17 +1297,30 @@ export async function* runAgentTurn(input) {
1101
1297
  // ledger the model asks about lives HERE, not behind an adapter, so
1102
1298
  // routing it through dispatch would only work on a device and would
1103
1299
  // record the undo as a fresh effect.
1104
- const result = action === SNAPSHOT_ACTION ? projectSnapshot(toolId, await requireSnapshot(readSnapshot, toolId), params) : toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch) : await (prefetch.get(call.id) ?? dispatch(toolId, action, params));
1300
+ const result = action === SNAPSHOT_ACTION ? projectSnapshot(toolId, await requireSnapshot(readSnapshot, toolId), params) : toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch, evidence, params, procedures) : await (prefetch.get(call.id) ?? dispatch(toolId, action, params));
1105
1301
  const cleaned = redact(stripSelfTraffic(result));
1106
1302
  const refused = reportsFailure(result);
1107
- const encodedResult = encodeResult(cleaned);
1303
+
1304
+ // Kept in full — after redaction, never before — so the model can go
1305
+ // back to it. Not a retrieve's own result (a retrieve of a retrieve is
1306
+ // a loop), and not a refusal (nothing to go back to).
1307
+ const text = fullText(cleaned);
1308
+ const ref = evidence && !refused && !(toolId === "ask-buoy" && action === "retrieve") ? evidence.stash({
1309
+ toolId,
1310
+ action,
1311
+ params,
1312
+ capturedAt: Date.now(),
1313
+ text
1314
+ }) : undefined;
1315
+ const encodedResult = encodeResult(text, ref);
1316
+ const traceResult = encodedResult.length > MAX_TRACE_RESULT_CHARS ? `${encodedResult.slice(0, MAX_TRACE_RESULT_CHARS)}…` : encodedResult;
1108
1317
  yield {
1109
1318
  type: "tool-end",
1110
1319
  id: call.id,
1111
1320
  ok: !refused,
1112
1321
  summary: summarise(result),
1113
1322
  durationMs: Date.now() - startedAt,
1114
- result: encodedResult.length > MAX_TRACE_RESULT_CHARS ? `${encodedResult.slice(0, MAX_TRACE_RESULT_CHARS)}…` : encodedResult
1323
+ result: traceResult
1115
1324
  };
1116
1325
 
1117
1326
  // The receipt: what the user sees of this result, built from the
@@ -1135,7 +1344,12 @@ export async function* runAgentTurn(input) {
1135
1344
  before: stateBefore,
1136
1345
  after: stateAfter
1137
1346
  });
1138
- if (fx) ledger.record(toolId, action, fx, ledgerGeneration);
1347
+ if (fx) {
1348
+ const entry = ledger.record(toolId, action, fx, ledgerGeneration);
1349
+ // Ties the change to its round, so the budget keeps that round
1350
+ // while the change is live. See LedgerEntry.callId.
1351
+ if (entry) entry.callId = call.id;
1352
+ }
1139
1353
  }
1140
1354
  // No receipt for a call that changed nothing: a "what changed" card
1141
1355
  // under a refusal is the same lie as the ledger entry, drawn bigger.
@@ -1153,10 +1367,36 @@ export async function* runAgentTurn(input) {
1153
1367
  block: receipt,
1154
1368
  forToolCallId: call.id
1155
1369
  };
1370
+ /**
1371
+ * THE OUTCOME CHECK. A write's own `{ok:true}` says the adapter
1372
+ * accepted it; this reads the app back and says whether the requested
1373
+ * state is actually there. Only for actions that have a verifier, only
1374
+ * for writes that were not refused, and never itself a write. Its
1375
+ * verdict rides in the tool result as a trailer the model reads, and
1376
+ * on a second tool-end so the row says "verified" / "unverified" /
1377
+ * "check failed". See engine/verify.ts for the three meanings.
1378
+ */
1379
+ const verification = descriptor.effect !== "read" && !refused ? await verifyOutcome({
1380
+ toolId,
1381
+ action,
1382
+ params,
1383
+ result,
1384
+ dispatch,
1385
+ signal,
1386
+ after: stateAfter
1387
+ }) : undefined;
1388
+ if (verification) yield {
1389
+ type: "tool-verified",
1390
+ id: call.id,
1391
+ verification
1392
+ };
1156
1393
  for (const url of encodedResult.match(URL_RE) ?? []) seenUrls.add(url);
1157
1394
  results.push({
1158
1395
  toolCallId: call.id,
1159
- content: encodedResult + ownedKeyNote(toolId, descriptor, params, storeKeyOwners) + aliasTrailer + (salvagedIds.has(call.id) ? SALVAGE_NOTE : "")
1396
+ content: encodedResult + ownedKeyNote(toolId, descriptor, params, storeKeyOwners) + aliasTrailer + (verification ? verificationTrailer(verification) : "") + (salvagedIds.has(call.id) ? SALVAGE_NOTE : ""),
1397
+ ...(ref ? {
1398
+ ref
1399
+ } : {})
1160
1400
  });
1161
1401
  if (toolId === "app" && (action === "reloadApp" || action === "reload")) {
1162
1402
  realmWillDie = true;
@@ -1201,7 +1441,8 @@ export async function* runAgentTurn(input) {
1201
1441
  blockId: awaiting
1202
1442
  };
1203
1443
  yield {
1204
- type: "done"
1444
+ type: "done",
1445
+ stopReason: "awaiting-user"
1205
1446
  };
1206
1447
  return messages;
1207
1448
  }
@@ -1209,7 +1450,8 @@ export async function* runAgentTurn(input) {
1209
1450
  // Anything after this dies with the JS realm. Stop cleanly instead of
1210
1451
  // sending a request whose answer can never arrive.
1211
1452
  yield {
1212
- type: "done"
1453
+ type: "done",
1454
+ stopReason: "realm-died"
1213
1455
  };
1214
1456
  return messages;
1215
1457
  }
@@ -1229,7 +1471,8 @@ export async function* runAgentTurn(input) {
1229
1471
  }
1230
1472
  };
1231
1473
  yield {
1232
- type: "done"
1474
+ type: "done",
1475
+ stopReason: "step-cap"
1233
1476
  };
1234
1477
  return messages;
1235
1478
  }
@@ -1356,7 +1599,7 @@ export async function runGatedAction(input) {
1356
1599
  };
1357
1600
  }
1358
1601
  try {
1359
- const result = toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch) : await dispatch(toolId, action, params);
1602
+ const result = toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch, undefined, params) : await dispatch(toolId, action, params);
1360
1603
 
1361
1604
  // Same rule as the model's path: a resolved `{ok:false}` is a refusal, not
1362
1605
  // a change. Without this a block button (or an approval answered after a
@@ -1398,7 +1641,33 @@ export async function runGatedAction(input) {
1398
1641
  * adapter's same-named actions remain for desktop/MCP, where the ledger is
1399
1642
  * reached over the broker instead.
1400
1643
  */
1401
- async function runAskBuoyAction(action, ledger, dispatch) {
1644
+ async function runAskBuoyAction(action, ledger, dispatch, evidence, params, procedures) {
1645
+ if (action === "retrieve") {
1646
+ return retrieveEvidence(evidence, params ?? {});
1647
+ }
1648
+ if (action === "openProcedure") {
1649
+ const id = typeof params?.id === "string" ? params.id : "";
1650
+ const found = procedures?.find(p => p.id === id);
1651
+ if (!found) {
1652
+ const ids = (procedures ?? []).map(p => p.id);
1653
+ return {
1654
+ ok: false,
1655
+ error: ids.length ? `No procedure "${id}". This app's procedures: ${ids.join(", ")}.` : "This app has no procedures written for it."
1656
+ };
1657
+ }
1658
+ return {
1659
+ id: found.id,
1660
+ title: found.title,
1661
+ ...(found.version ? {
1662
+ version: found.version
1663
+ } : {}),
1664
+ ...(found.requires?.length ? {
1665
+ requires: found.requires
1666
+ } : {}),
1667
+ body: found.body,
1668
+ note: "Follow these steps with the ordinary tools. Every write still goes through the same policy, approval and undo as any other call; a step that is refused stays refused."
1669
+ };
1670
+ }
1402
1671
  if (action === "listChanges") {
1403
1672
  const changes = ledger.list().filter(e => !e.undoneAt).map(e => ({
1404
1673
  toolId: e.toolId,