@buoy-gg/agent-core 7.0.36 → 7.0.39

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/README.md +13 -9
  2. package/lib/commonjs/catalog/catalog.g.js +93 -6
  3. package/lib/commonjs/catalog/catalog.g.js.map +1 -1
  4. package/lib/commonjs/catalog/catalog.source.json +106 -4
  5. package/lib/commonjs/catalog/catalog.types.g.js +3 -3
  6. package/lib/commonjs/catalog/catalog.types.g.js.map +1 -1
  7. package/lib/commonjs/catalog/toProviderTools.js +3 -1
  8. package/lib/commonjs/catalog/toProviderTools.js.map +1 -1
  9. package/lib/commonjs/effects/ledger.js +13 -1
  10. package/lib/commonjs/effects/ledger.js.map +1 -1
  11. package/lib/commonjs/engine/evidence.js +113 -0
  12. package/lib/commonjs/engine/evidence.js.map +1 -0
  13. package/lib/commonjs/engine/historyBudget.js +158 -55
  14. package/lib/commonjs/engine/historyBudget.js.map +1 -1
  15. package/lib/commonjs/engine/retrieve.js +214 -0
  16. package/lib/commonjs/engine/retrieve.js.map +1 -0
  17. package/lib/commonjs/engine/runAgentTurn.js +364 -94
  18. package/lib/commonjs/engine/runAgentTurn.js.map +1 -1
  19. package/lib/commonjs/engine/systemPrompt.js +31 -1
  20. package/lib/commonjs/engine/systemPrompt.js.map +1 -1
  21. package/lib/commonjs/engine/tokenCalibration.js +81 -0
  22. package/lib/commonjs/engine/tokenCalibration.js.map +1 -0
  23. package/lib/commonjs/engine/verify.js +300 -0
  24. package/lib/commonjs/engine/verify.js.map +1 -0
  25. package/lib/commonjs/index.js +16 -0
  26. package/lib/commonjs/index.js.map +1 -1
  27. package/lib/commonjs/providers/anthropic.js +12 -3
  28. package/lib/commonjs/providers/anthropic.js.map +1 -1
  29. package/lib/commonjs/providers/openai.js +11 -3
  30. package/lib/commonjs/providers/openai.js.map +1 -1
  31. package/lib/commonjs/providers/problem.js +98 -0
  32. package/lib/commonjs/providers/problem.js.map +1 -0
  33. package/lib/commonjs/providers/sse.js +6 -5
  34. package/lib/commonjs/providers/sse.js.map +1 -1
  35. package/lib/commonjs/providers/transport.js +20 -4
  36. package/lib/commonjs/providers/transport.js.map +1 -1
  37. package/lib/commonjs/providers/xhrStream.js +3 -0
  38. package/lib/commonjs/providers/xhrStream.js.map +1 -1
  39. package/lib/commonjs/session.js +32 -5
  40. package/lib/commonjs/session.js.map +1 -1
  41. package/lib/module/catalog/catalog.g.js +93 -6
  42. package/lib/module/catalog/catalog.g.js.map +1 -1
  43. package/lib/module/catalog/catalog.source.json +106 -4
  44. package/lib/module/catalog/catalog.types.g.js +3 -3
  45. package/lib/module/catalog/catalog.types.g.js.map +1 -1
  46. package/lib/module/catalog/toProviderTools.js +3 -1
  47. package/lib/module/catalog/toProviderTools.js.map +1 -1
  48. package/lib/module/effects/ledger.js +13 -1
  49. package/lib/module/effects/ledger.js.map +1 -1
  50. package/lib/module/engine/evidence.js +107 -0
  51. package/lib/module/engine/evidence.js.map +1 -0
  52. package/lib/module/engine/historyBudget.js +158 -54
  53. package/lib/module/engine/historyBudget.js.map +1 -1
  54. package/lib/module/engine/retrieve.js +210 -0
  55. package/lib/module/engine/retrieve.js.map +1 -0
  56. package/lib/module/engine/runAgentTurn.js +365 -96
  57. package/lib/module/engine/runAgentTurn.js.map +1 -1
  58. package/lib/module/engine/systemPrompt.js +30 -1
  59. package/lib/module/engine/systemPrompt.js.map +1 -1
  60. package/lib/module/engine/tokenCalibration.js +75 -0
  61. package/lib/module/engine/tokenCalibration.js.map +1 -0
  62. package/lib/module/engine/verify.js +292 -0
  63. package/lib/module/engine/verify.js.map +1 -0
  64. package/lib/module/index.js +2 -0
  65. package/lib/module/index.js.map +1 -1
  66. package/lib/module/providers/anthropic.js +12 -3
  67. package/lib/module/providers/anthropic.js.map +1 -1
  68. package/lib/module/providers/openai.js +11 -3
  69. package/lib/module/providers/openai.js.map +1 -1
  70. package/lib/module/providers/problem.js +92 -0
  71. package/lib/module/providers/problem.js.map +1 -0
  72. package/lib/module/providers/sse.js +6 -5
  73. package/lib/module/providers/sse.js.map +1 -1
  74. package/lib/module/providers/transport.js +20 -4
  75. package/lib/module/providers/transport.js.map +1 -1
  76. package/lib/module/providers/xhrStream.js +3 -0
  77. package/lib/module/providers/xhrStream.js.map +1 -1
  78. package/lib/module/session.js +33 -6
  79. package/lib/module/session.js.map +1 -1
  80. package/lib/typescript/catalog/catalog.g.d.ts +2 -2
  81. package/lib/typescript/catalog/catalog.g.d.ts.map +1 -1
  82. package/lib/typescript/catalog/catalog.types.g.d.ts +3 -3
  83. package/lib/typescript/catalog/catalog.types.g.d.ts.map +1 -1
  84. package/lib/typescript/catalog/toProviderTools.d.ts.map +1 -1
  85. package/lib/typescript/effects/ledger.d.ts +18 -0
  86. package/lib/typescript/effects/ledger.d.ts.map +1 -1
  87. package/lib/typescript/engine/evidence.d.ts +67 -0
  88. package/lib/typescript/engine/evidence.d.ts.map +1 -0
  89. package/lib/typescript/engine/historyBudget.d.ts +31 -8
  90. package/lib/typescript/engine/historyBudget.d.ts.map +1 -1
  91. package/lib/typescript/engine/retrieve.d.ts +32 -0
  92. package/lib/typescript/engine/retrieve.d.ts.map +1 -0
  93. package/lib/typescript/engine/runAgentTurn.d.ts +96 -2
  94. package/lib/typescript/engine/runAgentTurn.d.ts.map +1 -1
  95. package/lib/typescript/engine/systemPrompt.d.ts +39 -0
  96. package/lib/typescript/engine/systemPrompt.d.ts.map +1 -1
  97. package/lib/typescript/engine/tokenCalibration.d.ts +46 -0
  98. package/lib/typescript/engine/tokenCalibration.d.ts.map +1 -0
  99. package/lib/typescript/engine/verify.d.ts +67 -0
  100. package/lib/typescript/engine/verify.d.ts.map +1 -0
  101. package/lib/typescript/index.d.ts +4 -2
  102. package/lib/typescript/index.d.ts.map +1 -1
  103. package/lib/typescript/providers/anthropic.d.ts.map +1 -1
  104. package/lib/typescript/providers/openai.d.ts.map +1 -1
  105. package/lib/typescript/providers/problem.d.ts +50 -0
  106. package/lib/typescript/providers/problem.d.ts.map +1 -0
  107. package/lib/typescript/providers/sse.d.ts +4 -1
  108. package/lib/typescript/providers/sse.d.ts.map +1 -1
  109. package/lib/typescript/providers/transport.d.ts +8 -0
  110. package/lib/typescript/providers/transport.d.ts.map +1 -1
  111. package/lib/typescript/providers/types.d.ts +14 -0
  112. package/lib/typescript/providers/types.d.ts.map +1 -1
  113. package/lib/typescript/providers/xhrStream.d.ts +2 -0
  114. package/lib/typescript/providers/xhrStream.d.ts.map +1 -1
  115. package/lib/typescript/session.d.ts +15 -2
  116. package/lib/typescript/session.d.ts.map +1 -1
  117. package/lib/typescript/types.d.ts +6 -0
  118. package/lib/typescript/types.d.ts.map +1 -1
  119. package/package.json +1 -1
@@ -28,6 +28,10 @@ var _types = require("../blocks/types");
28
28
  var _askGate = require("./askGate");
29
29
  var _textToolCalls = require("./textToolCalls");
30
30
  var _historyBudget = require("./historyBudget");
31
+ var _tokenCalibration = require("./tokenCalibration");
32
+ var _evidence = require("./evidence");
33
+ var _retrieve = require("./retrieve");
34
+ var _verify = require("./verify");
31
35
  /**
32
36
  * One turn: user sentence in, tool calls and an answer out.
33
37
  *
@@ -81,6 +85,73 @@ function callSignature(calls) {
81
85
  const LOOP_WARNING = "[Buoy] You have now made this exact call three times in a row and gotten the same thing back. Repeating it again will not change the answer. Read the result you already have, then either try a different tool or different arguments, or tell the user plainly what you could not get and what you tried.";
82
86
  const DEFAULT_MAX_STEPS = 12;
83
87
  const DEFAULT_TURN_MS = 180_000;
88
+
89
+ /**
90
+ * Retrying a model request — narrowly.
91
+ *
92
+ * A 429 from a shared gateway, a 529 / `overloaded_error` from Anthropic, or
93
+ * a "prompt is too long" 400 used to end the turn with an error card and Try
94
+ * again — and Try again re-sends the QUESTION, restarting an investigation
95
+ * that may already have written things. All three fail before a model has
96
+ * generated anything, so asking again is safe: nothing runs twice, nothing
97
+ * is billed twice. The rule that makes it safe is enforced, not assumed —
98
+ * a request is only retried when its stream produced NO text, tool call or
99
+ * thinking; once anything has come back, the existing no-replay handling
100
+ * stands (see providers/transport.ts on consumed-exactly-once).
101
+ *
102
+ * Throttled: up to two retries, 1 s then 2 s (plus jitter), a `Retry-After`
103
+ * winning when the endpoint sent one. Overflow: one retry, with the history
104
+ * ceiling halved first — the proactive budget is chars ÷ 4 and the provider
105
+ * just said that was optimistic. Both wait inside the turn deadline and stop
106
+ * on the user's Stop. Small numbers on purpose: this is a phone, not a
107
+ * server queue.
108
+ */
109
+ const THROTTLE_RETRIES = 2;
110
+ const OVERFLOW_RETRIES = 1;
111
+ const RETRY_BASE_MS = 1_000;
112
+ const RETRY_JITTER_MS = 250;
113
+ const RETRY_AFTER_CAP_MS = 30_000;
114
+ function sleep(ms, signal) {
115
+ return new Promise(resolve => {
116
+ if (signal?.aborted) return resolve();
117
+ const t = setTimeout(done, ms);
118
+ function done() {
119
+ clearTimeout(t);
120
+ signal?.removeEventListener("abort", done);
121
+ resolve();
122
+ }
123
+ signal?.addEventListener("abort", done, {
124
+ once: true
125
+ });
126
+ });
127
+ }
128
+
129
+ /**
130
+ * Why the turn ended — the machine-readable half of the notices. The text in
131
+ * the notice blocks is unchanged (the eval classifier pins it); this is for a
132
+ * host, the bank and the desktop, which used to have to grep the prose.
133
+ */
134
+
135
+ /** Normalise an answer to its parts, so both call sites read one shape. */
136
+ function readAnswer(answer) {
137
+ if (typeof answer === "boolean") return {
138
+ approved: answer,
139
+ trust: false
140
+ };
141
+ const reason = answer.reason?.trim();
142
+ return {
143
+ approved: answer.approved,
144
+ ...(reason ? {
145
+ reason
146
+ } : {}),
147
+ trust: answer.trust === true && answer.approved
148
+ };
149
+ }
150
+
151
+ /** What the model is told when the user says no — with their note, when they left one. */
152
+ function declinedResult(reason) {
153
+ return reason ? `The user declined this change and said: "${reason}". Do not retry it as proposed; take their note into account and, if a different change would fit, propose that instead.` : "The user declined this change. Do not retry it; ask what they would prefer.";
154
+ }
84
155
  function findDescriptor(catalog, toolId, action) {
85
156
  return catalog.find(t => t.toolId === toolId)?.actions.find(a => a.action === action);
86
157
  }
@@ -95,15 +166,20 @@ function findDescriptor(catalog, toolId, action) {
95
166
  */
96
167
  const URL_PROVENANCE_REJECTION = "Rejected: every image URL must be one you read from a tool result in THIS conversation (images.list, a storage value, a response body). Never invent or remember URLs — read them first, then show them.";
97
168
  const droppedImageNote = n => ` NOTE: ${n} image ${n === 1 ? "URL was" : "URLs were"} dropped — they were not read from a tool result in this conversation, so the card the user is looking at has no picture there. Do not tell them it does; read the URLs and show it again, or say the artwork is missing.`;
98
- function encodeResult(value) {
99
- let text;
169
+
170
+ /** The whole result as text — what the evidence store keeps. */
171
+ function fullText(value) {
100
172
  try {
101
- text = JSON.stringify(value ?? null);
173
+ return JSON.stringify(value ?? null);
102
174
  } catch {
103
- text = String(value);
175
+ return String(value);
104
176
  }
177
+ }
178
+
179
+ /** What the model is sent: the text, cut at the cap with a marker that names where the rest is. */
180
+ function encodeResult(text, ref) {
105
181
  if (text.length <= MAX_RESULT_CHARS) return text;
106
- return `${text.slice(0, MAX_RESULT_CHARS)}\n\n[truncated — ${text.length} characters total. Ask for a narrower slice if you need more.]`;
182
+ return `${text.slice(0, MAX_RESULT_CHARS)}${(0, _evidence.truncationMarker)(text.length, ref)}`;
107
183
  }
108
184
 
109
185
  /** A short line for the transcript row, so the UI never renders raw payloads. */
@@ -172,8 +248,14 @@ async function* runAgentTurn(input) {
172
248
  policy,
173
249
  isRelease,
174
250
  requestApproval,
175
- signal
251
+ signal,
252
+ evidence,
253
+ trusted,
254
+ procedures
176
255
  } = input;
256
+ /** Notes the user left with a decline, by call id — see declinedResult. */
257
+ const declineReasons = new Map();
258
+ const calibration = input.calibration ?? new _tokenCalibration.TokenCalibration();
177
259
  const messages = [...input.messages];
178
260
  const tools = (0, _toProviderTools.toProviderTools)(catalog, {
179
261
  availableToolIds: input.availableToolIds,
@@ -197,7 +279,9 @@ async function* runAgentTurn(input) {
197
279
  * much to leave out of a budget. `maxTokens` is multiplied by four because
198
280
  * the budget is in characters.
199
281
  */
200
- const requestReserve = system.length + (systemVolatile?.length ?? 0) + tools.reduce((n, t) => n + t.description.length + JSON.stringify(t.inputSchema).length, 0) + maxTokens * 4;
282
+ const fixedChars = system.length + (systemVolatile?.length ?? 0) + tools.reduce((n, t) => n + t.description.length + JSON.stringify(t.inputSchema).length, 0);
283
+ /** The reserve at the ratio believed RIGHT NOW — it moves as usage reports come in. */
284
+ const requestReserve = () => fixedChars + calibration.chars(maxTokens);
201
285
  /**
202
286
  * Caps the AGENT's wall-clock, not the user's. Time spent parked on an
203
287
  * approval sheet is pushed onto the deadline as it is spent (see the
@@ -206,6 +290,13 @@ async function* runAgentTurn(input) {
206
290
  * for the turn that applied it.
207
291
  */
208
292
  let deadline = Date.now() + DEFAULT_TURN_MS;
293
+ /** Retries spent this turn, per kind — the budget is per turn, not per step. */
294
+ const retries = {
295
+ throttled: 0,
296
+ overflow: 0
297
+ };
298
+ /** Tightened after an overflow; see budgetForRequest's `cap`. In tokens, so a calibration change re-scales it. */
299
+ let historyCapTokens = _historyBudget.MAX_HISTORY_TOKENS;
209
300
 
210
301
  // Every http(s) URL that appeared in a tool result in this CONVERSATION.
211
302
  // The prompt tells the model image URLs must come from here; this Set is
@@ -251,7 +342,8 @@ async function* runAgentTurn(input) {
251
342
  for (let step = 0; step < maxSteps; step++) {
252
343
  if (signal?.aborted) {
253
344
  yield {
254
- type: "done"
345
+ type: "done",
346
+ stopReason: "stopped"
255
347
  };
256
348
  return messages;
257
349
  }
@@ -278,7 +370,8 @@ async function* runAgentTurn(input) {
278
370
  }
279
371
  };
280
372
  yield {
281
- type: "done"
373
+ type: "done",
374
+ stopReason: "time-cap"
282
375
  };
283
376
  return messages;
284
377
  }
@@ -296,13 +389,14 @@ async function* runAgentTurn(input) {
296
389
  *
297
390
  * The current round is never touched: it is the question being answered.
298
391
  */
299
- const budgeted = (0, _historyBudget.budgetForRequest)(messages, requestReserve);
392
+ const budgeted = (0, _historyBudget.budgetForRequest)(messages, requestReserve(), calibration.chars(historyCapTokens), calibration.charsPerToken, ledger.liveCallIds());
300
393
  if (budgeted.droppedRounds > 0) {
301
394
  messages.length = 0;
302
395
  messages.push(...budgeted.messages);
303
396
  yield {
304
397
  type: "history-trimmed",
305
- droppedRounds: budgeted.droppedRounds
398
+ droppedRounds: budgeted.droppedRounds,
399
+ droppedPinned: budgeted.droppedPinned
306
400
  };
307
401
  } else if (budgeted.messages !== messages) {
308
402
  messages.length = 0;
@@ -328,7 +422,7 @@ async function* runAgentTurn(input) {
328
422
  delta
329
423
  };
330
424
  };
331
- const calls = [];
425
+ let calls = [];
332
426
  /**
333
427
  * Recovered from text rather than emitted as calls — see textToolCalls.ts.
334
428
  * Tracked by id so each one's result can tell the model to stop doing that;
@@ -345,7 +439,7 @@ async function* runAgentTurn(input) {
345
439
  let holding = true;
346
440
  // Carried, never read: Anthropic requires the turn's thinking blocks back
347
441
  // verbatim with its tool results. See providers/anthropic.ts note 5.
348
- const thinking = [];
442
+ let thinking = [];
349
443
  let failed = false;
350
444
  /**
351
445
  * What the provider said about how the stream ended.
@@ -358,80 +452,166 @@ async function* runAgentTurn(input) {
358
452
  let outcome;
359
453
  /** Set for the two kinds that are recoverable rather than a hard failure. */
360
454
  let incomplete;
361
- for await (const ev of provider.send({
362
- messages,
363
- system,
364
- systemVolatile,
365
- tools,
366
- model,
367
- maxTokens,
368
- signal
369
- })) {
370
- if (ev.type === "text") {
371
- text += ev.delta;
372
- if (!holding) {
373
- yield textEvent(ev.delta);
374
- } else if ((0, _textToolCalls.couldBeToolCallEnvelope)(text)) {
375
- held += ev.delta;
376
- } else {
377
- // Not an envelope after all. Release everything at once and stream
378
- // the rest as usual — the reader loses nothing but a few characters
379
- // of latency at the very start of the answer.
380
- holding = false;
381
- held = "";
382
- yield textEvent(text);
383
- }
384
- } else if (ev.type === "tool-call") {
385
- calls.push(ev.call);
386
- } else if (ev.type === "thinking") {
387
- thinking.push(ev.block);
388
- // Surfaced as well as carried: the block goes back to the provider
389
- // verbatim (note above), and the readable half goes to the UI so a
390
- // tester can see WHY a turn did what it did. Redacted blocks have no
391
- // readable half and are carried only.
392
- if (ev.block.type === "thinking" && ev.block.thinking.trim()) {
393
- yield {
394
- type: "reasoning",
395
- text: ev.block.thinking
396
- };
455
+
456
+ /**
457
+ * THE REQUEST, with its retries. One request per pass; a pass that fails
458
+ * before the model generated anything may be sent again (see
459
+ * THROTTLE_RETRIES). Everything the pass accumulated is reset first —
460
+ * there is nothing to keep, by the rule that made the retry safe.
461
+ */
462
+ for (;;) {
463
+ text = "";
464
+ calls = [];
465
+ held = "";
466
+ holding = true;
467
+ thinking = [];
468
+ failed = false;
469
+ outcome = undefined;
470
+ incomplete = undefined;
471
+ /** The failure that ended this pass, when it is one the engine may retry. */
472
+ let retryable;
473
+ /** What this pass sends, in characters — the numerator of the calibration sample. */
474
+ const sentChars = fixedChars + messages.reduce((n, m) => n + (0, _historyBudget.messageSize)(m), 0);
475
+ for await (const ev of provider.send({
476
+ messages,
477
+ system,
478
+ systemVolatile,
479
+ tools,
480
+ model,
481
+ maxTokens,
482
+ signal
483
+ })) {
484
+ if (ev.type === "text") {
485
+ text += ev.delta;
486
+ if (!holding) {
487
+ yield textEvent(ev.delta);
488
+ } else if ((0, _textToolCalls.couldBeToolCallEnvelope)(text)) {
489
+ held += ev.delta;
490
+ } else {
491
+ // Not an envelope after all. Release everything at once and stream
492
+ // the rest as usual — the reader loses nothing but a few characters
493
+ // of latency at the very start of the answer.
494
+ holding = false;
495
+ held = "";
496
+ yield textEvent(text);
497
+ }
498
+ } else if (ev.type === "tool-call") {
499
+ calls.push(ev.call);
500
+ } else if (ev.type === "thinking") {
501
+ thinking.push(ev.block);
502
+ // Surfaced as well as carried: the block goes back to the provider
503
+ // verbatim (note above), and the readable half goes to the UI so a
504
+ // tester can see WHY a turn did what it did. Redacted blocks have no
505
+ // readable half and are carried only.
506
+ if (ev.block.type === "thinking" && ev.block.thinking.trim()) {
507
+ yield {
508
+ type: "reasoning",
509
+ text: ev.block.thinking
510
+ };
511
+ }
512
+ } else if (ev.type === "error") {
513
+ if (ev.kind === "truncated" || ev.kind === "timeout") {
514
+ // Not a failure the user did anything about, and not one Try again
515
+ // can fix — re-sending the QUESTION restarts an investigation that
516
+ // may already have written things. Held back from the error card
517
+ // (which is what draws Try again) and handled below as a recoverable
518
+ // stop with a Continue button.
519
+ incomplete = {
520
+ message: ev.message
521
+ };
522
+ } else if (ev.problem && (ev.problem.kind === "throttled" || ev.problem.kind === "overflow") && text === "" && calls.length === 0 && thinking.length === 0) {
523
+ // Nothing generated, and a kind that a wait or a trim can fix.
524
+ // Decided after the stream closes; the adapter returns on error.
525
+ retryable = {
526
+ problem: ev.problem,
527
+ message: ev.message
528
+ };
529
+ } else {
530
+ // A mid-stream error can arrive on an already-committed 200.
531
+ yield {
532
+ type: "error",
533
+ message: ev.message
534
+ };
535
+ failed = true;
536
+ }
537
+ } else if (ev.type === "done") {
538
+ outcome = ev.outcome;
539
+ if (ev.usage) {
540
+ // The provider just counted this prompt. One sample per request
541
+ // keeps the chars↔tokens ratio honest for the NEXT budget.
542
+ calibration.observe(sentChars, (0, _tokenCalibration.promptTokensOf)(ev.usage, provider.protocol));
543
+ // Surfaced so a host can meter cost per seat. Parsed for a long time;
544
+ // went nowhere anyone could see.
545
+ yield {
546
+ type: "usage",
547
+ ...ev.usage,
548
+ model: ev.model
549
+ };
550
+ }
397
551
  }
398
- } else if (ev.type === "error") {
399
- if (ev.kind === "truncated" || ev.kind === "timeout") {
400
- // Not a failure the user did anything about, and not one Try again
401
- // can fix — re-sending the QUESTION restarts an investigation that
402
- // may already have written things. Held back from the error card
403
- // (which is what draws Try again) and handled below as a recoverable
404
- // stop with a Continue button.
405
- incomplete = {
406
- message: ev.message
407
- };
552
+ }
553
+ if (!retryable) break;
554
+ const kind = retryable.problem.kind;
555
+ const maxAttempts = 1 + (kind === "throttled" ? THROTTLE_RETRIES : OVERFLOW_RETRIES);
556
+ const spent = retries[kind];
557
+ let waitMs = kind === "throttled" ? Math.min(retryable.problem.retryAfterMs ?? RETRY_BASE_MS * 2 ** spent, RETRY_AFTER_CAP_MS) + Math.floor(Math.random() * RETRY_JITTER_MS) : 0;
558
+ let giveUp = spent >= maxAttempts - 1 || signal?.aborted === true || Date.now() + waitMs > deadline;
559
+ if (!giveUp && kind === "overflow") {
560
+ // Halve the ceiling and trim again. If that changes nothing — the
561
+ // current round alone is over the limit — a retry would only repeat
562
+ // the refusal, so report it instead.
563
+ historyCapTokens = Math.floor(historyCapTokens / 2);
564
+ const before = messages.reduce((n, m) => n + (0, _historyBudget.messageSize)(m), 0);
565
+ const again = (0, _historyBudget.budgetForRequest)(messages, requestReserve(), calibration.chars(historyCapTokens), calibration.charsPerToken, ledger.liveCallIds());
566
+ const after = again.messages.reduce((n, m) => n + (0, _historyBudget.messageSize)(m), 0);
567
+ if (after >= before) {
568
+ giveUp = true;
408
569
  } else {
409
- // A mid-stream error can arrive on an already-committed 200.
410
- yield {
411
- type: "error",
412
- message: ev.message
413
- };
414
- failed = true;
415
- }
416
- } else if (ev.type === "done") {
417
- outcome = ev.outcome;
418
- if (ev.usage) {
419
- // Surfaced so a host can meter cost per seat. Parsed for a long time;
420
- // went nowhere anyone could see.
421
- yield {
422
- type: "usage",
423
- ...ev.usage,
424
- model: ev.model
570
+ messages.length = 0;
571
+ messages.push(...again.messages);
572
+ if (again.droppedRounds > 0) yield {
573
+ type: "history-trimmed",
574
+ droppedRounds: again.droppedRounds,
575
+ droppedPinned: again.droppedPinned
425
576
  };
426
577
  }
427
578
  }
579
+ if (giveUp) {
580
+ // Plain words first; the endpoint's own text after, for whoever files
581
+ // the bug. A user reading "prompt is too long: 213000 tokens" has no
582
+ // move; "start a new conversation" is one.
583
+ const plain = kind === "throttled" ? spent > 0 ? `The AI endpoint is rate-limiting requests — tried ${spent + 1} times.` : "The AI endpoint is rate-limiting requests." : "This conversation is too large for the AI endpoint and there was nothing older to trim. Start a new conversation, or ask a shorter question.";
584
+ yield {
585
+ type: "error",
586
+ message: `${plain} (${retryable.message})`
587
+ };
588
+ failed = true;
589
+ break;
590
+ }
591
+ retries[kind] += 1;
592
+ yield {
593
+ type: "retrying",
594
+ reason: kind,
595
+ attempt: retries[kind] + 1,
596
+ maxAttempts,
597
+ inMs: waitMs
598
+ };
599
+ if (waitMs > 0) await sleep(waitMs, signal);
600
+ if (signal?.aborted) {
601
+ yield {
602
+ type: "done",
603
+ stopReason: "stopped"
604
+ };
605
+ return messages;
606
+ }
428
607
  }
429
608
  if (failed) {
430
609
  // Whatever was withheld is the model's own words on the way to an error.
431
610
  // Show it rather than swallowing it.
432
611
  if (held) yield textEvent(held);
433
612
  yield {
434
- type: "done"
613
+ type: "done",
614
+ stopReason: "error"
435
615
  };
436
616
  return messages;
437
617
  }
@@ -487,7 +667,8 @@ async function* runAgentTurn(input) {
487
667
  }
488
668
  };
489
669
  yield {
490
- type: "done"
670
+ type: "done",
671
+ stopReason: "incomplete"
491
672
  };
492
673
  return messages;
493
674
  }
@@ -545,7 +726,8 @@ async function* runAgentTurn(input) {
545
726
  };
546
727
  }
547
728
  yield {
548
- type: "done"
729
+ type: "done",
730
+ stopReason: asked ? "awaiting-user" : "answered"
549
731
  };
550
732
  return messages;
551
733
  }
@@ -698,6 +880,11 @@ async function* runAgentTurn(input) {
698
880
  barrierRun = true;
699
881
  for (const p of planned) {
700
882
  if (!p?.gated || decided.has(p.call.id)) continue;
883
+ if (trusted?.has(`${p.toolId}.${p.action}`)) {
884
+ // Waived for this conversation: runs like any allowed write, no card.
885
+ decided.set(p.call.id, true);
886
+ continue;
887
+ }
701
888
  yield {
702
889
  type: "approval-required",
703
890
  id: p.call.id,
@@ -715,7 +902,7 @@ async function* runAgentTurn(input) {
715
902
  dispatch,
716
903
  catalog
717
904
  });
718
- const approved = requestApproval ? await requestApproval({
905
+ const answer = readAnswer(trusted?.has(`${p.toolId}.${p.action}`) ? true : requestApproval ? await requestApproval({
719
906
  id: p.call.id,
720
907
  toolId: p.toolId,
721
908
  action: p.action,
@@ -724,10 +911,13 @@ async function* runAgentTurn(input) {
724
911
  reason: p.reason,
725
912
  params: p.params,
726
913
  targetDigest
727
- }) : false;
914
+ }) : false);
728
915
  // The user's deliberation is not the agent's runtime.
729
916
  deadline += Date.now() - askedAt;
917
+ const approved = answer.approved;
730
918
  decided.set(p.call.id, approved);
919
+ if (answer.trust) trusted?.add(`${p.toolId}.${p.action}`);
920
+ if (answer.reason) declineReasons.set(p.call.id, answer.reason);
731
921
  if (!approved) {
732
922
  // One refusal ends the PLAN, not just the call. The remaining
733
923
  // changes were proposed together and nothing has run yet, so
@@ -979,7 +1169,7 @@ async function* runAgentTurn(input) {
979
1169
  for (const ev of unrunStep("declined")) yield ev;
980
1170
  results.push({
981
1171
  toolCallId: call.id,
982
- content: decided.get(call.id) === false ? "The user declined this change. Do not retry it; ask what they would prefer." : "Not run — the user declined another change in this same batch, so none of it was applied. Ask what they would prefer before proposing it again.",
1172
+ content: decided.get(call.id) === false ? declinedResult(declineReasons.get(call.id)) : "Not run — the user declined another change in this same batch, so none of it was applied. Ask what they would prefer before proposing it again.",
983
1173
  isError: true
984
1174
  });
985
1175
  continue;
@@ -990,6 +1180,10 @@ async function* runAgentTurn(input) {
990
1180
  // The barrier already put this card up, before the batch's first
991
1181
  // mutation. Asking again would be the same question twice.
992
1182
  approved = decided.get(call.id);
1183
+ } else if (trusted?.has(`${toolId}.${action}`)) {
1184
+ // Waived for this conversation — see RunTurnInput.trusted.
1185
+ approved = true;
1186
+ decided.set(call.id, true);
993
1187
  } else {
994
1188
  // The barrier could not resolve this call (see `planned`), so it was
995
1189
  // never offered. Ask here, exactly as this always did — a gate that
@@ -1014,7 +1208,7 @@ async function* runAgentTurn(input) {
1014
1208
  dispatch,
1015
1209
  catalog
1016
1210
  });
1017
- approved = requestApproval ? await requestApproval({
1211
+ const answer = readAnswer(requestApproval ? await requestApproval({
1018
1212
  id: call.id,
1019
1213
  toolId,
1020
1214
  action,
@@ -1023,11 +1217,14 @@ async function* runAgentTurn(input) {
1023
1217
  reason: verdict.reason,
1024
1218
  params,
1025
1219
  targetDigest
1026
- }) : false;
1220
+ }) : false);
1027
1221
  // The user's deliberation is not the agent's runtime. Give the clock
1028
1222
  // back before anything else can trip the deadline check.
1029
1223
  deadline += Date.now() - askedAt;
1224
+ approved = answer.approved;
1030
1225
  decided.set(call.id, approved);
1226
+ if (answer.trust) trusted?.add(`${toolId}.${action}`);
1227
+ if (answer.reason) declineReasons.set(call.id, answer.reason);
1031
1228
  }
1032
1229
  if (!approved) {
1033
1230
  // The row is the whole point here. Without it the bubble read as a
@@ -1037,7 +1234,7 @@ async function* runAgentTurn(input) {
1037
1234
  for (const ev of unrunStep("declined")) yield ev;
1038
1235
  results.push({
1039
1236
  toolCallId: call.id,
1040
- content: "The user declined this change. Do not retry it; ask what they would prefer.",
1237
+ content: declinedResult(declineReasons.get(call.id)),
1041
1238
  isError: true
1042
1239
  });
1043
1240
  continue;
@@ -1111,17 +1308,30 @@ async function* runAgentTurn(input) {
1111
1308
  // ledger the model asks about lives HERE, not behind an adapter, so
1112
1309
  // routing it through dispatch would only work on a device and would
1113
1310
  // record the undo as a fresh effect.
1114
- const result = action === _snapshotReads.SNAPSHOT_ACTION ? (0, _snapshotReads.projectSnapshot)(toolId, await requireSnapshot(readSnapshot, toolId), params) : toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch) : await (prefetch.get(call.id) ?? dispatch(toolId, action, params));
1311
+ const result = action === _snapshotReads.SNAPSHOT_ACTION ? (0, _snapshotReads.projectSnapshot)(toolId, await requireSnapshot(readSnapshot, toolId), params) : toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch, evidence, params, procedures) : await (prefetch.get(call.id) ?? dispatch(toolId, action, params));
1115
1312
  const cleaned = (0, _redact.redact)((0, _redact.stripSelfTraffic)(result));
1116
1313
  const refused = reportsFailure(result);
1117
- const encodedResult = encodeResult(cleaned);
1314
+
1315
+ // Kept in full — after redaction, never before — so the model can go
1316
+ // back to it. Not a retrieve's own result (a retrieve of a retrieve is
1317
+ // a loop), and not a refusal (nothing to go back to).
1318
+ const text = fullText(cleaned);
1319
+ const ref = evidence && !refused && !(toolId === "ask-buoy" && action === "retrieve") ? evidence.stash({
1320
+ toolId,
1321
+ action,
1322
+ params,
1323
+ capturedAt: Date.now(),
1324
+ text
1325
+ }) : undefined;
1326
+ const encodedResult = encodeResult(text, ref);
1327
+ const traceResult = encodedResult.length > MAX_TRACE_RESULT_CHARS ? `${encodedResult.slice(0, MAX_TRACE_RESULT_CHARS)}…` : encodedResult;
1118
1328
  yield {
1119
1329
  type: "tool-end",
1120
1330
  id: call.id,
1121
1331
  ok: !refused,
1122
1332
  summary: summarise(result),
1123
1333
  durationMs: Date.now() - startedAt,
1124
- result: encodedResult.length > MAX_TRACE_RESULT_CHARS ? `${encodedResult.slice(0, MAX_TRACE_RESULT_CHARS)}…` : encodedResult
1334
+ result: traceResult
1125
1335
  };
1126
1336
 
1127
1337
  // The receipt: what the user sees of this result, built from the
@@ -1145,7 +1355,12 @@ async function* runAgentTurn(input) {
1145
1355
  before: stateBefore,
1146
1356
  after: stateAfter
1147
1357
  });
1148
- if (fx) ledger.record(toolId, action, fx, ledgerGeneration);
1358
+ if (fx) {
1359
+ const entry = ledger.record(toolId, action, fx, ledgerGeneration);
1360
+ // Ties the change to its round, so the budget keeps that round
1361
+ // while the change is live. See LedgerEntry.callId.
1362
+ if (entry) entry.callId = call.id;
1363
+ }
1149
1364
  }
1150
1365
  // No receipt for a call that changed nothing: a "what changed" card
1151
1366
  // under a refusal is the same lie as the ledger entry, drawn bigger.
@@ -1163,10 +1378,36 @@ async function* runAgentTurn(input) {
1163
1378
  block: receipt,
1164
1379
  forToolCallId: call.id
1165
1380
  };
1381
+ /**
1382
+ * THE OUTCOME CHECK. A write's own `{ok:true}` says the adapter
1383
+ * accepted it; this reads the app back and says whether the requested
1384
+ * state is actually there. Only for actions that have a verifier, only
1385
+ * for writes that were not refused, and never itself a write. Its
1386
+ * verdict rides in the tool result as a trailer the model reads, and
1387
+ * on a second tool-end so the row says "verified" / "unverified" /
1388
+ * "check failed". See engine/verify.ts for the three meanings.
1389
+ */
1390
+ const verification = descriptor.effect !== "read" && !refused ? await (0, _verify.verifyOutcome)({
1391
+ toolId,
1392
+ action,
1393
+ params,
1394
+ result,
1395
+ dispatch,
1396
+ signal,
1397
+ after: stateAfter
1398
+ }) : undefined;
1399
+ if (verification) yield {
1400
+ type: "tool-verified",
1401
+ id: call.id,
1402
+ verification
1403
+ };
1166
1404
  for (const url of encodedResult.match(URL_RE) ?? []) seenUrls.add(url);
1167
1405
  results.push({
1168
1406
  toolCallId: call.id,
1169
- content: encodedResult + ownedKeyNote(toolId, descriptor, params, storeKeyOwners) + aliasTrailer + (salvagedIds.has(call.id) ? _textToolCalls.SALVAGE_NOTE : "")
1407
+ content: encodedResult + ownedKeyNote(toolId, descriptor, params, storeKeyOwners) + aliasTrailer + (verification ? (0, _verify.verificationTrailer)(verification) : "") + (salvagedIds.has(call.id) ? _textToolCalls.SALVAGE_NOTE : ""),
1408
+ ...(ref ? {
1409
+ ref
1410
+ } : {})
1170
1411
  });
1171
1412
  if (toolId === "app" && (action === "reloadApp" || action === "reload")) {
1172
1413
  realmWillDie = true;
@@ -1211,7 +1452,8 @@ async function* runAgentTurn(input) {
1211
1452
  blockId: awaiting
1212
1453
  };
1213
1454
  yield {
1214
- type: "done"
1455
+ type: "done",
1456
+ stopReason: "awaiting-user"
1215
1457
  };
1216
1458
  return messages;
1217
1459
  }
@@ -1219,7 +1461,8 @@ async function* runAgentTurn(input) {
1219
1461
  // Anything after this dies with the JS realm. Stop cleanly instead of
1220
1462
  // sending a request whose answer can never arrive.
1221
1463
  yield {
1222
- type: "done"
1464
+ type: "done",
1465
+ stopReason: "realm-died"
1223
1466
  };
1224
1467
  return messages;
1225
1468
  }
@@ -1239,7 +1482,8 @@ async function* runAgentTurn(input) {
1239
1482
  }
1240
1483
  };
1241
1484
  yield {
1242
- type: "done"
1485
+ type: "done",
1486
+ stopReason: "step-cap"
1243
1487
  };
1244
1488
  return messages;
1245
1489
  }
@@ -1364,7 +1608,7 @@ async function runGatedAction(input) {
1364
1608
  };
1365
1609
  }
1366
1610
  try {
1367
- const result = toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch) : await dispatch(toolId, action, params);
1611
+ const result = toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch, undefined, params) : await dispatch(toolId, action, params);
1368
1612
 
1369
1613
  // Same rule as the model's path: a resolved `{ok:false}` is a refusal, not
1370
1614
  // a change. Without this a block button (or an approval answered after a
@@ -1406,7 +1650,33 @@ async function runGatedAction(input) {
1406
1650
  * adapter's same-named actions remain for desktop/MCP, where the ledger is
1407
1651
  * reached over the broker instead.
1408
1652
  */
1409
- async function runAskBuoyAction(action, ledger, dispatch) {
1653
+ async function runAskBuoyAction(action, ledger, dispatch, evidence, params, procedures) {
1654
+ if (action === "retrieve") {
1655
+ return (0, _retrieve.retrieveEvidence)(evidence, params ?? {});
1656
+ }
1657
+ if (action === "openProcedure") {
1658
+ const id = typeof params?.id === "string" ? params.id : "";
1659
+ const found = procedures?.find(p => p.id === id);
1660
+ if (!found) {
1661
+ const ids = (procedures ?? []).map(p => p.id);
1662
+ return {
1663
+ ok: false,
1664
+ error: ids.length ? `No procedure "${id}". This app's procedures: ${ids.join(", ")}.` : "This app has no procedures written for it."
1665
+ };
1666
+ }
1667
+ return {
1668
+ id: found.id,
1669
+ title: found.title,
1670
+ ...(found.version ? {
1671
+ version: found.version
1672
+ } : {}),
1673
+ ...(found.requires?.length ? {
1674
+ requires: found.requires
1675
+ } : {}),
1676
+ body: found.body,
1677
+ note: "Follow these steps with the ordinary tools. Every write still goes through the same policy, approval and undo as any other call; a step that is refused stays refused."
1678
+ };
1679
+ }
1410
1680
  if (action === "listChanges") {
1411
1681
  const changes = ledger.list().filter(e => !e.undoneAt).map(e => ({
1412
1682
  toolId: e.toolId,