@buoy-gg/agent-core 7.0.35 → 7.0.38

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (208) hide show
  1. package/README.md +21 -14
  2. package/lib/commonjs/blocks/receipts.js +57 -1
  3. package/lib/commonjs/blocks/receipts.js.map +1 -1
  4. package/lib/commonjs/blocks/types.js +23 -2
  5. package/lib/commonjs/blocks/types.js.map +1 -1
  6. package/lib/commonjs/blocks/uiTool.js +423 -10
  7. package/lib/commonjs/blocks/uiTool.js.map +1 -1
  8. package/lib/commonjs/catalog/catalog.g.js +213 -48
  9. package/lib/commonjs/catalog/catalog.g.js.map +1 -1
  10. package/lib/commonjs/catalog/catalog.source.json +226 -48
  11. package/lib/commonjs/catalog/catalog.types.g.js +44 -0
  12. package/lib/commonjs/catalog/catalog.types.g.js.map +1 -0
  13. package/lib/commonjs/catalog/normalizeParams.js +333 -0
  14. package/lib/commonjs/catalog/normalizeParams.js.map +1 -0
  15. package/lib/commonjs/catalog/signature.js +68 -0
  16. package/lib/commonjs/catalog/signature.js.map +1 -0
  17. package/lib/commonjs/catalog/snapshotReads.js +13 -6
  18. package/lib/commonjs/catalog/snapshotReads.js.map +1 -1
  19. package/lib/commonjs/catalog/toProviderTools.js +44 -6
  20. package/lib/commonjs/catalog/toProviderTools.js.map +1 -1
  21. package/lib/commonjs/context/buildContextPack.js +367 -19
  22. package/lib/commonjs/context/buildContextPack.js.map +1 -1
  23. package/lib/commonjs/effects/digest.js +34 -0
  24. package/lib/commonjs/effects/digest.js.map +1 -0
  25. package/lib/commonjs/effects/ledger.js +98 -2
  26. package/lib/commonjs/effects/ledger.js.map +1 -1
  27. package/lib/commonjs/engine/askGate.js +287 -0
  28. package/lib/commonjs/engine/askGate.js.map +1 -0
  29. package/lib/commonjs/engine/effectFor.js +82 -1
  30. package/lib/commonjs/engine/effectFor.js.map +1 -1
  31. package/lib/commonjs/engine/evidence.js +113 -0
  32. package/lib/commonjs/engine/evidence.js.map +1 -0
  33. package/lib/commonjs/engine/historyBudget.js +363 -0
  34. package/lib/commonjs/engine/historyBudget.js.map +1 -0
  35. package/lib/commonjs/engine/retrieve.js +214 -0
  36. package/lib/commonjs/engine/retrieve.js.map +1 -0
  37. package/lib/commonjs/engine/runAgentTurn.js +1231 -120
  38. package/lib/commonjs/engine/runAgentTurn.js.map +1 -1
  39. package/lib/commonjs/engine/systemPrompt.js +103 -9
  40. package/lib/commonjs/engine/systemPrompt.js.map +1 -1
  41. package/lib/commonjs/engine/textToolCalls.js +276 -0
  42. package/lib/commonjs/engine/textToolCalls.js.map +1 -0
  43. package/lib/commonjs/engine/tokenCalibration.js +81 -0
  44. package/lib/commonjs/engine/tokenCalibration.js.map +1 -0
  45. package/lib/commonjs/engine/verify.js +300 -0
  46. package/lib/commonjs/engine/verify.js.map +1 -0
  47. package/lib/commonjs/index.js +74 -0
  48. package/lib/commonjs/index.js.map +1 -1
  49. package/lib/commonjs/policy/labels.js +19 -3
  50. package/lib/commonjs/policy/labels.js.map +1 -1
  51. package/lib/commonjs/policy/policy.js +6 -1
  52. package/lib/commonjs/policy/policy.js.map +1 -1
  53. package/lib/commonjs/policy/redact.js +42 -5
  54. package/lib/commonjs/policy/redact.js.map +1 -1
  55. package/lib/commonjs/providers/anthropic.js +340 -208
  56. package/lib/commonjs/providers/anthropic.js.map +1 -1
  57. package/lib/commonjs/providers/openai.js +237 -103
  58. package/lib/commonjs/providers/openai.js.map +1 -1
  59. package/lib/commonjs/providers/problem.js +98 -0
  60. package/lib/commonjs/providers/problem.js.map +1 -0
  61. package/lib/commonjs/providers/sse.js +170 -34
  62. package/lib/commonjs/providers/sse.js.map +1 -1
  63. package/lib/commonjs/providers/streamTimer.js +123 -0
  64. package/lib/commonjs/providers/streamTimer.js.map +1 -0
  65. package/lib/commonjs/providers/transport.js +123 -0
  66. package/lib/commonjs/providers/transport.js.map +1 -0
  67. package/lib/commonjs/providers/xhrStream.js +212 -0
  68. package/lib/commonjs/providers/xhrStream.js.map +1 -0
  69. package/lib/commonjs/session.js +242 -58
  70. package/lib/commonjs/session.js.map +1 -1
  71. package/lib/module/blocks/receipts.js +57 -1
  72. package/lib/module/blocks/receipts.js.map +1 -1
  73. package/lib/module/blocks/types.js +23 -2
  74. package/lib/module/blocks/types.js.map +1 -1
  75. package/lib/module/blocks/uiTool.js +421 -10
  76. package/lib/module/blocks/uiTool.js.map +1 -1
  77. package/lib/module/catalog/catalog.g.js +213 -48
  78. package/lib/module/catalog/catalog.g.js.map +1 -1
  79. package/lib/module/catalog/catalog.source.json +226 -48
  80. package/lib/module/catalog/catalog.types.g.js +40 -0
  81. package/lib/module/catalog/catalog.types.g.js.map +1 -0
  82. package/lib/module/catalog/normalizeParams.js +327 -0
  83. package/lib/module/catalog/normalizeParams.js.map +1 -0
  84. package/lib/module/catalog/signature.js +63 -0
  85. package/lib/module/catalog/signature.js.map +1 -0
  86. package/lib/module/catalog/snapshotReads.js +13 -6
  87. package/lib/module/catalog/snapshotReads.js.map +1 -1
  88. package/lib/module/catalog/toProviderTools.js +44 -6
  89. package/lib/module/catalog/toProviderTools.js.map +1 -1
  90. package/lib/module/context/buildContextPack.js +366 -19
  91. package/lib/module/context/buildContextPack.js.map +1 -1
  92. package/lib/module/effects/digest.js +29 -0
  93. package/lib/module/effects/digest.js.map +1 -0
  94. package/lib/module/effects/ledger.js +98 -2
  95. package/lib/module/effects/ledger.js.map +1 -1
  96. package/lib/module/engine/askGate.js +280 -0
  97. package/lib/module/engine/askGate.js.map +1 -0
  98. package/lib/module/engine/effectFor.js +83 -1
  99. package/lib/module/engine/effectFor.js.map +1 -1
  100. package/lib/module/engine/evidence.js +107 -0
  101. package/lib/module/engine/evidence.js.map +1 -0
  102. package/lib/module/engine/historyBudget.js +355 -0
  103. package/lib/module/engine/historyBudget.js.map +1 -0
  104. package/lib/module/engine/retrieve.js +210 -0
  105. package/lib/module/engine/retrieve.js.map +1 -0
  106. package/lib/module/engine/runAgentTurn.js +1227 -121
  107. package/lib/module/engine/runAgentTurn.js.map +1 -1
  108. package/lib/module/engine/systemPrompt.js +101 -9
  109. package/lib/module/engine/systemPrompt.js.map +1 -1
  110. package/lib/module/engine/textToolCalls.js +270 -0
  111. package/lib/module/engine/textToolCalls.js.map +1 -0
  112. package/lib/module/engine/tokenCalibration.js +75 -0
  113. package/lib/module/engine/tokenCalibration.js.map +1 -0
  114. package/lib/module/engine/verify.js +292 -0
  115. package/lib/module/engine/verify.js.map +1 -0
  116. package/lib/module/index.js +7 -3
  117. package/lib/module/index.js.map +1 -1
  118. package/lib/module/policy/labels.js +19 -3
  119. package/lib/module/policy/labels.js.map +1 -1
  120. package/lib/module/policy/policy.js +6 -1
  121. package/lib/module/policy/policy.js.map +1 -1
  122. package/lib/module/policy/redact.js +42 -5
  123. package/lib/module/policy/redact.js.map +1 -1
  124. package/lib/module/providers/anthropic.js +341 -209
  125. package/lib/module/providers/anthropic.js.map +1 -1
  126. package/lib/module/providers/openai.js +238 -104
  127. package/lib/module/providers/openai.js.map +1 -1
  128. package/lib/module/providers/problem.js +92 -0
  129. package/lib/module/providers/problem.js.map +1 -0
  130. package/lib/module/providers/sse.js +166 -34
  131. package/lib/module/providers/sse.js.map +1 -1
  132. package/lib/module/providers/streamTimer.js +118 -0
  133. package/lib/module/providers/streamTimer.js.map +1 -0
  134. package/lib/module/providers/transport.js +119 -0
  135. package/lib/module/providers/transport.js.map +1 -0
  136. package/lib/module/providers/xhrStream.js +207 -0
  137. package/lib/module/providers/xhrStream.js.map +1 -0
  138. package/lib/module/session.js +226 -60
  139. package/lib/module/session.js.map +1 -1
  140. package/lib/typescript/blocks/receipts.d.ts +5 -0
  141. package/lib/typescript/blocks/receipts.d.ts.map +1 -1
  142. package/lib/typescript/blocks/types.d.ts +23 -2
  143. package/lib/typescript/blocks/types.d.ts.map +1 -1
  144. package/lib/typescript/blocks/uiTool.d.ts +11 -2
  145. package/lib/typescript/blocks/uiTool.d.ts.map +1 -1
  146. package/lib/typescript/catalog/catalog.g.d.ts +2 -2
  147. package/lib/typescript/catalog/catalog.g.d.ts.map +1 -1
  148. package/lib/typescript/catalog/catalog.types.g.d.ts +40 -0
  149. package/lib/typescript/catalog/catalog.types.g.d.ts.map +1 -0
  150. package/lib/typescript/catalog/normalizeParams.d.ts +43 -0
  151. package/lib/typescript/catalog/normalizeParams.d.ts.map +1 -0
  152. package/lib/typescript/catalog/signature.d.ts +29 -0
  153. package/lib/typescript/catalog/signature.d.ts.map +1 -0
  154. package/lib/typescript/catalog/snapshotReads.d.ts.map +1 -1
  155. package/lib/typescript/catalog/toProviderTools.d.ts +28 -15
  156. package/lib/typescript/catalog/toProviderTools.d.ts.map +1 -1
  157. package/lib/typescript/context/buildContextPack.d.ts +24 -2
  158. package/lib/typescript/context/buildContextPack.d.ts.map +1 -1
  159. package/lib/typescript/effects/digest.d.ts +10 -0
  160. package/lib/typescript/effects/digest.d.ts.map +1 -0
  161. package/lib/typescript/effects/ledger.d.ts +73 -0
  162. package/lib/typescript/effects/ledger.d.ts.map +1 -1
  163. package/lib/typescript/engine/askGate.d.ts +29 -0
  164. package/lib/typescript/engine/askGate.d.ts.map +1 -0
  165. package/lib/typescript/engine/effectFor.d.ts +6 -1
  166. package/lib/typescript/engine/effectFor.d.ts.map +1 -1
  167. package/lib/typescript/engine/evidence.d.ts +67 -0
  168. package/lib/typescript/engine/evidence.d.ts.map +1 -0
  169. package/lib/typescript/engine/historyBudget.d.ts +115 -0
  170. package/lib/typescript/engine/historyBudget.d.ts.map +1 -0
  171. package/lib/typescript/engine/retrieve.d.ts +32 -0
  172. package/lib/typescript/engine/retrieve.d.ts.map +1 -0
  173. package/lib/typescript/engine/runAgentTurn.d.ts +179 -3
  174. package/lib/typescript/engine/runAgentTurn.d.ts.map +1 -1
  175. package/lib/typescript/engine/systemPrompt.d.ts +80 -0
  176. package/lib/typescript/engine/systemPrompt.d.ts.map +1 -1
  177. package/lib/typescript/engine/textToolCalls.d.ts +58 -0
  178. package/lib/typescript/engine/textToolCalls.d.ts.map +1 -0
  179. package/lib/typescript/engine/tokenCalibration.d.ts +46 -0
  180. package/lib/typescript/engine/tokenCalibration.d.ts.map +1 -0
  181. package/lib/typescript/engine/verify.d.ts +67 -0
  182. package/lib/typescript/engine/verify.d.ts.map +1 -0
  183. package/lib/typescript/index.d.ts +7 -3
  184. package/lib/typescript/index.d.ts.map +1 -1
  185. package/lib/typescript/policy/labels.d.ts.map +1 -1
  186. package/lib/typescript/policy/policy.d.ts.map +1 -1
  187. package/lib/typescript/policy/redact.d.ts.map +1 -1
  188. package/lib/typescript/providers/anthropic.d.ts.map +1 -1
  189. package/lib/typescript/providers/openai.d.ts +0 -16
  190. package/lib/typescript/providers/openai.d.ts.map +1 -1
  191. package/lib/typescript/providers/problem.d.ts +50 -0
  192. package/lib/typescript/providers/problem.d.ts.map +1 -0
  193. package/lib/typescript/providers/sse.d.ts +86 -2
  194. package/lib/typescript/providers/sse.d.ts.map +1 -1
  195. package/lib/typescript/providers/streamTimer.d.ts +59 -0
  196. package/lib/typescript/providers/streamTimer.d.ts.map +1 -0
  197. package/lib/typescript/providers/transport.d.ts +48 -0
  198. package/lib/typescript/providers/transport.d.ts.map +1 -0
  199. package/lib/typescript/providers/types.d.ts +105 -3
  200. package/lib/typescript/providers/types.d.ts.map +1 -1
  201. package/lib/typescript/providers/xhrStream.d.ts +45 -0
  202. package/lib/typescript/providers/xhrStream.d.ts.map +1 -0
  203. package/lib/typescript/session.d.ts +55 -4
  204. package/lib/typescript/session.d.ts.map +1 -1
  205. package/lib/typescript/types.d.ts +6 -0
  206. package/lib/typescript/types.d.ts.map +1 -1
  207. package/package.json +10 -10
  208. package/LICENSE +0 -58
@@ -3,19 +3,35 @@
3
3
  Object.defineProperty(exports, "__esModule", {
4
4
  value: true
5
5
  });
6
+ Object.defineProperty(exports, "digestOf", {
7
+ enumerable: true,
8
+ get: function () {
9
+ return _digest.digestOf;
10
+ }
11
+ });
12
+ exports.readTargetDigest = readTargetDigest;
6
13
  exports.runAgentTurn = runAgentTurn;
7
14
  exports.runGatedAction = runGatedAction;
8
15
  var _catalog = require("../catalog/catalog.g");
9
16
  var _validateParams = require("../catalog/validateParams");
17
+ var _normalizeParams = require("../catalog/normalizeParams");
10
18
  var _toProviderTools = require("../catalog/toProviderTools");
11
19
  var _snapshotReads = require("../catalog/snapshotReads");
12
20
  var _ledger = require("../effects/ledger");
13
21
  var _policy = require("../policy/policy");
14
22
  var _redact = require("../policy/redact");
15
23
  var _effectFor = require("./effectFor");
24
+ var _digest = require("../effects/digest");
16
25
  var _uiTool = require("../blocks/uiTool");
17
26
  var _receipts = require("../blocks/receipts");
18
27
  var _types = require("../blocks/types");
28
+ var _askGate = require("./askGate");
29
+ var _textToolCalls = require("./textToolCalls");
30
+ var _historyBudget = require("./historyBudget");
31
+ var _tokenCalibration = require("./tokenCalibration");
32
+ var _evidence = require("./evidence");
33
+ var _retrieve = require("./retrieve");
34
+ var _verify = require("./verify");
19
35
  /**
20
36
  * One turn: user sentence in, tool calls and an answer out.
21
37
  *
@@ -43,20 +59,127 @@ const MAX_RESULT_CHARS = 24_000;
43
59
  * store dump per step makes the trace unreadable rather than more useful.
44
60
  */
45
61
  const MAX_TRACE_RESULT_CHARS = 2_000;
62
+ /**
63
+ * How many identical rounds before the model is told it is looping.
64
+ *
65
+ * `maxSteps` alone was the whole answer, and it is the wrong shape: a model
66
+ * that misread a result retries the same call until the cap, then the turn
67
+ * ends with nothing — the user watches twelve identical rows go by and gets
68
+ * no answer. The cap is a backstop, not feedback. Telling the model what it
69
+ * is doing gives it the chance to change approach, and costs one sentence.
70
+ *
71
+ * Signature is name + action + arguments, key-sorted so `{b,a}` and `{a,b}`
72
+ * are the same call. Three, not two: a legitimate retry after a transient
73
+ * failure is normal, and warning on it would be noise.
74
+ */
75
+ const LOOP_REPEATS = 3;
76
+ function callSignature(calls) {
77
+ const sort = v => {
78
+ if (v === null || typeof v !== "object") return v;
79
+ if (Array.isArray(v)) return v.map(sort);
80
+ const o = v;
81
+ return Object.keys(o).sort().reduce((a, k) => (a[k] = sort(o[k]), a), {});
82
+ };
83
+ return calls.map(c => `${c.name}:${JSON.stringify(sort(c.input ?? {}))}`).join("|");
84
+ }
85
+ const LOOP_WARNING = "[Buoy] You have now made this exact call three times in a row and gotten the same thing back. Repeating it again will not change the answer. Read the result you already have, then either try a different tool or different arguments, or tell the user plainly what you could not get and what you tried.";
46
86
  const DEFAULT_MAX_STEPS = 12;
47
87
  const DEFAULT_TURN_MS = 180_000;
88
+
89
+ /**
90
+ * Retrying a model request — narrowly.
91
+ *
92
+ * A 429 from a shared gateway, a 529 / `overloaded_error` from Anthropic, or
93
+ * a "prompt is too long" 400 used to end the turn with an error card and Try
94
+ * again — and Try again re-sends the QUESTION, restarting an investigation
95
+ * that may already have written things. All three fail before a model has
96
+ * generated anything, so asking again is safe: nothing runs twice, nothing
97
+ * is billed twice. The rule that makes it safe is enforced, not assumed —
98
+ * a request is only retried when its stream produced NO text, tool call or
99
+ * thinking; once anything has come back, the existing no-replay handling
100
+ * stands (see providers/transport.ts on consumed-exactly-once).
101
+ *
102
+ * Throttled: up to two retries, 1 s then 2 s (plus jitter), a `Retry-After`
103
+ * winning when the endpoint sent one. Overflow: one retry, with the history
104
+ * ceiling halved first — the proactive budget is chars ÷ 4 and the provider
105
+ * just said that was optimistic. Both wait inside the turn deadline and stop
106
+ * on the user's Stop. Small numbers on purpose: this is a phone, not a
107
+ * server queue.
108
+ */
109
+ const THROTTLE_RETRIES = 2;
110
+ const OVERFLOW_RETRIES = 1;
111
+ const RETRY_BASE_MS = 1_000;
112
+ const RETRY_JITTER_MS = 250;
113
+ const RETRY_AFTER_CAP_MS = 30_000;
114
+ function sleep(ms, signal) {
115
+ return new Promise(resolve => {
116
+ if (signal?.aborted) return resolve();
117
+ const t = setTimeout(done, ms);
118
+ function done() {
119
+ clearTimeout(t);
120
+ signal?.removeEventListener("abort", done);
121
+ resolve();
122
+ }
123
+ signal?.addEventListener("abort", done, {
124
+ once: true
125
+ });
126
+ });
127
+ }
128
+
129
+ /**
130
+ * Why the turn ended — the machine-readable half of the notices. The text in
131
+ * the notice blocks is unchanged (the eval classifier pins it); this is for a
132
+ * host, the bank and the desktop, which used to have to grep the prose.
133
+ */
134
+
135
+ /** Normalise an answer to its parts, so both call sites read one shape. */
136
+ function readAnswer(answer) {
137
+ if (typeof answer === "boolean") return {
138
+ approved: answer,
139
+ trust: false
140
+ };
141
+ const reason = answer.reason?.trim();
142
+ return {
143
+ approved: answer.approved,
144
+ ...(reason ? {
145
+ reason
146
+ } : {}),
147
+ trust: answer.trust === true && answer.approved
148
+ };
149
+ }
150
+
151
+ /** What the model is told when the user says no — with their note, when they left one. */
152
+ function declinedResult(reason) {
153
+ return reason ? `The user declined this change and said: "${reason}". Do not retry it as proposed; take their note into account and, if a different change would fit, propose that instead.` : "The user declined this change. Do not retry it; ask what they would prefer.";
154
+ }
48
155
  function findDescriptor(catalog, toolId, action) {
49
156
  return catalog.find(t => t.toolId === toolId)?.actions.find(a => a.action === action);
50
157
  }
51
- function encodeResult(value) {
52
- let text;
158
+
159
+ /**
160
+ * Why a picture did not render, in the model's own terms.
161
+ *
162
+ * Two messages, not one, because they answer different questions: the first
163
+ * is "your block was refused", the second is "your block was shown but with
164
+ * holes in it". The literal opening of the rejection is pinned by the
165
+ * trajectory classifier (evals/ask-buoy/lib/trajectory.ts) — keep it.
166
+ */
167
+ const URL_PROVENANCE_REJECTION = "Rejected: every image URL must be one you read from a tool result in THIS conversation (images.list, a storage value, a response body). Never invent or remember URLs — read them first, then show them.";
168
+ const droppedImageNote = n => ` NOTE: ${n} image ${n === 1 ? "URL was" : "URLs were"} dropped — they were not read from a tool result in this conversation, so the card the user is looking at has no picture there. Do not tell them it does; read the URLs and show it again, or say the artwork is missing.`;
169
+
170
+ /** The whole result as text — what the evidence store keeps. */
171
+ function fullText(value) {
53
172
  try {
54
- text = JSON.stringify(value ?? null);
173
+ return JSON.stringify(value ?? null);
55
174
  } catch {
56
- text = String(value);
175
+ return String(value);
57
176
  }
177
+ }
178
+
179
+ /** What the model is sent: the text, cut at the cap with a marker that names where the rest is. */
180
+ function encodeResult(text, ref) {
58
181
  if (text.length <= MAX_RESULT_CHARS) return text;
59
- return `${text.slice(0, MAX_RESULT_CHARS)}\n\n[truncated — ${text.length} characters total. Ask for a narrower slice if you need more.]`;
182
+ return `${text.slice(0, MAX_RESULT_CHARS)}${(0, _evidence.truncationMarker)(text.length, ref)}`;
60
183
  }
61
184
 
62
185
  /** A short line for the transcript row, so the UI never renders raw payloads. */
@@ -92,11 +215,30 @@ function summarise(value) {
92
215
  function countOf(n) {
93
216
  return n === 1 ? "1 item" : `${n} items`;
94
217
  }
218
+
219
+ /**
220
+ * A call that came back reporting its own failure.
221
+ *
222
+ * Buoy adapters RESOLVE with `{ok:false, error}` rather than throwing, so the
223
+ * dispatch's try/catch never fires and a refused write looked like a completed
224
+ * one: the row read "Done", and — much worse — the effect ledger recorded a
225
+ * change that had not happened, so the changes bar offered to undo nothing.
226
+ * Caught on device: a typed-edit refusal ("qty is a number, 'many' is text")
227
+ * left the bag untouched and the bar claiming one permanent change.
228
+ *
229
+ * `ok === false` is the same signal `summarise` above already trusts to turn a
230
+ * result into the row's text, so this reads it no more liberally than the UI
231
+ * already does.
232
+ */
233
+ function reportsFailure(value) {
234
+ return typeof value === "object" && value !== null && value.ok === false;
235
+ }
95
236
  async function* runAgentTurn(input) {
96
237
  const {
97
238
  provider,
98
239
  catalog,
99
240
  system,
241
+ systemVolatile,
100
242
  model,
101
243
  maxTokens,
102
244
  dispatch,
@@ -106,8 +248,14 @@ async function* runAgentTurn(input) {
106
248
  policy,
107
249
  isRelease,
108
250
  requestApproval,
109
- signal
251
+ signal,
252
+ evidence,
253
+ trusted,
254
+ procedures
110
255
  } = input;
256
+ /** Notes the user left with a decline, by call id — see declinedResult. */
257
+ const declineReasons = new Map();
258
+ const calibration = input.calibration ?? new _tokenCalibration.TokenCalibration();
111
259
  const messages = [...input.messages];
112
260
  const tools = (0, _toProviderTools.toProviderTools)(catalog, {
113
261
  availableToolIds: input.availableToolIds,
@@ -116,105 +264,697 @@ async function* runAgentTurn(input) {
116
264
  });
117
265
  tools.push((0, _uiTool.uiProviderTool)());
118
266
  const maxSteps = policy.maxSteps ?? DEFAULT_MAX_STEPS;
119
- const deadline = Date.now() + DEFAULT_TURN_MS;
267
+ /**
268
+ * The conversation this turn belongs to. Handed back on every `record` so a
269
+ * dispatch that settles after the user started a new conversation lands in
270
+ * the app (nothing can pull it back) but not in the new chat's undo bar.
271
+ */
272
+ const ledgerGeneration = ledger.generation;
273
+ /**
274
+ * What every request costs before a single message is added: the tool
275
+ * definitions, the system prompt, and the room the model needs to reply.
276
+ *
277
+ * Measured, not guessed — the tool block alone is ~32k characters for a
278
+ * typical install and ~94k with all 24 tools available, which is far too
279
+ * much to leave out of a budget. `maxTokens` is multiplied by four because
280
+ * the budget is in characters.
281
+ */
282
+ const fixedChars = system.length + (systemVolatile?.length ?? 0) + tools.reduce((n, t) => n + t.description.length + JSON.stringify(t.inputSchema).length, 0);
283
+ /** The reserve at the ratio believed RIGHT NOW — it moves as usage reports come in. */
284
+ const requestReserve = () => fixedChars + calibration.chars(maxTokens);
285
+ /**
286
+ * Caps the AGENT's wall-clock, not the user's. Time spent parked on an
287
+ * approval sheet is pushed onto the deadline as it is spent (see the
288
+ * `requestApproval` await below) — otherwise a user who takes three minutes
289
+ * to tap Approve gets the change applied and then "this is taking too long"
290
+ * for the turn that applied it.
291
+ */
292
+ let deadline = Date.now() + DEFAULT_TURN_MS;
293
+ /** Retries spent this turn, per kind — the budget is per turn, not per step. */
294
+ const retries = {
295
+ throttled: 0,
296
+ overflow: 0
297
+ };
298
+ /** Tightened after an overflow; see budgetForRequest's `cap`. In tokens, so a calibration change re-scales it. */
299
+ let historyCapTokens = _historyBudget.MAX_HISTORY_TOKENS;
120
300
 
121
- // Every http(s) URL that appeared in a tool result THIS turn. The prompt
122
- // tells the model image URLs must come from here; this Set is the
123
- // enforcement — a model-remembered or injected URL in an image block is
301
+ // Every http(s) URL that appeared in a tool result in this CONVERSATION.
302
+ // The prompt tells the model image URLs must come from here; this Set is
303
+ // the enforcement — a model-remembered or injected URL in an image block is
124
304
  // both a fabrication vector and a data-exfil beacon (the GET carries
125
- // whatever is encoded into the URL).
126
- const seenUrls = new Set();
305
+ // whatever is encoded into the URL). Session-owned when the caller passes
306
+ // one; see RunTurnInput.seenUrls for why it is not per-turn.
307
+ const seenUrls = input.seenUrls ?? new Set();
127
308
  const URL_RE = /https?:\/\/[^\s"'\\)\]}>]+/g;
309
+
310
+ /** Whether any text has gone out THIS TURN — see `breakBefore`. */
311
+ let answerStarted = false;
312
+
313
+ /**
314
+ * The last few rounds' call signatures. See LOOP_REPEATS — a run of
315
+ * identical rounds gets one warning appended to the results, once, so the
316
+ * model is told rather than silently capped.
317
+ */
318
+ const recentSignatures = [];
319
+ let loopWarned = false;
320
+
321
+ /**
322
+ * Everything the model SAID this turn, in order. Read once, at the end, by
323
+ * the ask gate — a question asked in round one and left hanging is the same
324
+ * dead end as one asked in the last round. See askGate.ts.
325
+ */
326
+ const spoken = [];
327
+ /**
328
+ * Whether the user already has something to tap. An `actions` or
329
+ * `suggestions` block is the model doing this job itself, and a second card
330
+ * asking the same thing underneath it is worse than none.
331
+ *
332
+ * Note this does NOT suppress the shortfall gate below — that one asks
333
+ * whether the user can tap the GAP, which unrelated chips do not answer.
334
+ */
335
+ let offeredTap = false;
336
+ /**
337
+ * Whether this turn actually went and looked — navigated, described the
338
+ * screen, tapped something, refetched. The shortfall gate stays quiet when
339
+ * it did: an agent that tried and came up short has earned its answer.
340
+ */
341
+ let attemptedAcquisition = false;
128
342
  for (let step = 0; step < maxSteps; step++) {
129
343
  if (signal?.aborted) {
130
344
  yield {
131
- type: "done"
345
+ type: "done",
346
+ stopReason: "stopped"
132
347
  };
133
348
  return messages;
134
349
  }
135
350
  if (Date.now() > deadline) {
351
+ // Same shape as the step cap below. This used to be an `error`, which
352
+ // drew the failed-turn card with Try again — and Try again re-sends the
353
+ // question, restarting the whole investigation that just ran out of
354
+ // time. Continue picks up with everything so far still in history.
136
355
  yield {
137
- type: "error",
138
- message: "This is taking too long — stopping here."
356
+ type: "block",
357
+ block: {
358
+ id: `cap${Date.now()}`,
359
+ kind: "notice",
360
+ tone: "warning",
361
+ // Keeps the "This is taking too long" prefix: the eval classifier
362
+ // (evals/ask-buoy/lib/trajectory.ts) tells a time-out from a step
363
+ // cap by it.
364
+ text: `This is taking too long — stopped after ${Math.round(DEFAULT_TURN_MS / 1000)}s. What it found so far is above.`,
365
+ actions: [{
366
+ label: "Continue",
367
+ primary: true,
368
+ send: "Continue where you left off."
369
+ }]
370
+ }
139
371
  };
140
372
  yield {
141
- type: "done"
373
+ type: "done",
374
+ stopReason: "time-cap"
142
375
  };
143
376
  return messages;
144
377
  }
378
+
379
+ /**
380
+ * BUDGET, EVERY REQUEST — not once per turn.
381
+ *
382
+ * The session trims when a turn opens, and that used to be the only
383
+ * check. But a turn is not one request: each step appends an assistant
384
+ * message and a tool-results message, and a single result can be 24,000
385
+ * characters. A twelve-step investigation could therefore add six figures
386
+ * of history AFTER the only check had run, and the turn died on the
387
+ * provider's context limit with an opaque error — in exactly the long
388
+ * sessions the tool is for.
389
+ *
390
+ * The current round is never touched: it is the question being answered.
391
+ */
392
+ const budgeted = (0, _historyBudget.budgetForRequest)(messages, requestReserve(), calibration.chars(historyCapTokens), calibration.charsPerToken, ledger.liveCallIds());
393
+ if (budgeted.droppedRounds > 0) {
394
+ messages.length = 0;
395
+ messages.push(...budgeted.messages);
396
+ yield {
397
+ type: "history-trimmed",
398
+ droppedRounds: budgeted.droppedRounds,
399
+ droppedPinned: budgeted.droppedPinned
400
+ };
401
+ } else if (budgeted.messages !== messages) {
402
+ messages.length = 0;
403
+ messages.push(...budgeted.messages);
404
+ }
145
405
  let text = "";
146
- const calls = [];
406
+ /** See `breakBefore`: set once the first text of THIS round has gone out. */
407
+ let textThisRound = false;
408
+ const textEvent = delta => {
409
+ // Only when this round opens a NEW paragraph of an answer that has
410
+ // already started. A turn whose first words arrive in round three has
411
+ // nothing to be separated from, and marking that would put the flag on
412
+ // events where it means nothing.
413
+ const opensNewRound = answerStarted && !textThisRound;
414
+ textThisRound = true;
415
+ answerStarted = true;
416
+ return opensNewRound ? {
417
+ type: "text",
418
+ delta,
419
+ breakBefore: true
420
+ } : {
421
+ type: "text",
422
+ delta
423
+ };
424
+ };
425
+ let calls = [];
426
+ /**
427
+ * Recovered from text rather than emitted as calls — see textToolCalls.ts.
428
+ * Tracked by id so each one's result can tell the model to stop doing that;
429
+ * they are otherwise ordinary calls and get every check the others get.
430
+ */
431
+ const salvagedIds = new Set();
432
+ /**
433
+ * Text withheld from the screen while it could still be a tool-call
434
+ * envelope. Streamed text cannot be un-shown, so the choice has to be made
435
+ * before it leaves — and a model that writes its call as JSON must not
436
+ * have that JSON become the answer the user reads.
437
+ */
438
+ let held = "";
439
+ let holding = true;
147
440
  // Carried, never read: Anthropic requires the turn's thinking blocks back
148
441
  // verbatim with its tool results. See providers/anthropic.ts note 5.
149
- const thinking = [];
442
+ let thinking = [];
150
443
  let failed = false;
151
- for await (const ev of provider.send({
152
- messages,
153
- system,
154
- tools,
155
- model,
156
- maxTokens,
157
- signal
158
- })) {
159
- if (ev.type === "text") {
160
- text += ev.delta;
161
- yield {
162
- type: "text",
163
- delta: ev.delta
164
- };
165
- } else if (ev.type === "tool-call") {
166
- calls.push(ev.call);
167
- } else if (ev.type === "thinking") {
168
- thinking.push(ev.block);
169
- // Surfaced as well as carried: the block goes back to the provider
170
- // verbatim (note above), and the readable half goes to the UI so a
171
- // tester can see WHY a turn did what it did. Redacted blocks have no
172
- // readable half and are carried only.
173
- if (ev.block.type === "thinking" && ev.block.thinking.trim()) {
174
- yield {
175
- type: "reasoning",
176
- text: ev.block.thinking
444
+ /**
445
+ * What the provider said about how the stream ended.
446
+ *
447
+ * Undefined means it never said — a bare EOF, or a host-supplied provider
448
+ * whose generator simply returned. Both are treated as incomplete, so the
449
+ * fail-safe direction is the default rather than something each adapter
450
+ * has to remember to opt into.
451
+ */
452
+ let outcome;
453
+ /** Set for the two kinds that are recoverable rather than a hard failure. */
454
+ let incomplete;
455
+
456
+ /**
457
+ * THE REQUEST, with its retries. One request per pass; a pass that fails
458
+ * before the model generated anything may be sent again (see
459
+ * THROTTLE_RETRIES). Everything the pass accumulated is reset first —
460
+ * there is nothing to keep, by the rule that made the retry safe.
461
+ */
462
+ for (;;) {
463
+ text = "";
464
+ calls = [];
465
+ held = "";
466
+ holding = true;
467
+ thinking = [];
468
+ failed = false;
469
+ outcome = undefined;
470
+ incomplete = undefined;
471
+ /** The failure that ended this pass, when it is one the engine may retry. */
472
+ let retryable;
473
+ /** What this pass sends, in characters — the numerator of the calibration sample. */
474
+ const sentChars = fixedChars + messages.reduce((n, m) => n + (0, _historyBudget.messageSize)(m), 0);
475
+ for await (const ev of provider.send({
476
+ messages,
477
+ system,
478
+ systemVolatile,
479
+ tools,
480
+ model,
481
+ maxTokens,
482
+ signal
483
+ })) {
484
+ if (ev.type === "text") {
485
+ text += ev.delta;
486
+ if (!holding) {
487
+ yield textEvent(ev.delta);
488
+ } else if ((0, _textToolCalls.couldBeToolCallEnvelope)(text)) {
489
+ held += ev.delta;
490
+ } else {
491
+ // Not an envelope after all. Release everything at once and stream
492
+ // the rest as usual — the reader loses nothing but a few characters
493
+ // of latency at the very start of the answer.
494
+ holding = false;
495
+ held = "";
496
+ yield textEvent(text);
497
+ }
498
+ } else if (ev.type === "tool-call") {
499
+ calls.push(ev.call);
500
+ } else if (ev.type === "thinking") {
501
+ thinking.push(ev.block);
502
+ // Surfaced as well as carried: the block goes back to the provider
503
+ // verbatim (note above), and the readable half goes to the UI so a
504
+ // tester can see WHY a turn did what it did. Redacted blocks have no
505
+ // readable half and are carried only.
506
+ if (ev.block.type === "thinking" && ev.block.thinking.trim()) {
507
+ yield {
508
+ type: "reasoning",
509
+ text: ev.block.thinking
510
+ };
511
+ }
512
+ } else if (ev.type === "error") {
513
+ if (ev.kind === "truncated" || ev.kind === "timeout") {
514
+ // Not a failure the user did anything about, and not one Try again
515
+ // can fix — re-sending the QUESTION restarts an investigation that
516
+ // may already have written things. Held back from the error card
517
+ // (which is what draws Try again) and handled below as a recoverable
518
+ // stop with a Continue button.
519
+ incomplete = {
520
+ message: ev.message
521
+ };
522
+ } else if (ev.problem && (ev.problem.kind === "throttled" || ev.problem.kind === "overflow") && text === "" && calls.length === 0 && thinking.length === 0) {
523
+ // Nothing generated, and a kind that a wait or a trim can fix.
524
+ // Decided after the stream closes; the adapter returns on error.
525
+ retryable = {
526
+ problem: ev.problem,
527
+ message: ev.message
528
+ };
529
+ } else {
530
+ // A mid-stream error can arrive on an already-committed 200.
531
+ yield {
532
+ type: "error",
533
+ message: ev.message
534
+ };
535
+ failed = true;
536
+ }
537
+ } else if (ev.type === "done") {
538
+ outcome = ev.outcome;
539
+ if (ev.usage) {
540
+ // The provider just counted this prompt. One sample per request
541
+ // keeps the chars↔tokens ratio honest for the NEXT budget.
542
+ calibration.observe(sentChars, (0, _tokenCalibration.promptTokensOf)(ev.usage, provider.protocol));
543
+ // Surfaced so a host can meter cost per seat. Parsed for a long time;
544
+ // went nowhere anyone could see.
545
+ yield {
546
+ type: "usage",
547
+ ...ev.usage,
548
+ model: ev.model
549
+ };
550
+ }
551
+ }
552
+ }
553
+ if (!retryable) break;
554
+ const kind = retryable.problem.kind;
555
+ const maxAttempts = 1 + (kind === "throttled" ? THROTTLE_RETRIES : OVERFLOW_RETRIES);
556
+ const spent = retries[kind];
557
+ let waitMs = kind === "throttled" ? Math.min(retryable.problem.retryAfterMs ?? RETRY_BASE_MS * 2 ** spent, RETRY_AFTER_CAP_MS) + Math.floor(Math.random() * RETRY_JITTER_MS) : 0;
558
+ let giveUp = spent >= maxAttempts - 1 || signal?.aborted === true || Date.now() + waitMs > deadline;
559
+ if (!giveUp && kind === "overflow") {
560
+ // Halve the ceiling and trim again. If that changes nothing — the
561
+ // current round alone is over the limit — a retry would only repeat
562
+ // the refusal, so report it instead.
563
+ historyCapTokens = Math.floor(historyCapTokens / 2);
564
+ const before = messages.reduce((n, m) => n + (0, _historyBudget.messageSize)(m), 0);
565
+ const again = (0, _historyBudget.budgetForRequest)(messages, requestReserve(), calibration.chars(historyCapTokens), calibration.charsPerToken, ledger.liveCallIds());
566
+ const after = again.messages.reduce((n, m) => n + (0, _historyBudget.messageSize)(m), 0);
567
+ if (after >= before) {
568
+ giveUp = true;
569
+ } else {
570
+ messages.length = 0;
571
+ messages.push(...again.messages);
572
+ if (again.droppedRounds > 0) yield {
573
+ type: "history-trimmed",
574
+ droppedRounds: again.droppedRounds,
575
+ droppedPinned: again.droppedPinned
177
576
  };
178
577
  }
179
- } else if (ev.type === "error") {
180
- // A mid-stream error can arrive on an already-committed 200.
578
+ }
579
+ if (giveUp) {
580
+ // Plain words first; the endpoint's own text after, for whoever files
581
+ // the bug. A user reading "prompt is too long: 213000 tokens" has no
582
+ // move; "start a new conversation" is one.
583
+ const plain = kind === "throttled" ? spent > 0 ? `The AI endpoint is rate-limiting requests — tried ${spent + 1} times.` : "The AI endpoint is rate-limiting requests." : "This conversation is too large for the AI endpoint and there was nothing older to trim. Start a new conversation, or ask a shorter question.";
181
584
  yield {
182
585
  type: "error",
183
- message: ev.message
586
+ message: `${plain} (${retryable.message})`
184
587
  };
185
588
  failed = true;
186
- } else if (ev.type === "done" && ev.usage) {
187
- // Surfaced so a host can meter cost per seat. Parsed for a long time;
188
- // went nowhere anyone could see.
589
+ break;
590
+ }
591
+ retries[kind] += 1;
592
+ yield {
593
+ type: "retrying",
594
+ reason: kind,
595
+ attempt: retries[kind] + 1,
596
+ maxAttempts,
597
+ inMs: waitMs
598
+ };
599
+ if (waitMs > 0) await sleep(waitMs, signal);
600
+ if (signal?.aborted) {
189
601
  yield {
190
- type: "usage",
191
- input: ev.usage.input,
192
- output: ev.usage.output
602
+ type: "done",
603
+ stopReason: "stopped"
193
604
  };
605
+ return messages;
194
606
  }
195
607
  }
196
608
  if (failed) {
609
+ // Whatever was withheld is the model's own words on the way to an error.
610
+ // Show it rather than swallowing it.
611
+ if (held) yield textEvent(held);
612
+ yield {
613
+ type: "done",
614
+ stopReason: "error"
615
+ };
616
+ return messages;
617
+ }
618
+
619
+ /**
620
+ * THE COMPLETION GATE. Nothing below this runs a tool unless the provider
621
+ * said the response was whole.
622
+ *
623
+ * The failure it exists for is not theoretical and does not look like a
624
+ * failure: a connection cut after the model emitted
625
+ * `{"key":"cart","value":[]}` leaves valid JSON, a plausible-looking plan,
626
+ * and no error anywhere. The old code parsed those arguments, dispatched
627
+ * the write, sent the result back and carried on — a real mutation from
628
+ * half a sentence. Bare EOF is not consent.
629
+ *
630
+ * `output-limited` is the same shape from a different cause: the model hit
631
+ * max_tokens partway through planning. What it managed to SAY is real and
632
+ * is shown; what it was part-way through DOING is not run.
633
+ */
634
+ if (!incomplete && outcome !== "completed") {
635
+ incomplete = {
636
+ message: outcome === "output-limited" ? calls.length ? "The model ran out of room mid-plan, so nothing was run. What it found so far is above." : "The model ran out of room before finishing this answer." : "The answer ended before the endpoint said it was finished, so nothing was run."
637
+ };
638
+ }
639
+ if (incomplete) {
640
+ // Held text is the model's own words on the way out. Show it: the user
641
+ // watched it arrive and hiding it now reads as the app losing work.
642
+ if (held) yield textEvent(held);
643
+ /**
644
+ * Text only — never the tool calls, and never the thinking.
645
+ *
646
+ * An assistant turn carrying `tool_use` with no matching `tool_result`
647
+ * is an instant 400 on the next request, and these calls are exactly the
648
+ * ones that must not be answered. Keeping the text preserves what the
649
+ * user can see on screen for whatever comes next.
650
+ */
651
+ if (text.trim()) messages.push({
652
+ role: "assistant",
653
+ text
654
+ });
197
655
  yield {
198
- type: "done"
656
+ type: "block",
657
+ block: {
658
+ id: `cut${Date.now()}`,
659
+ kind: "notice",
660
+ tone: "warning",
661
+ text: incomplete.message,
662
+ actions: [{
663
+ label: "Continue",
664
+ primary: true,
665
+ send: "Continue where you left off."
666
+ }]
667
+ }
668
+ };
669
+ yield {
670
+ type: "done",
671
+ stopReason: "incomplete"
199
672
  };
200
673
  return messages;
201
674
  }
675
+ if (held) {
676
+ // A model that wrote its call out as JSON instead of calling it. Run it,
677
+ // and show its `reply` — never the blob. See textToolCalls.ts.
678
+ const salvaged = calls.length === 0 ? (0, _textToolCalls.parseTextToolCalls)(held, catalog) : undefined;
679
+ if (salvaged) {
680
+ for (const call of salvaged.calls) {
681
+ calls.push(call);
682
+ salvagedIds.add(call.id);
683
+ }
684
+ text = salvaged.reply;
685
+ if (salvaged.reply) yield textEvent(salvaged.reply);
686
+ } else {
687
+ yield textEvent(held);
688
+ }
689
+ held = "";
690
+ }
202
691
  messages.push({
203
692
  role: "assistant",
204
693
  text,
205
694
  toolCalls: calls.length ? calls : undefined,
206
695
  thinking: thinking.length ? thinking : undefined
207
696
  });
697
+ if (text.trim()) spoken.push(text.trim());
208
698
  if (calls.length === 0) {
699
+ // The turn is over and the model wrote its own words. Two ways those
700
+ // words end badly, and they are different failures — see askGate.ts.
701
+ const said = spoken.join("\n\n");
702
+
703
+ // 1. It answered a smaller question than it was asked and handed the
704
+ // rest back. Fires even when chips were offered: the question is
705
+ // whether the user can tap the GAP, not whether they can tap.
706
+ const gap = (0, _askGate.synthesizeShortfallBlock)({
707
+ text: said,
708
+ attempted: attemptedAcquisition
709
+ });
710
+ if (gap) yield {
711
+ type: "block",
712
+ block: gap
713
+ };
714
+
715
+ // 2. It ended waiting on the user in prose. Suppressed once anything
716
+ // tappable is on screen, the gap button included.
717
+ const asked = offeredTap || gap ? null : (0, _askGate.synthesizeAskBlock)(said);
718
+ if (asked) {
719
+ yield {
720
+ type: "block",
721
+ block: asked
722
+ };
723
+ yield {
724
+ type: "awaiting-user",
725
+ blockId: asked.id
726
+ };
727
+ }
209
728
  yield {
210
- type: "done"
729
+ type: "done",
730
+ stopReason: asked ? "awaiting-user" : "answered"
211
731
  };
212
732
  return messages;
213
733
  }
214
734
  const results = [];
215
735
  let realmWillDie = false;
736
+
737
+ /**
738
+ * Reads, already in flight.
739
+ *
740
+ * The loop below stays strictly serial — every yield, every ledger entry
741
+ * and every approval keeps its order — but a round of independent READS
742
+ * used to pay each round trip end to end. The prompt tells the model to
743
+ * chain reads freely ("reading is cheap; do it"), and a three-read round
744
+ * cost three times the device latency for no reason.
745
+ *
746
+ * ONLY when the whole batch is safe to overlap: every call a catalog read,
747
+ * none needing approval, no `buoy_ui`, nothing snapshot-only or
748
+ * session-local. One write, one approval or one unknown action in the
749
+ * batch and nothing is prefetched — the mixed case is where ordering
750
+ * actually matters, and it is not worth the risk for the latency.
751
+ */
752
+ const prefetch = new Map();
753
+ if (calls.length > 1) {
754
+ const safe = calls.every(c => {
755
+ if (c.name === _uiTool.UI_TOOL_NAME) return false;
756
+ const id = (0, _toProviderTools.fromProviderToolName)(c.name, catalog);
757
+ if (!id || id === "ask-buoy") return false;
758
+ const act = c.input?.action;
759
+ if (typeof act !== "string" || act === _snapshotReads.SNAPSHOT_ACTION) return false;
760
+ const d = findDescriptor(catalog, id, act);
761
+ if (!d || d.effect !== "read") return false;
762
+ return (0, _policy.decide)({
763
+ descriptor: d,
764
+ toolId: id,
765
+ policy,
766
+ isRelease
767
+ }).verdict === "allow";
768
+ });
769
+ if (safe) {
770
+ for (const c of calls) {
771
+ const id = (0, _toProviderTools.fromProviderToolName)(c.name, catalog);
772
+ const act = String(c.input.action);
773
+ const raw = {
774
+ ...(c.input.params ?? {})
775
+ };
776
+ const d = findDescriptor(catalog, id, act);
777
+ const norm = (0, _normalizeParams.normalizeParams)(id, act, raw);
778
+ const withDefaults = (0, _validateParams.applyDefaults)(norm.params, d.params);
779
+ if (!(0, _validateParams.validateParams)(withDefaults, d.params).ok) continue;
780
+ // Rejections are swallowed here and re-awaited in the loop, where
781
+ // they are turned into the same tool_result they always were.
782
+ const p = Promise.resolve(dispatch(id, act, withDefaults)).catch(e => {
783
+ throw e;
784
+ });
785
+ p.catch(() => {});
786
+ prefetch.set(c.id, p);
787
+ }
788
+ }
789
+ }
216
790
  let awaiting = null;
791
+
792
+ /**
793
+ * A call Stop got to first — as a row, so the transcript says which ones.
794
+ *
795
+ * Resolved leniently: the call has not been validated yet, and a label
796
+ * that falls back to `tool.action` is better than no row for a call the
797
+ * user needs to know did not run.
798
+ */
799
+ const stoppedStep = call => {
800
+ const toolId = (0, _toProviderTools.fromProviderToolName)(call.name, catalog) ?? call.name;
801
+ const action = typeof call.input?.action === "string" ? call.input.action : "";
802
+ const descriptor = action ? findDescriptor(catalog, toolId, action) : undefined;
803
+ const params = call.input?.params ?? {};
804
+ return [{
805
+ type: "tool-start",
806
+ id: call.id,
807
+ toolId,
808
+ action,
809
+ label: descriptor ? (0, _policy.describeCall)(toolId, descriptor, params) : `${toolId}.${action}`,
810
+ description: descriptor?.summary ?? "",
811
+ params,
812
+ effect: descriptor?.effect ?? "read"
813
+ }, {
814
+ type: "tool-end",
815
+ id: call.id,
816
+ ok: false,
817
+ summary: "stopped",
818
+ durationMs: 0
819
+ }];
820
+ };
821
+
822
+ /**
823
+ * THE BATCH BARRIER — what the whole batch will do, decided before any of
824
+ * it does anything.
825
+ *
826
+ * A model answers with several calls at once, and they used to run one at
827
+ * a time with each approval asked when its own call came up. So
828
+ * `[set the flag, delete the cache]` wrote the flag, THEN asked about the
829
+ * delete — and a user who declined had already changed the app, with the
830
+ * bubble reading "changed the flag" directly above "left the cache alone".
831
+ * Declining is supposed to mean the plan does not happen.
832
+ *
833
+ * The fix is not to ask again later or to refuse afterwards, neither of
834
+ * which can unrun a write. It is to ask FIRST: every gated call in the
835
+ * batch is put to the user before the batch's first mutation dispatches,
836
+ * so consent is given with the whole plan visible.
837
+ *
838
+ * Resolution here is pure and duplicates the loop's own — deliberately.
839
+ * Anything it cannot resolve (an unknown tool, a bad shape) comes back as
840
+ * neither mutating nor gated, so the loop reports it exactly as it always
841
+ * did and the barrier simply does not fire. Failing back to today's
842
+ * behaviour is the right failure for a safety gate to have.
843
+ */
844
+ const planned = calls.map(call => {
845
+ if (call.name === _uiTool.UI_TOOL_NAME) return undefined;
846
+ const toolId = (0, _toProviderTools.fromProviderToolName)(call.name, catalog);
847
+ const action = typeof call.input?.action === "string" ? call.input.action : undefined;
848
+ if (!toolId || !action) return undefined;
849
+ const descriptor = findDescriptor(catalog, toolId, action);
850
+ if (!descriptor) return undefined;
851
+ const raw = call.input.params ?? {};
852
+ const params = (0, _validateParams.applyDefaults)((0, _normalizeParams.normalizeParams)(toolId, action, raw).params, descriptor.params);
853
+ if (!(0, _validateParams.validateParams)(params, descriptor.params).ok) return undefined;
854
+ const verdict = (0, _policy.decide)({
855
+ descriptor,
856
+ toolId,
857
+ policy,
858
+ isRelease
859
+ });
860
+ return {
861
+ call,
862
+ toolId,
863
+ action,
864
+ descriptor,
865
+ params,
866
+ label: (0, _policy.describeCall)(toolId, descriptor, params),
867
+ mutates: descriptor.effect !== "read",
868
+ gated: verdict.verdict === "needs-approval",
869
+ reason: verdict.verdict === "needs-approval" ? verdict.reason : ""
870
+ };
871
+ });
872
+ const gatedInBatch = planned.some(p => p?.gated);
873
+ /** Decisions the barrier has already taken, so no call is asked twice. */
874
+ const decided = new Map();
875
+ let barrierRun = false;
876
+ let batchDeclined = false;
877
+
878
+ /** Put every gated call in the batch to the user, in the order written. */
879
+ async function* runBarrier() {
880
+ barrierRun = true;
881
+ for (const p of planned) {
882
+ if (!p?.gated || decided.has(p.call.id)) continue;
883
+ if (trusted?.has(`${p.toolId}.${p.action}`)) {
884
+ // Waived for this conversation: runs like any allowed write, no card.
885
+ decided.set(p.call.id, true);
886
+ continue;
887
+ }
888
+ yield {
889
+ type: "approval-required",
890
+ id: p.call.id,
891
+ toolId: p.toolId,
892
+ action: p.action,
893
+ label: p.label,
894
+ description: p.descriptor.summary,
895
+ reason: p.reason
896
+ };
897
+ const askedAt = Date.now();
898
+ const targetDigest = await readTargetDigest({
899
+ toolId: p.toolId,
900
+ action: p.action,
901
+ params: p.params,
902
+ dispatch,
903
+ catalog
904
+ });
905
+ const answer = readAnswer(trusted?.has(`${p.toolId}.${p.action}`) ? true : requestApproval ? await requestApproval({
906
+ id: p.call.id,
907
+ toolId: p.toolId,
908
+ action: p.action,
909
+ label: p.label,
910
+ description: p.descriptor.summary,
911
+ reason: p.reason,
912
+ params: p.params,
913
+ targetDigest
914
+ }) : false);
915
+ // The user's deliberation is not the agent's runtime.
916
+ deadline += Date.now() - askedAt;
917
+ const approved = answer.approved;
918
+ decided.set(p.call.id, approved);
919
+ if (answer.trust) trusted?.add(`${p.toolId}.${p.action}`);
920
+ if (answer.reason) declineReasons.set(p.call.id, answer.reason);
921
+ if (!approved) {
922
+ // One refusal ends the PLAN, not just the call. The remaining
923
+ // changes were proposed together and nothing has run yet, so
924
+ // carrying on with the rest would apply half of something the user
925
+ // has just turned down.
926
+ batchDeclined = true;
927
+ return;
928
+ }
929
+ }
930
+ }
217
931
  for (const call of calls) {
932
+ /**
933
+ * Stop was pressed while this batch was running.
934
+ *
935
+ * The abort signal used to be read only at the top of a ROUND, so a
936
+ * response carrying three calls ran all three after the tap — and the
937
+ * bubble then said "Stopped." over two writes the user believed it had
938
+ * prevented. Each call now checks before it starts. A dispatch already
939
+ * in flight is left to finish (nothing can pull a store write back out
940
+ * of an adapter) and keeps its receipt; everything after it is reported
941
+ * as not run, each as its own row, so the transcript says exactly which
942
+ * changes landed. A result is still pushed for every call — the provider
943
+ * requires one per tool_use or the next request is rejected.
944
+ */
945
+ if (signal?.aborted) {
946
+ // A `buoy_ui` card that was never shown is not a step; no row for it.
947
+ if (call.name !== _uiTool.UI_TOOL_NAME) {
948
+ for (const ev of stoppedStep(call)) yield ev;
949
+ }
950
+ results.push({
951
+ toolCallId: call.id,
952
+ content: "Not run — the user pressed Stop before this call started.",
953
+ isError: true
954
+ });
955
+ continue;
956
+ }
957
+
218
958
  // `buoy_ui`: show or ask. Validated like any action; an Ask block ends
219
959
  // the turn after this batch — the answer comes back as a user message.
220
960
  if (call.name === _uiTool.UI_TOOL_NAME) {
@@ -222,19 +962,35 @@ async function* runAgentTurn(input) {
222
962
  if (!built.ok || !built.block) {
223
963
  results.push({
224
964
  toolCallId: call.id,
225
- content: `Invalid ${_uiTool.UI_TOOL_NAME} block:\n${built.errors.join("\n")}`,
965
+ /**
966
+ * The trailing sentence is the expensive half.
967
+ *
968
+ * A model that gets a block bounced rewrites its WHOLE answer on
969
+ * the retry, and one assistant bubble spans every round of a turn
970
+ * (see `breakBefore`), so the user reads the same paragraph again
971
+ * — measured at four near-identical closings from three schema
972
+ * misses in a row on one live turn. The block is wrong; the
973
+ * sentence under it was fine.
974
+ */
975
+ content: `Invalid ${_uiTool.UI_TOOL_NAME} block:\n${built.errors.join("\n")}\n\nResend ONLY the corrected block. Do not rewrite your answer — whatever you already said this turn has been shown to the user, and saying it again repeats it on their screen.`,
226
976
  isError: true
227
977
  });
228
978
  continue;
229
979
  }
230
980
  // Provenance for pictures, ENFORCED: an image URL the model did not
231
981
  // read from a tool result this turn does not render. Live or nothing.
982
+ // Dropping pictures SILENTLY is its own bug: the model then writes
983
+ // "here they are, with artwork" over a card that has none, and the
984
+ // user is told something untrue about their own screen. Every drop
985
+ // comes back in the tool result so the next sentence can be honest.
986
+ let droppedImages = 0;
232
987
  if (built.block.kind === "imageGrid") {
233
988
  const kept = built.block.images.filter(img => seenUrls.has(img.url));
989
+ droppedImages = built.block.images.length - kept.length;
234
990
  if (kept.length === 0) {
235
991
  results.push({
236
992
  toolCallId: call.id,
237
- content: "Rejected: every image URL must be one you read from a tool result THIS turn (images.list, a storage value, a response body). Never invent or remember URLs — read them first.",
993
+ content: URL_PROVENANCE_REJECTION,
238
994
  isError: true
239
995
  });
240
996
  continue;
@@ -245,12 +1001,17 @@ async function* runAgentTurn(input) {
245
1001
  };
246
1002
  }
247
1003
  if (built.block.kind === "list") {
248
- built.block = {
249
- ...built.block,
250
- items: built.block.items.map(item => item.image && !seenUrls.has(item.image) ? {
1004
+ const items = built.block.items.map(item => {
1005
+ if (!item.image || seenUrls.has(item.image)) return item;
1006
+ droppedImages += 1;
1007
+ return {
251
1008
  ...item,
252
1009
  image: undefined
253
- } : item)
1010
+ };
1011
+ });
1012
+ built.block = {
1013
+ ...built.block,
1014
+ items
254
1015
  };
255
1016
  }
256
1017
  yield {
@@ -258,6 +1019,7 @@ async function* runAgentTurn(input) {
258
1019
  block: built.block,
259
1020
  forToolCallId: call.id
260
1021
  };
1022
+ if (built.block.kind === "actions" || built.block.kind === "suggestions") offeredTap = true;
261
1023
  if ((0, _types.isAskBlock)(built.block)) {
262
1024
  awaiting = built.block.id;
263
1025
  results.push({
@@ -267,7 +1029,7 @@ async function* runAgentTurn(input) {
267
1029
  } else {
268
1030
  results.push({
269
1031
  toolCallId: call.id,
270
- content: "Shown to the user. Don't repeat its contents in text."
1032
+ content: "Shown to the user. Don't repeat its contents in text." + (droppedImages > 0 ? droppedImageNote(droppedImages) : "")
271
1033
  });
272
1034
  }
273
1035
  continue;
@@ -275,12 +1037,40 @@ async function* runAgentTurn(input) {
275
1037
  const toolId = (0, _toProviderTools.fromProviderToolName)(call.name, catalog);
276
1038
  const action = call.input.action;
277
1039
 
1040
+ /**
1041
+ * A call that never reached a tool at all.
1042
+ *
1043
+ * Same rule as a refusal: the model made a move, the move went nowhere,
1044
+ * and a transcript that hides it leaves the reader watching the agent
1045
+ * change its mind for no visible reason. Caught on device — a model
1046
+ * guessed three navigation tool names that do not exist, spent 41
1047
+ * seconds doing it, and the only trace was a sentence it chose to write.
1048
+ * `read` because nothing was touched.
1049
+ */
1050
+ const deadCall = (label, summary) => [{
1051
+ type: "tool-start",
1052
+ id: call.id,
1053
+ toolId: toolId ?? call.name,
1054
+ action: action ?? "",
1055
+ label,
1056
+ description: "",
1057
+ params: call.input.params ?? {},
1058
+ effect: "read"
1059
+ }, {
1060
+ type: "tool-end",
1061
+ id: call.id,
1062
+ ok: false,
1063
+ summary,
1064
+ durationMs: 0
1065
+ }];
1066
+
278
1067
  // Two different mistakes, kept apart on purpose. Merging them tells a
279
1068
  // model that forgot `action` that the TOOL does not exist, and it stops
280
1069
  // reaching for a tool that was fine — the same distinction
281
1070
  // `dispatchToolAction` draws between an unknown tool and an unknown
282
1071
  // action, for the same reason.
283
1072
  if (!toolId) {
1073
+ for (const ev of deadCall(call.name, "No such tool")) yield ev;
284
1074
  results.push({
285
1075
  toolCallId: call.id,
286
1076
  content: `There is no tool called "${call.name}". Use one of the tools you were given.`,
@@ -289,6 +1079,7 @@ async function* runAgentTurn(input) {
289
1079
  continue;
290
1080
  }
291
1081
  if (!action) {
1082
+ for (const ev of deadCall(call.name, "No action given")) yield ev;
292
1083
  results.push({
293
1084
  toolCallId: call.id,
294
1085
  content: `"${call.name}" needs an "action" — it is required, and it names which of this tool's actions to run. See the action list in the tool's description.`,
@@ -302,6 +1093,7 @@ async function* runAgentTurn(input) {
302
1093
  // has nowhere to go but guess again, and the next guess is no better
303
1094
  // informed than the last.
304
1095
  const known = catalog.find(t => t.toolId === toolId)?.actions.map(a => a.action).join(", ");
1096
+ for (const ev of deadCall(`${toolId}.${action}`, "No such action")) yield ev;
305
1097
  results.push({
306
1098
  toolCallId: call.id,
307
1099
  content: `"${action}" is not an action on ${toolId}.${known ? ` Valid actions: ${known}.` : ""}`,
@@ -310,7 +1102,11 @@ async function* runAgentTurn(input) {
310
1102
  continue;
311
1103
  }
312
1104
  const raw = call.input.params ?? {};
313
- const params = (0, _validateParams.applyDefaults)(raw, descriptor.params);
1105
+ // Accept the parameter-name misses every model makes (queryKey for
1106
+ // queryHash, requestId for id, flat rule fields, …) before validation,
1107
+ // so a mechanical alias never costs a turn. See normalizeParams.ts.
1108
+ const normalized = (0, _normalizeParams.normalizeParams)(toolId, action, raw);
1109
+ const params = (0, _validateParams.applyDefaults)(normalized.params, descriptor.params);
314
1110
  const check = (0, _validateParams.validateParams)(params, descriptor.params);
315
1111
  if (!check.ok) {
316
1112
  results.push({
@@ -320,6 +1116,7 @@ async function* runAgentTurn(input) {
320
1116
  });
321
1117
  continue;
322
1118
  }
1119
+ const aliasTrailer = (0, _normalizeParams.aliasNote)(normalized.notes);
323
1120
  const label = (0, _policy.describeCall)(toolId, descriptor, params);
324
1121
  const verdict = (0, _policy.decide)({
325
1122
  descriptor,
@@ -327,14 +1124,29 @@ async function* runAgentTurn(input) {
327
1124
  policy,
328
1125
  isRelease
329
1126
  });
1127
+
1128
+ // A step the user can SEE, for a call that never ran. `tool-end` alone
1129
+ // updates a row that was never opened, so a refusal used to leave no
1130
+ // trace in the transcript at all — the model simply changed its mind
1131
+ // between one sentence and the next.
1132
+ const unrunStep = summary => [{
1133
+ type: "tool-start",
1134
+ id: call.id,
1135
+ toolId,
1136
+ action,
1137
+ label,
1138
+ description: descriptor.summary,
1139
+ params,
1140
+ effect: descriptor.effect
1141
+ }, {
1142
+ type: "tool-end",
1143
+ id: call.id,
1144
+ ok: false,
1145
+ summary,
1146
+ durationMs: 0
1147
+ }];
330
1148
  if (verdict.verdict === "refuse") {
331
- yield {
332
- type: "tool-end",
333
- id: call.id,
334
- ok: false,
335
- summary: "refused",
336
- durationMs: 0
337
- };
1149
+ for (const ev of unrunStep("refused")) yield ev;
338
1150
  results.push({
339
1151
  toolCallId: call.id,
340
1152
  content: verdict.reason,
@@ -342,33 +1154,93 @@ async function* runAgentTurn(input) {
342
1154
  });
343
1155
  continue;
344
1156
  }
345
- if (verdict.verdict === "needs-approval") {
346
- yield {
347
- type: "approval-required",
348
- id: call.id,
349
- toolId,
350
- action,
351
- label,
352
- description: descriptor.summary,
353
- reason: verdict.reason
354
- };
355
- const approved = requestApproval ? await requestApproval({
356
- id: call.id,
357
- toolId,
358
- action,
359
- label,
360
- description: descriptor.summary,
361
- reason: verdict.reason
362
- }) : false;
1157
+
1158
+ /**
1159
+ * Nothing in this batch changes anything until every card in it has been
1160
+ * answered. See `runBarrier` — this is the line that makes a decline
1161
+ * mean "the plan does not happen" rather than "the rest of the plan does
1162
+ * not happen".
1163
+ */
1164
+ const gatedHere = verdict.verdict === "needs-approval";
1165
+ if (gatedInBatch && !barrierRun && (descriptor.effect !== "read" || gatedHere)) {
1166
+ yield* runBarrier();
1167
+ }
1168
+ if (batchDeclined && (descriptor.effect !== "read" || gatedHere)) {
1169
+ for (const ev of unrunStep("declined")) yield ev;
1170
+ results.push({
1171
+ toolCallId: call.id,
1172
+ content: decided.get(call.id) === false ? declinedResult(declineReasons.get(call.id)) : "Not run — the user declined another change in this same batch, so none of it was applied. Ask what they would prefer before proposing it again.",
1173
+ isError: true
1174
+ });
1175
+ continue;
1176
+ }
1177
+ if (gatedHere) {
1178
+ let approved;
1179
+ if (decided.has(call.id)) {
1180
+ // The barrier already put this card up, before the batch's first
1181
+ // mutation. Asking again would be the same question twice.
1182
+ approved = decided.get(call.id);
1183
+ } else if (trusted?.has(`${toolId}.${action}`)) {
1184
+ // Waived for this conversation — see RunTurnInput.trusted.
1185
+ approved = true;
1186
+ decided.set(call.id, true);
1187
+ } else {
1188
+ // The barrier could not resolve this call (see `planned`), so it was
1189
+ // never offered. Ask here, exactly as this always did — a gate that
1190
+ // fails closed by silently declining would be worse than one that
1191
+ // asks late.
1192
+ yield {
1193
+ type: "approval-required",
1194
+ id: call.id,
1195
+ toolId,
1196
+ action,
1197
+ label,
1198
+ description: descriptor.summary,
1199
+ reason: verdict.reason
1200
+ };
1201
+ const askedAt = Date.now();
1202
+ // What the write is about to replace, as of NOW. Only a restored
1203
+ // card ever reads it back — see readTargetDigest.
1204
+ const targetDigest = await readTargetDigest({
1205
+ toolId,
1206
+ action,
1207
+ params,
1208
+ dispatch,
1209
+ catalog
1210
+ });
1211
+ const answer = readAnswer(requestApproval ? await requestApproval({
1212
+ id: call.id,
1213
+ toolId,
1214
+ action,
1215
+ label,
1216
+ description: descriptor.summary,
1217
+ reason: verdict.reason,
1218
+ params,
1219
+ targetDigest
1220
+ }) : false);
1221
+ // The user's deliberation is not the agent's runtime. Give the clock
1222
+ // back before anything else can trip the deadline check.
1223
+ deadline += Date.now() - askedAt;
1224
+ approved = answer.approved;
1225
+ decided.set(call.id, approved);
1226
+ if (answer.trust) trusted?.add(`${toolId}.${action}`);
1227
+ if (answer.reason) declineReasons.set(call.id, answer.reason);
1228
+ }
363
1229
  if (!approved) {
1230
+ // The row is the whole point here. Without it the bubble read as a
1231
+ // contradiction — "Bumped the line from 5 to 9." straight into "I
1232
+ // left it as it was" — with nothing on screen to say a change had
1233
+ // been proposed and turned down.
1234
+ for (const ev of unrunStep("declined")) yield ev;
364
1235
  results.push({
365
1236
  toolCallId: call.id,
366
- content: "The user declined this change. Do not retry it; ask what they would prefer.",
1237
+ content: declinedResult(declineReasons.get(call.id)),
367
1238
  isError: true
368
1239
  });
369
1240
  continue;
370
1241
  }
371
1242
  }
1243
+ if (_askGate.ACQUISITION_CALLS.has(`${toolId}.${action}`)) attemptedAcquisition = true;
372
1244
  yield {
373
1245
  type: "tool-start",
374
1246
  id: call.id,
@@ -385,7 +1257,50 @@ async function* runAgentTurn(input) {
385
1257
  // exactly. This is the one thing that makes storage reversible here when
386
1258
  // the Scenarios engine has to treat it as permanent.
387
1259
  const before = await captureBefore(descriptor, toolId, params, dispatch);
388
- const stateBefore = await captureStateBefore(descriptor, toolId, params, dispatch);
1260
+ const stateBefore = await captureStoreState(descriptor, toolId, params, dispatch);
1261
+
1262
+ /**
1263
+ * THE FINAL GUARD. Everything above this line happened in the past.
1264
+ *
1265
+ * `decide()` ran before the approval card went up, and between there and
1266
+ * here are three awaits: the target read, the person deciding, and two
1267
+ * pre-reads. Each one yields the thread, and a person can do a lot in
1268
+ * that gap — press Stop, or flip the read-only toggle in the settings
1269
+ * sheet while the card is still on screen. Both were reproduced: the
1270
+ * write went ahead on the verdict it captured before they touched
1271
+ * anything, which is precisely the moment the product promises it will
1272
+ * not. `live` is mutated in place by the session (see `refreshLive`), so
1273
+ * asking again reads the toggles as they are NOW.
1274
+ *
1275
+ * Below this there is no `await` before `dispatch` is CALLED. That is
1276
+ * load-bearing: an await here would reopen the same gap one line lower.
1277
+ */
1278
+ if (signal?.aborted) {
1279
+ for (const ev of unrunStep("stopped")) yield ev;
1280
+ results.push({
1281
+ toolCallId: call.id,
1282
+ content: "Not run — the user pressed Stop before this call started.",
1283
+ isError: true
1284
+ });
1285
+ continue;
1286
+ }
1287
+ const stillAllowed = (0, _policy.decide)({
1288
+ descriptor,
1289
+ toolId,
1290
+ policy,
1291
+ isRelease
1292
+ });
1293
+ if (stillAllowed.verdict === "refuse") {
1294
+ // `needs-approval` is NOT re-refused: reaching here means the tap
1295
+ // already happened, and asking twice for one call is its own bug.
1296
+ for (const ev of unrunStep("refused")) yield ev;
1297
+ results.push({
1298
+ toolCallId: call.id,
1299
+ content: stillAllowed.reason,
1300
+ isError: true
1301
+ });
1302
+ continue;
1303
+ }
389
1304
  try {
390
1305
  // `getSnapshot` is reserved: it is not an adapter action, so it goes to
391
1306
  // the snapshot reader and is trimmed to the fields worth sending. See
@@ -393,34 +1308,69 @@ async function* runAgentTurn(input) {
393
1308
  // ledger the model asks about lives HERE, not behind an adapter, so
394
1309
  // routing it through dispatch would only work on a device and would
395
1310
  // record the undo as a fresh effect.
396
- const result = action === _snapshotReads.SNAPSHOT_ACTION ? (0, _snapshotReads.projectSnapshot)(toolId, await requireSnapshot(readSnapshot, toolId), params) : toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch) : await dispatch(toolId, action, params);
1311
+ const result = action === _snapshotReads.SNAPSHOT_ACTION ? (0, _snapshotReads.projectSnapshot)(toolId, await requireSnapshot(readSnapshot, toolId), params) : toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch, evidence, params, procedures) : await (prefetch.get(call.id) ?? dispatch(toolId, action, params));
397
1312
  const cleaned = (0, _redact.redact)((0, _redact.stripSelfTraffic)(result));
398
- if (descriptor.effect !== "read" && toolId !== "ask-buoy") {
399
- // A delete of something we created IS the undo, whoever asked for
400
- // it — mark the matching entry undone instead of leaving the bar
401
- // promising an undo against a rule that no longer exists.
402
- ledger.noteExternalRevert(toolId, action, params);
403
- const fx = (0, _effectFor.effectFor)(toolId, descriptor, params, result, before);
404
- if (fx) ledger.record(toolId, action, fx);
405
- }
406
- const encodedResult = encodeResult(cleaned);
1313
+ const refused = reportsFailure(result);
1314
+
1315
+ // Kept in full — after redaction, never before — so the model can go
1316
+ // back to it. Not a retrieve's own result (a retrieve of a retrieve is
1317
+ // a loop), and not a refusal (nothing to go back to).
1318
+ const text = fullText(cleaned);
1319
+ const ref = evidence && !refused && !(toolId === "ask-buoy" && action === "retrieve") ? evidence.stash({
1320
+ toolId,
1321
+ action,
1322
+ params,
1323
+ capturedAt: Date.now(),
1324
+ text
1325
+ }) : undefined;
1326
+ const encodedResult = encodeResult(text, ref);
1327
+ const traceResult = encodedResult.length > MAX_TRACE_RESULT_CHARS ? `${encodedResult.slice(0, MAX_TRACE_RESULT_CHARS)}…` : encodedResult;
407
1328
  yield {
408
1329
  type: "tool-end",
409
1330
  id: call.id,
410
- ok: true,
1331
+ ok: !refused,
411
1332
  summary: summarise(result),
412
1333
  durationMs: Date.now() - startedAt,
413
- result: encodedResult.length > MAX_TRACE_RESULT_CHARS ? `${encodedResult.slice(0, MAX_TRACE_RESULT_CHARS)}…` : encodedResult
1334
+ result: traceResult
414
1335
  };
415
1336
 
416
1337
  // The receipt: what the user sees of this result, built from the
417
1338
  // same redacted data the model gets. See blocks/receipts.ts.
418
- const receipt = (0, _receipts.projectReceipt)({
1339
+ //
1340
+ // The store is read back a second time so the diff shows what the
1341
+ // write ACTUALLY did rather than what it asked for — the two differ on
1342
+ // every `path` edit and every keyed list edit, and the request-shaped
1343
+ // version reported untouched fields as deleted.
1344
+ const stateAfter = stateBefore === undefined || refused ? undefined : await captureStoreState(descriptor, toolId, params, dispatch);
1345
+ if (descriptor.effect !== "read" && toolId !== "ask-buoy" && !refused) {
1346
+ // A delete of something we created IS the undo, whoever asked for
1347
+ // it — mark the matching entry undone instead of leaving the bar
1348
+ // promising an undo against a rule that no longer exists.
1349
+ ledger.noteExternalRevert(toolId, action, params);
1350
+ // Recorded AFTER the read-back so a cache edit's entry can carry both
1351
+ // halves: the data to restore and a fingerprint of what the write
1352
+ // left behind, which is how undo knows whether the app has since
1353
+ // replaced it. See effectFor / ledger "query-write".
1354
+ const fx = (0, _effectFor.effectFor)(toolId, descriptor, params, result, before, {
1355
+ before: stateBefore,
1356
+ after: stateAfter
1357
+ });
1358
+ if (fx) {
1359
+ const entry = ledger.record(toolId, action, fx, ledgerGeneration);
1360
+ // Ties the change to its round, so the budget keeps that round
1361
+ // while the change is live. See LedgerEntry.callId.
1362
+ if (entry) entry.callId = call.id;
1363
+ }
1364
+ }
1365
+ // No receipt for a call that changed nothing: a "what changed" card
1366
+ // under a refusal is the same lie as the ledger entry, drawn bigger.
1367
+ const receipt = refused ? undefined : (0, _receipts.projectReceipt)({
419
1368
  toolId,
420
1369
  action,
421
1370
  params,
422
1371
  result: cleaned,
423
1372
  before: stateBefore ?? before,
1373
+ after: stateAfter,
424
1374
  effect: descriptor.effect
425
1375
  });
426
1376
  if (receipt) yield {
@@ -428,10 +1378,36 @@ async function* runAgentTurn(input) {
428
1378
  block: receipt,
429
1379
  forToolCallId: call.id
430
1380
  };
1381
+ /**
1382
+ * THE OUTCOME CHECK. A write's own `{ok:true}` says the adapter
1383
+ * accepted it; this reads the app back and says whether the requested
1384
+ * state is actually there. Only for actions that have a verifier, only
1385
+ * for writes that were not refused, and never itself a write. Its
1386
+ * verdict rides in the tool result as a trailer the model reads, and
1387
+ * on a second tool-end so the row says "verified" / "unverified" /
1388
+ * "check failed". See engine/verify.ts for the three meanings.
1389
+ */
1390
+ const verification = descriptor.effect !== "read" && !refused ? await (0, _verify.verifyOutcome)({
1391
+ toolId,
1392
+ action,
1393
+ params,
1394
+ result,
1395
+ dispatch,
1396
+ signal,
1397
+ after: stateAfter
1398
+ }) : undefined;
1399
+ if (verification) yield {
1400
+ type: "tool-verified",
1401
+ id: call.id,
1402
+ verification
1403
+ };
431
1404
  for (const url of encodedResult.match(URL_RE) ?? []) seenUrls.add(url);
432
1405
  results.push({
433
1406
  toolCallId: call.id,
434
- content: encodedResult + ownedKeyNote(toolId, descriptor, params, storeKeyOwners)
1407
+ content: encodedResult + ownedKeyNote(toolId, descriptor, params, storeKeyOwners) + aliasTrailer + (verification ? (0, _verify.verificationTrailer)(verification) : "") + (salvagedIds.has(call.id) ? _textToolCalls.SALVAGE_NOTE : ""),
1408
+ ...(ref ? {
1409
+ ref
1410
+ } : {})
435
1411
  });
436
1412
  if (toolId === "app" && (action === "reloadApp" || action === "reload")) {
437
1413
  realmWillDie = true;
@@ -452,6 +1428,20 @@ async function* runAgentTurn(input) {
452
1428
  });
453
1429
  }
454
1430
  }
1431
+
1432
+ // Looping? Say so, once, on the LAST result of this round — so it arrives
1433
+ // attached to the thing being repeated rather than as a free-floating
1434
+ // user turn, and the model reads it before deciding what to do next.
1435
+ if (calls.length) {
1436
+ recentSignatures.push(callSignature(calls));
1437
+ if (recentSignatures.length > LOOP_REPEATS) recentSignatures.shift();
1438
+ const looping = !loopWarned && recentSignatures.length === LOOP_REPEATS && recentSignatures.every(sig => sig === recentSignatures[0]);
1439
+ if (looping) {
1440
+ loopWarned = true;
1441
+ const last = results[results.length - 1];
1442
+ if (last) last.content = `${last.content}\n\n${LOOP_WARNING}`;
1443
+ }
1444
+ }
455
1445
  messages.push({
456
1446
  role: "tool-results",
457
1447
  results
@@ -462,7 +1452,8 @@ async function* runAgentTurn(input) {
462
1452
  blockId: awaiting
463
1453
  };
464
1454
  yield {
465
- type: "done"
1455
+ type: "done",
1456
+ stopReason: "awaiting-user"
466
1457
  };
467
1458
  return messages;
468
1459
  }
@@ -470,7 +1461,8 @@ async function* runAgentTurn(input) {
470
1461
  // Anything after this dies with the JS realm. Stop cleanly instead of
471
1462
  // sending a request whose answer can never arrive.
472
1463
  yield {
473
- type: "done"
1464
+ type: "done",
1465
+ stopReason: "realm-died"
474
1466
  };
475
1467
  return messages;
476
1468
  }
@@ -490,7 +1482,8 @@ async function* runAgentTurn(input) {
490
1482
  }
491
1483
  };
492
1484
  yield {
493
- type: "done"
1485
+ type: "done",
1486
+ stopReason: "step-cap"
494
1487
  };
495
1488
  return messages;
496
1489
  }
@@ -516,6 +1509,27 @@ function ownedKeyNote(toolId, descriptor, params, storeKeyOwners) {
516
1509
  if (!storeName) return "";
517
1510
  return `\n\n[Buoy] This wrote to disk, but "${params.key}" is the saved copy of the live "${storeName}" store, so nothing on screen has changed — the app read that key once at startup and has held the state in memory since. Call zustand.rehydrate({"storeName":"${storeName}"}) to make the app pick this up, or use zustand.setState next time to change the store directly. If you meant to set the value for the app's next launch, this is already done and no further action is needed.`;
518
1511
  }
1512
+ /**
1513
+ * Fingerprint the value a write is about to replace.
1514
+ *
1515
+ * Built for the approval card that outlives its turn: the app crashed with
1516
+ * "Set bag line L-01 qty to 14" still asking, relaunched, and the card came
1517
+ * back — and Allow wrote qty 14 against whatever the bag was NOW, maybe a
1518
+ * different account, maybe a line that no longer existed. A hash of the
1519
+ * params alone would not catch that (same label, same params, different
1520
+ * world). So the card carries a digest of the TARGET at ask time, and a
1521
+ * restored Allow re-reads and compares before it writes. Undefined when the
1522
+ * target cannot be read — then the card can only warn, not check.
1523
+ */
1524
+ async function readTargetDigest(input) {
1525
+ const descriptor = findDescriptor(input.catalog ?? _catalog.CATALOG, input.toolId, input.action);
1526
+ if (!descriptor || descriptor.effect === "read") return undefined;
1527
+ const store = await captureStoreState(descriptor, input.toolId, input.params, input.dispatch);
1528
+ if (store !== undefined) return (0, _digest.digestOf)(store);
1529
+ const before = await captureBefore(descriptor, input.toolId, input.params, input.dispatch);
1530
+ if (before !== undefined) return (0, _digest.digestOf)(before);
1531
+ return undefined;
1532
+ }
519
1533
  /**
520
1534
  * One human-initiated action through EVERY gate the model's calls go through:
521
1535
  * catalog lookup, param validation, policy, pre-read, effect ledger.
@@ -548,7 +1562,8 @@ async function runGatedAction(input) {
548
1562
  error: `${toolId}.${action} is not in Buoy's catalog.`
549
1563
  };
550
1564
  }
551
- const params = (0, _validateParams.applyDefaults)(input.params ?? {}, descriptor.params);
1565
+ const normalized = (0, _normalizeParams.normalizeParams)(toolId, action, input.params ?? {});
1566
+ const params = (0, _validateParams.applyDefaults)(normalized.params, descriptor.params);
552
1567
  const check = (0, _validateParams.validateParams)(params, descriptor.params);
553
1568
  if (!check.ok) {
554
1569
  return {
@@ -569,11 +1584,51 @@ async function runGatedAction(input) {
569
1584
  };
570
1585
  }
571
1586
  const before = await captureBefore(descriptor, toolId, params, dispatch);
1587
+ /**
1588
+ * The same read-back the model's path takes. It was missing here, and the
1589
+ * gap was invisible: a `setQueryData` through a block button or a restored
1590
+ * approval card recorded a ledger entry with no prior value, so Undo had
1591
+ * nothing to put back and the concurrent-writer check had no fingerprint to
1592
+ * compare. The bar said "1 undoable" over a change that could not be undone.
1593
+ */
1594
+ const stateBefore = await captureStoreState(descriptor, toolId, params, dispatch);
1595
+
1596
+ // Asked again after the pre-reads, for the same reason the model's path asks
1597
+ // again: those are awaits, and read-only can be switched on inside one.
1598
+ const stillAllowed = (0, _policy.decide)({
1599
+ descriptor,
1600
+ toolId,
1601
+ policy,
1602
+ isRelease
1603
+ });
1604
+ if (stillAllowed.verdict === "refuse") {
1605
+ return {
1606
+ ok: false,
1607
+ error: stillAllowed.reason
1608
+ };
1609
+ }
572
1610
  try {
573
- const result = toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch) : await dispatch(toolId, action, params);
1611
+ const result = toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch, undefined, params) : await dispatch(toolId, action, params);
1612
+
1613
+ // Same rule as the model's path: a resolved `{ok:false}` is a refusal, not
1614
+ // a change. Without this a block button (or an approval answered after a
1615
+ // reload) put a phantom entry in the undo bar too.
1616
+ if (reportsFailure(result)) {
1617
+ const err = result.error;
1618
+ return {
1619
+ ok: false,
1620
+ error: typeof err === "string" ? err : "The tool refused this."
1621
+ };
1622
+ }
574
1623
  if (descriptor.effect !== "read" && toolId !== "ask-buoy") {
575
1624
  ledger.noteExternalRevert(toolId, action, params);
576
- const fx = (0, _effectFor.effectFor)(toolId, descriptor, params, result, before);
1625
+ // Read back AFTER the write, exactly as the model's path does — the two
1626
+ // halves together are what make a cache edit undoable.
1627
+ const stateAfter = await captureStoreState(descriptor, toolId, params, dispatch);
1628
+ const fx = (0, _effectFor.effectFor)(toolId, descriptor, params, result, before, {
1629
+ before: stateBefore,
1630
+ after: stateAfter
1631
+ });
577
1632
  if (fx) ledger.record(toolId, action, fx);
578
1633
  }
579
1634
  return {
@@ -595,7 +1650,33 @@ async function runGatedAction(input) {
595
1650
  * adapter's same-named actions remain for desktop/MCP, where the ledger is
596
1651
  * reached over the broker instead.
597
1652
  */
598
- async function runAskBuoyAction(action, ledger, dispatch) {
1653
+ async function runAskBuoyAction(action, ledger, dispatch, evidence, params, procedures) {
1654
+ if (action === "retrieve") {
1655
+ return (0, _retrieve.retrieveEvidence)(evidence, params ?? {});
1656
+ }
1657
+ if (action === "openProcedure") {
1658
+ const id = typeof params?.id === "string" ? params.id : "";
1659
+ const found = procedures?.find(p => p.id === id);
1660
+ if (!found) {
1661
+ const ids = (procedures ?? []).map(p => p.id);
1662
+ return {
1663
+ ok: false,
1664
+ error: ids.length ? `No procedure "${id}". This app's procedures: ${ids.join(", ")}.` : "This app has no procedures written for it."
1665
+ };
1666
+ }
1667
+ return {
1668
+ id: found.id,
1669
+ title: found.title,
1670
+ ...(found.version ? {
1671
+ version: found.version
1672
+ } : {}),
1673
+ ...(found.requires?.length ? {
1674
+ requires: found.requires
1675
+ } : {}),
1676
+ body: found.body,
1677
+ note: "Follow these steps with the ordinary tools. Every write still goes through the same policy, approval and undo as any other call; a step that is refused stays refused."
1678
+ };
1679
+ }
599
1680
  if (action === "listChanges") {
600
1681
  const changes = ledger.list().filter(e => !e.undoneAt).map(e => ({
601
1682
  toolId: e.toolId,
@@ -645,16 +1726,46 @@ async function requireSnapshot(readSnapshot, toolId) {
645
1726
  * Zustand only for now: `getStoreState` is cheap and the write params carry
646
1727
  * the after-state, so the diff is exact without the store's cooperation.
647
1728
  */
648
- async function captureStateBefore(descriptor, toolId, params, dispatch) {
649
- if (toolId !== "zustand" || descriptor.action !== "setState") return undefined;
650
- if (typeof params.storeName !== "string") return undefined;
651
- try {
652
- return await dispatch(toolId, "getStoreState", {
653
- storeName: params.storeName
654
- });
655
- } catch {
656
- return undefined;
1729
+ /**
1730
+ * Read a zustand store whole, for the before/after halves of a write receipt.
1731
+ * Called on both sides of the dispatch — see the receipt above.
1732
+ */
1733
+ async function captureStoreState(descriptor, toolId, params, dispatch) {
1734
+ if (toolId === "zustand" && descriptor.action === "setState") {
1735
+ if (typeof params.storeName !== "string") return undefined;
1736
+ try {
1737
+ return await dispatch(toolId, "getStoreState", {
1738
+ storeName: params.storeName
1739
+ });
1740
+ } catch {
1741
+ return undefined;
1742
+ }
1743
+ }
1744
+ /**
1745
+ * The same read-both-sides trick for the CACHE.
1746
+ *
1747
+ * A zustand edit produced a "what changed" card and the identical edit to a
1748
+ * query produced nothing — for the write the prompt calls "THE way to change
1749
+ * what a server-backed screen shows". The QA tester's confirmation that the
1750
+ * right field moved was missing from the most-used tool in the product.
1751
+ */
1752
+ if (toolId === "query" && descriptor.action === "setQueryData") {
1753
+ // hashQueryKey, not JSON.stringify: React Query SORTS object keys when it
1754
+ // hashes, so a key carrying `{query:"",category:null}` hashes as
1755
+ // `{"category":null,"query":""}`. Stringifying in insertion order produced
1756
+ // a hash that matched nothing, the read came back empty, and the card
1757
+ // silently did not draw.
1758
+ const queryHash = typeof params.queryHash === "string" ? params.queryHash : Array.isArray(params.queryKey) ? (0, _normalizeParams.hashQueryKey)(params.queryKey) : undefined;
1759
+ if (!queryHash) return undefined;
1760
+ try {
1761
+ return await dispatch(toolId, "getQueryData", {
1762
+ queryHash
1763
+ });
1764
+ } catch {
1765
+ return undefined;
1766
+ }
657
1767
  }
1768
+ return undefined;
658
1769
  }
659
1770
  async function captureBefore(descriptor, toolId, params, dispatch) {
660
1771
  if (descriptor.effect === "read") return undefined;