@buoy-gg/agent-core 7.0.35 → 7.0.36

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/README.md +10 -7
  2. package/lib/commonjs/blocks/receipts.js +57 -1
  3. package/lib/commonjs/blocks/receipts.js.map +1 -1
  4. package/lib/commonjs/blocks/types.js +23 -2
  5. package/lib/commonjs/blocks/types.js.map +1 -1
  6. package/lib/commonjs/blocks/uiTool.js +423 -10
  7. package/lib/commonjs/blocks/uiTool.js.map +1 -1
  8. package/lib/commonjs/catalog/catalog.g.js +121 -43
  9. package/lib/commonjs/catalog/catalog.g.js.map +1 -1
  10. package/lib/commonjs/catalog/catalog.source.json +121 -45
  11. package/lib/commonjs/catalog/catalog.types.g.js +44 -0
  12. package/lib/commonjs/catalog/catalog.types.g.js.map +1 -0
  13. package/lib/commonjs/catalog/normalizeParams.js +333 -0
  14. package/lib/commonjs/catalog/normalizeParams.js.map +1 -0
  15. package/lib/commonjs/catalog/signature.js +68 -0
  16. package/lib/commonjs/catalog/signature.js.map +1 -0
  17. package/lib/commonjs/catalog/snapshotReads.js +13 -6
  18. package/lib/commonjs/catalog/snapshotReads.js.map +1 -1
  19. package/lib/commonjs/catalog/toProviderTools.js +41 -5
  20. package/lib/commonjs/catalog/toProviderTools.js.map +1 -1
  21. package/lib/commonjs/context/buildContextPack.js +367 -19
  22. package/lib/commonjs/context/buildContextPack.js.map +1 -1
  23. package/lib/commonjs/effects/digest.js +34 -0
  24. package/lib/commonjs/effects/digest.js.map +1 -0
  25. package/lib/commonjs/effects/ledger.js +85 -1
  26. package/lib/commonjs/effects/ledger.js.map +1 -1
  27. package/lib/commonjs/engine/askGate.js +287 -0
  28. package/lib/commonjs/engine/askGate.js.map +1 -0
  29. package/lib/commonjs/engine/effectFor.js +82 -1
  30. package/lib/commonjs/engine/effectFor.js.map +1 -1
  31. package/lib/commonjs/engine/historyBudget.js +260 -0
  32. package/lib/commonjs/engine/historyBudget.js.map +1 -0
  33. package/lib/commonjs/engine/runAgentTurn.js +924 -83
  34. package/lib/commonjs/engine/runAgentTurn.js.map +1 -1
  35. package/lib/commonjs/engine/systemPrompt.js +72 -8
  36. package/lib/commonjs/engine/systemPrompt.js.map +1 -1
  37. package/lib/commonjs/engine/textToolCalls.js +276 -0
  38. package/lib/commonjs/engine/textToolCalls.js.map +1 -0
  39. package/lib/commonjs/index.js +58 -0
  40. package/lib/commonjs/index.js.map +1 -1
  41. package/lib/commonjs/policy/labels.js +19 -3
  42. package/lib/commonjs/policy/labels.js.map +1 -1
  43. package/lib/commonjs/policy/policy.js +6 -1
  44. package/lib/commonjs/policy/policy.js.map +1 -1
  45. package/lib/commonjs/policy/redact.js +42 -5
  46. package/lib/commonjs/policy/redact.js.map +1 -1
  47. package/lib/commonjs/providers/anthropic.js +331 -208
  48. package/lib/commonjs/providers/anthropic.js.map +1 -1
  49. package/lib/commonjs/providers/openai.js +229 -103
  50. package/lib/commonjs/providers/openai.js.map +1 -1
  51. package/lib/commonjs/providers/sse.js +164 -29
  52. package/lib/commonjs/providers/sse.js.map +1 -1
  53. package/lib/commonjs/providers/streamTimer.js +123 -0
  54. package/lib/commonjs/providers/streamTimer.js.map +1 -0
  55. package/lib/commonjs/providers/transport.js +107 -0
  56. package/lib/commonjs/providers/transport.js.map +1 -0
  57. package/lib/commonjs/providers/xhrStream.js +209 -0
  58. package/lib/commonjs/providers/xhrStream.js.map +1 -0
  59. package/lib/commonjs/session.js +215 -58
  60. package/lib/commonjs/session.js.map +1 -1
  61. package/lib/module/blocks/receipts.js +57 -1
  62. package/lib/module/blocks/receipts.js.map +1 -1
  63. package/lib/module/blocks/types.js +23 -2
  64. package/lib/module/blocks/types.js.map +1 -1
  65. package/lib/module/blocks/uiTool.js +421 -10
  66. package/lib/module/blocks/uiTool.js.map +1 -1
  67. package/lib/module/catalog/catalog.g.js +121 -43
  68. package/lib/module/catalog/catalog.g.js.map +1 -1
  69. package/lib/module/catalog/catalog.source.json +121 -45
  70. package/lib/module/catalog/catalog.types.g.js +40 -0
  71. package/lib/module/catalog/catalog.types.g.js.map +1 -0
  72. package/lib/module/catalog/normalizeParams.js +327 -0
  73. package/lib/module/catalog/normalizeParams.js.map +1 -0
  74. package/lib/module/catalog/signature.js +63 -0
  75. package/lib/module/catalog/signature.js.map +1 -0
  76. package/lib/module/catalog/snapshotReads.js +13 -6
  77. package/lib/module/catalog/snapshotReads.js.map +1 -1
  78. package/lib/module/catalog/toProviderTools.js +41 -5
  79. package/lib/module/catalog/toProviderTools.js.map +1 -1
  80. package/lib/module/context/buildContextPack.js +366 -19
  81. package/lib/module/context/buildContextPack.js.map +1 -1
  82. package/lib/module/effects/digest.js +29 -0
  83. package/lib/module/effects/digest.js.map +1 -0
  84. package/lib/module/effects/ledger.js +85 -1
  85. package/lib/module/effects/ledger.js.map +1 -1
  86. package/lib/module/engine/askGate.js +280 -0
  87. package/lib/module/engine/askGate.js.map +1 -0
  88. package/lib/module/engine/effectFor.js +83 -1
  89. package/lib/module/engine/effectFor.js.map +1 -1
  90. package/lib/module/engine/historyBudget.js +251 -0
  91. package/lib/module/engine/historyBudget.js.map +1 -0
  92. package/lib/module/engine/runAgentTurn.js +920 -83
  93. package/lib/module/engine/runAgentTurn.js.map +1 -1
  94. package/lib/module/engine/systemPrompt.js +71 -8
  95. package/lib/module/engine/systemPrompt.js.map +1 -1
  96. package/lib/module/engine/textToolCalls.js +270 -0
  97. package/lib/module/engine/textToolCalls.js.map +1 -0
  98. package/lib/module/index.js +5 -3
  99. package/lib/module/index.js.map +1 -1
  100. package/lib/module/policy/labels.js +19 -3
  101. package/lib/module/policy/labels.js.map +1 -1
  102. package/lib/module/policy/policy.js +6 -1
  103. package/lib/module/policy/policy.js.map +1 -1
  104. package/lib/module/policy/redact.js +42 -5
  105. package/lib/module/policy/redact.js.map +1 -1
  106. package/lib/module/providers/anthropic.js +332 -209
  107. package/lib/module/providers/anthropic.js.map +1 -1
  108. package/lib/module/providers/openai.js +230 -104
  109. package/lib/module/providers/openai.js.map +1 -1
  110. package/lib/module/providers/sse.js +160 -29
  111. package/lib/module/providers/sse.js.map +1 -1
  112. package/lib/module/providers/streamTimer.js +118 -0
  113. package/lib/module/providers/streamTimer.js.map +1 -0
  114. package/lib/module/providers/transport.js +103 -0
  115. package/lib/module/providers/transport.js.map +1 -0
  116. package/lib/module/providers/xhrStream.js +204 -0
  117. package/lib/module/providers/xhrStream.js.map +1 -0
  118. package/lib/module/session.js +199 -60
  119. package/lib/module/session.js.map +1 -1
  120. package/lib/typescript/blocks/receipts.d.ts +5 -0
  121. package/lib/typescript/blocks/receipts.d.ts.map +1 -1
  122. package/lib/typescript/blocks/types.d.ts +23 -2
  123. package/lib/typescript/blocks/types.d.ts.map +1 -1
  124. package/lib/typescript/blocks/uiTool.d.ts +11 -2
  125. package/lib/typescript/blocks/uiTool.d.ts.map +1 -1
  126. package/lib/typescript/catalog/catalog.g.d.ts.map +1 -1
  127. package/lib/typescript/catalog/catalog.types.g.d.ts +40 -0
  128. package/lib/typescript/catalog/catalog.types.g.d.ts.map +1 -0
  129. package/lib/typescript/catalog/normalizeParams.d.ts +43 -0
  130. package/lib/typescript/catalog/normalizeParams.d.ts.map +1 -0
  131. package/lib/typescript/catalog/signature.d.ts +29 -0
  132. package/lib/typescript/catalog/signature.d.ts.map +1 -0
  133. package/lib/typescript/catalog/snapshotReads.d.ts.map +1 -1
  134. package/lib/typescript/catalog/toProviderTools.d.ts +28 -15
  135. package/lib/typescript/catalog/toProviderTools.d.ts.map +1 -1
  136. package/lib/typescript/context/buildContextPack.d.ts +24 -2
  137. package/lib/typescript/context/buildContextPack.d.ts.map +1 -1
  138. package/lib/typescript/effects/digest.d.ts +10 -0
  139. package/lib/typescript/effects/digest.d.ts.map +1 -0
  140. package/lib/typescript/effects/ledger.d.ts +55 -0
  141. package/lib/typescript/effects/ledger.d.ts.map +1 -1
  142. package/lib/typescript/engine/askGate.d.ts +29 -0
  143. package/lib/typescript/engine/askGate.d.ts.map +1 -0
  144. package/lib/typescript/engine/effectFor.d.ts +6 -1
  145. package/lib/typescript/engine/effectFor.d.ts.map +1 -1
  146. package/lib/typescript/engine/historyBudget.d.ts +92 -0
  147. package/lib/typescript/engine/historyBudget.d.ts.map +1 -0
  148. package/lib/typescript/engine/runAgentTurn.d.ts +83 -1
  149. package/lib/typescript/engine/runAgentTurn.d.ts.map +1 -1
  150. package/lib/typescript/engine/systemPrompt.d.ts +41 -0
  151. package/lib/typescript/engine/systemPrompt.d.ts.map +1 -1
  152. package/lib/typescript/engine/textToolCalls.d.ts +58 -0
  153. package/lib/typescript/engine/textToolCalls.d.ts.map +1 -0
  154. package/lib/typescript/index.d.ts +5 -3
  155. package/lib/typescript/index.d.ts.map +1 -1
  156. package/lib/typescript/policy/labels.d.ts.map +1 -1
  157. package/lib/typescript/policy/policy.d.ts.map +1 -1
  158. package/lib/typescript/policy/redact.d.ts.map +1 -1
  159. package/lib/typescript/providers/anthropic.d.ts.map +1 -1
  160. package/lib/typescript/providers/openai.d.ts +0 -16
  161. package/lib/typescript/providers/openai.d.ts.map +1 -1
  162. package/lib/typescript/providers/sse.d.ts +82 -1
  163. package/lib/typescript/providers/sse.d.ts.map +1 -1
  164. package/lib/typescript/providers/streamTimer.d.ts +59 -0
  165. package/lib/typescript/providers/streamTimer.d.ts.map +1 -0
  166. package/lib/typescript/providers/transport.d.ts +40 -0
  167. package/lib/typescript/providers/transport.d.ts.map +1 -0
  168. package/lib/typescript/providers/types.d.ts +91 -3
  169. package/lib/typescript/providers/types.d.ts.map +1 -1
  170. package/lib/typescript/providers/xhrStream.d.ts +43 -0
  171. package/lib/typescript/providers/xhrStream.d.ts.map +1 -0
  172. package/lib/typescript/session.d.ts +40 -2
  173. package/lib/typescript/session.d.ts.map +1 -1
  174. package/package.json +2 -2
@@ -21,15 +21,21 @@
21
21
 
22
22
  import { CATALOG } from "../catalog/catalog.g";
23
23
  import { applyDefaults, validateParams } from "../catalog/validateParams";
24
+ import { normalizeParams, aliasNote } from "../catalog/normalizeParams";
24
25
  import { fromProviderToolName, toProviderTools } from "../catalog/toProviderTools";
26
+ import { hashQueryKey } from "../catalog/normalizeParams";
25
27
  import { projectSnapshot, SNAPSHOT_ACTION } from "../catalog/snapshotReads";
26
28
  import { isReversible } from "../effects/ledger";
27
29
  import { describeCall, decide } from "../policy/policy";
28
30
  import { redact, stripSelfTraffic } from "../policy/redact";
29
31
  import { effectFor } from "./effectFor";
32
+ import { digestOf } from "../effects/digest";
30
33
  import { buildUiBlock, UI_TOOL_NAME, uiProviderTool } from "../blocks/uiTool";
31
34
  import { projectReceipt } from "../blocks/receipts";
32
35
  import { isAskBlock } from "../blocks/types";
36
+ import { ACQUISITION_CALLS, synthesizeAskBlock, synthesizeShortfallBlock } from "./askGate";
37
+ import { couldBeToolCallEnvelope, parseTextToolCalls, SALVAGE_NOTE } from "./textToolCalls";
38
+ import { budgetForRequest } from "./historyBudget";
33
39
 
34
40
  /** Beyond this a tool result costs more in tokens than it can possibly be worth. */
35
41
  const MAX_RESULT_CHARS = 24_000;
@@ -39,11 +45,46 @@ const MAX_RESULT_CHARS = 24_000;
39
45
  * store dump per step makes the trace unreadable rather than more useful.
40
46
  */
41
47
  const MAX_TRACE_RESULT_CHARS = 2_000;
48
+ /**
49
+ * How many identical rounds before the model is told it is looping.
50
+ *
51
+ * `maxSteps` alone was the whole answer, and it is the wrong shape: a model
52
+ * that misread a result retries the same call until the cap, then the turn
53
+ * ends with nothing — the user watches twelve identical rows go by and gets
54
+ * no answer. The cap is a backstop, not feedback. Telling the model what it
55
+ * is doing gives it the chance to change approach, and costs one sentence.
56
+ *
57
+ * Signature is name + action + arguments, key-sorted so `{b,a}` and `{a,b}`
58
+ * are the same call. Three, not two: a legitimate retry after a transient
59
+ * failure is normal, and warning on it would be noise.
60
+ */
61
+ const LOOP_REPEATS = 3;
62
+ function callSignature(calls) {
63
+ const sort = v => {
64
+ if (v === null || typeof v !== "object") return v;
65
+ if (Array.isArray(v)) return v.map(sort);
66
+ const o = v;
67
+ return Object.keys(o).sort().reduce((a, k) => (a[k] = sort(o[k]), a), {});
68
+ };
69
+ return calls.map(c => `${c.name}:${JSON.stringify(sort(c.input ?? {}))}`).join("|");
70
+ }
71
+ const LOOP_WARNING = "[Buoy] You have now made this exact call three times in a row and gotten the same thing back. Repeating it again will not change the answer. Read the result you already have, then either try a different tool or different arguments, or tell the user plainly what you could not get and what you tried.";
42
72
  const DEFAULT_MAX_STEPS = 12;
43
73
  const DEFAULT_TURN_MS = 180_000;
44
74
  function findDescriptor(catalog, toolId, action) {
45
75
  return catalog.find(t => t.toolId === toolId)?.actions.find(a => a.action === action);
46
76
  }
77
+
78
+ /**
79
+ * Why a picture did not render, in the model's own terms.
80
+ *
81
+ * Two messages, not one, because they answer different questions: the first
82
+ * is "your block was refused", the second is "your block was shown but with
83
+ * holes in it". The literal opening of the rejection is pinned by the
84
+ * trajectory classifier (evals/ask-buoy/lib/trajectory.ts) — keep it.
85
+ */
86
+ const URL_PROVENANCE_REJECTION = "Rejected: every image URL must be one you read from a tool result in THIS conversation (images.list, a storage value, a response body). Never invent or remember URLs — read them first, then show them.";
87
+ const droppedImageNote = n => ` NOTE: ${n} image ${n === 1 ? "URL was" : "URLs were"} dropped — they were not read from a tool result in this conversation, so the card the user is looking at has no picture there. Do not tell them it does; read the URLs and show it again, or say the artwork is missing.`;
47
88
  function encodeResult(value) {
48
89
  let text;
49
90
  try {
@@ -88,11 +129,30 @@ function summarise(value) {
88
129
  function countOf(n) {
89
130
  return n === 1 ? "1 item" : `${n} items`;
90
131
  }
132
+
133
+ /**
134
+ * A call that came back reporting its own failure.
135
+ *
136
+ * Buoy adapters RESOLVE with `{ok:false, error}` rather than throwing, so the
137
+ * dispatch's try/catch never fires and a refused write looked like a completed
138
+ * one: the row read "Done", and — much worse — the effect ledger recorded a
139
+ * change that had not happened, so the changes bar offered to undo nothing.
140
+ * Caught on device: a typed-edit refusal ("qty is a number, 'many' is text")
141
+ * left the bag untouched and the bar claiming one permanent change.
142
+ *
143
+ * `ok === false` is the same signal `summarise` above already trusts to turn a
144
+ * result into the row's text, so this reads it no more liberally than the UI
145
+ * already does.
146
+ */
147
+ function reportsFailure(value) {
148
+ return typeof value === "object" && value !== null && value.ok === false;
149
+ }
91
150
  export async function* runAgentTurn(input) {
92
151
  const {
93
152
  provider,
94
153
  catalog,
95
154
  system,
155
+ systemVolatile,
96
156
  model,
97
157
  maxTokens,
98
158
  dispatch,
@@ -112,15 +172,72 @@ export async function* runAgentTurn(input) {
112
172
  });
113
173
  tools.push(uiProviderTool());
114
174
  const maxSteps = policy.maxSteps ?? DEFAULT_MAX_STEPS;
115
- const deadline = Date.now() + DEFAULT_TURN_MS;
175
+ /**
176
+ * The conversation this turn belongs to. Handed back on every `record` so a
177
+ * dispatch that settles after the user started a new conversation lands in
178
+ * the app (nothing can pull it back) but not in the new chat's undo bar.
179
+ */
180
+ const ledgerGeneration = ledger.generation;
181
+ /**
182
+ * What every request costs before a single message is added: the tool
183
+ * definitions, the system prompt, and the room the model needs to reply.
184
+ *
185
+ * Measured, not guessed — the tool block alone is ~32k characters for a
186
+ * typical install and ~94k with all 24 tools available, which is far too
187
+ * much to leave out of a budget. `maxTokens` is multiplied by four because
188
+ * the budget is in characters.
189
+ */
190
+ const requestReserve = system.length + (systemVolatile?.length ?? 0) + tools.reduce((n, t) => n + t.description.length + JSON.stringify(t.inputSchema).length, 0) + maxTokens * 4;
191
+ /**
192
+ * Caps the AGENT's wall-clock, not the user's. Time spent parked on an
193
+ * approval sheet is pushed onto the deadline as it is spent (see the
194
+ * `requestApproval` await below) — otherwise a user who takes three minutes
195
+ * to tap Approve gets the change applied and then "this is taking too long"
196
+ * for the turn that applied it.
197
+ */
198
+ let deadline = Date.now() + DEFAULT_TURN_MS;
116
199
 
117
- // Every http(s) URL that appeared in a tool result THIS turn. The prompt
118
- // tells the model image URLs must come from here; this Set is the
119
- // enforcement — a model-remembered or injected URL in an image block is
200
+ // Every http(s) URL that appeared in a tool result in this CONVERSATION.
201
+ // The prompt tells the model image URLs must come from here; this Set is
202
+ // the enforcement — a model-remembered or injected URL in an image block is
120
203
  // both a fabrication vector and a data-exfil beacon (the GET carries
121
- // whatever is encoded into the URL).
122
- const seenUrls = new Set();
204
+ // whatever is encoded into the URL). Session-owned when the caller passes
205
+ // one; see RunTurnInput.seenUrls for why it is not per-turn.
206
+ const seenUrls = input.seenUrls ?? new Set();
123
207
  const URL_RE = /https?:\/\/[^\s"'\\)\]}>]+/g;
208
+
209
+ /** Whether any text has gone out THIS TURN — see `breakBefore`. */
210
+ let answerStarted = false;
211
+
212
+ /**
213
+ * The last few rounds' call signatures. See LOOP_REPEATS — a run of
214
+ * identical rounds gets one warning appended to the results, once, so the
215
+ * model is told rather than silently capped.
216
+ */
217
+ const recentSignatures = [];
218
+ let loopWarned = false;
219
+
220
+ /**
221
+ * Everything the model SAID this turn, in order. Read once, at the end, by
222
+ * the ask gate — a question asked in round one and left hanging is the same
223
+ * dead end as one asked in the last round. See askGate.ts.
224
+ */
225
+ const spoken = [];
226
+ /**
227
+ * Whether the user already has something to tap. An `actions` or
228
+ * `suggestions` block is the model doing this job itself, and a second card
229
+ * asking the same thing underneath it is worse than none.
230
+ *
231
+ * Note this does NOT suppress the shortfall gate below — that one asks
232
+ * whether the user can tap the GAP, which unrelated chips do not answer.
233
+ */
234
+ let offeredTap = false;
235
+ /**
236
+ * Whether this turn actually went and looked — navigated, described the
237
+ * screen, tapped something, refetched. The shortfall gate stays quiet when
238
+ * it did: an agent that tried and came up short has earned its answer.
239
+ */
240
+ let attemptedAcquisition = false;
124
241
  for (let step = 0; step < maxSteps; step++) {
125
242
  if (signal?.aborted) {
126
243
  yield {
@@ -129,24 +246,112 @@ export async function* runAgentTurn(input) {
129
246
  return messages;
130
247
  }
131
248
  if (Date.now() > deadline) {
249
+ // Same shape as the step cap below. This used to be an `error`, which
250
+ // drew the failed-turn card with Try again — and Try again re-sends the
251
+ // question, restarting the whole investigation that just ran out of
252
+ // time. Continue picks up with everything so far still in history.
132
253
  yield {
133
- type: "error",
134
- message: "This is taking too long — stopping here."
254
+ type: "block",
255
+ block: {
256
+ id: `cap${Date.now()}`,
257
+ kind: "notice",
258
+ tone: "warning",
259
+ // Keeps the "This is taking too long" prefix: the eval classifier
260
+ // (evals/ask-buoy/lib/trajectory.ts) tells a time-out from a step
261
+ // cap by it.
262
+ text: `This is taking too long — stopped after ${Math.round(DEFAULT_TURN_MS / 1000)}s. What it found so far is above.`,
263
+ actions: [{
264
+ label: "Continue",
265
+ primary: true,
266
+ send: "Continue where you left off."
267
+ }]
268
+ }
135
269
  };
136
270
  yield {
137
271
  type: "done"
138
272
  };
139
273
  return messages;
140
274
  }
275
+
276
+ /**
277
+ * BUDGET, EVERY REQUEST — not once per turn.
278
+ *
279
+ * The session trims when a turn opens, and that used to be the only
280
+ * check. But a turn is not one request: each step appends an assistant
281
+ * message and a tool-results message, and a single result can be 24,000
282
+ * characters. A twelve-step investigation could therefore add six figures
283
+ * of history AFTER the only check had run, and the turn died on the
284
+ * provider's context limit with an opaque error — in exactly the long
285
+ * sessions the tool is for.
286
+ *
287
+ * The current round is never touched: it is the question being answered.
288
+ */
289
+ const budgeted = budgetForRequest(messages, requestReserve);
290
+ if (budgeted.droppedRounds > 0) {
291
+ messages.length = 0;
292
+ messages.push(...budgeted.messages);
293
+ yield {
294
+ type: "history-trimmed",
295
+ droppedRounds: budgeted.droppedRounds
296
+ };
297
+ } else if (budgeted.messages !== messages) {
298
+ messages.length = 0;
299
+ messages.push(...budgeted.messages);
300
+ }
141
301
  let text = "";
302
+ /** See `breakBefore`: set once the first text of THIS round has gone out. */
303
+ let textThisRound = false;
304
+ const textEvent = delta => {
305
+ // Only when this round opens a NEW paragraph of an answer that has
306
+ // already started. A turn whose first words arrive in round three has
307
+ // nothing to be separated from, and marking that would put the flag on
308
+ // events where it means nothing.
309
+ const opensNewRound = answerStarted && !textThisRound;
310
+ textThisRound = true;
311
+ answerStarted = true;
312
+ return opensNewRound ? {
313
+ type: "text",
314
+ delta,
315
+ breakBefore: true
316
+ } : {
317
+ type: "text",
318
+ delta
319
+ };
320
+ };
142
321
  const calls = [];
322
+ /**
323
+ * Recovered from text rather than emitted as calls — see textToolCalls.ts.
324
+ * Tracked by id so each one's result can tell the model to stop doing that;
325
+ * they are otherwise ordinary calls and get every check the others get.
326
+ */
327
+ const salvagedIds = new Set();
328
+ /**
329
+ * Text withheld from the screen while it could still be a tool-call
330
+ * envelope. Streamed text cannot be un-shown, so the choice has to be made
331
+ * before it leaves — and a model that writes its call as JSON must not
332
+ * have that JSON become the answer the user reads.
333
+ */
334
+ let held = "";
335
+ let holding = true;
143
336
  // Carried, never read: Anthropic requires the turn's thinking blocks back
144
337
  // verbatim with its tool results. See providers/anthropic.ts note 5.
145
338
  const thinking = [];
146
339
  let failed = false;
340
+ /**
341
+ * What the provider said about how the stream ended.
342
+ *
343
+ * Undefined means it never said — a bare EOF, or a host-supplied provider
344
+ * whose generator simply returned. Both are treated as incomplete, so the
345
+ * fail-safe direction is the default rather than something each adapter
346
+ * has to remember to opt into.
347
+ */
348
+ let outcome;
349
+ /** Set for the two kinds that are recoverable rather than a hard failure. */
350
+ let incomplete;
147
351
  for await (const ev of provider.send({
148
352
  messages,
149
353
  system,
354
+ systemVolatile,
150
355
  tools,
151
356
  model,
152
357
  maxTokens,
@@ -154,10 +359,18 @@ export async function* runAgentTurn(input) {
154
359
  })) {
155
360
  if (ev.type === "text") {
156
361
  text += ev.delta;
157
- yield {
158
- type: "text",
159
- delta: ev.delta
160
- };
362
+ if (!holding) {
363
+ yield textEvent(ev.delta);
364
+ } else if (couldBeToolCallEnvelope(text)) {
365
+ held += ev.delta;
366
+ } else {
367
+ // Not an envelope after all. Release everything at once and stream
368
+ // the rest as usual — the reader loses nothing but a few characters
369
+ // of latency at the very start of the answer.
370
+ holding = false;
371
+ held = "";
372
+ yield textEvent(text);
373
+ }
161
374
  } else if (ev.type === "tool-call") {
162
375
  calls.push(ev.call);
163
376
  } else if (ev.type === "thinking") {
@@ -173,35 +386,154 @@ export async function* runAgentTurn(input) {
173
386
  };
174
387
  }
175
388
  } else if (ev.type === "error") {
176
- // A mid-stream error can arrive on an already-committed 200.
177
- yield {
178
- type: "error",
179
- message: ev.message
180
- };
181
- failed = true;
182
- } else if (ev.type === "done" && ev.usage) {
183
- // Surfaced so a host can meter cost per seat. Parsed for a long time;
184
- // went nowhere anyone could see.
185
- yield {
186
- type: "usage",
187
- input: ev.usage.input,
188
- output: ev.usage.output
189
- };
389
+ if (ev.kind === "truncated" || ev.kind === "timeout") {
390
+ // Not a failure the user did anything about, and not one Try again
391
+ // can fix — re-sending the QUESTION restarts an investigation that
392
+ // may already have written things. Held back from the error card
393
+ // (which is what draws Try again) and handled below as a recoverable
394
+ // stop with a Continue button.
395
+ incomplete = {
396
+ message: ev.message
397
+ };
398
+ } else {
399
+ // A mid-stream error can arrive on an already-committed 200.
400
+ yield {
401
+ type: "error",
402
+ message: ev.message
403
+ };
404
+ failed = true;
405
+ }
406
+ } else if (ev.type === "done") {
407
+ outcome = ev.outcome;
408
+ if (ev.usage) {
409
+ // Surfaced so a host can meter cost per seat. Parsed for a long time;
410
+ // went nowhere anyone could see.
411
+ yield {
412
+ type: "usage",
413
+ ...ev.usage,
414
+ model: ev.model
415
+ };
416
+ }
190
417
  }
191
418
  }
192
419
  if (failed) {
420
+ // Whatever was withheld is the model's own words on the way to an error.
421
+ // Show it rather than swallowing it.
422
+ if (held) yield textEvent(held);
423
+ yield {
424
+ type: "done"
425
+ };
426
+ return messages;
427
+ }
428
+
429
+ /**
430
+ * THE COMPLETION GATE. Nothing below this runs a tool unless the provider
431
+ * said the response was whole.
432
+ *
433
+ * The failure it exists for is not theoretical and does not look like a
434
+ * failure: a connection cut after the model emitted
435
+ * `{"key":"cart","value":[]}` leaves valid JSON, a plausible-looking plan,
436
+ * and no error anywhere. The old code parsed those arguments, dispatched
437
+ * the write, sent the result back and carried on — a real mutation from
438
+ * half a sentence. Bare EOF is not consent.
439
+ *
440
+ * `output-limited` is the same shape from a different cause: the model hit
441
+ * max_tokens partway through planning. What it managed to SAY is real and
442
+ * is shown; what it was part-way through DOING is not run.
443
+ */
444
+ if (!incomplete && outcome !== "completed") {
445
+ incomplete = {
446
+ message: outcome === "output-limited" ? calls.length ? "The model ran out of room mid-plan, so nothing was run. What it found so far is above." : "The model ran out of room before finishing this answer." : "The answer ended before the endpoint said it was finished, so nothing was run."
447
+ };
448
+ }
449
+ if (incomplete) {
450
+ // Held text is the model's own words on the way out. Show it: the user
451
+ // watched it arrive and hiding it now reads as the app losing work.
452
+ if (held) yield textEvent(held);
453
+ /**
454
+ * Text only — never the tool calls, and never the thinking.
455
+ *
456
+ * An assistant turn carrying `tool_use` with no matching `tool_result`
457
+ * is an instant 400 on the next request, and these calls are exactly the
458
+ * ones that must not be answered. Keeping the text preserves what the
459
+ * user can see on screen for whatever comes next.
460
+ */
461
+ if (text.trim()) messages.push({
462
+ role: "assistant",
463
+ text
464
+ });
465
+ yield {
466
+ type: "block",
467
+ block: {
468
+ id: `cut${Date.now()}`,
469
+ kind: "notice",
470
+ tone: "warning",
471
+ text: incomplete.message,
472
+ actions: [{
473
+ label: "Continue",
474
+ primary: true,
475
+ send: "Continue where you left off."
476
+ }]
477
+ }
478
+ };
193
479
  yield {
194
480
  type: "done"
195
481
  };
196
482
  return messages;
197
483
  }
484
+ if (held) {
485
+ // A model that wrote its call out as JSON instead of calling it. Run it,
486
+ // and show its `reply` — never the blob. See textToolCalls.ts.
487
+ const salvaged = calls.length === 0 ? parseTextToolCalls(held, catalog) : undefined;
488
+ if (salvaged) {
489
+ for (const call of salvaged.calls) {
490
+ calls.push(call);
491
+ salvagedIds.add(call.id);
492
+ }
493
+ text = salvaged.reply;
494
+ if (salvaged.reply) yield textEvent(salvaged.reply);
495
+ } else {
496
+ yield textEvent(held);
497
+ }
498
+ held = "";
499
+ }
198
500
  messages.push({
199
501
  role: "assistant",
200
502
  text,
201
503
  toolCalls: calls.length ? calls : undefined,
202
504
  thinking: thinking.length ? thinking : undefined
203
505
  });
506
+ if (text.trim()) spoken.push(text.trim());
204
507
  if (calls.length === 0) {
508
+ // The turn is over and the model wrote its own words. Two ways those
509
+ // words end badly, and they are different failures — see askGate.ts.
510
+ const said = spoken.join("\n\n");
511
+
512
+ // 1. It answered a smaller question than it was asked and handed the
513
+ // rest back. Fires even when chips were offered: the question is
514
+ // whether the user can tap the GAP, not whether they can tap.
515
+ const gap = synthesizeShortfallBlock({
516
+ text: said,
517
+ attempted: attemptedAcquisition
518
+ });
519
+ if (gap) yield {
520
+ type: "block",
521
+ block: gap
522
+ };
523
+
524
+ // 2. It ended waiting on the user in prose. Suppressed once anything
525
+ // tappable is on screen, the gap button included.
526
+ const asked = offeredTap || gap ? null : synthesizeAskBlock(said);
527
+ if (asked) {
528
+ yield {
529
+ type: "block",
530
+ block: asked
531
+ };
532
+ yield {
533
+ type: "awaiting-user",
534
+ blockId: asked.id
535
+ };
536
+ }
205
537
  yield {
206
538
  type: "done"
207
539
  };
@@ -209,8 +541,220 @@ export async function* runAgentTurn(input) {
209
541
  }
210
542
  const results = [];
211
543
  let realmWillDie = false;
544
+
545
+ /**
546
+ * Reads, already in flight.
547
+ *
548
+ * The loop below stays strictly serial — every yield, every ledger entry
549
+ * and every approval keeps its order — but a round of independent READS
550
+ * used to pay each round trip end to end. The prompt tells the model to
551
+ * chain reads freely ("reading is cheap; do it"), and a three-read round
552
+ * cost three times the device latency for no reason.
553
+ *
554
+ * ONLY when the whole batch is safe to overlap: every call a catalog read,
555
+ * none needing approval, no `buoy_ui`, nothing snapshot-only or
556
+ * session-local. One write, one approval or one unknown action in the
557
+ * batch and nothing is prefetched — the mixed case is where ordering
558
+ * actually matters, and it is not worth the risk for the latency.
559
+ */
560
+ const prefetch = new Map();
561
+ if (calls.length > 1) {
562
+ const safe = calls.every(c => {
563
+ if (c.name === UI_TOOL_NAME) return false;
564
+ const id = fromProviderToolName(c.name, catalog);
565
+ if (!id || id === "ask-buoy") return false;
566
+ const act = c.input?.action;
567
+ if (typeof act !== "string" || act === SNAPSHOT_ACTION) return false;
568
+ const d = findDescriptor(catalog, id, act);
569
+ if (!d || d.effect !== "read") return false;
570
+ return decide({
571
+ descriptor: d,
572
+ toolId: id,
573
+ policy,
574
+ isRelease
575
+ }).verdict === "allow";
576
+ });
577
+ if (safe) {
578
+ for (const c of calls) {
579
+ const id = fromProviderToolName(c.name, catalog);
580
+ const act = String(c.input.action);
581
+ const raw = {
582
+ ...(c.input.params ?? {})
583
+ };
584
+ const d = findDescriptor(catalog, id, act);
585
+ const norm = normalizeParams(id, act, raw);
586
+ const withDefaults = applyDefaults(norm.params, d.params);
587
+ if (!validateParams(withDefaults, d.params).ok) continue;
588
+ // Rejections are swallowed here and re-awaited in the loop, where
589
+ // they are turned into the same tool_result they always were.
590
+ const p = Promise.resolve(dispatch(id, act, withDefaults)).catch(e => {
591
+ throw e;
592
+ });
593
+ p.catch(() => {});
594
+ prefetch.set(c.id, p);
595
+ }
596
+ }
597
+ }
212
598
  let awaiting = null;
599
+
600
+ /**
601
+ * A call Stop got to first — as a row, so the transcript says which ones.
602
+ *
603
+ * Resolved leniently: the call has not been validated yet, and a label
604
+ * that falls back to `tool.action` is better than no row for a call the
605
+ * user needs to know did not run.
606
+ */
607
+ const stoppedStep = call => {
608
+ const toolId = fromProviderToolName(call.name, catalog) ?? call.name;
609
+ const action = typeof call.input?.action === "string" ? call.input.action : "";
610
+ const descriptor = action ? findDescriptor(catalog, toolId, action) : undefined;
611
+ const params = call.input?.params ?? {};
612
+ return [{
613
+ type: "tool-start",
614
+ id: call.id,
615
+ toolId,
616
+ action,
617
+ label: descriptor ? describeCall(toolId, descriptor, params) : `${toolId}.${action}`,
618
+ description: descriptor?.summary ?? "",
619
+ params,
620
+ effect: descriptor?.effect ?? "read"
621
+ }, {
622
+ type: "tool-end",
623
+ id: call.id,
624
+ ok: false,
625
+ summary: "stopped",
626
+ durationMs: 0
627
+ }];
628
+ };
629
+
630
+ /**
631
+ * THE BATCH BARRIER — what the whole batch will do, decided before any of
632
+ * it does anything.
633
+ *
634
+ * A model answers with several calls at once, and they used to run one at
635
+ * a time with each approval asked when its own call came up. So
636
+ * `[set the flag, delete the cache]` wrote the flag, THEN asked about the
637
+ * delete — and a user who declined had already changed the app, with the
638
+ * bubble reading "changed the flag" directly above "left the cache alone".
639
+ * Declining is supposed to mean the plan does not happen.
640
+ *
641
+ * The fix is not to ask again later or to refuse afterwards, neither of
642
+ * which can unrun a write. It is to ask FIRST: every gated call in the
643
+ * batch is put to the user before the batch's first mutation dispatches,
644
+ * so consent is given with the whole plan visible.
645
+ *
646
+ * Resolution here is pure and duplicates the loop's own — deliberately.
647
+ * Anything it cannot resolve (an unknown tool, a bad shape) comes back as
648
+ * neither mutating nor gated, so the loop reports it exactly as it always
649
+ * did and the barrier simply does not fire. Failing back to today's
650
+ * behaviour is the right failure for a safety gate to have.
651
+ */
652
+ const planned = calls.map(call => {
653
+ if (call.name === UI_TOOL_NAME) return undefined;
654
+ const toolId = fromProviderToolName(call.name, catalog);
655
+ const action = typeof call.input?.action === "string" ? call.input.action : undefined;
656
+ if (!toolId || !action) return undefined;
657
+ const descriptor = findDescriptor(catalog, toolId, action);
658
+ if (!descriptor) return undefined;
659
+ const raw = call.input.params ?? {};
660
+ const params = applyDefaults(normalizeParams(toolId, action, raw).params, descriptor.params);
661
+ if (!validateParams(params, descriptor.params).ok) return undefined;
662
+ const verdict = decide({
663
+ descriptor,
664
+ toolId,
665
+ policy,
666
+ isRelease
667
+ });
668
+ return {
669
+ call,
670
+ toolId,
671
+ action,
672
+ descriptor,
673
+ params,
674
+ label: describeCall(toolId, descriptor, params),
675
+ mutates: descriptor.effect !== "read",
676
+ gated: verdict.verdict === "needs-approval",
677
+ reason: verdict.verdict === "needs-approval" ? verdict.reason : ""
678
+ };
679
+ });
680
+ const gatedInBatch = planned.some(p => p?.gated);
681
+ /** Decisions the barrier has already taken, so no call is asked twice. */
682
+ const decided = new Map();
683
+ let barrierRun = false;
684
+ let batchDeclined = false;
685
+
686
+ /** Put every gated call in the batch to the user, in the order written. */
687
+ async function* runBarrier() {
688
+ barrierRun = true;
689
+ for (const p of planned) {
690
+ if (!p?.gated || decided.has(p.call.id)) continue;
691
+ yield {
692
+ type: "approval-required",
693
+ id: p.call.id,
694
+ toolId: p.toolId,
695
+ action: p.action,
696
+ label: p.label,
697
+ description: p.descriptor.summary,
698
+ reason: p.reason
699
+ };
700
+ const askedAt = Date.now();
701
+ const targetDigest = await readTargetDigest({
702
+ toolId: p.toolId,
703
+ action: p.action,
704
+ params: p.params,
705
+ dispatch,
706
+ catalog
707
+ });
708
+ const approved = requestApproval ? await requestApproval({
709
+ id: p.call.id,
710
+ toolId: p.toolId,
711
+ action: p.action,
712
+ label: p.label,
713
+ description: p.descriptor.summary,
714
+ reason: p.reason,
715
+ params: p.params,
716
+ targetDigest
717
+ }) : false;
718
+ // The user's deliberation is not the agent's runtime.
719
+ deadline += Date.now() - askedAt;
720
+ decided.set(p.call.id, approved);
721
+ if (!approved) {
722
+ // One refusal ends the PLAN, not just the call. The remaining
723
+ // changes were proposed together and nothing has run yet, so
724
+ // carrying on with the rest would apply half of something the user
725
+ // has just turned down.
726
+ batchDeclined = true;
727
+ return;
728
+ }
729
+ }
730
+ }
213
731
  for (const call of calls) {
732
+ /**
733
+ * Stop was pressed while this batch was running.
734
+ *
735
+ * The abort signal used to be read only at the top of a ROUND, so a
736
+ * response carrying three calls ran all three after the tap — and the
737
+ * bubble then said "Stopped." over two writes the user believed it had
738
+ * prevented. Each call now checks before it starts. A dispatch already
739
+ * in flight is left to finish (nothing can pull a store write back out
740
+ * of an adapter) and keeps its receipt; everything after it is reported
741
+ * as not run, each as its own row, so the transcript says exactly which
742
+ * changes landed. A result is still pushed for every call — the provider
743
+ * requires one per tool_use or the next request is rejected.
744
+ */
745
+ if (signal?.aborted) {
746
+ // A `buoy_ui` card that was never shown is not a step; no row for it.
747
+ if (call.name !== UI_TOOL_NAME) {
748
+ for (const ev of stoppedStep(call)) yield ev;
749
+ }
750
+ results.push({
751
+ toolCallId: call.id,
752
+ content: "Not run — the user pressed Stop before this call started.",
753
+ isError: true
754
+ });
755
+ continue;
756
+ }
757
+
214
758
  // `buoy_ui`: show or ask. Validated like any action; an Ask block ends
215
759
  // the turn after this batch — the answer comes back as a user message.
216
760
  if (call.name === UI_TOOL_NAME) {
@@ -218,19 +762,35 @@ export async function* runAgentTurn(input) {
218
762
  if (!built.ok || !built.block) {
219
763
  results.push({
220
764
  toolCallId: call.id,
221
- content: `Invalid ${UI_TOOL_NAME} block:\n${built.errors.join("\n")}`,
765
+ /**
766
+ * The trailing sentence is the expensive half.
767
+ *
768
+ * A model that gets a block bounced rewrites its WHOLE answer on
769
+ * the retry, and one assistant bubble spans every round of a turn
770
+ * (see `breakBefore`), so the user reads the same paragraph again
771
+ * — measured at four near-identical closings from three schema
772
+ * misses in a row on one live turn. The block is wrong; the
773
+ * sentence under it was fine.
774
+ */
775
+ content: `Invalid ${UI_TOOL_NAME} block:\n${built.errors.join("\n")}\n\nResend ONLY the corrected block. Do not rewrite your answer — whatever you already said this turn has been shown to the user, and saying it again repeats it on their screen.`,
222
776
  isError: true
223
777
  });
224
778
  continue;
225
779
  }
226
780
  // Provenance for pictures, ENFORCED: an image URL the model did not
227
781
  // read from a tool result this turn does not render. Live or nothing.
782
+ // Dropping pictures SILENTLY is its own bug: the model then writes
783
+ // "here they are, with artwork" over a card that has none, and the
784
+ // user is told something untrue about their own screen. Every drop
785
+ // comes back in the tool result so the next sentence can be honest.
786
+ let droppedImages = 0;
228
787
  if (built.block.kind === "imageGrid") {
229
788
  const kept = built.block.images.filter(img => seenUrls.has(img.url));
789
+ droppedImages = built.block.images.length - kept.length;
230
790
  if (kept.length === 0) {
231
791
  results.push({
232
792
  toolCallId: call.id,
233
- content: "Rejected: every image URL must be one you read from a tool result THIS turn (images.list, a storage value, a response body). Never invent or remember URLs — read them first.",
793
+ content: URL_PROVENANCE_REJECTION,
234
794
  isError: true
235
795
  });
236
796
  continue;
@@ -241,12 +801,17 @@ export async function* runAgentTurn(input) {
241
801
  };
242
802
  }
243
803
  if (built.block.kind === "list") {
244
- built.block = {
245
- ...built.block,
246
- items: built.block.items.map(item => item.image && !seenUrls.has(item.image) ? {
804
+ const items = built.block.items.map(item => {
805
+ if (!item.image || seenUrls.has(item.image)) return item;
806
+ droppedImages += 1;
807
+ return {
247
808
  ...item,
248
809
  image: undefined
249
- } : item)
810
+ };
811
+ });
812
+ built.block = {
813
+ ...built.block,
814
+ items
250
815
  };
251
816
  }
252
817
  yield {
@@ -254,6 +819,7 @@ export async function* runAgentTurn(input) {
254
819
  block: built.block,
255
820
  forToolCallId: call.id
256
821
  };
822
+ if (built.block.kind === "actions" || built.block.kind === "suggestions") offeredTap = true;
257
823
  if (isAskBlock(built.block)) {
258
824
  awaiting = built.block.id;
259
825
  results.push({
@@ -263,7 +829,7 @@ export async function* runAgentTurn(input) {
263
829
  } else {
264
830
  results.push({
265
831
  toolCallId: call.id,
266
- content: "Shown to the user. Don't repeat its contents in text."
832
+ content: "Shown to the user. Don't repeat its contents in text." + (droppedImages > 0 ? droppedImageNote(droppedImages) : "")
267
833
  });
268
834
  }
269
835
  continue;
@@ -271,12 +837,40 @@ export async function* runAgentTurn(input) {
271
837
  const toolId = fromProviderToolName(call.name, catalog);
272
838
  const action = call.input.action;
273
839
 
840
+ /**
841
+ * A call that never reached a tool at all.
842
+ *
843
+ * Same rule as a refusal: the model made a move, the move went nowhere,
844
+ * and a transcript that hides it leaves the reader watching the agent
845
+ * change its mind for no visible reason. Caught on device — a model
846
+ * guessed three navigation tool names that do not exist, spent 41
847
+ * seconds doing it, and the only trace was a sentence it chose to write.
848
+ * `read` because nothing was touched.
849
+ */
850
+ const deadCall = (label, summary) => [{
851
+ type: "tool-start",
852
+ id: call.id,
853
+ toolId: toolId ?? call.name,
854
+ action: action ?? "",
855
+ label,
856
+ description: "",
857
+ params: call.input.params ?? {},
858
+ effect: "read"
859
+ }, {
860
+ type: "tool-end",
861
+ id: call.id,
862
+ ok: false,
863
+ summary,
864
+ durationMs: 0
865
+ }];
866
+
274
867
  // Two different mistakes, kept apart on purpose. Merging them tells a
275
868
  // model that forgot `action` that the TOOL does not exist, and it stops
276
869
  // reaching for a tool that was fine — the same distinction
277
870
  // `dispatchToolAction` draws between an unknown tool and an unknown
278
871
  // action, for the same reason.
279
872
  if (!toolId) {
873
+ for (const ev of deadCall(call.name, "No such tool")) yield ev;
280
874
  results.push({
281
875
  toolCallId: call.id,
282
876
  content: `There is no tool called "${call.name}". Use one of the tools you were given.`,
@@ -285,6 +879,7 @@ export async function* runAgentTurn(input) {
285
879
  continue;
286
880
  }
287
881
  if (!action) {
882
+ for (const ev of deadCall(call.name, "No action given")) yield ev;
288
883
  results.push({
289
884
  toolCallId: call.id,
290
885
  content: `"${call.name}" needs an "action" — it is required, and it names which of this tool's actions to run. See the action list in the tool's description.`,
@@ -298,6 +893,7 @@ export async function* runAgentTurn(input) {
298
893
  // has nowhere to go but guess again, and the next guess is no better
299
894
  // informed than the last.
300
895
  const known = catalog.find(t => t.toolId === toolId)?.actions.map(a => a.action).join(", ");
896
+ for (const ev of deadCall(`${toolId}.${action}`, "No such action")) yield ev;
301
897
  results.push({
302
898
  toolCallId: call.id,
303
899
  content: `"${action}" is not an action on ${toolId}.${known ? ` Valid actions: ${known}.` : ""}`,
@@ -306,7 +902,11 @@ export async function* runAgentTurn(input) {
306
902
  continue;
307
903
  }
308
904
  const raw = call.input.params ?? {};
309
- const params = applyDefaults(raw, descriptor.params);
905
+ // Accept the parameter-name misses every model makes (queryKey for
906
+ // queryHash, requestId for id, flat rule fields, …) before validation,
907
+ // so a mechanical alias never costs a turn. See normalizeParams.ts.
908
+ const normalized = normalizeParams(toolId, action, raw);
909
+ const params = applyDefaults(normalized.params, descriptor.params);
310
910
  const check = validateParams(params, descriptor.params);
311
911
  if (!check.ok) {
312
912
  results.push({
@@ -316,6 +916,7 @@ export async function* runAgentTurn(input) {
316
916
  });
317
917
  continue;
318
918
  }
919
+ const aliasTrailer = aliasNote(normalized.notes);
319
920
  const label = describeCall(toolId, descriptor, params);
320
921
  const verdict = decide({
321
922
  descriptor,
@@ -323,14 +924,29 @@ export async function* runAgentTurn(input) {
323
924
  policy,
324
925
  isRelease
325
926
  });
927
+
928
+ // A step the user can SEE, for a call that never ran. `tool-end` alone
929
+ // updates a row that was never opened, so a refusal used to leave no
930
+ // trace in the transcript at all — the model simply changed its mind
931
+ // between one sentence and the next.
932
+ const unrunStep = summary => [{
933
+ type: "tool-start",
934
+ id: call.id,
935
+ toolId,
936
+ action,
937
+ label,
938
+ description: descriptor.summary,
939
+ params,
940
+ effect: descriptor.effect
941
+ }, {
942
+ type: "tool-end",
943
+ id: call.id,
944
+ ok: false,
945
+ summary,
946
+ durationMs: 0
947
+ }];
326
948
  if (verdict.verdict === "refuse") {
327
- yield {
328
- type: "tool-end",
329
- id: call.id,
330
- ok: false,
331
- summary: "refused",
332
- durationMs: 0
333
- };
949
+ for (const ev of unrunStep("refused")) yield ev;
334
950
  results.push({
335
951
  toolCallId: call.id,
336
952
  content: verdict.reason,
@@ -338,25 +954,77 @@ export async function* runAgentTurn(input) {
338
954
  });
339
955
  continue;
340
956
  }
341
- if (verdict.verdict === "needs-approval") {
342
- yield {
343
- type: "approval-required",
344
- id: call.id,
345
- toolId,
346
- action,
347
- label,
348
- description: descriptor.summary,
349
- reason: verdict.reason
350
- };
351
- const approved = requestApproval ? await requestApproval({
352
- id: call.id,
353
- toolId,
354
- action,
355
- label,
356
- description: descriptor.summary,
357
- reason: verdict.reason
358
- }) : false;
957
+
958
+ /**
959
+ * Nothing in this batch changes anything until every card in it has been
960
+ * answered. See `runBarrier` — this is the line that makes a decline
961
+ * mean "the plan does not happen" rather than "the rest of the plan does
962
+ * not happen".
963
+ */
964
+ const gatedHere = verdict.verdict === "needs-approval";
965
+ if (gatedInBatch && !barrierRun && (descriptor.effect !== "read" || gatedHere)) {
966
+ yield* runBarrier();
967
+ }
968
+ if (batchDeclined && (descriptor.effect !== "read" || gatedHere)) {
969
+ for (const ev of unrunStep("declined")) yield ev;
970
+ results.push({
971
+ toolCallId: call.id,
972
+ content: decided.get(call.id) === false ? "The user declined this change. Do not retry it; ask what they would prefer." : "Not run — the user declined another change in this same batch, so none of it was applied. Ask what they would prefer before proposing it again.",
973
+ isError: true
974
+ });
975
+ continue;
976
+ }
977
+ if (gatedHere) {
978
+ let approved;
979
+ if (decided.has(call.id)) {
980
+ // The barrier already put this card up, before the batch's first
981
+ // mutation. Asking again would be the same question twice.
982
+ approved = decided.get(call.id);
983
+ } else {
984
+ // The barrier could not resolve this call (see `planned`), so it was
985
+ // never offered. Ask here, exactly as this always did — a gate that
986
+ // fails closed by silently declining would be worse than one that
987
+ // asks late.
988
+ yield {
989
+ type: "approval-required",
990
+ id: call.id,
991
+ toolId,
992
+ action,
993
+ label,
994
+ description: descriptor.summary,
995
+ reason: verdict.reason
996
+ };
997
+ const askedAt = Date.now();
998
+ // What the write is about to replace, as of NOW. Only a restored
999
+ // card ever reads it back — see readTargetDigest.
1000
+ const targetDigest = await readTargetDigest({
1001
+ toolId,
1002
+ action,
1003
+ params,
1004
+ dispatch,
1005
+ catalog
1006
+ });
1007
+ approved = requestApproval ? await requestApproval({
1008
+ id: call.id,
1009
+ toolId,
1010
+ action,
1011
+ label,
1012
+ description: descriptor.summary,
1013
+ reason: verdict.reason,
1014
+ params,
1015
+ targetDigest
1016
+ }) : false;
1017
+ // The user's deliberation is not the agent's runtime. Give the clock
1018
+ // back before anything else can trip the deadline check.
1019
+ deadline += Date.now() - askedAt;
1020
+ decided.set(call.id, approved);
1021
+ }
359
1022
  if (!approved) {
1023
+ // The row is the whole point here. Without it the bubble read as a
1024
+ // contradiction — "Bumped the line from 5 to 9." straight into "I
1025
+ // left it as it was" — with nothing on screen to say a change had
1026
+ // been proposed and turned down.
1027
+ for (const ev of unrunStep("declined")) yield ev;
360
1028
  results.push({
361
1029
  toolCallId: call.id,
362
1030
  content: "The user declined this change. Do not retry it; ask what they would prefer.",
@@ -365,6 +1033,7 @@ export async function* runAgentTurn(input) {
365
1033
  continue;
366
1034
  }
367
1035
  }
1036
+ if (ACQUISITION_CALLS.has(`${toolId}.${action}`)) attemptedAcquisition = true;
368
1037
  yield {
369
1038
  type: "tool-start",
370
1039
  id: call.id,
@@ -381,7 +1050,50 @@ export async function* runAgentTurn(input) {
381
1050
  // exactly. This is the one thing that makes storage reversible here when
382
1051
  // the Scenarios engine has to treat it as permanent.
383
1052
  const before = await captureBefore(descriptor, toolId, params, dispatch);
384
- const stateBefore = await captureStateBefore(descriptor, toolId, params, dispatch);
1053
+ const stateBefore = await captureStoreState(descriptor, toolId, params, dispatch);
1054
+
1055
+ /**
1056
+ * THE FINAL GUARD. Everything above this line happened in the past.
1057
+ *
1058
+ * `decide()` ran before the approval card went up, and between there and
1059
+ * here are three awaits: the target read, the person deciding, and two
1060
+ * pre-reads. Each one yields the thread, and a person can do a lot in
1061
+ * that gap — press Stop, or flip the read-only toggle in the settings
1062
+ * sheet while the card is still on screen. Both were reproduced: the
1063
+ * write went ahead on the verdict it captured before they touched
1064
+ * anything, which is precisely the moment the product promises it will
1065
+ * not. `live` is mutated in place by the session (see `refreshLive`), so
1066
+ * asking again reads the toggles as they are NOW.
1067
+ *
1068
+ * Below this there is no `await` before `dispatch` is CALLED. That is
1069
+ * load-bearing: an await here would reopen the same gap one line lower.
1070
+ */
1071
+ if (signal?.aborted) {
1072
+ for (const ev of unrunStep("stopped")) yield ev;
1073
+ results.push({
1074
+ toolCallId: call.id,
1075
+ content: "Not run — the user pressed Stop before this call started.",
1076
+ isError: true
1077
+ });
1078
+ continue;
1079
+ }
1080
+ const stillAllowed = decide({
1081
+ descriptor,
1082
+ toolId,
1083
+ policy,
1084
+ isRelease
1085
+ });
1086
+ if (stillAllowed.verdict === "refuse") {
1087
+ // `needs-approval` is NOT re-refused: reaching here means the tap
1088
+ // already happened, and asking twice for one call is its own bug.
1089
+ for (const ev of unrunStep("refused")) yield ev;
1090
+ results.push({
1091
+ toolCallId: call.id,
1092
+ content: stillAllowed.reason,
1093
+ isError: true
1094
+ });
1095
+ continue;
1096
+ }
385
1097
  try {
386
1098
  // `getSnapshot` is reserved: it is not an adapter action, so it goes to
387
1099
  // the snapshot reader and is trimmed to the fields worth sending. See
@@ -389,21 +1101,14 @@ export async function* runAgentTurn(input) {
389
1101
  // ledger the model asks about lives HERE, not behind an adapter, so
390
1102
  // routing it through dispatch would only work on a device and would
391
1103
  // record the undo as a fresh effect.
392
- const result = action === SNAPSHOT_ACTION ? projectSnapshot(toolId, await requireSnapshot(readSnapshot, toolId), params) : toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch) : await dispatch(toolId, action, params);
1104
+ const result = action === SNAPSHOT_ACTION ? projectSnapshot(toolId, await requireSnapshot(readSnapshot, toolId), params) : toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch) : await (prefetch.get(call.id) ?? dispatch(toolId, action, params));
393
1105
  const cleaned = redact(stripSelfTraffic(result));
394
- if (descriptor.effect !== "read" && toolId !== "ask-buoy") {
395
- // A delete of something we created IS the undo, whoever asked for
396
- // it — mark the matching entry undone instead of leaving the bar
397
- // promising an undo against a rule that no longer exists.
398
- ledger.noteExternalRevert(toolId, action, params);
399
- const fx = effectFor(toolId, descriptor, params, result, before);
400
- if (fx) ledger.record(toolId, action, fx);
401
- }
1106
+ const refused = reportsFailure(result);
402
1107
  const encodedResult = encodeResult(cleaned);
403
1108
  yield {
404
1109
  type: "tool-end",
405
1110
  id: call.id,
406
- ok: true,
1111
+ ok: !refused,
407
1112
  summary: summarise(result),
408
1113
  durationMs: Date.now() - startedAt,
409
1114
  result: encodedResult.length > MAX_TRACE_RESULT_CHARS ? `${encodedResult.slice(0, MAX_TRACE_RESULT_CHARS)}…` : encodedResult
@@ -411,12 +1116,36 @@ export async function* runAgentTurn(input) {
411
1116
 
412
1117
  // The receipt: what the user sees of this result, built from the
413
1118
  // same redacted data the model gets. See blocks/receipts.ts.
414
- const receipt = projectReceipt({
1119
+ //
1120
+ // The store is read back a second time so the diff shows what the
1121
+ // write ACTUALLY did rather than what it asked for — the two differ on
1122
+ // every `path` edit and every keyed list edit, and the request-shaped
1123
+ // version reported untouched fields as deleted.
1124
+ const stateAfter = stateBefore === undefined || refused ? undefined : await captureStoreState(descriptor, toolId, params, dispatch);
1125
+ if (descriptor.effect !== "read" && toolId !== "ask-buoy" && !refused) {
1126
+ // A delete of something we created IS the undo, whoever asked for
1127
+ // it — mark the matching entry undone instead of leaving the bar
1128
+ // promising an undo against a rule that no longer exists.
1129
+ ledger.noteExternalRevert(toolId, action, params);
1130
+ // Recorded AFTER the read-back so a cache edit's entry can carry both
1131
+ // halves: the data to restore and a fingerprint of what the write
1132
+ // left behind, which is how undo knows whether the app has since
1133
+ // replaced it. See effectFor / ledger "query-write".
1134
+ const fx = effectFor(toolId, descriptor, params, result, before, {
1135
+ before: stateBefore,
1136
+ after: stateAfter
1137
+ });
1138
+ if (fx) ledger.record(toolId, action, fx, ledgerGeneration);
1139
+ }
1140
+ // No receipt for a call that changed nothing: a "what changed" card
1141
+ // under a refusal is the same lie as the ledger entry, drawn bigger.
1142
+ const receipt = refused ? undefined : projectReceipt({
415
1143
  toolId,
416
1144
  action,
417
1145
  params,
418
1146
  result: cleaned,
419
1147
  before: stateBefore ?? before,
1148
+ after: stateAfter,
420
1149
  effect: descriptor.effect
421
1150
  });
422
1151
  if (receipt) yield {
@@ -427,7 +1156,7 @@ export async function* runAgentTurn(input) {
427
1156
  for (const url of encodedResult.match(URL_RE) ?? []) seenUrls.add(url);
428
1157
  results.push({
429
1158
  toolCallId: call.id,
430
- content: encodedResult + ownedKeyNote(toolId, descriptor, params, storeKeyOwners)
1159
+ content: encodedResult + ownedKeyNote(toolId, descriptor, params, storeKeyOwners) + aliasTrailer + (salvagedIds.has(call.id) ? SALVAGE_NOTE : "")
431
1160
  });
432
1161
  if (toolId === "app" && (action === "reloadApp" || action === "reload")) {
433
1162
  realmWillDie = true;
@@ -448,6 +1177,20 @@ export async function* runAgentTurn(input) {
448
1177
  });
449
1178
  }
450
1179
  }
1180
+
1181
+ // Looping? Say so, once, on the LAST result of this round — so it arrives
1182
+ // attached to the thing being repeated rather than as a free-floating
1183
+ // user turn, and the model reads it before deciding what to do next.
1184
+ if (calls.length) {
1185
+ recentSignatures.push(callSignature(calls));
1186
+ if (recentSignatures.length > LOOP_REPEATS) recentSignatures.shift();
1187
+ const looping = !loopWarned && recentSignatures.length === LOOP_REPEATS && recentSignatures.every(sig => sig === recentSignatures[0]);
1188
+ if (looping) {
1189
+ loopWarned = true;
1190
+ const last = results[results.length - 1];
1191
+ if (last) last.content = `${last.content}\n\n${LOOP_WARNING}`;
1192
+ }
1193
+ }
451
1194
  messages.push({
452
1195
  role: "tool-results",
453
1196
  results
@@ -512,6 +1255,29 @@ function ownedKeyNote(toolId, descriptor, params, storeKeyOwners) {
512
1255
  if (!storeName) return "";
513
1256
  return `\n\n[Buoy] This wrote to disk, but "${params.key}" is the saved copy of the live "${storeName}" store, so nothing on screen has changed — the app read that key once at startup and has held the state in memory since. Call zustand.rehydrate({"storeName":"${storeName}"}) to make the app pick this up, or use zustand.setState next time to change the store directly. If you meant to set the value for the app's next launch, this is already done and no further action is needed.`;
514
1257
  }
1258
+ export { digestOf };
1259
+
1260
+ /**
1261
+ * Fingerprint the value a write is about to replace.
1262
+ *
1263
+ * Built for the approval card that outlives its turn: the app crashed with
1264
+ * "Set bag line L-01 qty to 14" still asking, relaunched, and the card came
1265
+ * back — and Allow wrote qty 14 against whatever the bag was NOW, maybe a
1266
+ * different account, maybe a line that no longer existed. A hash of the
1267
+ * params alone would not catch that (same label, same params, different
1268
+ * world). So the card carries a digest of the TARGET at ask time, and a
1269
+ * restored Allow re-reads and compares before it writes. Undefined when the
1270
+ * target cannot be read — then the card can only warn, not check.
1271
+ */
1272
+ export async function readTargetDigest(input) {
1273
+ const descriptor = findDescriptor(input.catalog ?? CATALOG, input.toolId, input.action);
1274
+ if (!descriptor || descriptor.effect === "read") return undefined;
1275
+ const store = await captureStoreState(descriptor, input.toolId, input.params, input.dispatch);
1276
+ if (store !== undefined) return digestOf(store);
1277
+ const before = await captureBefore(descriptor, input.toolId, input.params, input.dispatch);
1278
+ if (before !== undefined) return digestOf(before);
1279
+ return undefined;
1280
+ }
515
1281
  /**
516
1282
  * One human-initiated action through EVERY gate the model's calls go through:
517
1283
  * catalog lookup, param validation, policy, pre-read, effect ledger.
@@ -544,7 +1310,8 @@ export async function runGatedAction(input) {
544
1310
  error: `${toolId}.${action} is not in Buoy's catalog.`
545
1311
  };
546
1312
  }
547
- const params = applyDefaults(input.params ?? {}, descriptor.params);
1313
+ const normalized = normalizeParams(toolId, action, input.params ?? {});
1314
+ const params = applyDefaults(normalized.params, descriptor.params);
548
1315
  const check = validateParams(params, descriptor.params);
549
1316
  if (!check.ok) {
550
1317
  return {
@@ -565,11 +1332,51 @@ export async function runGatedAction(input) {
565
1332
  };
566
1333
  }
567
1334
  const before = await captureBefore(descriptor, toolId, params, dispatch);
1335
+ /**
1336
+ * The same read-back the model's path takes. It was missing here, and the
1337
+ * gap was invisible: a `setQueryData` through a block button or a restored
1338
+ * approval card recorded a ledger entry with no prior value, so Undo had
1339
+ * nothing to put back and the concurrent-writer check had no fingerprint to
1340
+ * compare. The bar said "1 undoable" over a change that could not be undone.
1341
+ */
1342
+ const stateBefore = await captureStoreState(descriptor, toolId, params, dispatch);
1343
+
1344
+ // Asked again after the pre-reads, for the same reason the model's path asks
1345
+ // again: those are awaits, and read-only can be switched on inside one.
1346
+ const stillAllowed = decide({
1347
+ descriptor,
1348
+ toolId,
1349
+ policy,
1350
+ isRelease
1351
+ });
1352
+ if (stillAllowed.verdict === "refuse") {
1353
+ return {
1354
+ ok: false,
1355
+ error: stillAllowed.reason
1356
+ };
1357
+ }
568
1358
  try {
569
1359
  const result = toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch) : await dispatch(toolId, action, params);
1360
+
1361
+ // Same rule as the model's path: a resolved `{ok:false}` is a refusal, not
1362
+ // a change. Without this a block button (or an approval answered after a
1363
+ // reload) put a phantom entry in the undo bar too.
1364
+ if (reportsFailure(result)) {
1365
+ const err = result.error;
1366
+ return {
1367
+ ok: false,
1368
+ error: typeof err === "string" ? err : "The tool refused this."
1369
+ };
1370
+ }
570
1371
  if (descriptor.effect !== "read" && toolId !== "ask-buoy") {
571
1372
  ledger.noteExternalRevert(toolId, action, params);
572
- const fx = effectFor(toolId, descriptor, params, result, before);
1373
+ // Read back AFTER the write, exactly as the model's path does — the two
1374
+ // halves together are what make a cache edit undoable.
1375
+ const stateAfter = await captureStoreState(descriptor, toolId, params, dispatch);
1376
+ const fx = effectFor(toolId, descriptor, params, result, before, {
1377
+ before: stateBefore,
1378
+ after: stateAfter
1379
+ });
573
1380
  if (fx) ledger.record(toolId, action, fx);
574
1381
  }
575
1382
  return {
@@ -641,16 +1448,46 @@ async function requireSnapshot(readSnapshot, toolId) {
641
1448
  * Zustand only for now: `getStoreState` is cheap and the write params carry
642
1449
  * the after-state, so the diff is exact without the store's cooperation.
643
1450
  */
644
- async function captureStateBefore(descriptor, toolId, params, dispatch) {
645
- if (toolId !== "zustand" || descriptor.action !== "setState") return undefined;
646
- if (typeof params.storeName !== "string") return undefined;
647
- try {
648
- return await dispatch(toolId, "getStoreState", {
649
- storeName: params.storeName
650
- });
651
- } catch {
652
- return undefined;
1451
+ /**
1452
+ * Read a zustand store whole, for the before/after halves of a write receipt.
1453
+ * Called on both sides of the dispatch — see the receipt above.
1454
+ */
1455
+ async function captureStoreState(descriptor, toolId, params, dispatch) {
1456
+ if (toolId === "zustand" && descriptor.action === "setState") {
1457
+ if (typeof params.storeName !== "string") return undefined;
1458
+ try {
1459
+ return await dispatch(toolId, "getStoreState", {
1460
+ storeName: params.storeName
1461
+ });
1462
+ } catch {
1463
+ return undefined;
1464
+ }
1465
+ }
1466
+ /**
1467
+ * The same read-both-sides trick for the CACHE.
1468
+ *
1469
+ * A zustand edit produced a "what changed" card and the identical edit to a
1470
+ * query produced nothing — for the write the prompt calls "THE way to change
1471
+ * what a server-backed screen shows". The QA tester's confirmation that the
1472
+ * right field moved was missing from the most-used tool in the product.
1473
+ */
1474
+ if (toolId === "query" && descriptor.action === "setQueryData") {
1475
+ // hashQueryKey, not JSON.stringify: React Query SORTS object keys when it
1476
+ // hashes, so a key carrying `{query:"",category:null}` hashes as
1477
+ // `{"category":null,"query":""}`. Stringifying in insertion order produced
1478
+ // a hash that matched nothing, the read came back empty, and the card
1479
+ // silently did not draw.
1480
+ const queryHash = typeof params.queryHash === "string" ? params.queryHash : Array.isArray(params.queryKey) ? hashQueryKey(params.queryKey) : undefined;
1481
+ if (!queryHash) return undefined;
1482
+ try {
1483
+ return await dispatch(toolId, "getQueryData", {
1484
+ queryHash
1485
+ });
1486
+ } catch {
1487
+ return undefined;
1488
+ }
653
1489
  }
1490
+ return undefined;
654
1491
  }
655
1492
  async function captureBefore(descriptor, toolId, params, dispatch) {
656
1493
  if (descriptor.effect === "read") return undefined;