@buoy-gg/agent-core 7.0.34 → 7.0.36

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (181) hide show
  1. package/README.md +10 -7
  2. package/lib/commonjs/blocks/receipts.js +83 -24
  3. package/lib/commonjs/blocks/receipts.js.map +1 -1
  4. package/lib/commonjs/blocks/runId.js +31 -0
  5. package/lib/commonjs/blocks/runId.js.map +1 -0
  6. package/lib/commonjs/blocks/types.js +23 -2
  7. package/lib/commonjs/blocks/types.js.map +1 -1
  8. package/lib/commonjs/blocks/uiTool.js +426 -12
  9. package/lib/commonjs/blocks/uiTool.js.map +1 -1
  10. package/lib/commonjs/catalog/catalog.g.js +121 -43
  11. package/lib/commonjs/catalog/catalog.g.js.map +1 -1
  12. package/lib/commonjs/catalog/catalog.source.json +121 -45
  13. package/lib/commonjs/catalog/catalog.types.g.js +44 -0
  14. package/lib/commonjs/catalog/catalog.types.g.js.map +1 -0
  15. package/lib/commonjs/catalog/normalizeParams.js +333 -0
  16. package/lib/commonjs/catalog/normalizeParams.js.map +1 -0
  17. package/lib/commonjs/catalog/signature.js +68 -0
  18. package/lib/commonjs/catalog/signature.js.map +1 -0
  19. package/lib/commonjs/catalog/snapshotReads.js +13 -6
  20. package/lib/commonjs/catalog/snapshotReads.js.map +1 -1
  21. package/lib/commonjs/catalog/toProviderTools.js +41 -5
  22. package/lib/commonjs/catalog/toProviderTools.js.map +1 -1
  23. package/lib/commonjs/context/buildContextPack.js +367 -19
  24. package/lib/commonjs/context/buildContextPack.js.map +1 -1
  25. package/lib/commonjs/effects/digest.js +34 -0
  26. package/lib/commonjs/effects/digest.js.map +1 -0
  27. package/lib/commonjs/effects/ledger.js +85 -1
  28. package/lib/commonjs/effects/ledger.js.map +1 -1
  29. package/lib/commonjs/engine/askGate.js +287 -0
  30. package/lib/commonjs/engine/askGate.js.map +1 -0
  31. package/lib/commonjs/engine/effectFor.js +82 -1
  32. package/lib/commonjs/engine/effectFor.js.map +1 -1
  33. package/lib/commonjs/engine/historyBudget.js +260 -0
  34. package/lib/commonjs/engine/historyBudget.js.map +1 -0
  35. package/lib/commonjs/engine/runAgentTurn.js +958 -88
  36. package/lib/commonjs/engine/runAgentTurn.js.map +1 -1
  37. package/lib/commonjs/engine/systemPrompt.js +72 -8
  38. package/lib/commonjs/engine/systemPrompt.js.map +1 -1
  39. package/lib/commonjs/engine/textToolCalls.js +276 -0
  40. package/lib/commonjs/engine/textToolCalls.js.map +1 -0
  41. package/lib/commonjs/index.js +81 -1
  42. package/lib/commonjs/index.js.map +1 -1
  43. package/lib/commonjs/policy/labels.js +19 -3
  44. package/lib/commonjs/policy/labels.js.map +1 -1
  45. package/lib/commonjs/policy/policy.js +6 -1
  46. package/lib/commonjs/policy/policy.js.map +1 -1
  47. package/lib/commonjs/policy/redact.js +42 -5
  48. package/lib/commonjs/policy/redact.js.map +1 -1
  49. package/lib/commonjs/providers/anthropic.js +340 -174
  50. package/lib/commonjs/providers/anthropic.js.map +1 -1
  51. package/lib/commonjs/providers/openai.js +229 -103
  52. package/lib/commonjs/providers/openai.js.map +1 -1
  53. package/lib/commonjs/providers/sse.js +164 -29
  54. package/lib/commonjs/providers/sse.js.map +1 -1
  55. package/lib/commonjs/providers/streamTimer.js +123 -0
  56. package/lib/commonjs/providers/streamTimer.js.map +1 -0
  57. package/lib/commonjs/providers/transport.js +107 -0
  58. package/lib/commonjs/providers/transport.js.map +1 -0
  59. package/lib/commonjs/providers/xhrStream.js +209 -0
  60. package/lib/commonjs/providers/xhrStream.js.map +1 -0
  61. package/lib/commonjs/session.js +245 -52
  62. package/lib/commonjs/session.js.map +1 -1
  63. package/lib/module/blocks/receipts.js +83 -24
  64. package/lib/module/blocks/receipts.js.map +1 -1
  65. package/lib/module/blocks/runId.js +26 -0
  66. package/lib/module/blocks/runId.js.map +1 -0
  67. package/lib/module/blocks/types.js +23 -2
  68. package/lib/module/blocks/types.js.map +1 -1
  69. package/lib/module/blocks/uiTool.js +424 -12
  70. package/lib/module/blocks/uiTool.js.map +1 -1
  71. package/lib/module/catalog/catalog.g.js +121 -43
  72. package/lib/module/catalog/catalog.g.js.map +1 -1
  73. package/lib/module/catalog/catalog.source.json +121 -45
  74. package/lib/module/catalog/catalog.types.g.js +40 -0
  75. package/lib/module/catalog/catalog.types.g.js.map +1 -0
  76. package/lib/module/catalog/normalizeParams.js +327 -0
  77. package/lib/module/catalog/normalizeParams.js.map +1 -0
  78. package/lib/module/catalog/signature.js +63 -0
  79. package/lib/module/catalog/signature.js.map +1 -0
  80. package/lib/module/catalog/snapshotReads.js +13 -6
  81. package/lib/module/catalog/snapshotReads.js.map +1 -1
  82. package/lib/module/catalog/toProviderTools.js +41 -5
  83. package/lib/module/catalog/toProviderTools.js.map +1 -1
  84. package/lib/module/context/buildContextPack.js +366 -19
  85. package/lib/module/context/buildContextPack.js.map +1 -1
  86. package/lib/module/effects/digest.js +29 -0
  87. package/lib/module/effects/digest.js.map +1 -0
  88. package/lib/module/effects/ledger.js +85 -1
  89. package/lib/module/effects/ledger.js.map +1 -1
  90. package/lib/module/engine/askGate.js +280 -0
  91. package/lib/module/engine/askGate.js.map +1 -0
  92. package/lib/module/engine/effectFor.js +83 -1
  93. package/lib/module/engine/effectFor.js.map +1 -1
  94. package/lib/module/engine/historyBudget.js +251 -0
  95. package/lib/module/engine/historyBudget.js.map +1 -0
  96. package/lib/module/engine/runAgentTurn.js +954 -88
  97. package/lib/module/engine/runAgentTurn.js.map +1 -1
  98. package/lib/module/engine/systemPrompt.js +71 -8
  99. package/lib/module/engine/systemPrompt.js.map +1 -1
  100. package/lib/module/engine/textToolCalls.js +270 -0
  101. package/lib/module/engine/textToolCalls.js.map +1 -0
  102. package/lib/module/index.js +7 -4
  103. package/lib/module/index.js.map +1 -1
  104. package/lib/module/policy/labels.js +19 -3
  105. package/lib/module/policy/labels.js.map +1 -1
  106. package/lib/module/policy/policy.js +6 -1
  107. package/lib/module/policy/policy.js.map +1 -1
  108. package/lib/module/policy/redact.js +42 -5
  109. package/lib/module/policy/redact.js.map +1 -1
  110. package/lib/module/providers/anthropic.js +341 -175
  111. package/lib/module/providers/anthropic.js.map +1 -1
  112. package/lib/module/providers/openai.js +230 -104
  113. package/lib/module/providers/openai.js.map +1 -1
  114. package/lib/module/providers/sse.js +160 -29
  115. package/lib/module/providers/sse.js.map +1 -1
  116. package/lib/module/providers/streamTimer.js +118 -0
  117. package/lib/module/providers/streamTimer.js.map +1 -0
  118. package/lib/module/providers/transport.js +103 -0
  119. package/lib/module/providers/transport.js.map +1 -0
  120. package/lib/module/providers/xhrStream.js +204 -0
  121. package/lib/module/providers/xhrStream.js.map +1 -0
  122. package/lib/module/session.js +228 -54
  123. package/lib/module/session.js.map +1 -1
  124. package/lib/typescript/blocks/receipts.d.ts +5 -10
  125. package/lib/typescript/blocks/receipts.d.ts.map +1 -1
  126. package/lib/typescript/blocks/runId.d.ts +20 -0
  127. package/lib/typescript/blocks/runId.d.ts.map +1 -0
  128. package/lib/typescript/blocks/types.d.ts +23 -2
  129. package/lib/typescript/blocks/types.d.ts.map +1 -1
  130. package/lib/typescript/blocks/uiTool.d.ts +11 -2
  131. package/lib/typescript/blocks/uiTool.d.ts.map +1 -1
  132. package/lib/typescript/catalog/catalog.g.d.ts.map +1 -1
  133. package/lib/typescript/catalog/catalog.types.g.d.ts +40 -0
  134. package/lib/typescript/catalog/catalog.types.g.d.ts.map +1 -0
  135. package/lib/typescript/catalog/normalizeParams.d.ts +43 -0
  136. package/lib/typescript/catalog/normalizeParams.d.ts.map +1 -0
  137. package/lib/typescript/catalog/signature.d.ts +29 -0
  138. package/lib/typescript/catalog/signature.d.ts.map +1 -0
  139. package/lib/typescript/catalog/snapshotReads.d.ts.map +1 -1
  140. package/lib/typescript/catalog/toProviderTools.d.ts +28 -15
  141. package/lib/typescript/catalog/toProviderTools.d.ts.map +1 -1
  142. package/lib/typescript/context/buildContextPack.d.ts +24 -2
  143. package/lib/typescript/context/buildContextPack.d.ts.map +1 -1
  144. package/lib/typescript/effects/digest.d.ts +10 -0
  145. package/lib/typescript/effects/digest.d.ts.map +1 -0
  146. package/lib/typescript/effects/ledger.d.ts +55 -0
  147. package/lib/typescript/effects/ledger.d.ts.map +1 -1
  148. package/lib/typescript/engine/askGate.d.ts +29 -0
  149. package/lib/typescript/engine/askGate.d.ts.map +1 -0
  150. package/lib/typescript/engine/effectFor.d.ts +6 -1
  151. package/lib/typescript/engine/effectFor.d.ts.map +1 -1
  152. package/lib/typescript/engine/historyBudget.d.ts +92 -0
  153. package/lib/typescript/engine/historyBudget.d.ts.map +1 -0
  154. package/lib/typescript/engine/runAgentTurn.d.ts +115 -1
  155. package/lib/typescript/engine/runAgentTurn.d.ts.map +1 -1
  156. package/lib/typescript/engine/systemPrompt.d.ts +41 -0
  157. package/lib/typescript/engine/systemPrompt.d.ts.map +1 -1
  158. package/lib/typescript/engine/textToolCalls.d.ts +58 -0
  159. package/lib/typescript/engine/textToolCalls.d.ts.map +1 -0
  160. package/lib/typescript/index.d.ts +8 -5
  161. package/lib/typescript/index.d.ts.map +1 -1
  162. package/lib/typescript/policy/labels.d.ts.map +1 -1
  163. package/lib/typescript/policy/policy.d.ts.map +1 -1
  164. package/lib/typescript/policy/redact.d.ts.map +1 -1
  165. package/lib/typescript/providers/anthropic.d.ts +5 -0
  166. package/lib/typescript/providers/anthropic.d.ts.map +1 -1
  167. package/lib/typescript/providers/openai.d.ts +0 -16
  168. package/lib/typescript/providers/openai.d.ts.map +1 -1
  169. package/lib/typescript/providers/sse.d.ts +82 -1
  170. package/lib/typescript/providers/sse.d.ts.map +1 -1
  171. package/lib/typescript/providers/streamTimer.d.ts +59 -0
  172. package/lib/typescript/providers/streamTimer.d.ts.map +1 -0
  173. package/lib/typescript/providers/transport.d.ts +40 -0
  174. package/lib/typescript/providers/transport.d.ts.map +1 -0
  175. package/lib/typescript/providers/types.d.ts +116 -3
  176. package/lib/typescript/providers/types.d.ts.map +1 -1
  177. package/lib/typescript/providers/xhrStream.d.ts +43 -0
  178. package/lib/typescript/providers/xhrStream.d.ts.map +1 -0
  179. package/lib/typescript/session.d.ts +63 -2
  180. package/lib/typescript/session.d.ts.map +1 -1
  181. package/package.json +2 -2
@@ -3,19 +3,31 @@
3
3
  Object.defineProperty(exports, "__esModule", {
4
4
  value: true
5
5
  });
6
+ Object.defineProperty(exports, "digestOf", {
7
+ enumerable: true,
8
+ get: function () {
9
+ return _digest.digestOf;
10
+ }
11
+ });
12
+ exports.readTargetDigest = readTargetDigest;
6
13
  exports.runAgentTurn = runAgentTurn;
7
14
  exports.runGatedAction = runGatedAction;
8
15
  var _catalog = require("../catalog/catalog.g");
9
16
  var _validateParams = require("../catalog/validateParams");
17
+ var _normalizeParams = require("../catalog/normalizeParams");
10
18
  var _toProviderTools = require("../catalog/toProviderTools");
11
19
  var _snapshotReads = require("../catalog/snapshotReads");
12
20
  var _ledger = require("../effects/ledger");
13
21
  var _policy = require("../policy/policy");
14
22
  var _redact = require("../policy/redact");
15
23
  var _effectFor = require("./effectFor");
24
+ var _digest = require("../effects/digest");
16
25
  var _uiTool = require("../blocks/uiTool");
17
26
  var _receipts = require("../blocks/receipts");
18
27
  var _types = require("../blocks/types");
28
+ var _askGate = require("./askGate");
29
+ var _textToolCalls = require("./textToolCalls");
30
+ var _historyBudget = require("./historyBudget");
19
31
  /**
20
32
  * One turn: user sentence in, tool calls and an answer out.
21
33
  *
@@ -37,11 +49,52 @@ var _types = require("../blocks/types");
37
49
 
38
50
  /** Beyond this a tool result costs more in tokens than it can possibly be worth. */
39
51
  const MAX_RESULT_CHARS = 24_000;
52
+ /**
53
+ * The result kept for the trace. Far smaller than what the MODEL gets — this
54
+ * one is read by a person in a chat bubble or a pasted JSON file, and a 24k
55
+ * store dump per step makes the trace unreadable rather than more useful.
56
+ */
57
+ const MAX_TRACE_RESULT_CHARS = 2_000;
58
+ /**
59
+ * How many identical rounds before the model is told it is looping.
60
+ *
61
+ * `maxSteps` alone was the whole answer, and it is the wrong shape: a model
62
+ * that misread a result retries the same call until the cap, then the turn
63
+ * ends with nothing — the user watches twelve identical rows go by and gets
64
+ * no answer. The cap is a backstop, not feedback. Telling the model what it
65
+ * is doing gives it the chance to change approach, and costs one sentence.
66
+ *
67
+ * Signature is name + action + arguments, key-sorted so `{b,a}` and `{a,b}`
68
+ * are the same call. Three, not two: a legitimate retry after a transient
69
+ * failure is normal, and warning on it would be noise.
70
+ */
71
+ const LOOP_REPEATS = 3;
72
+ function callSignature(calls) {
73
+ const sort = v => {
74
+ if (v === null || typeof v !== "object") return v;
75
+ if (Array.isArray(v)) return v.map(sort);
76
+ const o = v;
77
+ return Object.keys(o).sort().reduce((a, k) => (a[k] = sort(o[k]), a), {});
78
+ };
79
+ return calls.map(c => `${c.name}:${JSON.stringify(sort(c.input ?? {}))}`).join("|");
80
+ }
81
+ const LOOP_WARNING = "[Buoy] You have now made this exact call three times in a row and gotten the same thing back. Repeating it again will not change the answer. Read the result you already have, then either try a different tool or different arguments, or tell the user plainly what you could not get and what you tried.";
40
82
  const DEFAULT_MAX_STEPS = 12;
41
83
  const DEFAULT_TURN_MS = 180_000;
42
84
  function findDescriptor(catalog, toolId, action) {
43
85
  return catalog.find(t => t.toolId === toolId)?.actions.find(a => a.action === action);
44
86
  }
87
+
88
+ /**
89
+ * Why a picture did not render, in the model's own terms.
90
+ *
91
+ * Two messages, not one, because they answer different questions: the first
92
+ * is "your block was refused", the second is "your block was shown but with
93
+ * holes in it". The literal opening of the rejection is pinned by the
94
+ * trajectory classifier (evals/ask-buoy/lib/trajectory.ts) — keep it.
95
+ */
96
+ const URL_PROVENANCE_REJECTION = "Rejected: every image URL must be one you read from a tool result in THIS conversation (images.list, a storage value, a response body). Never invent or remember URLs — read them first, then show them.";
97
+ const droppedImageNote = n => ` NOTE: ${n} image ${n === 1 ? "URL was" : "URLs were"} dropped — they were not read from a tool result in this conversation, so the card the user is looking at has no picture there. Do not tell them it does; read the URLs and show it again, or say the artwork is missing.`;
45
98
  function encodeResult(value) {
46
99
  let text;
47
100
  try {
@@ -86,11 +139,30 @@ function summarise(value) {
86
139
  function countOf(n) {
87
140
  return n === 1 ? "1 item" : `${n} items`;
88
141
  }
142
+
143
+ /**
144
+ * A call that came back reporting its own failure.
145
+ *
146
+ * Buoy adapters RESOLVE with `{ok:false, error}` rather than throwing, so the
147
+ * dispatch's try/catch never fires and a refused write looked like a completed
148
+ * one: the row read "Done", and — much worse — the effect ledger recorded a
149
+ * change that had not happened, so the changes bar offered to undo nothing.
150
+ * Caught on device: a typed-edit refusal ("qty is a number, 'many' is text")
151
+ * left the bag untouched and the bar claiming one permanent change.
152
+ *
153
+ * `ok === false` is the same signal `summarise` above already trusts to turn a
154
+ * result into the row's text, so this reads it no more liberally than the UI
155
+ * already does.
156
+ */
157
+ function reportsFailure(value) {
158
+ return typeof value === "object" && value !== null && value.ok === false;
159
+ }
89
160
  async function* runAgentTurn(input) {
90
161
  const {
91
162
  provider,
92
163
  catalog,
93
164
  system,
165
+ systemVolatile,
94
166
  model,
95
167
  maxTokens,
96
168
  dispatch,
@@ -110,15 +182,72 @@ async function* runAgentTurn(input) {
110
182
  });
111
183
  tools.push((0, _uiTool.uiProviderTool)());
112
184
  const maxSteps = policy.maxSteps ?? DEFAULT_MAX_STEPS;
113
- const deadline = Date.now() + DEFAULT_TURN_MS;
185
+ /**
186
+ * The conversation this turn belongs to. Handed back on every `record` so a
187
+ * dispatch that settles after the user started a new conversation lands in
188
+ * the app (nothing can pull it back) but not in the new chat's undo bar.
189
+ */
190
+ const ledgerGeneration = ledger.generation;
191
+ /**
192
+ * What every request costs before a single message is added: the tool
193
+ * definitions, the system prompt, and the room the model needs to reply.
194
+ *
195
+ * Measured, not guessed — the tool block alone is ~32k characters for a
196
+ * typical install and ~94k with all 24 tools available, which is far too
197
+ * much to leave out of a budget. `maxTokens` is multiplied by four because
198
+ * the budget is in characters.
199
+ */
200
+ const requestReserve = system.length + (systemVolatile?.length ?? 0) + tools.reduce((n, t) => n + t.description.length + JSON.stringify(t.inputSchema).length, 0) + maxTokens * 4;
201
+ /**
202
+ * Caps the AGENT's wall-clock, not the user's. Time spent parked on an
203
+ * approval sheet is pushed onto the deadline as it is spent (see the
204
+ * `requestApproval` await below) — otherwise a user who takes three minutes
205
+ * to tap Approve gets the change applied and then "this is taking too long"
206
+ * for the turn that applied it.
207
+ */
208
+ let deadline = Date.now() + DEFAULT_TURN_MS;
114
209
 
115
- // Every http(s) URL that appeared in a tool result THIS turn. The prompt
116
- // tells the model image URLs must come from here; this Set is the
117
- // enforcement — a model-remembered or injected URL in an image block is
210
+ // Every http(s) URL that appeared in a tool result in this CONVERSATION.
211
+ // The prompt tells the model image URLs must come from here; this Set is
212
+ // the enforcement — a model-remembered or injected URL in an image block is
118
213
  // both a fabrication vector and a data-exfil beacon (the GET carries
119
- // whatever is encoded into the URL).
120
- const seenUrls = new Set();
214
+ // whatever is encoded into the URL). Session-owned when the caller passes
215
+ // one; see RunTurnInput.seenUrls for why it is not per-turn.
216
+ const seenUrls = input.seenUrls ?? new Set();
121
217
  const URL_RE = /https?:\/\/[^\s"'\\)\]}>]+/g;
218
+
219
+ /** Whether any text has gone out THIS TURN — see `breakBefore`. */
220
+ let answerStarted = false;
221
+
222
+ /**
223
+ * The last few rounds' call signatures. See LOOP_REPEATS — a run of
224
+ * identical rounds gets one warning appended to the results, once, so the
225
+ * model is told rather than silently capped.
226
+ */
227
+ const recentSignatures = [];
228
+ let loopWarned = false;
229
+
230
+ /**
231
+ * Everything the model SAID this turn, in order. Read once, at the end, by
232
+ * the ask gate — a question asked in round one and left hanging is the same
233
+ * dead end as one asked in the last round. See askGate.ts.
234
+ */
235
+ const spoken = [];
236
+ /**
237
+ * Whether the user already has something to tap. An `actions` or
238
+ * `suggestions` block is the model doing this job itself, and a second card
239
+ * asking the same thing underneath it is worse than none.
240
+ *
241
+ * Note this does NOT suppress the shortfall gate below — that one asks
242
+ * whether the user can tap the GAP, which unrelated chips do not answer.
243
+ */
244
+ let offeredTap = false;
245
+ /**
246
+ * Whether this turn actually went and looked — navigated, described the
247
+ * screen, tapped something, refetched. The shortfall gate stays quiet when
248
+ * it did: an agent that tried and came up short has earned its answer.
249
+ */
250
+ let attemptedAcquisition = false;
122
251
  for (let step = 0; step < maxSteps; step++) {
123
252
  if (signal?.aborted) {
124
253
  yield {
@@ -127,21 +256,112 @@ async function* runAgentTurn(input) {
127
256
  return messages;
128
257
  }
129
258
  if (Date.now() > deadline) {
259
+ // Same shape as the step cap below. This used to be an `error`, which
260
+ // drew the failed-turn card with Try again — and Try again re-sends the
261
+ // question, restarting the whole investigation that just ran out of
262
+ // time. Continue picks up with everything so far still in history.
130
263
  yield {
131
- type: "error",
132
- message: "This is taking too long — stopping here."
264
+ type: "block",
265
+ block: {
266
+ id: `cap${Date.now()}`,
267
+ kind: "notice",
268
+ tone: "warning",
269
+ // Keeps the "This is taking too long" prefix: the eval classifier
270
+ // (evals/ask-buoy/lib/trajectory.ts) tells a time-out from a step
271
+ // cap by it.
272
+ text: `This is taking too long — stopped after ${Math.round(DEFAULT_TURN_MS / 1000)}s. What it found so far is above.`,
273
+ actions: [{
274
+ label: "Continue",
275
+ primary: true,
276
+ send: "Continue where you left off."
277
+ }]
278
+ }
133
279
  };
134
280
  yield {
135
281
  type: "done"
136
282
  };
137
283
  return messages;
138
284
  }
285
+
286
+ /**
287
+ * BUDGET, EVERY REQUEST — not once per turn.
288
+ *
289
+ * The session trims when a turn opens, and that used to be the only
290
+ * check. But a turn is not one request: each step appends an assistant
291
+ * message and a tool-results message, and a single result can be 24,000
292
+ * characters. A twelve-step investigation could therefore add six figures
293
+ * of history AFTER the only check had run, and the turn died on the
294
+ * provider's context limit with an opaque error — in exactly the long
295
+ * sessions the tool is for.
296
+ *
297
+ * The current round is never touched: it is the question being answered.
298
+ */
299
+ const budgeted = (0, _historyBudget.budgetForRequest)(messages, requestReserve);
300
+ if (budgeted.droppedRounds > 0) {
301
+ messages.length = 0;
302
+ messages.push(...budgeted.messages);
303
+ yield {
304
+ type: "history-trimmed",
305
+ droppedRounds: budgeted.droppedRounds
306
+ };
307
+ } else if (budgeted.messages !== messages) {
308
+ messages.length = 0;
309
+ messages.push(...budgeted.messages);
310
+ }
139
311
  let text = "";
312
+ /** See `breakBefore`: set once the first text of THIS round has gone out. */
313
+ let textThisRound = false;
314
+ const textEvent = delta => {
315
+ // Only when this round opens a NEW paragraph of an answer that has
316
+ // already started. A turn whose first words arrive in round three has
317
+ // nothing to be separated from, and marking that would put the flag on
318
+ // events where it means nothing.
319
+ const opensNewRound = answerStarted && !textThisRound;
320
+ textThisRound = true;
321
+ answerStarted = true;
322
+ return opensNewRound ? {
323
+ type: "text",
324
+ delta,
325
+ breakBefore: true
326
+ } : {
327
+ type: "text",
328
+ delta
329
+ };
330
+ };
140
331
  const calls = [];
332
+ /**
333
+ * Recovered from text rather than emitted as calls — see textToolCalls.ts.
334
+ * Tracked by id so each one's result can tell the model to stop doing that;
335
+ * they are otherwise ordinary calls and get every check the others get.
336
+ */
337
+ const salvagedIds = new Set();
338
+ /**
339
+ * Text withheld from the screen while it could still be a tool-call
340
+ * envelope. Streamed text cannot be un-shown, so the choice has to be made
341
+ * before it leaves — and a model that writes its call as JSON must not
342
+ * have that JSON become the answer the user reads.
343
+ */
344
+ let held = "";
345
+ let holding = true;
346
+ // Carried, never read: Anthropic requires the turn's thinking blocks back
347
+ // verbatim with its tool results. See providers/anthropic.ts note 5.
348
+ const thinking = [];
141
349
  let failed = false;
350
+ /**
351
+ * What the provider said about how the stream ended.
352
+ *
353
+ * Undefined means it never said — a bare EOF, or a host-supplied provider
354
+ * whose generator simply returned. Both are treated as incomplete, so the
355
+ * fail-safe direction is the default rather than something each adapter
356
+ * has to remember to opt into.
357
+ */
358
+ let outcome;
359
+ /** Set for the two kinds that are recoverable rather than a hard failure. */
360
+ let incomplete;
142
361
  for await (const ev of provider.send({
143
362
  messages,
144
363
  system,
364
+ systemVolatile,
145
365
  tools,
146
366
  model,
147
367
  maxTokens,
@@ -149,41 +369,181 @@ async function* runAgentTurn(input) {
149
369
  })) {
150
370
  if (ev.type === "text") {
151
371
  text += ev.delta;
152
- yield {
153
- type: "text",
154
- delta: ev.delta
155
- };
372
+ if (!holding) {
373
+ yield textEvent(ev.delta);
374
+ } else if ((0, _textToolCalls.couldBeToolCallEnvelope)(text)) {
375
+ held += ev.delta;
376
+ } else {
377
+ // Not an envelope after all. Release everything at once and stream
378
+ // the rest as usual — the reader loses nothing but a few characters
379
+ // of latency at the very start of the answer.
380
+ holding = false;
381
+ held = "";
382
+ yield textEvent(text);
383
+ }
156
384
  } else if (ev.type === "tool-call") {
157
385
  calls.push(ev.call);
386
+ } else if (ev.type === "thinking") {
387
+ thinking.push(ev.block);
388
+ // Surfaced as well as carried: the block goes back to the provider
389
+ // verbatim (note above), and the readable half goes to the UI so a
390
+ // tester can see WHY a turn did what it did. Redacted blocks have no
391
+ // readable half and are carried only.
392
+ if (ev.block.type === "thinking" && ev.block.thinking.trim()) {
393
+ yield {
394
+ type: "reasoning",
395
+ text: ev.block.thinking
396
+ };
397
+ }
158
398
  } else if (ev.type === "error") {
159
- // A mid-stream error can arrive on an already-committed 200.
160
- yield {
161
- type: "error",
162
- message: ev.message
163
- };
164
- failed = true;
165
- } else if (ev.type === "done" && ev.usage) {
166
- // Surfaced so a host can meter cost per seat. Parsed for a long time;
167
- // went nowhere anyone could see.
168
- yield {
169
- type: "usage",
170
- input: ev.usage.input,
171
- output: ev.usage.output
172
- };
399
+ if (ev.kind === "truncated" || ev.kind === "timeout") {
400
+ // Not a failure the user did anything about, and not one Try again
401
+ // can fix — re-sending the QUESTION restarts an investigation that
402
+ // may already have written things. Held back from the error card
403
+ // (which is what draws Try again) and handled below as a recoverable
404
+ // stop with a Continue button.
405
+ incomplete = {
406
+ message: ev.message
407
+ };
408
+ } else {
409
+ // A mid-stream error can arrive on an already-committed 200.
410
+ yield {
411
+ type: "error",
412
+ message: ev.message
413
+ };
414
+ failed = true;
415
+ }
416
+ } else if (ev.type === "done") {
417
+ outcome = ev.outcome;
418
+ if (ev.usage) {
419
+ // Surfaced so a host can meter cost per seat. Parsed for a long time;
420
+ // went nowhere anyone could see.
421
+ yield {
422
+ type: "usage",
423
+ ...ev.usage,
424
+ model: ev.model
425
+ };
426
+ }
173
427
  }
174
428
  }
175
429
  if (failed) {
430
+ // Whatever was withheld is the model's own words on the way to an error.
431
+ // Show it rather than swallowing it.
432
+ if (held) yield textEvent(held);
176
433
  yield {
177
434
  type: "done"
178
435
  };
179
436
  return messages;
180
437
  }
438
+
439
+ /**
440
+ * THE COMPLETION GATE. Nothing below this runs a tool unless the provider
441
+ * said the response was whole.
442
+ *
443
+ * The failure it exists for is not theoretical and does not look like a
444
+ * failure: a connection cut after the model emitted
445
+ * `{"key":"cart","value":[]}` leaves valid JSON, a plausible-looking plan,
446
+ * and no error anywhere. The old code parsed those arguments, dispatched
447
+ * the write, sent the result back and carried on — a real mutation from
448
+ * half a sentence. Bare EOF is not consent.
449
+ *
450
+ * `output-limited` is the same shape from a different cause: the model hit
451
+ * max_tokens partway through planning. What it managed to SAY is real and
452
+ * is shown; what it was part-way through DOING is not run.
453
+ */
454
+ if (!incomplete && outcome !== "completed") {
455
+ incomplete = {
456
+ message: outcome === "output-limited" ? calls.length ? "The model ran out of room mid-plan, so nothing was run. What it found so far is above." : "The model ran out of room before finishing this answer." : "The answer ended before the endpoint said it was finished, so nothing was run."
457
+ };
458
+ }
459
+ if (incomplete) {
460
+ // Held text is the model's own words on the way out. Show it: the user
461
+ // watched it arrive and hiding it now reads as the app losing work.
462
+ if (held) yield textEvent(held);
463
+ /**
464
+ * Text only — never the tool calls, and never the thinking.
465
+ *
466
+ * An assistant turn carrying `tool_use` with no matching `tool_result`
467
+ * is an instant 400 on the next request, and these calls are exactly the
468
+ * ones that must not be answered. Keeping the text preserves what the
469
+ * user can see on screen for whatever comes next.
470
+ */
471
+ if (text.trim()) messages.push({
472
+ role: "assistant",
473
+ text
474
+ });
475
+ yield {
476
+ type: "block",
477
+ block: {
478
+ id: `cut${Date.now()}`,
479
+ kind: "notice",
480
+ tone: "warning",
481
+ text: incomplete.message,
482
+ actions: [{
483
+ label: "Continue",
484
+ primary: true,
485
+ send: "Continue where you left off."
486
+ }]
487
+ }
488
+ };
489
+ yield {
490
+ type: "done"
491
+ };
492
+ return messages;
493
+ }
494
+ if (held) {
495
+ // A model that wrote its call out as JSON instead of calling it. Run it,
496
+ // and show its `reply` — never the blob. See textToolCalls.ts.
497
+ const salvaged = calls.length === 0 ? (0, _textToolCalls.parseTextToolCalls)(held, catalog) : undefined;
498
+ if (salvaged) {
499
+ for (const call of salvaged.calls) {
500
+ calls.push(call);
501
+ salvagedIds.add(call.id);
502
+ }
503
+ text = salvaged.reply;
504
+ if (salvaged.reply) yield textEvent(salvaged.reply);
505
+ } else {
506
+ yield textEvent(held);
507
+ }
508
+ held = "";
509
+ }
181
510
  messages.push({
182
511
  role: "assistant",
183
512
  text,
184
- toolCalls: calls.length ? calls : undefined
513
+ toolCalls: calls.length ? calls : undefined,
514
+ thinking: thinking.length ? thinking : undefined
185
515
  });
516
+ if (text.trim()) spoken.push(text.trim());
186
517
  if (calls.length === 0) {
518
+ // The turn is over and the model wrote its own words. Two ways those
519
+ // words end badly, and they are different failures — see askGate.ts.
520
+ const said = spoken.join("\n\n");
521
+
522
+ // 1. It answered a smaller question than it was asked and handed the
523
+ // rest back. Fires even when chips were offered: the question is
524
+ // whether the user can tap the GAP, not whether they can tap.
525
+ const gap = (0, _askGate.synthesizeShortfallBlock)({
526
+ text: said,
527
+ attempted: attemptedAcquisition
528
+ });
529
+ if (gap) yield {
530
+ type: "block",
531
+ block: gap
532
+ };
533
+
534
+ // 2. It ended waiting on the user in prose. Suppressed once anything
535
+ // tappable is on screen, the gap button included.
536
+ const asked = offeredTap || gap ? null : (0, _askGate.synthesizeAskBlock)(said);
537
+ if (asked) {
538
+ yield {
539
+ type: "block",
540
+ block: asked
541
+ };
542
+ yield {
543
+ type: "awaiting-user",
544
+ blockId: asked.id
545
+ };
546
+ }
187
547
  yield {
188
548
  type: "done"
189
549
  };
@@ -191,8 +551,220 @@ async function* runAgentTurn(input) {
191
551
  }
192
552
  const results = [];
193
553
  let realmWillDie = false;
554
+
555
+ /**
556
+ * Reads, already in flight.
557
+ *
558
+ * The loop below stays strictly serial — every yield, every ledger entry
559
+ * and every approval keeps its order — but a round of independent READS
560
+ * used to pay each round trip end to end. The prompt tells the model to
561
+ * chain reads freely ("reading is cheap; do it"), and a three-read round
562
+ * cost three times the device latency for no reason.
563
+ *
564
+ * ONLY when the whole batch is safe to overlap: every call a catalog read,
565
+ * none needing approval, no `buoy_ui`, nothing snapshot-only or
566
+ * session-local. One write, one approval or one unknown action in the
567
+ * batch and nothing is prefetched — the mixed case is where ordering
568
+ * actually matters, and it is not worth the risk for the latency.
569
+ */
570
+ const prefetch = new Map();
571
+ if (calls.length > 1) {
572
+ const safe = calls.every(c => {
573
+ if (c.name === _uiTool.UI_TOOL_NAME) return false;
574
+ const id = (0, _toProviderTools.fromProviderToolName)(c.name, catalog);
575
+ if (!id || id === "ask-buoy") return false;
576
+ const act = c.input?.action;
577
+ if (typeof act !== "string" || act === _snapshotReads.SNAPSHOT_ACTION) return false;
578
+ const d = findDescriptor(catalog, id, act);
579
+ if (!d || d.effect !== "read") return false;
580
+ return (0, _policy.decide)({
581
+ descriptor: d,
582
+ toolId: id,
583
+ policy,
584
+ isRelease
585
+ }).verdict === "allow";
586
+ });
587
+ if (safe) {
588
+ for (const c of calls) {
589
+ const id = (0, _toProviderTools.fromProviderToolName)(c.name, catalog);
590
+ const act = String(c.input.action);
591
+ const raw = {
592
+ ...(c.input.params ?? {})
593
+ };
594
+ const d = findDescriptor(catalog, id, act);
595
+ const norm = (0, _normalizeParams.normalizeParams)(id, act, raw);
596
+ const withDefaults = (0, _validateParams.applyDefaults)(norm.params, d.params);
597
+ if (!(0, _validateParams.validateParams)(withDefaults, d.params).ok) continue;
598
+ // Rejections are swallowed here and re-awaited in the loop, where
599
+ // they are turned into the same tool_result they always were.
600
+ const p = Promise.resolve(dispatch(id, act, withDefaults)).catch(e => {
601
+ throw e;
602
+ });
603
+ p.catch(() => {});
604
+ prefetch.set(c.id, p);
605
+ }
606
+ }
607
+ }
194
608
  let awaiting = null;
609
+
610
+ /**
611
+ * A call Stop got to first — as a row, so the transcript says which ones.
612
+ *
613
+ * Resolved leniently: the call has not been validated yet, and a label
614
+ * that falls back to `tool.action` is better than no row for a call the
615
+ * user needs to know did not run.
616
+ */
617
+ const stoppedStep = call => {
618
+ const toolId = (0, _toProviderTools.fromProviderToolName)(call.name, catalog) ?? call.name;
619
+ const action = typeof call.input?.action === "string" ? call.input.action : "";
620
+ const descriptor = action ? findDescriptor(catalog, toolId, action) : undefined;
621
+ const params = call.input?.params ?? {};
622
+ return [{
623
+ type: "tool-start",
624
+ id: call.id,
625
+ toolId,
626
+ action,
627
+ label: descriptor ? (0, _policy.describeCall)(toolId, descriptor, params) : `${toolId}.${action}`,
628
+ description: descriptor?.summary ?? "",
629
+ params,
630
+ effect: descriptor?.effect ?? "read"
631
+ }, {
632
+ type: "tool-end",
633
+ id: call.id,
634
+ ok: false,
635
+ summary: "stopped",
636
+ durationMs: 0
637
+ }];
638
+ };
639
+
640
+ /**
641
+ * THE BATCH BARRIER — what the whole batch will do, decided before any of
642
+ * it does anything.
643
+ *
644
+ * A model answers with several calls at once, and they used to run one at
645
+ * a time with each approval asked when its own call came up. So
646
+ * `[set the flag, delete the cache]` wrote the flag, THEN asked about the
647
+ * delete — and a user who declined had already changed the app, with the
648
+ * bubble reading "changed the flag" directly above "left the cache alone".
649
+ * Declining is supposed to mean the plan does not happen.
650
+ *
651
+ * The fix is not to ask again later or to refuse afterwards, neither of
652
+ * which can unrun a write. It is to ask FIRST: every gated call in the
653
+ * batch is put to the user before the batch's first mutation dispatches,
654
+ * so consent is given with the whole plan visible.
655
+ *
656
+ * Resolution here is pure and duplicates the loop's own — deliberately.
657
+ * Anything it cannot resolve (an unknown tool, a bad shape) comes back as
658
+ * neither mutating nor gated, so the loop reports it exactly as it always
659
+ * did and the barrier simply does not fire. Failing back to today's
660
+ * behaviour is the right failure for a safety gate to have.
661
+ */
662
+ const planned = calls.map(call => {
663
+ if (call.name === _uiTool.UI_TOOL_NAME) return undefined;
664
+ const toolId = (0, _toProviderTools.fromProviderToolName)(call.name, catalog);
665
+ const action = typeof call.input?.action === "string" ? call.input.action : undefined;
666
+ if (!toolId || !action) return undefined;
667
+ const descriptor = findDescriptor(catalog, toolId, action);
668
+ if (!descriptor) return undefined;
669
+ const raw = call.input.params ?? {};
670
+ const params = (0, _validateParams.applyDefaults)((0, _normalizeParams.normalizeParams)(toolId, action, raw).params, descriptor.params);
671
+ if (!(0, _validateParams.validateParams)(params, descriptor.params).ok) return undefined;
672
+ const verdict = (0, _policy.decide)({
673
+ descriptor,
674
+ toolId,
675
+ policy,
676
+ isRelease
677
+ });
678
+ return {
679
+ call,
680
+ toolId,
681
+ action,
682
+ descriptor,
683
+ params,
684
+ label: (0, _policy.describeCall)(toolId, descriptor, params),
685
+ mutates: descriptor.effect !== "read",
686
+ gated: verdict.verdict === "needs-approval",
687
+ reason: verdict.verdict === "needs-approval" ? verdict.reason : ""
688
+ };
689
+ });
690
+ const gatedInBatch = planned.some(p => p?.gated);
691
+ /** Decisions the barrier has already taken, so no call is asked twice. */
692
+ const decided = new Map();
693
+ let barrierRun = false;
694
+ let batchDeclined = false;
695
+
696
+ /** Put every gated call in the batch to the user, in the order written. */
697
+ async function* runBarrier() {
698
+ barrierRun = true;
699
+ for (const p of planned) {
700
+ if (!p?.gated || decided.has(p.call.id)) continue;
701
+ yield {
702
+ type: "approval-required",
703
+ id: p.call.id,
704
+ toolId: p.toolId,
705
+ action: p.action,
706
+ label: p.label,
707
+ description: p.descriptor.summary,
708
+ reason: p.reason
709
+ };
710
+ const askedAt = Date.now();
711
+ const targetDigest = await readTargetDigest({
712
+ toolId: p.toolId,
713
+ action: p.action,
714
+ params: p.params,
715
+ dispatch,
716
+ catalog
717
+ });
718
+ const approved = requestApproval ? await requestApproval({
719
+ id: p.call.id,
720
+ toolId: p.toolId,
721
+ action: p.action,
722
+ label: p.label,
723
+ description: p.descriptor.summary,
724
+ reason: p.reason,
725
+ params: p.params,
726
+ targetDigest
727
+ }) : false;
728
+ // The user's deliberation is not the agent's runtime.
729
+ deadline += Date.now() - askedAt;
730
+ decided.set(p.call.id, approved);
731
+ if (!approved) {
732
+ // One refusal ends the PLAN, not just the call. The remaining
733
+ // changes were proposed together and nothing has run yet, so
734
+ // carrying on with the rest would apply half of something the user
735
+ // has just turned down.
736
+ batchDeclined = true;
737
+ return;
738
+ }
739
+ }
740
+ }
195
741
  for (const call of calls) {
742
+ /**
743
+ * Stop was pressed while this batch was running.
744
+ *
745
+ * The abort signal used to be read only at the top of a ROUND, so a
746
+ * response carrying three calls ran all three after the tap — and the
747
+ * bubble then said "Stopped." over two writes the user believed it had
748
+ * prevented. Each call now checks before it starts. A dispatch already
749
+ * in flight is left to finish (nothing can pull a store write back out
750
+ * of an adapter) and keeps its receipt; everything after it is reported
751
+ * as not run, each as its own row, so the transcript says exactly which
752
+ * changes landed. A result is still pushed for every call — the provider
753
+ * requires one per tool_use or the next request is rejected.
754
+ */
755
+ if (signal?.aborted) {
756
+ // A `buoy_ui` card that was never shown is not a step; no row for it.
757
+ if (call.name !== _uiTool.UI_TOOL_NAME) {
758
+ for (const ev of stoppedStep(call)) yield ev;
759
+ }
760
+ results.push({
761
+ toolCallId: call.id,
762
+ content: "Not run — the user pressed Stop before this call started.",
763
+ isError: true
764
+ });
765
+ continue;
766
+ }
767
+
196
768
  // `buoy_ui`: show or ask. Validated like any action; an Ask block ends
197
769
  // the turn after this batch — the answer comes back as a user message.
198
770
  if (call.name === _uiTool.UI_TOOL_NAME) {
@@ -200,19 +772,35 @@ async function* runAgentTurn(input) {
200
772
  if (!built.ok || !built.block) {
201
773
  results.push({
202
774
  toolCallId: call.id,
203
- content: `Invalid ${_uiTool.UI_TOOL_NAME} block:\n${built.errors.join("\n")}`,
775
+ /**
776
+ * The trailing sentence is the expensive half.
777
+ *
778
+ * A model that gets a block bounced rewrites its WHOLE answer on
779
+ * the retry, and one assistant bubble spans every round of a turn
780
+ * (see `breakBefore`), so the user reads the same paragraph again
781
+ * — measured at four near-identical closings from three schema
782
+ * misses in a row on one live turn. The block is wrong; the
783
+ * sentence under it was fine.
784
+ */
785
+ content: `Invalid ${_uiTool.UI_TOOL_NAME} block:\n${built.errors.join("\n")}\n\nResend ONLY the corrected block. Do not rewrite your answer — whatever you already said this turn has been shown to the user, and saying it again repeats it on their screen.`,
204
786
  isError: true
205
787
  });
206
788
  continue;
207
789
  }
208
790
  // Provenance for pictures, ENFORCED: an image URL the model did not
209
791
  // read from a tool result this turn does not render. Live or nothing.
792
+ // Dropping pictures SILENTLY is its own bug: the model then writes
793
+ // "here they are, with artwork" over a card that has none, and the
794
+ // user is told something untrue about their own screen. Every drop
795
+ // comes back in the tool result so the next sentence can be honest.
796
+ let droppedImages = 0;
210
797
  if (built.block.kind === "imageGrid") {
211
798
  const kept = built.block.images.filter(img => seenUrls.has(img.url));
799
+ droppedImages = built.block.images.length - kept.length;
212
800
  if (kept.length === 0) {
213
801
  results.push({
214
802
  toolCallId: call.id,
215
- content: "Rejected: every image URL must be one you read from a tool result THIS turn (images.list, a storage value, a response body). Never invent or remember URLs — read them first.",
803
+ content: URL_PROVENANCE_REJECTION,
216
804
  isError: true
217
805
  });
218
806
  continue;
@@ -223,12 +811,17 @@ async function* runAgentTurn(input) {
223
811
  };
224
812
  }
225
813
  if (built.block.kind === "list") {
226
- built.block = {
227
- ...built.block,
228
- items: built.block.items.map(item => item.image && !seenUrls.has(item.image) ? {
814
+ const items = built.block.items.map(item => {
815
+ if (!item.image || seenUrls.has(item.image)) return item;
816
+ droppedImages += 1;
817
+ return {
229
818
  ...item,
230
819
  image: undefined
231
- } : item)
820
+ };
821
+ });
822
+ built.block = {
823
+ ...built.block,
824
+ items
232
825
  };
233
826
  }
234
827
  yield {
@@ -236,6 +829,7 @@ async function* runAgentTurn(input) {
236
829
  block: built.block,
237
830
  forToolCallId: call.id
238
831
  };
832
+ if (built.block.kind === "actions" || built.block.kind === "suggestions") offeredTap = true;
239
833
  if ((0, _types.isAskBlock)(built.block)) {
240
834
  awaiting = built.block.id;
241
835
  results.push({
@@ -245,7 +839,7 @@ async function* runAgentTurn(input) {
245
839
  } else {
246
840
  results.push({
247
841
  toolCallId: call.id,
248
- content: "Shown to the user. Don't repeat its contents in text."
842
+ content: "Shown to the user. Don't repeat its contents in text." + (droppedImages > 0 ? droppedImageNote(droppedImages) : "")
249
843
  });
250
844
  }
251
845
  continue;
@@ -253,12 +847,40 @@ async function* runAgentTurn(input) {
253
847
  const toolId = (0, _toProviderTools.fromProviderToolName)(call.name, catalog);
254
848
  const action = call.input.action;
255
849
 
850
+ /**
851
+ * A call that never reached a tool at all.
852
+ *
853
+ * Same rule as a refusal: the model made a move, the move went nowhere,
854
+ * and a transcript that hides it leaves the reader watching the agent
855
+ * change its mind for no visible reason. Caught on device — a model
856
+ * guessed three navigation tool names that do not exist, spent 41
857
+ * seconds doing it, and the only trace was a sentence it chose to write.
858
+ * `read` because nothing was touched.
859
+ */
860
+ const deadCall = (label, summary) => [{
861
+ type: "tool-start",
862
+ id: call.id,
863
+ toolId: toolId ?? call.name,
864
+ action: action ?? "",
865
+ label,
866
+ description: "",
867
+ params: call.input.params ?? {},
868
+ effect: "read"
869
+ }, {
870
+ type: "tool-end",
871
+ id: call.id,
872
+ ok: false,
873
+ summary,
874
+ durationMs: 0
875
+ }];
876
+
256
877
  // Two different mistakes, kept apart on purpose. Merging them tells a
257
878
  // model that forgot `action` that the TOOL does not exist, and it stops
258
879
  // reaching for a tool that was fine — the same distinction
259
880
  // `dispatchToolAction` draws between an unknown tool and an unknown
260
881
  // action, for the same reason.
261
882
  if (!toolId) {
883
+ for (const ev of deadCall(call.name, "No such tool")) yield ev;
262
884
  results.push({
263
885
  toolCallId: call.id,
264
886
  content: `There is no tool called "${call.name}". Use one of the tools you were given.`,
@@ -267,6 +889,7 @@ async function* runAgentTurn(input) {
267
889
  continue;
268
890
  }
269
891
  if (!action) {
892
+ for (const ev of deadCall(call.name, "No action given")) yield ev;
270
893
  results.push({
271
894
  toolCallId: call.id,
272
895
  content: `"${call.name}" needs an "action" — it is required, and it names which of this tool's actions to run. See the action list in the tool's description.`,
@@ -280,6 +903,7 @@ async function* runAgentTurn(input) {
280
903
  // has nowhere to go but guess again, and the next guess is no better
281
904
  // informed than the last.
282
905
  const known = catalog.find(t => t.toolId === toolId)?.actions.map(a => a.action).join(", ");
906
+ for (const ev of deadCall(`${toolId}.${action}`, "No such action")) yield ev;
283
907
  results.push({
284
908
  toolCallId: call.id,
285
909
  content: `"${action}" is not an action on ${toolId}.${known ? ` Valid actions: ${known}.` : ""}`,
@@ -288,7 +912,11 @@ async function* runAgentTurn(input) {
288
912
  continue;
289
913
  }
290
914
  const raw = call.input.params ?? {};
291
- const params = (0, _validateParams.applyDefaults)(raw, descriptor.params);
915
+ // Accept the parameter-name misses every model makes (queryKey for
916
+ // queryHash, requestId for id, flat rule fields, …) before validation,
917
+ // so a mechanical alias never costs a turn. See normalizeParams.ts.
918
+ const normalized = (0, _normalizeParams.normalizeParams)(toolId, action, raw);
919
+ const params = (0, _validateParams.applyDefaults)(normalized.params, descriptor.params);
292
920
  const check = (0, _validateParams.validateParams)(params, descriptor.params);
293
921
  if (!check.ok) {
294
922
  results.push({
@@ -298,6 +926,7 @@ async function* runAgentTurn(input) {
298
926
  });
299
927
  continue;
300
928
  }
929
+ const aliasTrailer = (0, _normalizeParams.aliasNote)(normalized.notes);
301
930
  const label = (0, _policy.describeCall)(toolId, descriptor, params);
302
931
  const verdict = (0, _policy.decide)({
303
932
  descriptor,
@@ -305,13 +934,29 @@ async function* runAgentTurn(input) {
305
934
  policy,
306
935
  isRelease
307
936
  });
937
+
938
+ // A step the user can SEE, for a call that never ran. `tool-end` alone
939
+ // updates a row that was never opened, so a refusal used to leave no
940
+ // trace in the transcript at all — the model simply changed its mind
941
+ // between one sentence and the next.
942
+ const unrunStep = summary => [{
943
+ type: "tool-start",
944
+ id: call.id,
945
+ toolId,
946
+ action,
947
+ label,
948
+ description: descriptor.summary,
949
+ params,
950
+ effect: descriptor.effect
951
+ }, {
952
+ type: "tool-end",
953
+ id: call.id,
954
+ ok: false,
955
+ summary,
956
+ durationMs: 0
957
+ }];
308
958
  if (verdict.verdict === "refuse") {
309
- yield {
310
- type: "tool-end",
311
- id: call.id,
312
- ok: false,
313
- summary: "refused"
314
- };
959
+ for (const ev of unrunStep("refused")) yield ev;
315
960
  results.push({
316
961
  toolCallId: call.id,
317
962
  content: verdict.reason,
@@ -319,25 +964,77 @@ async function* runAgentTurn(input) {
319
964
  });
320
965
  continue;
321
966
  }
322
- if (verdict.verdict === "needs-approval") {
323
- yield {
324
- type: "approval-required",
325
- id: call.id,
326
- toolId,
327
- action,
328
- label,
329
- description: descriptor.summary,
330
- reason: verdict.reason
331
- };
332
- const approved = requestApproval ? await requestApproval({
333
- id: call.id,
334
- toolId,
335
- action,
336
- label,
337
- description: descriptor.summary,
338
- reason: verdict.reason
339
- }) : false;
967
+
968
+ /**
969
+ * Nothing in this batch changes anything until every card in it has been
970
+ * answered. See `runBarrier` — this is the line that makes a decline
971
+ * mean "the plan does not happen" rather than "the rest of the plan does
972
+ * not happen".
973
+ */
974
+ const gatedHere = verdict.verdict === "needs-approval";
975
+ if (gatedInBatch && !barrierRun && (descriptor.effect !== "read" || gatedHere)) {
976
+ yield* runBarrier();
977
+ }
978
+ if (batchDeclined && (descriptor.effect !== "read" || gatedHere)) {
979
+ for (const ev of unrunStep("declined")) yield ev;
980
+ results.push({
981
+ toolCallId: call.id,
982
+ content: decided.get(call.id) === false ? "The user declined this change. Do not retry it; ask what they would prefer." : "Not run — the user declined another change in this same batch, so none of it was applied. Ask what they would prefer before proposing it again.",
983
+ isError: true
984
+ });
985
+ continue;
986
+ }
987
+ if (gatedHere) {
988
+ let approved;
989
+ if (decided.has(call.id)) {
990
+ // The barrier already put this card up, before the batch's first
991
+ // mutation. Asking again would be the same question twice.
992
+ approved = decided.get(call.id);
993
+ } else {
994
+ // The barrier could not resolve this call (see `planned`), so it was
995
+ // never offered. Ask here, exactly as this always did — a gate that
996
+ // fails closed by silently declining would be worse than one that
997
+ // asks late.
998
+ yield {
999
+ type: "approval-required",
1000
+ id: call.id,
1001
+ toolId,
1002
+ action,
1003
+ label,
1004
+ description: descriptor.summary,
1005
+ reason: verdict.reason
1006
+ };
1007
+ const askedAt = Date.now();
1008
+ // What the write is about to replace, as of NOW. Only a restored
1009
+ // card ever reads it back — see readTargetDigest.
1010
+ const targetDigest = await readTargetDigest({
1011
+ toolId,
1012
+ action,
1013
+ params,
1014
+ dispatch,
1015
+ catalog
1016
+ });
1017
+ approved = requestApproval ? await requestApproval({
1018
+ id: call.id,
1019
+ toolId,
1020
+ action,
1021
+ label,
1022
+ description: descriptor.summary,
1023
+ reason: verdict.reason,
1024
+ params,
1025
+ targetDigest
1026
+ }) : false;
1027
+ // The user's deliberation is not the agent's runtime. Give the clock
1028
+ // back before anything else can trip the deadline check.
1029
+ deadline += Date.now() - askedAt;
1030
+ decided.set(call.id, approved);
1031
+ }
340
1032
  if (!approved) {
1033
+ // The row is the whole point here. Without it the bubble read as a
1034
+ // contradiction — "Bumped the line from 5 to 9." straight into "I
1035
+ // left it as it was" — with nothing on screen to say a change had
1036
+ // been proposed and turned down.
1037
+ for (const ev of unrunStep("declined")) yield ev;
341
1038
  results.push({
342
1039
  toolCallId: call.id,
343
1040
  content: "The user declined this change. Do not retry it; ask what they would prefer.",
@@ -346,20 +1043,67 @@ async function* runAgentTurn(input) {
346
1043
  continue;
347
1044
  }
348
1045
  }
1046
+ if (_askGate.ACQUISITION_CALLS.has(`${toolId}.${action}`)) attemptedAcquisition = true;
349
1047
  yield {
350
1048
  type: "tool-start",
351
1049
  id: call.id,
352
1050
  toolId,
353
1051
  action,
354
1052
  label,
355
- description: descriptor.summary
1053
+ description: descriptor.summary,
1054
+ params,
1055
+ effect: descriptor.effect
356
1056
  };
1057
+ const startedAt = Date.now();
357
1058
 
358
1059
  // Capture the prior value BEFORE a storage write, so undo can restore it
359
1060
  // exactly. This is the one thing that makes storage reversible here when
360
1061
  // the Scenarios engine has to treat it as permanent.
361
1062
  const before = await captureBefore(descriptor, toolId, params, dispatch);
362
- const stateBefore = await captureStateBefore(descriptor, toolId, params, dispatch);
1063
+ const stateBefore = await captureStoreState(descriptor, toolId, params, dispatch);
1064
+
1065
+ /**
1066
+ * THE FINAL GUARD. Everything above this line happened in the past.
1067
+ *
1068
+ * `decide()` ran before the approval card went up, and between there and
1069
+ * here are three awaits: the target read, the person deciding, and two
1070
+ * pre-reads. Each one yields the thread, and a person can do a lot in
1071
+ * that gap — press Stop, or flip the read-only toggle in the settings
1072
+ * sheet while the card is still on screen. Both were reproduced: the
1073
+ * write went ahead on the verdict it captured before they touched
1074
+ * anything, which is precisely the moment the product promises it will
1075
+ * not. `live` is mutated in place by the session (see `refreshLive`), so
1076
+ * asking again reads the toggles as they are NOW.
1077
+ *
1078
+ * Below this there is no `await` before `dispatch` is CALLED. That is
1079
+ * load-bearing: an await here would reopen the same gap one line lower.
1080
+ */
1081
+ if (signal?.aborted) {
1082
+ for (const ev of unrunStep("stopped")) yield ev;
1083
+ results.push({
1084
+ toolCallId: call.id,
1085
+ content: "Not run — the user pressed Stop before this call started.",
1086
+ isError: true
1087
+ });
1088
+ continue;
1089
+ }
1090
+ const stillAllowed = (0, _policy.decide)({
1091
+ descriptor,
1092
+ toolId,
1093
+ policy,
1094
+ isRelease
1095
+ });
1096
+ if (stillAllowed.verdict === "refuse") {
1097
+ // `needs-approval` is NOT re-refused: reaching here means the tap
1098
+ // already happened, and asking twice for one call is its own bug.
1099
+ for (const ev of unrunStep("refused")) yield ev;
1100
+ results.push({
1101
+ toolCallId: call.id,
1102
+ content: stillAllowed.reason,
1103
+ isError: true
1104
+ });
1105
+ continue;
1106
+ }
363
1107
  try {
364
1108
  // `getSnapshot` is reserved: it is not an adapter action, so it goes to
365
1109
  // the snapshot reader and is trimmed to the fields worth sending. See
@@ -367,31 +1111,51 @@ async function* runAgentTurn(input) {
367
1111
  // ledger the model asks about lives HERE, not behind an adapter, so
368
1112
  // routing it through dispatch would only work on a device and would
369
1113
  // record the undo as a fresh effect.
370
- const result = action === _snapshotReads.SNAPSHOT_ACTION ? (0, _snapshotReads.projectSnapshot)(toolId, await requireSnapshot(readSnapshot, toolId), params) : toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch) : await dispatch(toolId, action, params);
1114
+ const result = action === _snapshotReads.SNAPSHOT_ACTION ? (0, _snapshotReads.projectSnapshot)(toolId, await requireSnapshot(readSnapshot, toolId), params) : toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch) : await (prefetch.get(call.id) ?? dispatch(toolId, action, params));
371
1115
  const cleaned = (0, _redact.redact)((0, _redact.stripSelfTraffic)(result));
372
- if (descriptor.effect !== "read" && toolId !== "ask-buoy") {
373
- // A delete of something we created IS the undo, whoever asked for
374
- // it — mark the matching entry undone instead of leaving the bar
375
- // promising an undo against a rule that no longer exists.
376
- ledger.noteExternalRevert(toolId, action, params);
377
- const fx = (0, _effectFor.effectFor)(toolId, descriptor, params, result, before);
378
- if (fx) ledger.record(toolId, action, fx);
379
- }
1116
+ const refused = reportsFailure(result);
1117
+ const encodedResult = encodeResult(cleaned);
380
1118
  yield {
381
1119
  type: "tool-end",
382
1120
  id: call.id,
383
- ok: true,
384
- summary: summarise(result)
1121
+ ok: !refused,
1122
+ summary: summarise(result),
1123
+ durationMs: Date.now() - startedAt,
1124
+ result: encodedResult.length > MAX_TRACE_RESULT_CHARS ? `${encodedResult.slice(0, MAX_TRACE_RESULT_CHARS)}…` : encodedResult
385
1125
  };
386
1126
 
387
1127
  // The receipt: what the user sees of this result, built from the
388
1128
  // same redacted data the model gets. See blocks/receipts.ts.
389
- const receipt = (0, _receipts.projectReceipt)({
1129
+ //
1130
+ // The store is read back a second time so the diff shows what the
1131
+ // write ACTUALLY did rather than what it asked for — the two differ on
1132
+ // every `path` edit and every keyed list edit, and the request-shaped
1133
+ // version reported untouched fields as deleted.
1134
+ const stateAfter = stateBefore === undefined || refused ? undefined : await captureStoreState(descriptor, toolId, params, dispatch);
1135
+ if (descriptor.effect !== "read" && toolId !== "ask-buoy" && !refused) {
1136
+ // A delete of something we created IS the undo, whoever asked for
1137
+ // it — mark the matching entry undone instead of leaving the bar
1138
+ // promising an undo against a rule that no longer exists.
1139
+ ledger.noteExternalRevert(toolId, action, params);
1140
+ // Recorded AFTER the read-back so a cache edit's entry can carry both
1141
+ // halves: the data to restore and a fingerprint of what the write
1142
+ // left behind, which is how undo knows whether the app has since
1143
+ // replaced it. See effectFor / ledger "query-write".
1144
+ const fx = (0, _effectFor.effectFor)(toolId, descriptor, params, result, before, {
1145
+ before: stateBefore,
1146
+ after: stateAfter
1147
+ });
1148
+ if (fx) ledger.record(toolId, action, fx, ledgerGeneration);
1149
+ }
1150
+ // No receipt for a call that changed nothing: a "what changed" card
1151
+ // under a refusal is the same lie as the ledger entry, drawn bigger.
1152
+ const receipt = refused ? undefined : (0, _receipts.projectReceipt)({
390
1153
  toolId,
391
1154
  action,
392
1155
  params,
393
1156
  result: cleaned,
394
1157
  before: stateBefore ?? before,
1158
+ after: stateAfter,
395
1159
  effect: descriptor.effect
396
1160
  });
397
1161
  if (receipt) yield {
@@ -399,11 +1163,10 @@ async function* runAgentTurn(input) {
399
1163
  block: receipt,
400
1164
  forToolCallId: call.id
401
1165
  };
402
- const encoded = encodeResult(cleaned);
403
- for (const url of encoded.match(URL_RE) ?? []) seenUrls.add(url);
1166
+ for (const url of encodedResult.match(URL_RE) ?? []) seenUrls.add(url);
404
1167
  results.push({
405
1168
  toolCallId: call.id,
406
- content: encoded + ownedKeyNote(toolId, descriptor, params, storeKeyOwners)
1169
+ content: encodedResult + ownedKeyNote(toolId, descriptor, params, storeKeyOwners) + aliasTrailer + (salvagedIds.has(call.id) ? _textToolCalls.SALVAGE_NOTE : "")
407
1170
  });
408
1171
  if (toolId === "app" && (action === "reloadApp" || action === "reload")) {
409
1172
  realmWillDie = true;
@@ -414,7 +1177,8 @@ async function* runAgentTurn(input) {
414
1177
  type: "tool-end",
415
1178
  id: call.id,
416
1179
  ok: false,
417
- summary: message
1180
+ summary: message,
1181
+ durationMs: Date.now() - startedAt
418
1182
  };
419
1183
  results.push({
420
1184
  toolCallId: call.id,
@@ -423,6 +1187,20 @@ async function* runAgentTurn(input) {
423
1187
  });
424
1188
  }
425
1189
  }
1190
+
1191
+ // Looping? Say so, once, on the LAST result of this round — so it arrives
1192
+ // attached to the thing being repeated rather than as a free-floating
1193
+ // user turn, and the model reads it before deciding what to do next.
1194
+ if (calls.length) {
1195
+ recentSignatures.push(callSignature(calls));
1196
+ if (recentSignatures.length > LOOP_REPEATS) recentSignatures.shift();
1197
+ const looping = !loopWarned && recentSignatures.length === LOOP_REPEATS && recentSignatures.every(sig => sig === recentSignatures[0]);
1198
+ if (looping) {
1199
+ loopWarned = true;
1200
+ const last = results[results.length - 1];
1201
+ if (last) last.content = `${last.content}\n\n${LOOP_WARNING}`;
1202
+ }
1203
+ }
426
1204
  messages.push({
427
1205
  role: "tool-results",
428
1206
  results
@@ -487,6 +1265,27 @@ function ownedKeyNote(toolId, descriptor, params, storeKeyOwners) {
487
1265
  if (!storeName) return "";
488
1266
  return `\n\n[Buoy] This wrote to disk, but "${params.key}" is the saved copy of the live "${storeName}" store, so nothing on screen has changed — the app read that key once at startup and has held the state in memory since. Call zustand.rehydrate({"storeName":"${storeName}"}) to make the app pick this up, or use zustand.setState next time to change the store directly. If you meant to set the value for the app's next launch, this is already done and no further action is needed.`;
489
1267
  }
1268
+ /**
1269
+ * Fingerprint the value a write is about to replace.
1270
+ *
1271
+ * Built for the approval card that outlives its turn: the app crashed with
1272
+ * "Set bag line L-01 qty to 14" still asking, relaunched, and the card came
1273
+ * back — and Allow wrote qty 14 against whatever the bag was NOW, maybe a
1274
+ * different account, maybe a line that no longer existed. A hash of the
1275
+ * params alone would not catch that (same label, same params, different
1276
+ * world). So the card carries a digest of the TARGET at ask time, and a
1277
+ * restored Allow re-reads and compares before it writes. Undefined when the
1278
+ * target cannot be read — then the card can only warn, not check.
1279
+ */
1280
+ async function readTargetDigest(input) {
1281
+ const descriptor = findDescriptor(input.catalog ?? _catalog.CATALOG, input.toolId, input.action);
1282
+ if (!descriptor || descriptor.effect === "read") return undefined;
1283
+ const store = await captureStoreState(descriptor, input.toolId, input.params, input.dispatch);
1284
+ if (store !== undefined) return (0, _digest.digestOf)(store);
1285
+ const before = await captureBefore(descriptor, input.toolId, input.params, input.dispatch);
1286
+ if (before !== undefined) return (0, _digest.digestOf)(before);
1287
+ return undefined;
1288
+ }
490
1289
  /**
491
1290
  * One human-initiated action through EVERY gate the model's calls go through:
492
1291
  * catalog lookup, param validation, policy, pre-read, effect ledger.
@@ -519,7 +1318,8 @@ async function runGatedAction(input) {
519
1318
  error: `${toolId}.${action} is not in Buoy's catalog.`
520
1319
  };
521
1320
  }
522
- const params = (0, _validateParams.applyDefaults)(input.params ?? {}, descriptor.params);
1321
+ const normalized = (0, _normalizeParams.normalizeParams)(toolId, action, input.params ?? {});
1322
+ const params = (0, _validateParams.applyDefaults)(normalized.params, descriptor.params);
523
1323
  const check = (0, _validateParams.validateParams)(params, descriptor.params);
524
1324
  if (!check.ok) {
525
1325
  return {
@@ -540,11 +1340,51 @@ async function runGatedAction(input) {
540
1340
  };
541
1341
  }
542
1342
  const before = await captureBefore(descriptor, toolId, params, dispatch);
1343
+ /**
1344
+ * The same read-back the model's path takes. It was missing here, and the
1345
+ * gap was invisible: a `setQueryData` through a block button or a restored
1346
+ * approval card recorded a ledger entry with no prior value, so Undo had
1347
+ * nothing to put back and the concurrent-writer check had no fingerprint to
1348
+ * compare. The bar said "1 undoable" over a change that could not be undone.
1349
+ */
1350
+ const stateBefore = await captureStoreState(descriptor, toolId, params, dispatch);
1351
+
1352
+ // Asked again after the pre-reads, for the same reason the model's path asks
1353
+ // again: those are awaits, and read-only can be switched on inside one.
1354
+ const stillAllowed = (0, _policy.decide)({
1355
+ descriptor,
1356
+ toolId,
1357
+ policy,
1358
+ isRelease
1359
+ });
1360
+ if (stillAllowed.verdict === "refuse") {
1361
+ return {
1362
+ ok: false,
1363
+ error: stillAllowed.reason
1364
+ };
1365
+ }
543
1366
  try {
544
1367
  const result = toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch) : await dispatch(toolId, action, params);
1368
+
1369
+ // Same rule as the model's path: a resolved `{ok:false}` is a refusal, not
1370
+ // a change. Without this a block button (or an approval answered after a
1371
+ // reload) put a phantom entry in the undo bar too.
1372
+ if (reportsFailure(result)) {
1373
+ const err = result.error;
1374
+ return {
1375
+ ok: false,
1376
+ error: typeof err === "string" ? err : "The tool refused this."
1377
+ };
1378
+ }
545
1379
  if (descriptor.effect !== "read" && toolId !== "ask-buoy") {
546
1380
  ledger.noteExternalRevert(toolId, action, params);
547
- const fx = (0, _effectFor.effectFor)(toolId, descriptor, params, result, before);
1381
+ // Read back AFTER the write, exactly as the model's path does — the two
1382
+ // halves together are what make a cache edit undoable.
1383
+ const stateAfter = await captureStoreState(descriptor, toolId, params, dispatch);
1384
+ const fx = (0, _effectFor.effectFor)(toolId, descriptor, params, result, before, {
1385
+ before: stateBefore,
1386
+ after: stateAfter
1387
+ });
548
1388
  if (fx) ledger.record(toolId, action, fx);
549
1389
  }
550
1390
  return {
@@ -616,16 +1456,46 @@ async function requireSnapshot(readSnapshot, toolId) {
616
1456
  * Zustand only for now: `getStoreState` is cheap and the write params carry
617
1457
  * the after-state, so the diff is exact without the store's cooperation.
618
1458
  */
619
- async function captureStateBefore(descriptor, toolId, params, dispatch) {
620
- if (toolId !== "zustand" || descriptor.action !== "setState") return undefined;
621
- if (typeof params.storeName !== "string") return undefined;
622
- try {
623
- return await dispatch(toolId, "getStoreState", {
624
- storeName: params.storeName
625
- });
626
- } catch {
627
- return undefined;
1459
+ /**
1460
+ * Read a zustand store whole, for the before/after halves of a write receipt.
1461
+ * Called on both sides of the dispatch — see the receipt above.
1462
+ */
1463
+ async function captureStoreState(descriptor, toolId, params, dispatch) {
1464
+ if (toolId === "zustand" && descriptor.action === "setState") {
1465
+ if (typeof params.storeName !== "string") return undefined;
1466
+ try {
1467
+ return await dispatch(toolId, "getStoreState", {
1468
+ storeName: params.storeName
1469
+ });
1470
+ } catch {
1471
+ return undefined;
1472
+ }
1473
+ }
1474
+ /**
1475
+ * The same read-both-sides trick for the CACHE.
1476
+ *
1477
+ * A zustand edit produced a "what changed" card and the identical edit to a
1478
+ * query produced nothing — for the write the prompt calls "THE way to change
1479
+ * what a server-backed screen shows". The QA tester's confirmation that the
1480
+ * right field moved was missing from the most-used tool in the product.
1481
+ */
1482
+ if (toolId === "query" && descriptor.action === "setQueryData") {
1483
+ // hashQueryKey, not JSON.stringify: React Query SORTS object keys when it
1484
+ // hashes, so a key carrying `{query:"",category:null}` hashes as
1485
+ // `{"category":null,"query":""}`. Stringifying in insertion order produced
1486
+ // a hash that matched nothing, the read came back empty, and the card
1487
+ // silently did not draw.
1488
+ const queryHash = typeof params.queryHash === "string" ? params.queryHash : Array.isArray(params.queryKey) ? (0, _normalizeParams.hashQueryKey)(params.queryKey) : undefined;
1489
+ if (!queryHash) return undefined;
1490
+ try {
1491
+ return await dispatch(toolId, "getQueryData", {
1492
+ queryHash
1493
+ });
1494
+ } catch {
1495
+ return undefined;
1496
+ }
628
1497
  }
1498
+ return undefined;
629
1499
  }
630
1500
  async function captureBefore(descriptor, toolId, params, dispatch) {
631
1501
  if (descriptor.effect === "read") return undefined;