@buoy-gg/agent-core 7.0.35 → 7.0.38

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (208) hide show
  1. package/README.md +21 -14
  2. package/lib/commonjs/blocks/receipts.js +57 -1
  3. package/lib/commonjs/blocks/receipts.js.map +1 -1
  4. package/lib/commonjs/blocks/types.js +23 -2
  5. package/lib/commonjs/blocks/types.js.map +1 -1
  6. package/lib/commonjs/blocks/uiTool.js +423 -10
  7. package/lib/commonjs/blocks/uiTool.js.map +1 -1
  8. package/lib/commonjs/catalog/catalog.g.js +213 -48
  9. package/lib/commonjs/catalog/catalog.g.js.map +1 -1
  10. package/lib/commonjs/catalog/catalog.source.json +226 -48
  11. package/lib/commonjs/catalog/catalog.types.g.js +44 -0
  12. package/lib/commonjs/catalog/catalog.types.g.js.map +1 -0
  13. package/lib/commonjs/catalog/normalizeParams.js +333 -0
  14. package/lib/commonjs/catalog/normalizeParams.js.map +1 -0
  15. package/lib/commonjs/catalog/signature.js +68 -0
  16. package/lib/commonjs/catalog/signature.js.map +1 -0
  17. package/lib/commonjs/catalog/snapshotReads.js +13 -6
  18. package/lib/commonjs/catalog/snapshotReads.js.map +1 -1
  19. package/lib/commonjs/catalog/toProviderTools.js +44 -6
  20. package/lib/commonjs/catalog/toProviderTools.js.map +1 -1
  21. package/lib/commonjs/context/buildContextPack.js +367 -19
  22. package/lib/commonjs/context/buildContextPack.js.map +1 -1
  23. package/lib/commonjs/effects/digest.js +34 -0
  24. package/lib/commonjs/effects/digest.js.map +1 -0
  25. package/lib/commonjs/effects/ledger.js +98 -2
  26. package/lib/commonjs/effects/ledger.js.map +1 -1
  27. package/lib/commonjs/engine/askGate.js +287 -0
  28. package/lib/commonjs/engine/askGate.js.map +1 -0
  29. package/lib/commonjs/engine/effectFor.js +82 -1
  30. package/lib/commonjs/engine/effectFor.js.map +1 -1
  31. package/lib/commonjs/engine/evidence.js +113 -0
  32. package/lib/commonjs/engine/evidence.js.map +1 -0
  33. package/lib/commonjs/engine/historyBudget.js +363 -0
  34. package/lib/commonjs/engine/historyBudget.js.map +1 -0
  35. package/lib/commonjs/engine/retrieve.js +214 -0
  36. package/lib/commonjs/engine/retrieve.js.map +1 -0
  37. package/lib/commonjs/engine/runAgentTurn.js +1231 -120
  38. package/lib/commonjs/engine/runAgentTurn.js.map +1 -1
  39. package/lib/commonjs/engine/systemPrompt.js +103 -9
  40. package/lib/commonjs/engine/systemPrompt.js.map +1 -1
  41. package/lib/commonjs/engine/textToolCalls.js +276 -0
  42. package/lib/commonjs/engine/textToolCalls.js.map +1 -0
  43. package/lib/commonjs/engine/tokenCalibration.js +81 -0
  44. package/lib/commonjs/engine/tokenCalibration.js.map +1 -0
  45. package/lib/commonjs/engine/verify.js +300 -0
  46. package/lib/commonjs/engine/verify.js.map +1 -0
  47. package/lib/commonjs/index.js +74 -0
  48. package/lib/commonjs/index.js.map +1 -1
  49. package/lib/commonjs/policy/labels.js +19 -3
  50. package/lib/commonjs/policy/labels.js.map +1 -1
  51. package/lib/commonjs/policy/policy.js +6 -1
  52. package/lib/commonjs/policy/policy.js.map +1 -1
  53. package/lib/commonjs/policy/redact.js +42 -5
  54. package/lib/commonjs/policy/redact.js.map +1 -1
  55. package/lib/commonjs/providers/anthropic.js +340 -208
  56. package/lib/commonjs/providers/anthropic.js.map +1 -1
  57. package/lib/commonjs/providers/openai.js +237 -103
  58. package/lib/commonjs/providers/openai.js.map +1 -1
  59. package/lib/commonjs/providers/problem.js +98 -0
  60. package/lib/commonjs/providers/problem.js.map +1 -0
  61. package/lib/commonjs/providers/sse.js +170 -34
  62. package/lib/commonjs/providers/sse.js.map +1 -1
  63. package/lib/commonjs/providers/streamTimer.js +123 -0
  64. package/lib/commonjs/providers/streamTimer.js.map +1 -0
  65. package/lib/commonjs/providers/transport.js +123 -0
  66. package/lib/commonjs/providers/transport.js.map +1 -0
  67. package/lib/commonjs/providers/xhrStream.js +212 -0
  68. package/lib/commonjs/providers/xhrStream.js.map +1 -0
  69. package/lib/commonjs/session.js +242 -58
  70. package/lib/commonjs/session.js.map +1 -1
  71. package/lib/module/blocks/receipts.js +57 -1
  72. package/lib/module/blocks/receipts.js.map +1 -1
  73. package/lib/module/blocks/types.js +23 -2
  74. package/lib/module/blocks/types.js.map +1 -1
  75. package/lib/module/blocks/uiTool.js +421 -10
  76. package/lib/module/blocks/uiTool.js.map +1 -1
  77. package/lib/module/catalog/catalog.g.js +213 -48
  78. package/lib/module/catalog/catalog.g.js.map +1 -1
  79. package/lib/module/catalog/catalog.source.json +226 -48
  80. package/lib/module/catalog/catalog.types.g.js +40 -0
  81. package/lib/module/catalog/catalog.types.g.js.map +1 -0
  82. package/lib/module/catalog/normalizeParams.js +327 -0
  83. package/lib/module/catalog/normalizeParams.js.map +1 -0
  84. package/lib/module/catalog/signature.js +63 -0
  85. package/lib/module/catalog/signature.js.map +1 -0
  86. package/lib/module/catalog/snapshotReads.js +13 -6
  87. package/lib/module/catalog/snapshotReads.js.map +1 -1
  88. package/lib/module/catalog/toProviderTools.js +44 -6
  89. package/lib/module/catalog/toProviderTools.js.map +1 -1
  90. package/lib/module/context/buildContextPack.js +366 -19
  91. package/lib/module/context/buildContextPack.js.map +1 -1
  92. package/lib/module/effects/digest.js +29 -0
  93. package/lib/module/effects/digest.js.map +1 -0
  94. package/lib/module/effects/ledger.js +98 -2
  95. package/lib/module/effects/ledger.js.map +1 -1
  96. package/lib/module/engine/askGate.js +280 -0
  97. package/lib/module/engine/askGate.js.map +1 -0
  98. package/lib/module/engine/effectFor.js +83 -1
  99. package/lib/module/engine/effectFor.js.map +1 -1
  100. package/lib/module/engine/evidence.js +107 -0
  101. package/lib/module/engine/evidence.js.map +1 -0
  102. package/lib/module/engine/historyBudget.js +355 -0
  103. package/lib/module/engine/historyBudget.js.map +1 -0
  104. package/lib/module/engine/retrieve.js +210 -0
  105. package/lib/module/engine/retrieve.js.map +1 -0
  106. package/lib/module/engine/runAgentTurn.js +1227 -121
  107. package/lib/module/engine/runAgentTurn.js.map +1 -1
  108. package/lib/module/engine/systemPrompt.js +101 -9
  109. package/lib/module/engine/systemPrompt.js.map +1 -1
  110. package/lib/module/engine/textToolCalls.js +270 -0
  111. package/lib/module/engine/textToolCalls.js.map +1 -0
  112. package/lib/module/engine/tokenCalibration.js +75 -0
  113. package/lib/module/engine/tokenCalibration.js.map +1 -0
  114. package/lib/module/engine/verify.js +292 -0
  115. package/lib/module/engine/verify.js.map +1 -0
  116. package/lib/module/index.js +7 -3
  117. package/lib/module/index.js.map +1 -1
  118. package/lib/module/policy/labels.js +19 -3
  119. package/lib/module/policy/labels.js.map +1 -1
  120. package/lib/module/policy/policy.js +6 -1
  121. package/lib/module/policy/policy.js.map +1 -1
  122. package/lib/module/policy/redact.js +42 -5
  123. package/lib/module/policy/redact.js.map +1 -1
  124. package/lib/module/providers/anthropic.js +341 -209
  125. package/lib/module/providers/anthropic.js.map +1 -1
  126. package/lib/module/providers/openai.js +238 -104
  127. package/lib/module/providers/openai.js.map +1 -1
  128. package/lib/module/providers/problem.js +92 -0
  129. package/lib/module/providers/problem.js.map +1 -0
  130. package/lib/module/providers/sse.js +166 -34
  131. package/lib/module/providers/sse.js.map +1 -1
  132. package/lib/module/providers/streamTimer.js +118 -0
  133. package/lib/module/providers/streamTimer.js.map +1 -0
  134. package/lib/module/providers/transport.js +119 -0
  135. package/lib/module/providers/transport.js.map +1 -0
  136. package/lib/module/providers/xhrStream.js +207 -0
  137. package/lib/module/providers/xhrStream.js.map +1 -0
  138. package/lib/module/session.js +226 -60
  139. package/lib/module/session.js.map +1 -1
  140. package/lib/typescript/blocks/receipts.d.ts +5 -0
  141. package/lib/typescript/blocks/receipts.d.ts.map +1 -1
  142. package/lib/typescript/blocks/types.d.ts +23 -2
  143. package/lib/typescript/blocks/types.d.ts.map +1 -1
  144. package/lib/typescript/blocks/uiTool.d.ts +11 -2
  145. package/lib/typescript/blocks/uiTool.d.ts.map +1 -1
  146. package/lib/typescript/catalog/catalog.g.d.ts +2 -2
  147. package/lib/typescript/catalog/catalog.g.d.ts.map +1 -1
  148. package/lib/typescript/catalog/catalog.types.g.d.ts +40 -0
  149. package/lib/typescript/catalog/catalog.types.g.d.ts.map +1 -0
  150. package/lib/typescript/catalog/normalizeParams.d.ts +43 -0
  151. package/lib/typescript/catalog/normalizeParams.d.ts.map +1 -0
  152. package/lib/typescript/catalog/signature.d.ts +29 -0
  153. package/lib/typescript/catalog/signature.d.ts.map +1 -0
  154. package/lib/typescript/catalog/snapshotReads.d.ts.map +1 -1
  155. package/lib/typescript/catalog/toProviderTools.d.ts +28 -15
  156. package/lib/typescript/catalog/toProviderTools.d.ts.map +1 -1
  157. package/lib/typescript/context/buildContextPack.d.ts +24 -2
  158. package/lib/typescript/context/buildContextPack.d.ts.map +1 -1
  159. package/lib/typescript/effects/digest.d.ts +10 -0
  160. package/lib/typescript/effects/digest.d.ts.map +1 -0
  161. package/lib/typescript/effects/ledger.d.ts +73 -0
  162. package/lib/typescript/effects/ledger.d.ts.map +1 -1
  163. package/lib/typescript/engine/askGate.d.ts +29 -0
  164. package/lib/typescript/engine/askGate.d.ts.map +1 -0
  165. package/lib/typescript/engine/effectFor.d.ts +6 -1
  166. package/lib/typescript/engine/effectFor.d.ts.map +1 -1
  167. package/lib/typescript/engine/evidence.d.ts +67 -0
  168. package/lib/typescript/engine/evidence.d.ts.map +1 -0
  169. package/lib/typescript/engine/historyBudget.d.ts +115 -0
  170. package/lib/typescript/engine/historyBudget.d.ts.map +1 -0
  171. package/lib/typescript/engine/retrieve.d.ts +32 -0
  172. package/lib/typescript/engine/retrieve.d.ts.map +1 -0
  173. package/lib/typescript/engine/runAgentTurn.d.ts +179 -3
  174. package/lib/typescript/engine/runAgentTurn.d.ts.map +1 -1
  175. package/lib/typescript/engine/systemPrompt.d.ts +80 -0
  176. package/lib/typescript/engine/systemPrompt.d.ts.map +1 -1
  177. package/lib/typescript/engine/textToolCalls.d.ts +58 -0
  178. package/lib/typescript/engine/textToolCalls.d.ts.map +1 -0
  179. package/lib/typescript/engine/tokenCalibration.d.ts +46 -0
  180. package/lib/typescript/engine/tokenCalibration.d.ts.map +1 -0
  181. package/lib/typescript/engine/verify.d.ts +67 -0
  182. package/lib/typescript/engine/verify.d.ts.map +1 -0
  183. package/lib/typescript/index.d.ts +7 -3
  184. package/lib/typescript/index.d.ts.map +1 -1
  185. package/lib/typescript/policy/labels.d.ts.map +1 -1
  186. package/lib/typescript/policy/policy.d.ts.map +1 -1
  187. package/lib/typescript/policy/redact.d.ts.map +1 -1
  188. package/lib/typescript/providers/anthropic.d.ts.map +1 -1
  189. package/lib/typescript/providers/openai.d.ts +0 -16
  190. package/lib/typescript/providers/openai.d.ts.map +1 -1
  191. package/lib/typescript/providers/problem.d.ts +50 -0
  192. package/lib/typescript/providers/problem.d.ts.map +1 -0
  193. package/lib/typescript/providers/sse.d.ts +86 -2
  194. package/lib/typescript/providers/sse.d.ts.map +1 -1
  195. package/lib/typescript/providers/streamTimer.d.ts +59 -0
  196. package/lib/typescript/providers/streamTimer.d.ts.map +1 -0
  197. package/lib/typescript/providers/transport.d.ts +48 -0
  198. package/lib/typescript/providers/transport.d.ts.map +1 -0
  199. package/lib/typescript/providers/types.d.ts +105 -3
  200. package/lib/typescript/providers/types.d.ts.map +1 -1
  201. package/lib/typescript/providers/xhrStream.d.ts +45 -0
  202. package/lib/typescript/providers/xhrStream.d.ts.map +1 -0
  203. package/lib/typescript/session.d.ts +55 -4
  204. package/lib/typescript/session.d.ts.map +1 -1
  205. package/lib/typescript/types.d.ts +6 -0
  206. package/lib/typescript/types.d.ts.map +1 -1
  207. package/package.json +10 -10
  208. package/LICENSE +0 -58
@@ -21,16 +21,25 @@
21
21
 
22
22
  import { CATALOG } from "../catalog/catalog.g";
23
23
  import { applyDefaults, validateParams } from "../catalog/validateParams";
24
+ import { normalizeParams, aliasNote } from "../catalog/normalizeParams";
24
25
  import { fromProviderToolName, toProviderTools } from "../catalog/toProviderTools";
26
+ import { hashQueryKey } from "../catalog/normalizeParams";
25
27
  import { projectSnapshot, SNAPSHOT_ACTION } from "../catalog/snapshotReads";
26
28
  import { isReversible } from "../effects/ledger";
27
29
  import { describeCall, decide } from "../policy/policy";
28
30
  import { redact, stripSelfTraffic } from "../policy/redact";
29
31
  import { effectFor } from "./effectFor";
32
+ import { digestOf } from "../effects/digest";
30
33
  import { buildUiBlock, UI_TOOL_NAME, uiProviderTool } from "../blocks/uiTool";
31
34
  import { projectReceipt } from "../blocks/receipts";
32
35
  import { isAskBlock } from "../blocks/types";
33
-
36
+ import { ACQUISITION_CALLS, synthesizeAskBlock, synthesizeShortfallBlock } from "./askGate";
37
+ import { couldBeToolCallEnvelope, parseTextToolCalls, SALVAGE_NOTE } from "./textToolCalls";
38
+ import { budgetForRequest, MAX_HISTORY_TOKENS, messageSize } from "./historyBudget";
39
+ import { promptTokensOf, TokenCalibration } from "./tokenCalibration";
40
+ import { truncationMarker } from "./evidence";
41
+ import { retrieveEvidence } from "./retrieve";
42
+ import { verificationTrailer, verifyOutcome } from "./verify";
34
43
  /** Beyond this a tool result costs more in tokens than it can possibly be worth. */
35
44
  const MAX_RESULT_CHARS = 24_000;
36
45
  /**
@@ -39,20 +48,127 @@ const MAX_RESULT_CHARS = 24_000;
39
48
  * store dump per step makes the trace unreadable rather than more useful.
40
49
  */
41
50
  const MAX_TRACE_RESULT_CHARS = 2_000;
51
+ /**
52
+ * How many identical rounds before the model is told it is looping.
53
+ *
54
+ * `maxSteps` alone was the whole answer, and it is the wrong shape: a model
55
+ * that misread a result retries the same call until the cap, then the turn
56
+ * ends with nothing — the user watches twelve identical rows go by and gets
57
+ * no answer. The cap is a backstop, not feedback. Telling the model what it
58
+ * is doing gives it the chance to change approach, and costs one sentence.
59
+ *
60
+ * Signature is name + action + arguments, key-sorted so `{b,a}` and `{a,b}`
61
+ * are the same call. Three, not two: a legitimate retry after a transient
62
+ * failure is normal, and warning on it would be noise.
63
+ */
64
+ const LOOP_REPEATS = 3;
65
+ function callSignature(calls) {
66
+ const sort = v => {
67
+ if (v === null || typeof v !== "object") return v;
68
+ if (Array.isArray(v)) return v.map(sort);
69
+ const o = v;
70
+ return Object.keys(o).sort().reduce((a, k) => (a[k] = sort(o[k]), a), {});
71
+ };
72
+ return calls.map(c => `${c.name}:${JSON.stringify(sort(c.input ?? {}))}`).join("|");
73
+ }
74
+ const LOOP_WARNING = "[Buoy] You have now made this exact call three times in a row and gotten the same thing back. Repeating it again will not change the answer. Read the result you already have, then either try a different tool or different arguments, or tell the user plainly what you could not get and what you tried.";
42
75
  const DEFAULT_MAX_STEPS = 12;
43
76
  const DEFAULT_TURN_MS = 180_000;
77
+
78
+ /**
79
+ * Retrying a model request — narrowly.
80
+ *
81
+ * A 429 from a shared gateway, a 529 / `overloaded_error` from Anthropic, or
82
+ * a "prompt is too long" 400 used to end the turn with an error card and Try
83
+ * again — and Try again re-sends the QUESTION, restarting an investigation
84
+ * that may already have written things. All three fail before a model has
85
+ * generated anything, so asking again is safe: nothing runs twice, nothing
86
+ * is billed twice. The rule that makes it safe is enforced, not assumed —
87
+ * a request is only retried when its stream produced NO text, tool call or
88
+ * thinking; once anything has come back, the existing no-replay handling
89
+ * stands (see providers/transport.ts on consumed-exactly-once).
90
+ *
91
+ * Throttled: up to two retries, 1 s then 2 s (plus jitter), a `Retry-After`
92
+ * winning when the endpoint sent one. Overflow: one retry, with the history
93
+ * ceiling halved first — the proactive budget is chars ÷ 4 and the provider
94
+ * just said that was optimistic. Both wait inside the turn deadline and stop
95
+ * on the user's Stop. Small numbers on purpose: this is a phone, not a
96
+ * server queue.
97
+ */
98
+ const THROTTLE_RETRIES = 2;
99
+ const OVERFLOW_RETRIES = 1;
100
+ const RETRY_BASE_MS = 1_000;
101
+ const RETRY_JITTER_MS = 250;
102
+ const RETRY_AFTER_CAP_MS = 30_000;
103
+ function sleep(ms, signal) {
104
+ return new Promise(resolve => {
105
+ if (signal?.aborted) return resolve();
106
+ const t = setTimeout(done, ms);
107
+ function done() {
108
+ clearTimeout(t);
109
+ signal?.removeEventListener("abort", done);
110
+ resolve();
111
+ }
112
+ signal?.addEventListener("abort", done, {
113
+ once: true
114
+ });
115
+ });
116
+ }
117
+
118
+ /**
119
+ * Why the turn ended — the machine-readable half of the notices. The text in
120
+ * the notice blocks is unchanged (the eval classifier pins it); this is for a
121
+ * host, the bank and the desktop, which used to have to grep the prose.
122
+ */
123
+
124
+ /** Normalise an answer to its parts, so both call sites read one shape. */
125
+ function readAnswer(answer) {
126
+ if (typeof answer === "boolean") return {
127
+ approved: answer,
128
+ trust: false
129
+ };
130
+ const reason = answer.reason?.trim();
131
+ return {
132
+ approved: answer.approved,
133
+ ...(reason ? {
134
+ reason
135
+ } : {}),
136
+ trust: answer.trust === true && answer.approved
137
+ };
138
+ }
139
+
140
+ /** What the model is told when the user says no — with their note, when they left one. */
141
+ function declinedResult(reason) {
142
+ return reason ? `The user declined this change and said: "${reason}". Do not retry it as proposed; take their note into account and, if a different change would fit, propose that instead.` : "The user declined this change. Do not retry it; ask what they would prefer.";
143
+ }
44
144
  function findDescriptor(catalog, toolId, action) {
45
145
  return catalog.find(t => t.toolId === toolId)?.actions.find(a => a.action === action);
46
146
  }
47
- function encodeResult(value) {
48
- let text;
147
+
148
+ /**
149
+ * Why a picture did not render, in the model's own terms.
150
+ *
151
+ * Two messages, not one, because they answer different questions: the first
152
+ * is "your block was refused", the second is "your block was shown but with
153
+ * holes in it". The literal opening of the rejection is pinned by the
154
+ * trajectory classifier (evals/ask-buoy/lib/trajectory.ts) — keep it.
155
+ */
156
+ const URL_PROVENANCE_REJECTION = "Rejected: every image URL must be one you read from a tool result in THIS conversation (images.list, a storage value, a response body). Never invent or remember URLs — read them first, then show them.";
157
+ const droppedImageNote = n => ` NOTE: ${n} image ${n === 1 ? "URL was" : "URLs were"} dropped — they were not read from a tool result in this conversation, so the card the user is looking at has no picture there. Do not tell them it does; read the URLs and show it again, or say the artwork is missing.`;
158
+
159
+ /** The whole result as text — what the evidence store keeps. */
160
+ function fullText(value) {
49
161
  try {
50
- text = JSON.stringify(value ?? null);
162
+ return JSON.stringify(value ?? null);
51
163
  } catch {
52
- text = String(value);
164
+ return String(value);
53
165
  }
166
+ }
167
+
168
+ /** What the model is sent: the text, cut at the cap with a marker that names where the rest is. */
169
+ function encodeResult(text, ref) {
54
170
  if (text.length <= MAX_RESULT_CHARS) return text;
55
- return `${text.slice(0, MAX_RESULT_CHARS)}\n\n[truncated — ${text.length} characters total. Ask for a narrower slice if you need more.]`;
171
+ return `${text.slice(0, MAX_RESULT_CHARS)}${truncationMarker(text.length, ref)}`;
56
172
  }
57
173
 
58
174
  /** A short line for the transcript row, so the UI never renders raw payloads. */
@@ -88,11 +204,30 @@ function summarise(value) {
88
204
  function countOf(n) {
89
205
  return n === 1 ? "1 item" : `${n} items`;
90
206
  }
207
+
208
+ /**
209
+ * A call that came back reporting its own failure.
210
+ *
211
+ * Buoy adapters RESOLVE with `{ok:false, error}` rather than throwing, so the
212
+ * dispatch's try/catch never fires and a refused write looked like a completed
213
+ * one: the row read "Done", and — much worse — the effect ledger recorded a
214
+ * change that had not happened, so the changes bar offered to undo nothing.
215
+ * Caught on device: a typed-edit refusal ("qty is a number, 'many' is text")
216
+ * left the bag untouched and the bar claiming one permanent change.
217
+ *
218
+ * `ok === false` is the same signal `summarise` above already trusts to turn a
219
+ * result into the row's text, so this reads it no more liberally than the UI
220
+ * already does.
221
+ */
222
+ function reportsFailure(value) {
223
+ return typeof value === "object" && value !== null && value.ok === false;
224
+ }
91
225
  export async function* runAgentTurn(input) {
92
226
  const {
93
227
  provider,
94
228
  catalog,
95
229
  system,
230
+ systemVolatile,
96
231
  model,
97
232
  maxTokens,
98
233
  dispatch,
@@ -102,8 +237,14 @@ export async function* runAgentTurn(input) {
102
237
  policy,
103
238
  isRelease,
104
239
  requestApproval,
105
- signal
240
+ signal,
241
+ evidence,
242
+ trusted,
243
+ procedures
106
244
  } = input;
245
+ /** Notes the user left with a decline, by call id — see declinedResult. */
246
+ const declineReasons = new Map();
247
+ const calibration = input.calibration ?? new TokenCalibration();
107
248
  const messages = [...input.messages];
108
249
  const tools = toProviderTools(catalog, {
109
250
  availableToolIds: input.availableToolIds,
@@ -112,105 +253,697 @@ export async function* runAgentTurn(input) {
112
253
  });
113
254
  tools.push(uiProviderTool());
114
255
  const maxSteps = policy.maxSteps ?? DEFAULT_MAX_STEPS;
115
- const deadline = Date.now() + DEFAULT_TURN_MS;
256
+ /**
257
+ * The conversation this turn belongs to. Handed back on every `record` so a
258
+ * dispatch that settles after the user started a new conversation lands in
259
+ * the app (nothing can pull it back) but not in the new chat's undo bar.
260
+ */
261
+ const ledgerGeneration = ledger.generation;
262
+ /**
263
+ * What every request costs before a single message is added: the tool
264
+ * definitions, the system prompt, and the room the model needs to reply.
265
+ *
266
+ * Measured, not guessed — the tool block alone is ~32k characters for a
267
+ * typical install and ~94k with all 24 tools available, which is far too
268
+ * much to leave out of a budget. `maxTokens` is multiplied by four because
269
+ * the budget is in characters.
270
+ */
271
+ const fixedChars = system.length + (systemVolatile?.length ?? 0) + tools.reduce((n, t) => n + t.description.length + JSON.stringify(t.inputSchema).length, 0);
272
+ /** The reserve at the ratio believed RIGHT NOW — it moves as usage reports come in. */
273
+ const requestReserve = () => fixedChars + calibration.chars(maxTokens);
274
+ /**
275
+ * Caps the AGENT's wall-clock, not the user's. Time spent parked on an
276
+ * approval sheet is pushed onto the deadline as it is spent (see the
277
+ * `requestApproval` await below) — otherwise a user who takes three minutes
278
+ * to tap Approve gets the change applied and then "this is taking too long"
279
+ * for the turn that applied it.
280
+ */
281
+ let deadline = Date.now() + DEFAULT_TURN_MS;
282
+ /** Retries spent this turn, per kind — the budget is per turn, not per step. */
283
+ const retries = {
284
+ throttled: 0,
285
+ overflow: 0
286
+ };
287
+ /** Tightened after an overflow; see budgetForRequest's `cap`. In tokens, so a calibration change re-scales it. */
288
+ let historyCapTokens = MAX_HISTORY_TOKENS;
116
289
 
117
- // Every http(s) URL that appeared in a tool result THIS turn. The prompt
118
- // tells the model image URLs must come from here; this Set is the
119
- // enforcement — a model-remembered or injected URL in an image block is
290
+ // Every http(s) URL that appeared in a tool result in this CONVERSATION.
291
+ // The prompt tells the model image URLs must come from here; this Set is
292
+ // the enforcement — a model-remembered or injected URL in an image block is
120
293
  // both a fabrication vector and a data-exfil beacon (the GET carries
121
- // whatever is encoded into the URL).
122
- const seenUrls = new Set();
294
+ // whatever is encoded into the URL). Session-owned when the caller passes
295
+ // one; see RunTurnInput.seenUrls for why it is not per-turn.
296
+ const seenUrls = input.seenUrls ?? new Set();
123
297
  const URL_RE = /https?:\/\/[^\s"'\\)\]}>]+/g;
298
+
299
+ /** Whether any text has gone out THIS TURN — see `breakBefore`. */
300
+ let answerStarted = false;
301
+
302
+ /**
303
+ * The last few rounds' call signatures. See LOOP_REPEATS — a run of
304
+ * identical rounds gets one warning appended to the results, once, so the
305
+ * model is told rather than silently capped.
306
+ */
307
+ const recentSignatures = [];
308
+ let loopWarned = false;
309
+
310
+ /**
311
+ * Everything the model SAID this turn, in order. Read once, at the end, by
312
+ * the ask gate — a question asked in round one and left hanging is the same
313
+ * dead end as one asked in the last round. See askGate.ts.
314
+ */
315
+ const spoken = [];
316
+ /**
317
+ * Whether the user already has something to tap. An `actions` or
318
+ * `suggestions` block is the model doing this job itself, and a second card
319
+ * asking the same thing underneath it is worse than none.
320
+ *
321
+ * Note this does NOT suppress the shortfall gate below — that one asks
322
+ * whether the user can tap the GAP, which unrelated chips do not answer.
323
+ */
324
+ let offeredTap = false;
325
+ /**
326
+ * Whether this turn actually went and looked — navigated, described the
327
+ * screen, tapped something, refetched. The shortfall gate stays quiet when
328
+ * it did: an agent that tried and came up short has earned its answer.
329
+ */
330
+ let attemptedAcquisition = false;
124
331
  for (let step = 0; step < maxSteps; step++) {
125
332
  if (signal?.aborted) {
126
333
  yield {
127
- type: "done"
334
+ type: "done",
335
+ stopReason: "stopped"
128
336
  };
129
337
  return messages;
130
338
  }
131
339
  if (Date.now() > deadline) {
340
+ // Same shape as the step cap below. This used to be an `error`, which
341
+ // drew the failed-turn card with Try again — and Try again re-sends the
342
+ // question, restarting the whole investigation that just ran out of
343
+ // time. Continue picks up with everything so far still in history.
132
344
  yield {
133
- type: "error",
134
- message: "This is taking too long — stopping here."
345
+ type: "block",
346
+ block: {
347
+ id: `cap${Date.now()}`,
348
+ kind: "notice",
349
+ tone: "warning",
350
+ // Keeps the "This is taking too long" prefix: the eval classifier
351
+ // (evals/ask-buoy/lib/trajectory.ts) tells a time-out from a step
352
+ // cap by it.
353
+ text: `This is taking too long — stopped after ${Math.round(DEFAULT_TURN_MS / 1000)}s. What it found so far is above.`,
354
+ actions: [{
355
+ label: "Continue",
356
+ primary: true,
357
+ send: "Continue where you left off."
358
+ }]
359
+ }
135
360
  };
136
361
  yield {
137
- type: "done"
362
+ type: "done",
363
+ stopReason: "time-cap"
138
364
  };
139
365
  return messages;
140
366
  }
367
+
368
+ /**
369
+ * BUDGET, EVERY REQUEST — not once per turn.
370
+ *
371
+ * The session trims when a turn opens, and that used to be the only
372
+ * check. But a turn is not one request: each step appends an assistant
373
+ * message and a tool-results message, and a single result can be 24,000
374
+ * characters. A twelve-step investigation could therefore add six figures
375
+ * of history AFTER the only check had run, and the turn died on the
376
+ * provider's context limit with an opaque error — in exactly the long
377
+ * sessions the tool is for.
378
+ *
379
+ * The current round is never touched: it is the question being answered.
380
+ */
381
+ const budgeted = budgetForRequest(messages, requestReserve(), calibration.chars(historyCapTokens), calibration.charsPerToken, ledger.liveCallIds());
382
+ if (budgeted.droppedRounds > 0) {
383
+ messages.length = 0;
384
+ messages.push(...budgeted.messages);
385
+ yield {
386
+ type: "history-trimmed",
387
+ droppedRounds: budgeted.droppedRounds,
388
+ droppedPinned: budgeted.droppedPinned
389
+ };
390
+ } else if (budgeted.messages !== messages) {
391
+ messages.length = 0;
392
+ messages.push(...budgeted.messages);
393
+ }
141
394
  let text = "";
142
- const calls = [];
395
+ /** See `breakBefore`: set once the first text of THIS round has gone out. */
396
+ let textThisRound = false;
397
+ const textEvent = delta => {
398
+ // Only when this round opens a NEW paragraph of an answer that has
399
+ // already started. A turn whose first words arrive in round three has
400
+ // nothing to be separated from, and marking that would put the flag on
401
+ // events where it means nothing.
402
+ const opensNewRound = answerStarted && !textThisRound;
403
+ textThisRound = true;
404
+ answerStarted = true;
405
+ return opensNewRound ? {
406
+ type: "text",
407
+ delta,
408
+ breakBefore: true
409
+ } : {
410
+ type: "text",
411
+ delta
412
+ };
413
+ };
414
+ let calls = [];
415
+ /**
416
+ * Recovered from text rather than emitted as calls — see textToolCalls.ts.
417
+ * Tracked by id so each one's result can tell the model to stop doing that;
418
+ * they are otherwise ordinary calls and get every check the others get.
419
+ */
420
+ const salvagedIds = new Set();
421
+ /**
422
+ * Text withheld from the screen while it could still be a tool-call
423
+ * envelope. Streamed text cannot be un-shown, so the choice has to be made
424
+ * before it leaves — and a model that writes its call as JSON must not
425
+ * have that JSON become the answer the user reads.
426
+ */
427
+ let held = "";
428
+ let holding = true;
143
429
  // Carried, never read: Anthropic requires the turn's thinking blocks back
144
430
  // verbatim with its tool results. See providers/anthropic.ts note 5.
145
- const thinking = [];
431
+ let thinking = [];
146
432
  let failed = false;
147
- for await (const ev of provider.send({
148
- messages,
149
- system,
150
- tools,
151
- model,
152
- maxTokens,
153
- signal
154
- })) {
155
- if (ev.type === "text") {
156
- text += ev.delta;
157
- yield {
158
- type: "text",
159
- delta: ev.delta
160
- };
161
- } else if (ev.type === "tool-call") {
162
- calls.push(ev.call);
163
- } else if (ev.type === "thinking") {
164
- thinking.push(ev.block);
165
- // Surfaced as well as carried: the block goes back to the provider
166
- // verbatim (note above), and the readable half goes to the UI so a
167
- // tester can see WHY a turn did what it did. Redacted blocks have no
168
- // readable half and are carried only.
169
- if (ev.block.type === "thinking" && ev.block.thinking.trim()) {
170
- yield {
171
- type: "reasoning",
172
- text: ev.block.thinking
433
+ /**
434
+ * What the provider said about how the stream ended.
435
+ *
436
+ * Undefined means it never said — a bare EOF, or a host-supplied provider
437
+ * whose generator simply returned. Both are treated as incomplete, so the
438
+ * fail-safe direction is the default rather than something each adapter
439
+ * has to remember to opt into.
440
+ */
441
+ let outcome;
442
+ /** Set for the two kinds that are recoverable rather than a hard failure. */
443
+ let incomplete;
444
+
445
+ /**
446
+ * THE REQUEST, with its retries. One request per pass; a pass that fails
447
+ * before the model generated anything may be sent again (see
448
+ * THROTTLE_RETRIES). Everything the pass accumulated is reset first —
449
+ * there is nothing to keep, by the rule that made the retry safe.
450
+ */
451
+ for (;;) {
452
+ text = "";
453
+ calls = [];
454
+ held = "";
455
+ holding = true;
456
+ thinking = [];
457
+ failed = false;
458
+ outcome = undefined;
459
+ incomplete = undefined;
460
+ /** The failure that ended this pass, when it is one the engine may retry. */
461
+ let retryable;
462
+ /** What this pass sends, in characters — the numerator of the calibration sample. */
463
+ const sentChars = fixedChars + messages.reduce((n, m) => n + messageSize(m), 0);
464
+ for await (const ev of provider.send({
465
+ messages,
466
+ system,
467
+ systemVolatile,
468
+ tools,
469
+ model,
470
+ maxTokens,
471
+ signal
472
+ })) {
473
+ if (ev.type === "text") {
474
+ text += ev.delta;
475
+ if (!holding) {
476
+ yield textEvent(ev.delta);
477
+ } else if (couldBeToolCallEnvelope(text)) {
478
+ held += ev.delta;
479
+ } else {
480
+ // Not an envelope after all. Release everything at once and stream
481
+ // the rest as usual — the reader loses nothing but a few characters
482
+ // of latency at the very start of the answer.
483
+ holding = false;
484
+ held = "";
485
+ yield textEvent(text);
486
+ }
487
+ } else if (ev.type === "tool-call") {
488
+ calls.push(ev.call);
489
+ } else if (ev.type === "thinking") {
490
+ thinking.push(ev.block);
491
+ // Surfaced as well as carried: the block goes back to the provider
492
+ // verbatim (note above), and the readable half goes to the UI so a
493
+ // tester can see WHY a turn did what it did. Redacted blocks have no
494
+ // readable half and are carried only.
495
+ if (ev.block.type === "thinking" && ev.block.thinking.trim()) {
496
+ yield {
497
+ type: "reasoning",
498
+ text: ev.block.thinking
499
+ };
500
+ }
501
+ } else if (ev.type === "error") {
502
+ if (ev.kind === "truncated" || ev.kind === "timeout") {
503
+ // Not a failure the user did anything about, and not one Try again
504
+ // can fix — re-sending the QUESTION restarts an investigation that
505
+ // may already have written things. Held back from the error card
506
+ // (which is what draws Try again) and handled below as a recoverable
507
+ // stop with a Continue button.
508
+ incomplete = {
509
+ message: ev.message
510
+ };
511
+ } else if (ev.problem && (ev.problem.kind === "throttled" || ev.problem.kind === "overflow") && text === "" && calls.length === 0 && thinking.length === 0) {
512
+ // Nothing generated, and a kind that a wait or a trim can fix.
513
+ // Decided after the stream closes; the adapter returns on error.
514
+ retryable = {
515
+ problem: ev.problem,
516
+ message: ev.message
517
+ };
518
+ } else {
519
+ // A mid-stream error can arrive on an already-committed 200.
520
+ yield {
521
+ type: "error",
522
+ message: ev.message
523
+ };
524
+ failed = true;
525
+ }
526
+ } else if (ev.type === "done") {
527
+ outcome = ev.outcome;
528
+ if (ev.usage) {
529
+ // The provider just counted this prompt. One sample per request
530
+ // keeps the chars↔tokens ratio honest for the NEXT budget.
531
+ calibration.observe(sentChars, promptTokensOf(ev.usage, provider.protocol));
532
+ // Surfaced so a host can meter cost per seat. Parsed for a long time;
533
+ // went nowhere anyone could see.
534
+ yield {
535
+ type: "usage",
536
+ ...ev.usage,
537
+ model: ev.model
538
+ };
539
+ }
540
+ }
541
+ }
542
+ if (!retryable) break;
543
+ const kind = retryable.problem.kind;
544
+ const maxAttempts = 1 + (kind === "throttled" ? THROTTLE_RETRIES : OVERFLOW_RETRIES);
545
+ const spent = retries[kind];
546
+ let waitMs = kind === "throttled" ? Math.min(retryable.problem.retryAfterMs ?? RETRY_BASE_MS * 2 ** spent, RETRY_AFTER_CAP_MS) + Math.floor(Math.random() * RETRY_JITTER_MS) : 0;
547
+ let giveUp = spent >= maxAttempts - 1 || signal?.aborted === true || Date.now() + waitMs > deadline;
548
+ if (!giveUp && kind === "overflow") {
549
+ // Halve the ceiling and trim again. If that changes nothing — the
550
+ // current round alone is over the limit — a retry would only repeat
551
+ // the refusal, so report it instead.
552
+ historyCapTokens = Math.floor(historyCapTokens / 2);
553
+ const before = messages.reduce((n, m) => n + messageSize(m), 0);
554
+ const again = budgetForRequest(messages, requestReserve(), calibration.chars(historyCapTokens), calibration.charsPerToken, ledger.liveCallIds());
555
+ const after = again.messages.reduce((n, m) => n + messageSize(m), 0);
556
+ if (after >= before) {
557
+ giveUp = true;
558
+ } else {
559
+ messages.length = 0;
560
+ messages.push(...again.messages);
561
+ if (again.droppedRounds > 0) yield {
562
+ type: "history-trimmed",
563
+ droppedRounds: again.droppedRounds,
564
+ droppedPinned: again.droppedPinned
173
565
  };
174
566
  }
175
- } else if (ev.type === "error") {
176
- // A mid-stream error can arrive on an already-committed 200.
567
+ }
568
+ if (giveUp) {
569
+ // Plain words first; the endpoint's own text after, for whoever files
570
+ // the bug. A user reading "prompt is too long: 213000 tokens" has no
571
+ // move; "start a new conversation" is one.
572
+ const plain = kind === "throttled" ? spent > 0 ? `The AI endpoint is rate-limiting requests — tried ${spent + 1} times.` : "The AI endpoint is rate-limiting requests." : "This conversation is too large for the AI endpoint and there was nothing older to trim. Start a new conversation, or ask a shorter question.";
177
573
  yield {
178
574
  type: "error",
179
- message: ev.message
575
+ message: `${plain} (${retryable.message})`
180
576
  };
181
577
  failed = true;
182
- } else if (ev.type === "done" && ev.usage) {
183
- // Surfaced so a host can meter cost per seat. Parsed for a long time;
184
- // went nowhere anyone could see.
578
+ break;
579
+ }
580
+ retries[kind] += 1;
581
+ yield {
582
+ type: "retrying",
583
+ reason: kind,
584
+ attempt: retries[kind] + 1,
585
+ maxAttempts,
586
+ inMs: waitMs
587
+ };
588
+ if (waitMs > 0) await sleep(waitMs, signal);
589
+ if (signal?.aborted) {
185
590
  yield {
186
- type: "usage",
187
- input: ev.usage.input,
188
- output: ev.usage.output
591
+ type: "done",
592
+ stopReason: "stopped"
189
593
  };
594
+ return messages;
190
595
  }
191
596
  }
192
597
  if (failed) {
598
+ // Whatever was withheld is the model's own words on the way to an error.
599
+ // Show it rather than swallowing it.
600
+ if (held) yield textEvent(held);
601
+ yield {
602
+ type: "done",
603
+ stopReason: "error"
604
+ };
605
+ return messages;
606
+ }
607
+
608
+ /**
609
+ * THE COMPLETION GATE. Nothing below this runs a tool unless the provider
610
+ * said the response was whole.
611
+ *
612
+ * The failure it exists for is not theoretical and does not look like a
613
+ * failure: a connection cut after the model emitted
614
+ * `{"key":"cart","value":[]}` leaves valid JSON, a plausible-looking plan,
615
+ * and no error anywhere. The old code parsed those arguments, dispatched
616
+ * the write, sent the result back and carried on — a real mutation from
617
+ * half a sentence. Bare EOF is not consent.
618
+ *
619
+ * `output-limited` is the same shape from a different cause: the model hit
620
+ * max_tokens partway through planning. What it managed to SAY is real and
621
+ * is shown; what it was part-way through DOING is not run.
622
+ */
623
+ if (!incomplete && outcome !== "completed") {
624
+ incomplete = {
625
+ message: outcome === "output-limited" ? calls.length ? "The model ran out of room mid-plan, so nothing was run. What it found so far is above." : "The model ran out of room before finishing this answer." : "The answer ended before the endpoint said it was finished, so nothing was run."
626
+ };
627
+ }
628
+ if (incomplete) {
629
+ // Held text is the model's own words on the way out. Show it: the user
630
+ // watched it arrive and hiding it now reads as the app losing work.
631
+ if (held) yield textEvent(held);
632
+ /**
633
+ * Text only — never the tool calls, and never the thinking.
634
+ *
635
+ * An assistant turn carrying `tool_use` with no matching `tool_result`
636
+ * is an instant 400 on the next request, and these calls are exactly the
637
+ * ones that must not be answered. Keeping the text preserves what the
638
+ * user can see on screen for whatever comes next.
639
+ */
640
+ if (text.trim()) messages.push({
641
+ role: "assistant",
642
+ text
643
+ });
644
+ yield {
645
+ type: "block",
646
+ block: {
647
+ id: `cut${Date.now()}`,
648
+ kind: "notice",
649
+ tone: "warning",
650
+ text: incomplete.message,
651
+ actions: [{
652
+ label: "Continue",
653
+ primary: true,
654
+ send: "Continue where you left off."
655
+ }]
656
+ }
657
+ };
193
658
  yield {
194
- type: "done"
659
+ type: "done",
660
+ stopReason: "incomplete"
195
661
  };
196
662
  return messages;
197
663
  }
664
+ if (held) {
665
+ // A model that wrote its call out as JSON instead of calling it. Run it,
666
+ // and show its `reply` — never the blob. See textToolCalls.ts.
667
+ const salvaged = calls.length === 0 ? parseTextToolCalls(held, catalog) : undefined;
668
+ if (salvaged) {
669
+ for (const call of salvaged.calls) {
670
+ calls.push(call);
671
+ salvagedIds.add(call.id);
672
+ }
673
+ text = salvaged.reply;
674
+ if (salvaged.reply) yield textEvent(salvaged.reply);
675
+ } else {
676
+ yield textEvent(held);
677
+ }
678
+ held = "";
679
+ }
198
680
  messages.push({
199
681
  role: "assistant",
200
682
  text,
201
683
  toolCalls: calls.length ? calls : undefined,
202
684
  thinking: thinking.length ? thinking : undefined
203
685
  });
686
+ if (text.trim()) spoken.push(text.trim());
204
687
  if (calls.length === 0) {
688
+ // The turn is over and the model wrote its own words. Two ways those
689
+ // words end badly, and they are different failures — see askGate.ts.
690
+ const said = spoken.join("\n\n");
691
+
692
+ // 1. It answered a smaller question than it was asked and handed the
693
+ // rest back. Fires even when chips were offered: the question is
694
+ // whether the user can tap the GAP, not whether they can tap.
695
+ const gap = synthesizeShortfallBlock({
696
+ text: said,
697
+ attempted: attemptedAcquisition
698
+ });
699
+ if (gap) yield {
700
+ type: "block",
701
+ block: gap
702
+ };
703
+
704
+ // 2. It ended waiting on the user in prose. Suppressed once anything
705
+ // tappable is on screen, the gap button included.
706
+ const asked = offeredTap || gap ? null : synthesizeAskBlock(said);
707
+ if (asked) {
708
+ yield {
709
+ type: "block",
710
+ block: asked
711
+ };
712
+ yield {
713
+ type: "awaiting-user",
714
+ blockId: asked.id
715
+ };
716
+ }
205
717
  yield {
206
- type: "done"
718
+ type: "done",
719
+ stopReason: asked ? "awaiting-user" : "answered"
207
720
  };
208
721
  return messages;
209
722
  }
210
723
  const results = [];
211
724
  let realmWillDie = false;
725
+
726
+ /**
727
+ * Reads, already in flight.
728
+ *
729
+ * The loop below stays strictly serial — every yield, every ledger entry
730
+ * and every approval keeps its order — but a round of independent READS
731
+ * used to pay each round trip end to end. The prompt tells the model to
732
+ * chain reads freely ("reading is cheap; do it"), and a three-read round
733
+ * cost three times the device latency for no reason.
734
+ *
735
+ * ONLY when the whole batch is safe to overlap: every call a catalog read,
736
+ * none needing approval, no `buoy_ui`, nothing snapshot-only or
737
+ * session-local. One write, one approval or one unknown action in the
738
+ * batch and nothing is prefetched — the mixed case is where ordering
739
+ * actually matters, and it is not worth the risk for the latency.
740
+ */
741
+ const prefetch = new Map();
742
+ if (calls.length > 1) {
743
+ const safe = calls.every(c => {
744
+ if (c.name === UI_TOOL_NAME) return false;
745
+ const id = fromProviderToolName(c.name, catalog);
746
+ if (!id || id === "ask-buoy") return false;
747
+ const act = c.input?.action;
748
+ if (typeof act !== "string" || act === SNAPSHOT_ACTION) return false;
749
+ const d = findDescriptor(catalog, id, act);
750
+ if (!d || d.effect !== "read") return false;
751
+ return decide({
752
+ descriptor: d,
753
+ toolId: id,
754
+ policy,
755
+ isRelease
756
+ }).verdict === "allow";
757
+ });
758
+ if (safe) {
759
+ for (const c of calls) {
760
+ const id = fromProviderToolName(c.name, catalog);
761
+ const act = String(c.input.action);
762
+ const raw = {
763
+ ...(c.input.params ?? {})
764
+ };
765
+ const d = findDescriptor(catalog, id, act);
766
+ const norm = normalizeParams(id, act, raw);
767
+ const withDefaults = applyDefaults(norm.params, d.params);
768
+ if (!validateParams(withDefaults, d.params).ok) continue;
769
+ // Rejections are swallowed here and re-awaited in the loop, where
770
+ // they are turned into the same tool_result they always were.
771
+ const p = Promise.resolve(dispatch(id, act, withDefaults)).catch(e => {
772
+ throw e;
773
+ });
774
+ p.catch(() => {});
775
+ prefetch.set(c.id, p);
776
+ }
777
+ }
778
+ }
212
779
  let awaiting = null;
780
+
781
+ /**
782
+ * A call Stop got to first — as a row, so the transcript says which ones.
783
+ *
784
+ * Resolved leniently: the call has not been validated yet, and a label
785
+ * that falls back to `tool.action` is better than no row for a call the
786
+ * user needs to know did not run.
787
+ */
788
+ const stoppedStep = call => {
789
+ const toolId = fromProviderToolName(call.name, catalog) ?? call.name;
790
+ const action = typeof call.input?.action === "string" ? call.input.action : "";
791
+ const descriptor = action ? findDescriptor(catalog, toolId, action) : undefined;
792
+ const params = call.input?.params ?? {};
793
+ return [{
794
+ type: "tool-start",
795
+ id: call.id,
796
+ toolId,
797
+ action,
798
+ label: descriptor ? describeCall(toolId, descriptor, params) : `${toolId}.${action}`,
799
+ description: descriptor?.summary ?? "",
800
+ params,
801
+ effect: descriptor?.effect ?? "read"
802
+ }, {
803
+ type: "tool-end",
804
+ id: call.id,
805
+ ok: false,
806
+ summary: "stopped",
807
+ durationMs: 0
808
+ }];
809
+ };
810
+
811
+ /**
812
+ * THE BATCH BARRIER — what the whole batch will do, decided before any of
813
+ * it does anything.
814
+ *
815
+ * A model answers with several calls at once, and they used to run one at
816
+ * a time with each approval asked when its own call came up. So
817
+ * `[set the flag, delete the cache]` wrote the flag, THEN asked about the
818
+ * delete — and a user who declined had already changed the app, with the
819
+ * bubble reading "changed the flag" directly above "left the cache alone".
820
+ * Declining is supposed to mean the plan does not happen.
821
+ *
822
+ * The fix is not to ask again later or to refuse afterwards, neither of
823
+ * which can unrun a write. It is to ask FIRST: every gated call in the
824
+ * batch is put to the user before the batch's first mutation dispatches,
825
+ * so consent is given with the whole plan visible.
826
+ *
827
+ * Resolution here is pure and duplicates the loop's own — deliberately.
828
+ * Anything it cannot resolve (an unknown tool, a bad shape) comes back as
829
+ * neither mutating nor gated, so the loop reports it exactly as it always
830
+ * did and the barrier simply does not fire. Failing back to today's
831
+ * behaviour is the right failure for a safety gate to have.
832
+ */
833
+ const planned = calls.map(call => {
834
+ if (call.name === UI_TOOL_NAME) return undefined;
835
+ const toolId = fromProviderToolName(call.name, catalog);
836
+ const action = typeof call.input?.action === "string" ? call.input.action : undefined;
837
+ if (!toolId || !action) return undefined;
838
+ const descriptor = findDescriptor(catalog, toolId, action);
839
+ if (!descriptor) return undefined;
840
+ const raw = call.input.params ?? {};
841
+ const params = applyDefaults(normalizeParams(toolId, action, raw).params, descriptor.params);
842
+ if (!validateParams(params, descriptor.params).ok) return undefined;
843
+ const verdict = decide({
844
+ descriptor,
845
+ toolId,
846
+ policy,
847
+ isRelease
848
+ });
849
+ return {
850
+ call,
851
+ toolId,
852
+ action,
853
+ descriptor,
854
+ params,
855
+ label: describeCall(toolId, descriptor, params),
856
+ mutates: descriptor.effect !== "read",
857
+ gated: verdict.verdict === "needs-approval",
858
+ reason: verdict.verdict === "needs-approval" ? verdict.reason : ""
859
+ };
860
+ });
861
+ const gatedInBatch = planned.some(p => p?.gated);
862
+ /** Decisions the barrier has already taken, so no call is asked twice. */
863
+ const decided = new Map();
864
+ let barrierRun = false;
865
+ let batchDeclined = false;
866
+
867
+ /** Put every gated call in the batch to the user, in the order written. */
868
+ async function* runBarrier() {
869
+ barrierRun = true;
870
+ for (const p of planned) {
871
+ if (!p?.gated || decided.has(p.call.id)) continue;
872
+ if (trusted?.has(`${p.toolId}.${p.action}`)) {
873
+ // Waived for this conversation: runs like any allowed write, no card.
874
+ decided.set(p.call.id, true);
875
+ continue;
876
+ }
877
+ yield {
878
+ type: "approval-required",
879
+ id: p.call.id,
880
+ toolId: p.toolId,
881
+ action: p.action,
882
+ label: p.label,
883
+ description: p.descriptor.summary,
884
+ reason: p.reason
885
+ };
886
+ const askedAt = Date.now();
887
+ const targetDigest = await readTargetDigest({
888
+ toolId: p.toolId,
889
+ action: p.action,
890
+ params: p.params,
891
+ dispatch,
892
+ catalog
893
+ });
894
+ const answer = readAnswer(trusted?.has(`${p.toolId}.${p.action}`) ? true : requestApproval ? await requestApproval({
895
+ id: p.call.id,
896
+ toolId: p.toolId,
897
+ action: p.action,
898
+ label: p.label,
899
+ description: p.descriptor.summary,
900
+ reason: p.reason,
901
+ params: p.params,
902
+ targetDigest
903
+ }) : false);
904
+ // The user's deliberation is not the agent's runtime.
905
+ deadline += Date.now() - askedAt;
906
+ const approved = answer.approved;
907
+ decided.set(p.call.id, approved);
908
+ if (answer.trust) trusted?.add(`${p.toolId}.${p.action}`);
909
+ if (answer.reason) declineReasons.set(p.call.id, answer.reason);
910
+ if (!approved) {
911
+ // One refusal ends the PLAN, not just the call. The remaining
912
+ // changes were proposed together and nothing has run yet, so
913
+ // carrying on with the rest would apply half of something the user
914
+ // has just turned down.
915
+ batchDeclined = true;
916
+ return;
917
+ }
918
+ }
919
+ }
213
920
  for (const call of calls) {
921
+ /**
922
+ * Stop was pressed while this batch was running.
923
+ *
924
+ * The abort signal used to be read only at the top of a ROUND, so a
925
+ * response carrying three calls ran all three after the tap — and the
926
+ * bubble then said "Stopped." over two writes the user believed it had
927
+ * prevented. Each call now checks before it starts. A dispatch already
928
+ * in flight is left to finish (nothing can pull a store write back out
929
+ * of an adapter) and keeps its receipt; everything after it is reported
930
+ * as not run, each as its own row, so the transcript says exactly which
931
+ * changes landed. A result is still pushed for every call — the provider
932
+ * requires one per tool_use or the next request is rejected.
933
+ */
934
+ if (signal?.aborted) {
935
+ // A `buoy_ui` card that was never shown is not a step; no row for it.
936
+ if (call.name !== UI_TOOL_NAME) {
937
+ for (const ev of stoppedStep(call)) yield ev;
938
+ }
939
+ results.push({
940
+ toolCallId: call.id,
941
+ content: "Not run — the user pressed Stop before this call started.",
942
+ isError: true
943
+ });
944
+ continue;
945
+ }
946
+
214
947
  // `buoy_ui`: show or ask. Validated like any action; an Ask block ends
215
948
  // the turn after this batch — the answer comes back as a user message.
216
949
  if (call.name === UI_TOOL_NAME) {
@@ -218,19 +951,35 @@ export async function* runAgentTurn(input) {
218
951
  if (!built.ok || !built.block) {
219
952
  results.push({
220
953
  toolCallId: call.id,
221
- content: `Invalid ${UI_TOOL_NAME} block:\n${built.errors.join("\n")}`,
954
+ /**
955
+ * The trailing sentence is the expensive half.
956
+ *
957
+ * A model that gets a block bounced rewrites its WHOLE answer on
958
+ * the retry, and one assistant bubble spans every round of a turn
959
+ * (see `breakBefore`), so the user reads the same paragraph again
960
+ * — measured at four near-identical closings from three schema
961
+ * misses in a row on one live turn. The block is wrong; the
962
+ * sentence under it was fine.
963
+ */
964
+ content: `Invalid ${UI_TOOL_NAME} block:\n${built.errors.join("\n")}\n\nResend ONLY the corrected block. Do not rewrite your answer — whatever you already said this turn has been shown to the user, and saying it again repeats it on their screen.`,
222
965
  isError: true
223
966
  });
224
967
  continue;
225
968
  }
226
969
  // Provenance for pictures, ENFORCED: an image URL the model did not
227
970
  // read from a tool result this turn does not render. Live or nothing.
971
+ // Dropping pictures SILENTLY is its own bug: the model then writes
972
+ // "here they are, with artwork" over a card that has none, and the
973
+ // user is told something untrue about their own screen. Every drop
974
+ // comes back in the tool result so the next sentence can be honest.
975
+ let droppedImages = 0;
228
976
  if (built.block.kind === "imageGrid") {
229
977
  const kept = built.block.images.filter(img => seenUrls.has(img.url));
978
+ droppedImages = built.block.images.length - kept.length;
230
979
  if (kept.length === 0) {
231
980
  results.push({
232
981
  toolCallId: call.id,
233
- content: "Rejected: every image URL must be one you read from a tool result THIS turn (images.list, a storage value, a response body). Never invent or remember URLs — read them first.",
982
+ content: URL_PROVENANCE_REJECTION,
234
983
  isError: true
235
984
  });
236
985
  continue;
@@ -241,12 +990,17 @@ export async function* runAgentTurn(input) {
241
990
  };
242
991
  }
243
992
  if (built.block.kind === "list") {
244
- built.block = {
245
- ...built.block,
246
- items: built.block.items.map(item => item.image && !seenUrls.has(item.image) ? {
993
+ const items = built.block.items.map(item => {
994
+ if (!item.image || seenUrls.has(item.image)) return item;
995
+ droppedImages += 1;
996
+ return {
247
997
  ...item,
248
998
  image: undefined
249
- } : item)
999
+ };
1000
+ });
1001
+ built.block = {
1002
+ ...built.block,
1003
+ items
250
1004
  };
251
1005
  }
252
1006
  yield {
@@ -254,6 +1008,7 @@ export async function* runAgentTurn(input) {
254
1008
  block: built.block,
255
1009
  forToolCallId: call.id
256
1010
  };
1011
+ if (built.block.kind === "actions" || built.block.kind === "suggestions") offeredTap = true;
257
1012
  if (isAskBlock(built.block)) {
258
1013
  awaiting = built.block.id;
259
1014
  results.push({
@@ -263,7 +1018,7 @@ export async function* runAgentTurn(input) {
263
1018
  } else {
264
1019
  results.push({
265
1020
  toolCallId: call.id,
266
- content: "Shown to the user. Don't repeat its contents in text."
1021
+ content: "Shown to the user. Don't repeat its contents in text." + (droppedImages > 0 ? droppedImageNote(droppedImages) : "")
267
1022
  });
268
1023
  }
269
1024
  continue;
@@ -271,12 +1026,40 @@ export async function* runAgentTurn(input) {
271
1026
  const toolId = fromProviderToolName(call.name, catalog);
272
1027
  const action = call.input.action;
273
1028
 
1029
+ /**
1030
+ * A call that never reached a tool at all.
1031
+ *
1032
+ * Same rule as a refusal: the model made a move, the move went nowhere,
1033
+ * and a transcript that hides it leaves the reader watching the agent
1034
+ * change its mind for no visible reason. Caught on device — a model
1035
+ * guessed three navigation tool names that do not exist, spent 41
1036
+ * seconds doing it, and the only trace was a sentence it chose to write.
1037
+ * `read` because nothing was touched.
1038
+ */
1039
+ const deadCall = (label, summary) => [{
1040
+ type: "tool-start",
1041
+ id: call.id,
1042
+ toolId: toolId ?? call.name,
1043
+ action: action ?? "",
1044
+ label,
1045
+ description: "",
1046
+ params: call.input.params ?? {},
1047
+ effect: "read"
1048
+ }, {
1049
+ type: "tool-end",
1050
+ id: call.id,
1051
+ ok: false,
1052
+ summary,
1053
+ durationMs: 0
1054
+ }];
1055
+
274
1056
  // Two different mistakes, kept apart on purpose. Merging them tells a
275
1057
  // model that forgot `action` that the TOOL does not exist, and it stops
276
1058
  // reaching for a tool that was fine — the same distinction
277
1059
  // `dispatchToolAction` draws between an unknown tool and an unknown
278
1060
  // action, for the same reason.
279
1061
  if (!toolId) {
1062
+ for (const ev of deadCall(call.name, "No such tool")) yield ev;
280
1063
  results.push({
281
1064
  toolCallId: call.id,
282
1065
  content: `There is no tool called "${call.name}". Use one of the tools you were given.`,
@@ -285,6 +1068,7 @@ export async function* runAgentTurn(input) {
285
1068
  continue;
286
1069
  }
287
1070
  if (!action) {
1071
+ for (const ev of deadCall(call.name, "No action given")) yield ev;
288
1072
  results.push({
289
1073
  toolCallId: call.id,
290
1074
  content: `"${call.name}" needs an "action" — it is required, and it names which of this tool's actions to run. See the action list in the tool's description.`,
@@ -298,6 +1082,7 @@ export async function* runAgentTurn(input) {
298
1082
  // has nowhere to go but guess again, and the next guess is no better
299
1083
  // informed than the last.
300
1084
  const known = catalog.find(t => t.toolId === toolId)?.actions.map(a => a.action).join(", ");
1085
+ for (const ev of deadCall(`${toolId}.${action}`, "No such action")) yield ev;
301
1086
  results.push({
302
1087
  toolCallId: call.id,
303
1088
  content: `"${action}" is not an action on ${toolId}.${known ? ` Valid actions: ${known}.` : ""}`,
@@ -306,7 +1091,11 @@ export async function* runAgentTurn(input) {
306
1091
  continue;
307
1092
  }
308
1093
  const raw = call.input.params ?? {};
309
- const params = applyDefaults(raw, descriptor.params);
1094
+ // Accept the parameter-name misses every model makes (queryKey for
1095
+ // queryHash, requestId for id, flat rule fields, …) before validation,
1096
+ // so a mechanical alias never costs a turn. See normalizeParams.ts.
1097
+ const normalized = normalizeParams(toolId, action, raw);
1098
+ const params = applyDefaults(normalized.params, descriptor.params);
310
1099
  const check = validateParams(params, descriptor.params);
311
1100
  if (!check.ok) {
312
1101
  results.push({
@@ -316,6 +1105,7 @@ export async function* runAgentTurn(input) {
316
1105
  });
317
1106
  continue;
318
1107
  }
1108
+ const aliasTrailer = aliasNote(normalized.notes);
319
1109
  const label = describeCall(toolId, descriptor, params);
320
1110
  const verdict = decide({
321
1111
  descriptor,
@@ -323,14 +1113,29 @@ export async function* runAgentTurn(input) {
323
1113
  policy,
324
1114
  isRelease
325
1115
  });
1116
+
1117
+ // A step the user can SEE, for a call that never ran. `tool-end` alone
1118
+ // updates a row that was never opened, so a refusal used to leave no
1119
+ // trace in the transcript at all — the model simply changed its mind
1120
+ // between one sentence and the next.
1121
+ const unrunStep = summary => [{
1122
+ type: "tool-start",
1123
+ id: call.id,
1124
+ toolId,
1125
+ action,
1126
+ label,
1127
+ description: descriptor.summary,
1128
+ params,
1129
+ effect: descriptor.effect
1130
+ }, {
1131
+ type: "tool-end",
1132
+ id: call.id,
1133
+ ok: false,
1134
+ summary,
1135
+ durationMs: 0
1136
+ }];
326
1137
  if (verdict.verdict === "refuse") {
327
- yield {
328
- type: "tool-end",
329
- id: call.id,
330
- ok: false,
331
- summary: "refused",
332
- durationMs: 0
333
- };
1138
+ for (const ev of unrunStep("refused")) yield ev;
334
1139
  results.push({
335
1140
  toolCallId: call.id,
336
1141
  content: verdict.reason,
@@ -338,33 +1143,93 @@ export async function* runAgentTurn(input) {
338
1143
  });
339
1144
  continue;
340
1145
  }
341
- if (verdict.verdict === "needs-approval") {
342
- yield {
343
- type: "approval-required",
344
- id: call.id,
345
- toolId,
346
- action,
347
- label,
348
- description: descriptor.summary,
349
- reason: verdict.reason
350
- };
351
- const approved = requestApproval ? await requestApproval({
352
- id: call.id,
353
- toolId,
354
- action,
355
- label,
356
- description: descriptor.summary,
357
- reason: verdict.reason
358
- }) : false;
1146
+
1147
+ /**
1148
+ * Nothing in this batch changes anything until every card in it has been
1149
+ * answered. See `runBarrier` — this is the line that makes a decline
1150
+ * mean "the plan does not happen" rather than "the rest of the plan does
1151
+ * not happen".
1152
+ */
1153
+ const gatedHere = verdict.verdict === "needs-approval";
1154
+ if (gatedInBatch && !barrierRun && (descriptor.effect !== "read" || gatedHere)) {
1155
+ yield* runBarrier();
1156
+ }
1157
+ if (batchDeclined && (descriptor.effect !== "read" || gatedHere)) {
1158
+ for (const ev of unrunStep("declined")) yield ev;
1159
+ results.push({
1160
+ toolCallId: call.id,
1161
+ content: decided.get(call.id) === false ? declinedResult(declineReasons.get(call.id)) : "Not run — the user declined another change in this same batch, so none of it was applied. Ask what they would prefer before proposing it again.",
1162
+ isError: true
1163
+ });
1164
+ continue;
1165
+ }
1166
+ if (gatedHere) {
1167
+ let approved;
1168
+ if (decided.has(call.id)) {
1169
+ // The barrier already put this card up, before the batch's first
1170
+ // mutation. Asking again would be the same question twice.
1171
+ approved = decided.get(call.id);
1172
+ } else if (trusted?.has(`${toolId}.${action}`)) {
1173
+ // Waived for this conversation — see RunTurnInput.trusted.
1174
+ approved = true;
1175
+ decided.set(call.id, true);
1176
+ } else {
1177
+ // The barrier could not resolve this call (see `planned`), so it was
1178
+ // never offered. Ask here, exactly as this always did — a gate that
1179
+ // fails closed by silently declining would be worse than one that
1180
+ // asks late.
1181
+ yield {
1182
+ type: "approval-required",
1183
+ id: call.id,
1184
+ toolId,
1185
+ action,
1186
+ label,
1187
+ description: descriptor.summary,
1188
+ reason: verdict.reason
1189
+ };
1190
+ const askedAt = Date.now();
1191
+ // What the write is about to replace, as of NOW. Only a restored
1192
+ // card ever reads it back — see readTargetDigest.
1193
+ const targetDigest = await readTargetDigest({
1194
+ toolId,
1195
+ action,
1196
+ params,
1197
+ dispatch,
1198
+ catalog
1199
+ });
1200
+ const answer = readAnswer(requestApproval ? await requestApproval({
1201
+ id: call.id,
1202
+ toolId,
1203
+ action,
1204
+ label,
1205
+ description: descriptor.summary,
1206
+ reason: verdict.reason,
1207
+ params,
1208
+ targetDigest
1209
+ }) : false);
1210
+ // The user's deliberation is not the agent's runtime. Give the clock
1211
+ // back before anything else can trip the deadline check.
1212
+ deadline += Date.now() - askedAt;
1213
+ approved = answer.approved;
1214
+ decided.set(call.id, approved);
1215
+ if (answer.trust) trusted?.add(`${toolId}.${action}`);
1216
+ if (answer.reason) declineReasons.set(call.id, answer.reason);
1217
+ }
359
1218
  if (!approved) {
1219
+ // The row is the whole point here. Without it the bubble read as a
1220
+ // contradiction — "Bumped the line from 5 to 9." straight into "I
1221
+ // left it as it was" — with nothing on screen to say a change had
1222
+ // been proposed and turned down.
1223
+ for (const ev of unrunStep("declined")) yield ev;
360
1224
  results.push({
361
1225
  toolCallId: call.id,
362
- content: "The user declined this change. Do not retry it; ask what they would prefer.",
1226
+ content: declinedResult(declineReasons.get(call.id)),
363
1227
  isError: true
364
1228
  });
365
1229
  continue;
366
1230
  }
367
1231
  }
1232
+ if (ACQUISITION_CALLS.has(`${toolId}.${action}`)) attemptedAcquisition = true;
368
1233
  yield {
369
1234
  type: "tool-start",
370
1235
  id: call.id,
@@ -381,7 +1246,50 @@ export async function* runAgentTurn(input) {
381
1246
  // exactly. This is the one thing that makes storage reversible here when
382
1247
  // the Scenarios engine has to treat it as permanent.
383
1248
  const before = await captureBefore(descriptor, toolId, params, dispatch);
384
- const stateBefore = await captureStateBefore(descriptor, toolId, params, dispatch);
1249
+ const stateBefore = await captureStoreState(descriptor, toolId, params, dispatch);
1250
+
1251
+ /**
1252
+ * THE FINAL GUARD. Everything above this line happened in the past.
1253
+ *
1254
+ * `decide()` ran before the approval card went up, and between there and
1255
+ * here are three awaits: the target read, the person deciding, and two
1256
+ * pre-reads. Each one yields the thread, and a person can do a lot in
1257
+ * that gap — press Stop, or flip the read-only toggle in the settings
1258
+ * sheet while the card is still on screen. Both were reproduced: the
1259
+ * write went ahead on the verdict it captured before they touched
1260
+ * anything, which is precisely the moment the product promises it will
1261
+ * not. `live` is mutated in place by the session (see `refreshLive`), so
1262
+ * asking again reads the toggles as they are NOW.
1263
+ *
1264
+ * Below this there is no `await` before `dispatch` is CALLED. That is
1265
+ * load-bearing: an await here would reopen the same gap one line lower.
1266
+ */
1267
+ if (signal?.aborted) {
1268
+ for (const ev of unrunStep("stopped")) yield ev;
1269
+ results.push({
1270
+ toolCallId: call.id,
1271
+ content: "Not run — the user pressed Stop before this call started.",
1272
+ isError: true
1273
+ });
1274
+ continue;
1275
+ }
1276
+ const stillAllowed = decide({
1277
+ descriptor,
1278
+ toolId,
1279
+ policy,
1280
+ isRelease
1281
+ });
1282
+ if (stillAllowed.verdict === "refuse") {
1283
+ // `needs-approval` is NOT re-refused: reaching here means the tap
1284
+ // already happened, and asking twice for one call is its own bug.
1285
+ for (const ev of unrunStep("refused")) yield ev;
1286
+ results.push({
1287
+ toolCallId: call.id,
1288
+ content: stillAllowed.reason,
1289
+ isError: true
1290
+ });
1291
+ continue;
1292
+ }
385
1293
  try {
386
1294
  // `getSnapshot` is reserved: it is not an adapter action, so it goes to
387
1295
  // the snapshot reader and is trimmed to the fields worth sending. See
@@ -389,34 +1297,69 @@ export async function* runAgentTurn(input) {
389
1297
  // ledger the model asks about lives HERE, not behind an adapter, so
390
1298
  // routing it through dispatch would only work on a device and would
391
1299
  // record the undo as a fresh effect.
392
- const result = action === SNAPSHOT_ACTION ? projectSnapshot(toolId, await requireSnapshot(readSnapshot, toolId), params) : toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch) : await dispatch(toolId, action, params);
1300
+ const result = action === SNAPSHOT_ACTION ? projectSnapshot(toolId, await requireSnapshot(readSnapshot, toolId), params) : toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch, evidence, params, procedures) : await (prefetch.get(call.id) ?? dispatch(toolId, action, params));
393
1301
  const cleaned = redact(stripSelfTraffic(result));
394
- if (descriptor.effect !== "read" && toolId !== "ask-buoy") {
395
- // A delete of something we created IS the undo, whoever asked for
396
- // it — mark the matching entry undone instead of leaving the bar
397
- // promising an undo against a rule that no longer exists.
398
- ledger.noteExternalRevert(toolId, action, params);
399
- const fx = effectFor(toolId, descriptor, params, result, before);
400
- if (fx) ledger.record(toolId, action, fx);
401
- }
402
- const encodedResult = encodeResult(cleaned);
1302
+ const refused = reportsFailure(result);
1303
+
1304
+ // Kept in full — after redaction, never before — so the model can go
1305
+ // back to it. Not a retrieve's own result (a retrieve of a retrieve is
1306
+ // a loop), and not a refusal (nothing to go back to).
1307
+ const text = fullText(cleaned);
1308
+ const ref = evidence && !refused && !(toolId === "ask-buoy" && action === "retrieve") ? evidence.stash({
1309
+ toolId,
1310
+ action,
1311
+ params,
1312
+ capturedAt: Date.now(),
1313
+ text
1314
+ }) : undefined;
1315
+ const encodedResult = encodeResult(text, ref);
1316
+ const traceResult = encodedResult.length > MAX_TRACE_RESULT_CHARS ? `${encodedResult.slice(0, MAX_TRACE_RESULT_CHARS)}…` : encodedResult;
403
1317
  yield {
404
1318
  type: "tool-end",
405
1319
  id: call.id,
406
- ok: true,
1320
+ ok: !refused,
407
1321
  summary: summarise(result),
408
1322
  durationMs: Date.now() - startedAt,
409
- result: encodedResult.length > MAX_TRACE_RESULT_CHARS ? `${encodedResult.slice(0, MAX_TRACE_RESULT_CHARS)}…` : encodedResult
1323
+ result: traceResult
410
1324
  };
411
1325
 
412
1326
  // The receipt: what the user sees of this result, built from the
413
1327
  // same redacted data the model gets. See blocks/receipts.ts.
414
- const receipt = projectReceipt({
1328
+ //
1329
+ // The store is read back a second time so the diff shows what the
1330
+ // write ACTUALLY did rather than what it asked for — the two differ on
1331
+ // every `path` edit and every keyed list edit, and the request-shaped
1332
+ // version reported untouched fields as deleted.
1333
+ const stateAfter = stateBefore === undefined || refused ? undefined : await captureStoreState(descriptor, toolId, params, dispatch);
1334
+ if (descriptor.effect !== "read" && toolId !== "ask-buoy" && !refused) {
1335
+ // A delete of something we created IS the undo, whoever asked for
1336
+ // it — mark the matching entry undone instead of leaving the bar
1337
+ // promising an undo against a rule that no longer exists.
1338
+ ledger.noteExternalRevert(toolId, action, params);
1339
+ // Recorded AFTER the read-back so a cache edit's entry can carry both
1340
+ // halves: the data to restore and a fingerprint of what the write
1341
+ // left behind, which is how undo knows whether the app has since
1342
+ // replaced it. See effectFor / ledger "query-write".
1343
+ const fx = effectFor(toolId, descriptor, params, result, before, {
1344
+ before: stateBefore,
1345
+ after: stateAfter
1346
+ });
1347
+ if (fx) {
1348
+ const entry = ledger.record(toolId, action, fx, ledgerGeneration);
1349
+ // Ties the change to its round, so the budget keeps that round
1350
+ // while the change is live. See LedgerEntry.callId.
1351
+ if (entry) entry.callId = call.id;
1352
+ }
1353
+ }
1354
+ // No receipt for a call that changed nothing: a "what changed" card
1355
+ // under a refusal is the same lie as the ledger entry, drawn bigger.
1356
+ const receipt = refused ? undefined : projectReceipt({
415
1357
  toolId,
416
1358
  action,
417
1359
  params,
418
1360
  result: cleaned,
419
1361
  before: stateBefore ?? before,
1362
+ after: stateAfter,
420
1363
  effect: descriptor.effect
421
1364
  });
422
1365
  if (receipt) yield {
@@ -424,10 +1367,36 @@ export async function* runAgentTurn(input) {
424
1367
  block: receipt,
425
1368
  forToolCallId: call.id
426
1369
  };
1370
+ /**
1371
+ * THE OUTCOME CHECK. A write's own `{ok:true}` says the adapter
1372
+ * accepted it; this reads the app back and says whether the requested
1373
+ * state is actually there. Only for actions that have a verifier, only
1374
+ * for writes that were not refused, and never itself a write. Its
1375
+ * verdict rides in the tool result as a trailer the model reads, and
1376
+ * on a second tool-end so the row says "verified" / "unverified" /
1377
+ * "check failed". See engine/verify.ts for the three meanings.
1378
+ */
1379
+ const verification = descriptor.effect !== "read" && !refused ? await verifyOutcome({
1380
+ toolId,
1381
+ action,
1382
+ params,
1383
+ result,
1384
+ dispatch,
1385
+ signal,
1386
+ after: stateAfter
1387
+ }) : undefined;
1388
+ if (verification) yield {
1389
+ type: "tool-verified",
1390
+ id: call.id,
1391
+ verification
1392
+ };
427
1393
  for (const url of encodedResult.match(URL_RE) ?? []) seenUrls.add(url);
428
1394
  results.push({
429
1395
  toolCallId: call.id,
430
- content: encodedResult + ownedKeyNote(toolId, descriptor, params, storeKeyOwners)
1396
+ content: encodedResult + ownedKeyNote(toolId, descriptor, params, storeKeyOwners) + aliasTrailer + (verification ? verificationTrailer(verification) : "") + (salvagedIds.has(call.id) ? SALVAGE_NOTE : ""),
1397
+ ...(ref ? {
1398
+ ref
1399
+ } : {})
431
1400
  });
432
1401
  if (toolId === "app" && (action === "reloadApp" || action === "reload")) {
433
1402
  realmWillDie = true;
@@ -448,6 +1417,20 @@ export async function* runAgentTurn(input) {
448
1417
  });
449
1418
  }
450
1419
  }
1420
+
1421
+ // Looping? Say so, once, on the LAST result of this round — so it arrives
1422
+ // attached to the thing being repeated rather than as a free-floating
1423
+ // user turn, and the model reads it before deciding what to do next.
1424
+ if (calls.length) {
1425
+ recentSignatures.push(callSignature(calls));
1426
+ if (recentSignatures.length > LOOP_REPEATS) recentSignatures.shift();
1427
+ const looping = !loopWarned && recentSignatures.length === LOOP_REPEATS && recentSignatures.every(sig => sig === recentSignatures[0]);
1428
+ if (looping) {
1429
+ loopWarned = true;
1430
+ const last = results[results.length - 1];
1431
+ if (last) last.content = `${last.content}\n\n${LOOP_WARNING}`;
1432
+ }
1433
+ }
451
1434
  messages.push({
452
1435
  role: "tool-results",
453
1436
  results
@@ -458,7 +1441,8 @@ export async function* runAgentTurn(input) {
458
1441
  blockId: awaiting
459
1442
  };
460
1443
  yield {
461
- type: "done"
1444
+ type: "done",
1445
+ stopReason: "awaiting-user"
462
1446
  };
463
1447
  return messages;
464
1448
  }
@@ -466,7 +1450,8 @@ export async function* runAgentTurn(input) {
466
1450
  // Anything after this dies with the JS realm. Stop cleanly instead of
467
1451
  // sending a request whose answer can never arrive.
468
1452
  yield {
469
- type: "done"
1453
+ type: "done",
1454
+ stopReason: "realm-died"
470
1455
  };
471
1456
  return messages;
472
1457
  }
@@ -486,7 +1471,8 @@ export async function* runAgentTurn(input) {
486
1471
  }
487
1472
  };
488
1473
  yield {
489
- type: "done"
1474
+ type: "done",
1475
+ stopReason: "step-cap"
490
1476
  };
491
1477
  return messages;
492
1478
  }
@@ -512,6 +1498,29 @@ function ownedKeyNote(toolId, descriptor, params, storeKeyOwners) {
512
1498
  if (!storeName) return "";
513
1499
  return `\n\n[Buoy] This wrote to disk, but "${params.key}" is the saved copy of the live "${storeName}" store, so nothing on screen has changed — the app read that key once at startup and has held the state in memory since. Call zustand.rehydrate({"storeName":"${storeName}"}) to make the app pick this up, or use zustand.setState next time to change the store directly. If you meant to set the value for the app's next launch, this is already done and no further action is needed.`;
514
1500
  }
1501
+ export { digestOf };
1502
+
1503
+ /**
1504
+ * Fingerprint the value a write is about to replace.
1505
+ *
1506
+ * Built for the approval card that outlives its turn: the app crashed with
1507
+ * "Set bag line L-01 qty to 14" still asking, relaunched, and the card came
1508
+ * back — and Allow wrote qty 14 against whatever the bag was NOW, maybe a
1509
+ * different account, maybe a line that no longer existed. A hash of the
1510
+ * params alone would not catch that (same label, same params, different
1511
+ * world). So the card carries a digest of the TARGET at ask time, and a
1512
+ * restored Allow re-reads and compares before it writes. Undefined when the
1513
+ * target cannot be read — then the card can only warn, not check.
1514
+ */
1515
+ export async function readTargetDigest(input) {
1516
+ const descriptor = findDescriptor(input.catalog ?? CATALOG, input.toolId, input.action);
1517
+ if (!descriptor || descriptor.effect === "read") return undefined;
1518
+ const store = await captureStoreState(descriptor, input.toolId, input.params, input.dispatch);
1519
+ if (store !== undefined) return digestOf(store);
1520
+ const before = await captureBefore(descriptor, input.toolId, input.params, input.dispatch);
1521
+ if (before !== undefined) return digestOf(before);
1522
+ return undefined;
1523
+ }
515
1524
  /**
516
1525
  * One human-initiated action through EVERY gate the model's calls go through:
517
1526
  * catalog lookup, param validation, policy, pre-read, effect ledger.
@@ -544,7 +1553,8 @@ export async function runGatedAction(input) {
544
1553
  error: `${toolId}.${action} is not in Buoy's catalog.`
545
1554
  };
546
1555
  }
547
- const params = applyDefaults(input.params ?? {}, descriptor.params);
1556
+ const normalized = normalizeParams(toolId, action, input.params ?? {});
1557
+ const params = applyDefaults(normalized.params, descriptor.params);
548
1558
  const check = validateParams(params, descriptor.params);
549
1559
  if (!check.ok) {
550
1560
  return {
@@ -565,11 +1575,51 @@ export async function runGatedAction(input) {
565
1575
  };
566
1576
  }
567
1577
  const before = await captureBefore(descriptor, toolId, params, dispatch);
1578
+ /**
1579
+ * The same read-back the model's path takes. It was missing here, and the
1580
+ * gap was invisible: a `setQueryData` through a block button or a restored
1581
+ * approval card recorded a ledger entry with no prior value, so Undo had
1582
+ * nothing to put back and the concurrent-writer check had no fingerprint to
1583
+ * compare. The bar said "1 undoable" over a change that could not be undone.
1584
+ */
1585
+ const stateBefore = await captureStoreState(descriptor, toolId, params, dispatch);
1586
+
1587
+ // Asked again after the pre-reads, for the same reason the model's path asks
1588
+ // again: those are awaits, and read-only can be switched on inside one.
1589
+ const stillAllowed = decide({
1590
+ descriptor,
1591
+ toolId,
1592
+ policy,
1593
+ isRelease
1594
+ });
1595
+ if (stillAllowed.verdict === "refuse") {
1596
+ return {
1597
+ ok: false,
1598
+ error: stillAllowed.reason
1599
+ };
1600
+ }
568
1601
  try {
569
- const result = toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch) : await dispatch(toolId, action, params);
1602
+ const result = toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch, undefined, params) : await dispatch(toolId, action, params);
1603
+
1604
+ // Same rule as the model's path: a resolved `{ok:false}` is a refusal, not
1605
+ // a change. Without this a block button (or an approval answered after a
1606
+ // reload) put a phantom entry in the undo bar too.
1607
+ if (reportsFailure(result)) {
1608
+ const err = result.error;
1609
+ return {
1610
+ ok: false,
1611
+ error: typeof err === "string" ? err : "The tool refused this."
1612
+ };
1613
+ }
570
1614
  if (descriptor.effect !== "read" && toolId !== "ask-buoy") {
571
1615
  ledger.noteExternalRevert(toolId, action, params);
572
- const fx = effectFor(toolId, descriptor, params, result, before);
1616
+ // Read back AFTER the write, exactly as the model's path does — the two
1617
+ // halves together are what make a cache edit undoable.
1618
+ const stateAfter = await captureStoreState(descriptor, toolId, params, dispatch);
1619
+ const fx = effectFor(toolId, descriptor, params, result, before, {
1620
+ before: stateBefore,
1621
+ after: stateAfter
1622
+ });
573
1623
  if (fx) ledger.record(toolId, action, fx);
574
1624
  }
575
1625
  return {
@@ -591,7 +1641,33 @@ export async function runGatedAction(input) {
591
1641
  * adapter's same-named actions remain for desktop/MCP, where the ledger is
592
1642
  * reached over the broker instead.
593
1643
  */
594
- async function runAskBuoyAction(action, ledger, dispatch) {
1644
+ async function runAskBuoyAction(action, ledger, dispatch, evidence, params, procedures) {
1645
+ if (action === "retrieve") {
1646
+ return retrieveEvidence(evidence, params ?? {});
1647
+ }
1648
+ if (action === "openProcedure") {
1649
+ const id = typeof params?.id === "string" ? params.id : "";
1650
+ const found = procedures?.find(p => p.id === id);
1651
+ if (!found) {
1652
+ const ids = (procedures ?? []).map(p => p.id);
1653
+ return {
1654
+ ok: false,
1655
+ error: ids.length ? `No procedure "${id}". This app's procedures: ${ids.join(", ")}.` : "This app has no procedures written for it."
1656
+ };
1657
+ }
1658
+ return {
1659
+ id: found.id,
1660
+ title: found.title,
1661
+ ...(found.version ? {
1662
+ version: found.version
1663
+ } : {}),
1664
+ ...(found.requires?.length ? {
1665
+ requires: found.requires
1666
+ } : {}),
1667
+ body: found.body,
1668
+ note: "Follow these steps with the ordinary tools. Every write still goes through the same policy, approval and undo as any other call; a step that is refused stays refused."
1669
+ };
1670
+ }
595
1671
  if (action === "listChanges") {
596
1672
  const changes = ledger.list().filter(e => !e.undoneAt).map(e => ({
597
1673
  toolId: e.toolId,
@@ -641,16 +1717,46 @@ async function requireSnapshot(readSnapshot, toolId) {
641
1717
  * Zustand only for now: `getStoreState` is cheap and the write params carry
642
1718
  * the after-state, so the diff is exact without the store's cooperation.
643
1719
  */
644
- async function captureStateBefore(descriptor, toolId, params, dispatch) {
645
- if (toolId !== "zustand" || descriptor.action !== "setState") return undefined;
646
- if (typeof params.storeName !== "string") return undefined;
647
- try {
648
- return await dispatch(toolId, "getStoreState", {
649
- storeName: params.storeName
650
- });
651
- } catch {
652
- return undefined;
1720
+ /**
1721
+ * Read a zustand store whole, for the before/after halves of a write receipt.
1722
+ * Called on both sides of the dispatch — see the receipt above.
1723
+ */
1724
+ async function captureStoreState(descriptor, toolId, params, dispatch) {
1725
+ if (toolId === "zustand" && descriptor.action === "setState") {
1726
+ if (typeof params.storeName !== "string") return undefined;
1727
+ try {
1728
+ return await dispatch(toolId, "getStoreState", {
1729
+ storeName: params.storeName
1730
+ });
1731
+ } catch {
1732
+ return undefined;
1733
+ }
1734
+ }
1735
+ /**
1736
+ * The same read-both-sides trick for the CACHE.
1737
+ *
1738
+ * A zustand edit produced a "what changed" card and the identical edit to a
1739
+ * query produced nothing — for the write the prompt calls "THE way to change
1740
+ * what a server-backed screen shows". The QA tester's confirmation that the
1741
+ * right field moved was missing from the most-used tool in the product.
1742
+ */
1743
+ if (toolId === "query" && descriptor.action === "setQueryData") {
1744
+ // hashQueryKey, not JSON.stringify: React Query SORTS object keys when it
1745
+ // hashes, so a key carrying `{query:"",category:null}` hashes as
1746
+ // `{"category":null,"query":""}`. Stringifying in insertion order produced
1747
+ // a hash that matched nothing, the read came back empty, and the card
1748
+ // silently did not draw.
1749
+ const queryHash = typeof params.queryHash === "string" ? params.queryHash : Array.isArray(params.queryKey) ? hashQueryKey(params.queryKey) : undefined;
1750
+ if (!queryHash) return undefined;
1751
+ try {
1752
+ return await dispatch(toolId, "getQueryData", {
1753
+ queryHash
1754
+ });
1755
+ } catch {
1756
+ return undefined;
1757
+ }
653
1758
  }
1759
+ return undefined;
654
1760
  }
655
1761
  async function captureBefore(descriptor, toolId, params, dispatch) {
656
1762
  if (descriptor.effect === "read") return undefined;