@buoy-gg/agent-core 7.0.35 → 7.0.38
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +21 -14
- package/lib/commonjs/blocks/receipts.js +57 -1
- package/lib/commonjs/blocks/receipts.js.map +1 -1
- package/lib/commonjs/blocks/types.js +23 -2
- package/lib/commonjs/blocks/types.js.map +1 -1
- package/lib/commonjs/blocks/uiTool.js +423 -10
- package/lib/commonjs/blocks/uiTool.js.map +1 -1
- package/lib/commonjs/catalog/catalog.g.js +213 -48
- package/lib/commonjs/catalog/catalog.g.js.map +1 -1
- package/lib/commonjs/catalog/catalog.source.json +226 -48
- package/lib/commonjs/catalog/catalog.types.g.js +44 -0
- package/lib/commonjs/catalog/catalog.types.g.js.map +1 -0
- package/lib/commonjs/catalog/normalizeParams.js +333 -0
- package/lib/commonjs/catalog/normalizeParams.js.map +1 -0
- package/lib/commonjs/catalog/signature.js +68 -0
- package/lib/commonjs/catalog/signature.js.map +1 -0
- package/lib/commonjs/catalog/snapshotReads.js +13 -6
- package/lib/commonjs/catalog/snapshotReads.js.map +1 -1
- package/lib/commonjs/catalog/toProviderTools.js +44 -6
- package/lib/commonjs/catalog/toProviderTools.js.map +1 -1
- package/lib/commonjs/context/buildContextPack.js +367 -19
- package/lib/commonjs/context/buildContextPack.js.map +1 -1
- package/lib/commonjs/effects/digest.js +34 -0
- package/lib/commonjs/effects/digest.js.map +1 -0
- package/lib/commonjs/effects/ledger.js +98 -2
- package/lib/commonjs/effects/ledger.js.map +1 -1
- package/lib/commonjs/engine/askGate.js +287 -0
- package/lib/commonjs/engine/askGate.js.map +1 -0
- package/lib/commonjs/engine/effectFor.js +82 -1
- package/lib/commonjs/engine/effectFor.js.map +1 -1
- package/lib/commonjs/engine/evidence.js +113 -0
- package/lib/commonjs/engine/evidence.js.map +1 -0
- package/lib/commonjs/engine/historyBudget.js +363 -0
- package/lib/commonjs/engine/historyBudget.js.map +1 -0
- package/lib/commonjs/engine/retrieve.js +214 -0
- package/lib/commonjs/engine/retrieve.js.map +1 -0
- package/lib/commonjs/engine/runAgentTurn.js +1231 -120
- package/lib/commonjs/engine/runAgentTurn.js.map +1 -1
- package/lib/commonjs/engine/systemPrompt.js +103 -9
- package/lib/commonjs/engine/systemPrompt.js.map +1 -1
- package/lib/commonjs/engine/textToolCalls.js +276 -0
- package/lib/commonjs/engine/textToolCalls.js.map +1 -0
- package/lib/commonjs/engine/tokenCalibration.js +81 -0
- package/lib/commonjs/engine/tokenCalibration.js.map +1 -0
- package/lib/commonjs/engine/verify.js +300 -0
- package/lib/commonjs/engine/verify.js.map +1 -0
- package/lib/commonjs/index.js +74 -0
- package/lib/commonjs/index.js.map +1 -1
- package/lib/commonjs/policy/labels.js +19 -3
- package/lib/commonjs/policy/labels.js.map +1 -1
- package/lib/commonjs/policy/policy.js +6 -1
- package/lib/commonjs/policy/policy.js.map +1 -1
- package/lib/commonjs/policy/redact.js +42 -5
- package/lib/commonjs/policy/redact.js.map +1 -1
- package/lib/commonjs/providers/anthropic.js +340 -208
- package/lib/commonjs/providers/anthropic.js.map +1 -1
- package/lib/commonjs/providers/openai.js +237 -103
- package/lib/commonjs/providers/openai.js.map +1 -1
- package/lib/commonjs/providers/problem.js +98 -0
- package/lib/commonjs/providers/problem.js.map +1 -0
- package/lib/commonjs/providers/sse.js +170 -34
- package/lib/commonjs/providers/sse.js.map +1 -1
- package/lib/commonjs/providers/streamTimer.js +123 -0
- package/lib/commonjs/providers/streamTimer.js.map +1 -0
- package/lib/commonjs/providers/transport.js +123 -0
- package/lib/commonjs/providers/transport.js.map +1 -0
- package/lib/commonjs/providers/xhrStream.js +212 -0
- package/lib/commonjs/providers/xhrStream.js.map +1 -0
- package/lib/commonjs/session.js +242 -58
- package/lib/commonjs/session.js.map +1 -1
- package/lib/module/blocks/receipts.js +57 -1
- package/lib/module/blocks/receipts.js.map +1 -1
- package/lib/module/blocks/types.js +23 -2
- package/lib/module/blocks/types.js.map +1 -1
- package/lib/module/blocks/uiTool.js +421 -10
- package/lib/module/blocks/uiTool.js.map +1 -1
- package/lib/module/catalog/catalog.g.js +213 -48
- package/lib/module/catalog/catalog.g.js.map +1 -1
- package/lib/module/catalog/catalog.source.json +226 -48
- package/lib/module/catalog/catalog.types.g.js +40 -0
- package/lib/module/catalog/catalog.types.g.js.map +1 -0
- package/lib/module/catalog/normalizeParams.js +327 -0
- package/lib/module/catalog/normalizeParams.js.map +1 -0
- package/lib/module/catalog/signature.js +63 -0
- package/lib/module/catalog/signature.js.map +1 -0
- package/lib/module/catalog/snapshotReads.js +13 -6
- package/lib/module/catalog/snapshotReads.js.map +1 -1
- package/lib/module/catalog/toProviderTools.js +44 -6
- package/lib/module/catalog/toProviderTools.js.map +1 -1
- package/lib/module/context/buildContextPack.js +366 -19
- package/lib/module/context/buildContextPack.js.map +1 -1
- package/lib/module/effects/digest.js +29 -0
- package/lib/module/effects/digest.js.map +1 -0
- package/lib/module/effects/ledger.js +98 -2
- package/lib/module/effects/ledger.js.map +1 -1
- package/lib/module/engine/askGate.js +280 -0
- package/lib/module/engine/askGate.js.map +1 -0
- package/lib/module/engine/effectFor.js +83 -1
- package/lib/module/engine/effectFor.js.map +1 -1
- package/lib/module/engine/evidence.js +107 -0
- package/lib/module/engine/evidence.js.map +1 -0
- package/lib/module/engine/historyBudget.js +355 -0
- package/lib/module/engine/historyBudget.js.map +1 -0
- package/lib/module/engine/retrieve.js +210 -0
- package/lib/module/engine/retrieve.js.map +1 -0
- package/lib/module/engine/runAgentTurn.js +1227 -121
- package/lib/module/engine/runAgentTurn.js.map +1 -1
- package/lib/module/engine/systemPrompt.js +101 -9
- package/lib/module/engine/systemPrompt.js.map +1 -1
- package/lib/module/engine/textToolCalls.js +270 -0
- package/lib/module/engine/textToolCalls.js.map +1 -0
- package/lib/module/engine/tokenCalibration.js +75 -0
- package/lib/module/engine/tokenCalibration.js.map +1 -0
- package/lib/module/engine/verify.js +292 -0
- package/lib/module/engine/verify.js.map +1 -0
- package/lib/module/index.js +7 -3
- package/lib/module/index.js.map +1 -1
- package/lib/module/policy/labels.js +19 -3
- package/lib/module/policy/labels.js.map +1 -1
- package/lib/module/policy/policy.js +6 -1
- package/lib/module/policy/policy.js.map +1 -1
- package/lib/module/policy/redact.js +42 -5
- package/lib/module/policy/redact.js.map +1 -1
- package/lib/module/providers/anthropic.js +341 -209
- package/lib/module/providers/anthropic.js.map +1 -1
- package/lib/module/providers/openai.js +238 -104
- package/lib/module/providers/openai.js.map +1 -1
- package/lib/module/providers/problem.js +92 -0
- package/lib/module/providers/problem.js.map +1 -0
- package/lib/module/providers/sse.js +166 -34
- package/lib/module/providers/sse.js.map +1 -1
- package/lib/module/providers/streamTimer.js +118 -0
- package/lib/module/providers/streamTimer.js.map +1 -0
- package/lib/module/providers/transport.js +119 -0
- package/lib/module/providers/transport.js.map +1 -0
- package/lib/module/providers/xhrStream.js +207 -0
- package/lib/module/providers/xhrStream.js.map +1 -0
- package/lib/module/session.js +226 -60
- package/lib/module/session.js.map +1 -1
- package/lib/typescript/blocks/receipts.d.ts +5 -0
- package/lib/typescript/blocks/receipts.d.ts.map +1 -1
- package/lib/typescript/blocks/types.d.ts +23 -2
- package/lib/typescript/blocks/types.d.ts.map +1 -1
- package/lib/typescript/blocks/uiTool.d.ts +11 -2
- package/lib/typescript/blocks/uiTool.d.ts.map +1 -1
- package/lib/typescript/catalog/catalog.g.d.ts +2 -2
- package/lib/typescript/catalog/catalog.g.d.ts.map +1 -1
- package/lib/typescript/catalog/catalog.types.g.d.ts +40 -0
- package/lib/typescript/catalog/catalog.types.g.d.ts.map +1 -0
- package/lib/typescript/catalog/normalizeParams.d.ts +43 -0
- package/lib/typescript/catalog/normalizeParams.d.ts.map +1 -0
- package/lib/typescript/catalog/signature.d.ts +29 -0
- package/lib/typescript/catalog/signature.d.ts.map +1 -0
- package/lib/typescript/catalog/snapshotReads.d.ts.map +1 -1
- package/lib/typescript/catalog/toProviderTools.d.ts +28 -15
- package/lib/typescript/catalog/toProviderTools.d.ts.map +1 -1
- package/lib/typescript/context/buildContextPack.d.ts +24 -2
- package/lib/typescript/context/buildContextPack.d.ts.map +1 -1
- package/lib/typescript/effects/digest.d.ts +10 -0
- package/lib/typescript/effects/digest.d.ts.map +1 -0
- package/lib/typescript/effects/ledger.d.ts +73 -0
- package/lib/typescript/effects/ledger.d.ts.map +1 -1
- package/lib/typescript/engine/askGate.d.ts +29 -0
- package/lib/typescript/engine/askGate.d.ts.map +1 -0
- package/lib/typescript/engine/effectFor.d.ts +6 -1
- package/lib/typescript/engine/effectFor.d.ts.map +1 -1
- package/lib/typescript/engine/evidence.d.ts +67 -0
- package/lib/typescript/engine/evidence.d.ts.map +1 -0
- package/lib/typescript/engine/historyBudget.d.ts +115 -0
- package/lib/typescript/engine/historyBudget.d.ts.map +1 -0
- package/lib/typescript/engine/retrieve.d.ts +32 -0
- package/lib/typescript/engine/retrieve.d.ts.map +1 -0
- package/lib/typescript/engine/runAgentTurn.d.ts +179 -3
- package/lib/typescript/engine/runAgentTurn.d.ts.map +1 -1
- package/lib/typescript/engine/systemPrompt.d.ts +80 -0
- package/lib/typescript/engine/systemPrompt.d.ts.map +1 -1
- package/lib/typescript/engine/textToolCalls.d.ts +58 -0
- package/lib/typescript/engine/textToolCalls.d.ts.map +1 -0
- package/lib/typescript/engine/tokenCalibration.d.ts +46 -0
- package/lib/typescript/engine/tokenCalibration.d.ts.map +1 -0
- package/lib/typescript/engine/verify.d.ts +67 -0
- package/lib/typescript/engine/verify.d.ts.map +1 -0
- package/lib/typescript/index.d.ts +7 -3
- package/lib/typescript/index.d.ts.map +1 -1
- package/lib/typescript/policy/labels.d.ts.map +1 -1
- package/lib/typescript/policy/policy.d.ts.map +1 -1
- package/lib/typescript/policy/redact.d.ts.map +1 -1
- package/lib/typescript/providers/anthropic.d.ts.map +1 -1
- package/lib/typescript/providers/openai.d.ts +0 -16
- package/lib/typescript/providers/openai.d.ts.map +1 -1
- package/lib/typescript/providers/problem.d.ts +50 -0
- package/lib/typescript/providers/problem.d.ts.map +1 -0
- package/lib/typescript/providers/sse.d.ts +86 -2
- package/lib/typescript/providers/sse.d.ts.map +1 -1
- package/lib/typescript/providers/streamTimer.d.ts +59 -0
- package/lib/typescript/providers/streamTimer.d.ts.map +1 -0
- package/lib/typescript/providers/transport.d.ts +48 -0
- package/lib/typescript/providers/transport.d.ts.map +1 -0
- package/lib/typescript/providers/types.d.ts +105 -3
- package/lib/typescript/providers/types.d.ts.map +1 -1
- package/lib/typescript/providers/xhrStream.d.ts +45 -0
- package/lib/typescript/providers/xhrStream.d.ts.map +1 -0
- package/lib/typescript/session.d.ts +55 -4
- package/lib/typescript/session.d.ts.map +1 -1
- package/lib/typescript/types.d.ts +6 -0
- package/lib/typescript/types.d.ts.map +1 -1
- package/package.json +10 -10
- package/LICENSE +0 -58
|
@@ -21,16 +21,25 @@
|
|
|
21
21
|
|
|
22
22
|
import { CATALOG } from "../catalog/catalog.g";
|
|
23
23
|
import { applyDefaults, validateParams } from "../catalog/validateParams";
|
|
24
|
+
import { normalizeParams, aliasNote } from "../catalog/normalizeParams";
|
|
24
25
|
import { fromProviderToolName, toProviderTools } from "../catalog/toProviderTools";
|
|
26
|
+
import { hashQueryKey } from "../catalog/normalizeParams";
|
|
25
27
|
import { projectSnapshot, SNAPSHOT_ACTION } from "../catalog/snapshotReads";
|
|
26
28
|
import { isReversible } from "../effects/ledger";
|
|
27
29
|
import { describeCall, decide } from "../policy/policy";
|
|
28
30
|
import { redact, stripSelfTraffic } from "../policy/redact";
|
|
29
31
|
import { effectFor } from "./effectFor";
|
|
32
|
+
import { digestOf } from "../effects/digest";
|
|
30
33
|
import { buildUiBlock, UI_TOOL_NAME, uiProviderTool } from "../blocks/uiTool";
|
|
31
34
|
import { projectReceipt } from "../blocks/receipts";
|
|
32
35
|
import { isAskBlock } from "../blocks/types";
|
|
33
|
-
|
|
36
|
+
import { ACQUISITION_CALLS, synthesizeAskBlock, synthesizeShortfallBlock } from "./askGate";
|
|
37
|
+
import { couldBeToolCallEnvelope, parseTextToolCalls, SALVAGE_NOTE } from "./textToolCalls";
|
|
38
|
+
import { budgetForRequest, MAX_HISTORY_TOKENS, messageSize } from "./historyBudget";
|
|
39
|
+
import { promptTokensOf, TokenCalibration } from "./tokenCalibration";
|
|
40
|
+
import { truncationMarker } from "./evidence";
|
|
41
|
+
import { retrieveEvidence } from "./retrieve";
|
|
42
|
+
import { verificationTrailer, verifyOutcome } from "./verify";
|
|
34
43
|
/** Beyond this a tool result costs more in tokens than it can possibly be worth. */
|
|
35
44
|
const MAX_RESULT_CHARS = 24_000;
|
|
36
45
|
/**
|
|
@@ -39,20 +48,127 @@ const MAX_RESULT_CHARS = 24_000;
|
|
|
39
48
|
* store dump per step makes the trace unreadable rather than more useful.
|
|
40
49
|
*/
|
|
41
50
|
const MAX_TRACE_RESULT_CHARS = 2_000;
|
|
51
|
+
/**
|
|
52
|
+
* How many identical rounds before the model is told it is looping.
|
|
53
|
+
*
|
|
54
|
+
* `maxSteps` alone was the whole answer, and it is the wrong shape: a model
|
|
55
|
+
* that misread a result retries the same call until the cap, then the turn
|
|
56
|
+
* ends with nothing — the user watches twelve identical rows go by and gets
|
|
57
|
+
* no answer. The cap is a backstop, not feedback. Telling the model what it
|
|
58
|
+
* is doing gives it the chance to change approach, and costs one sentence.
|
|
59
|
+
*
|
|
60
|
+
* Signature is name + action + arguments, key-sorted so `{b,a}` and `{a,b}`
|
|
61
|
+
* are the same call. Three, not two: a legitimate retry after a transient
|
|
62
|
+
* failure is normal, and warning on it would be noise.
|
|
63
|
+
*/
|
|
64
|
+
const LOOP_REPEATS = 3;
|
|
65
|
+
function callSignature(calls) {
|
|
66
|
+
const sort = v => {
|
|
67
|
+
if (v === null || typeof v !== "object") return v;
|
|
68
|
+
if (Array.isArray(v)) return v.map(sort);
|
|
69
|
+
const o = v;
|
|
70
|
+
return Object.keys(o).sort().reduce((a, k) => (a[k] = sort(o[k]), a), {});
|
|
71
|
+
};
|
|
72
|
+
return calls.map(c => `${c.name}:${JSON.stringify(sort(c.input ?? {}))}`).join("|");
|
|
73
|
+
}
|
|
74
|
+
const LOOP_WARNING = "[Buoy] You have now made this exact call three times in a row and gotten the same thing back. Repeating it again will not change the answer. Read the result you already have, then either try a different tool or different arguments, or tell the user plainly what you could not get and what you tried.";
|
|
42
75
|
const DEFAULT_MAX_STEPS = 12;
|
|
43
76
|
const DEFAULT_TURN_MS = 180_000;
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* Retrying a model request — narrowly.
|
|
80
|
+
*
|
|
81
|
+
* A 429 from a shared gateway, a 529 / `overloaded_error` from Anthropic, or
|
|
82
|
+
* a "prompt is too long" 400 used to end the turn with an error card and Try
|
|
83
|
+
* again — and Try again re-sends the QUESTION, restarting an investigation
|
|
84
|
+
* that may already have written things. All three fail before a model has
|
|
85
|
+
* generated anything, so asking again is safe: nothing runs twice, nothing
|
|
86
|
+
* is billed twice. The rule that makes it safe is enforced, not assumed —
|
|
87
|
+
* a request is only retried when its stream produced NO text, tool call or
|
|
88
|
+
* thinking; once anything has come back, the existing no-replay handling
|
|
89
|
+
* stands (see providers/transport.ts on consumed-exactly-once).
|
|
90
|
+
*
|
|
91
|
+
* Throttled: up to two retries, 1 s then 2 s (plus jitter), a `Retry-After`
|
|
92
|
+
* winning when the endpoint sent one. Overflow: one retry, with the history
|
|
93
|
+
* ceiling halved first — the proactive budget is chars ÷ 4 and the provider
|
|
94
|
+
* just said that was optimistic. Both wait inside the turn deadline and stop
|
|
95
|
+
* on the user's Stop. Small numbers on purpose: this is a phone, not a
|
|
96
|
+
* server queue.
|
|
97
|
+
*/
|
|
98
|
+
const THROTTLE_RETRIES = 2;
|
|
99
|
+
const OVERFLOW_RETRIES = 1;
|
|
100
|
+
const RETRY_BASE_MS = 1_000;
|
|
101
|
+
const RETRY_JITTER_MS = 250;
|
|
102
|
+
const RETRY_AFTER_CAP_MS = 30_000;
|
|
103
|
+
function sleep(ms, signal) {
|
|
104
|
+
return new Promise(resolve => {
|
|
105
|
+
if (signal?.aborted) return resolve();
|
|
106
|
+
const t = setTimeout(done, ms);
|
|
107
|
+
function done() {
|
|
108
|
+
clearTimeout(t);
|
|
109
|
+
signal?.removeEventListener("abort", done);
|
|
110
|
+
resolve();
|
|
111
|
+
}
|
|
112
|
+
signal?.addEventListener("abort", done, {
|
|
113
|
+
once: true
|
|
114
|
+
});
|
|
115
|
+
});
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/**
|
|
119
|
+
* Why the turn ended — the machine-readable half of the notices. The text in
|
|
120
|
+
* the notice blocks is unchanged (the eval classifier pins it); this is for a
|
|
121
|
+
* host, the bank and the desktop, which used to have to grep the prose.
|
|
122
|
+
*/
|
|
123
|
+
|
|
124
|
+
/** Normalise an answer to its parts, so both call sites read one shape. */
|
|
125
|
+
function readAnswer(answer) {
|
|
126
|
+
if (typeof answer === "boolean") return {
|
|
127
|
+
approved: answer,
|
|
128
|
+
trust: false
|
|
129
|
+
};
|
|
130
|
+
const reason = answer.reason?.trim();
|
|
131
|
+
return {
|
|
132
|
+
approved: answer.approved,
|
|
133
|
+
...(reason ? {
|
|
134
|
+
reason
|
|
135
|
+
} : {}),
|
|
136
|
+
trust: answer.trust === true && answer.approved
|
|
137
|
+
};
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/** What the model is told when the user says no — with their note, when they left one. */
|
|
141
|
+
function declinedResult(reason) {
|
|
142
|
+
return reason ? `The user declined this change and said: "${reason}". Do not retry it as proposed; take their note into account and, if a different change would fit, propose that instead.` : "The user declined this change. Do not retry it; ask what they would prefer.";
|
|
143
|
+
}
|
|
44
144
|
function findDescriptor(catalog, toolId, action) {
|
|
45
145
|
return catalog.find(t => t.toolId === toolId)?.actions.find(a => a.action === action);
|
|
46
146
|
}
|
|
47
|
-
|
|
48
|
-
|
|
147
|
+
|
|
148
|
+
/**
|
|
149
|
+
* Why a picture did not render, in the model's own terms.
|
|
150
|
+
*
|
|
151
|
+
* Two messages, not one, because they answer different questions: the first
|
|
152
|
+
* is "your block was refused", the second is "your block was shown but with
|
|
153
|
+
* holes in it". The literal opening of the rejection is pinned by the
|
|
154
|
+
* trajectory classifier (evals/ask-buoy/lib/trajectory.ts) — keep it.
|
|
155
|
+
*/
|
|
156
|
+
const URL_PROVENANCE_REJECTION = "Rejected: every image URL must be one you read from a tool result in THIS conversation (images.list, a storage value, a response body). Never invent or remember URLs — read them first, then show them.";
|
|
157
|
+
const droppedImageNote = n => ` NOTE: ${n} image ${n === 1 ? "URL was" : "URLs were"} dropped — they were not read from a tool result in this conversation, so the card the user is looking at has no picture there. Do not tell them it does; read the URLs and show it again, or say the artwork is missing.`;
|
|
158
|
+
|
|
159
|
+
/** The whole result as text — what the evidence store keeps. */
|
|
160
|
+
function fullText(value) {
|
|
49
161
|
try {
|
|
50
|
-
|
|
162
|
+
return JSON.stringify(value ?? null);
|
|
51
163
|
} catch {
|
|
52
|
-
|
|
164
|
+
return String(value);
|
|
53
165
|
}
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/** What the model is sent: the text, cut at the cap with a marker that names where the rest is. */
|
|
169
|
+
function encodeResult(text, ref) {
|
|
54
170
|
if (text.length <= MAX_RESULT_CHARS) return text;
|
|
55
|
-
return `${text.slice(0, MAX_RESULT_CHARS)}
|
|
171
|
+
return `${text.slice(0, MAX_RESULT_CHARS)}${truncationMarker(text.length, ref)}`;
|
|
56
172
|
}
|
|
57
173
|
|
|
58
174
|
/** A short line for the transcript row, so the UI never renders raw payloads. */
|
|
@@ -88,11 +204,30 @@ function summarise(value) {
|
|
|
88
204
|
function countOf(n) {
|
|
89
205
|
return n === 1 ? "1 item" : `${n} items`;
|
|
90
206
|
}
|
|
207
|
+
|
|
208
|
+
/**
|
|
209
|
+
* A call that came back reporting its own failure.
|
|
210
|
+
*
|
|
211
|
+
* Buoy adapters RESOLVE with `{ok:false, error}` rather than throwing, so the
|
|
212
|
+
* dispatch's try/catch never fires and a refused write looked like a completed
|
|
213
|
+
* one: the row read "Done", and — much worse — the effect ledger recorded a
|
|
214
|
+
* change that had not happened, so the changes bar offered to undo nothing.
|
|
215
|
+
* Caught on device: a typed-edit refusal ("qty is a number, 'many' is text")
|
|
216
|
+
* left the bag untouched and the bar claiming one permanent change.
|
|
217
|
+
*
|
|
218
|
+
* `ok === false` is the same signal `summarise` above already trusts to turn a
|
|
219
|
+
* result into the row's text, so this reads it no more liberally than the UI
|
|
220
|
+
* already does.
|
|
221
|
+
*/
|
|
222
|
+
function reportsFailure(value) {
|
|
223
|
+
return typeof value === "object" && value !== null && value.ok === false;
|
|
224
|
+
}
|
|
91
225
|
export async function* runAgentTurn(input) {
|
|
92
226
|
const {
|
|
93
227
|
provider,
|
|
94
228
|
catalog,
|
|
95
229
|
system,
|
|
230
|
+
systemVolatile,
|
|
96
231
|
model,
|
|
97
232
|
maxTokens,
|
|
98
233
|
dispatch,
|
|
@@ -102,8 +237,14 @@ export async function* runAgentTurn(input) {
|
|
|
102
237
|
policy,
|
|
103
238
|
isRelease,
|
|
104
239
|
requestApproval,
|
|
105
|
-
signal
|
|
240
|
+
signal,
|
|
241
|
+
evidence,
|
|
242
|
+
trusted,
|
|
243
|
+
procedures
|
|
106
244
|
} = input;
|
|
245
|
+
/** Notes the user left with a decline, by call id — see declinedResult. */
|
|
246
|
+
const declineReasons = new Map();
|
|
247
|
+
const calibration = input.calibration ?? new TokenCalibration();
|
|
107
248
|
const messages = [...input.messages];
|
|
108
249
|
const tools = toProviderTools(catalog, {
|
|
109
250
|
availableToolIds: input.availableToolIds,
|
|
@@ -112,105 +253,697 @@ export async function* runAgentTurn(input) {
|
|
|
112
253
|
});
|
|
113
254
|
tools.push(uiProviderTool());
|
|
114
255
|
const maxSteps = policy.maxSteps ?? DEFAULT_MAX_STEPS;
|
|
115
|
-
|
|
256
|
+
/**
|
|
257
|
+
* The conversation this turn belongs to. Handed back on every `record` so a
|
|
258
|
+
* dispatch that settles after the user started a new conversation lands in
|
|
259
|
+
* the app (nothing can pull it back) but not in the new chat's undo bar.
|
|
260
|
+
*/
|
|
261
|
+
const ledgerGeneration = ledger.generation;
|
|
262
|
+
/**
|
|
263
|
+
* What every request costs before a single message is added: the tool
|
|
264
|
+
* definitions, the system prompt, and the room the model needs to reply.
|
|
265
|
+
*
|
|
266
|
+
* Measured, not guessed — the tool block alone is ~32k characters for a
|
|
267
|
+
* typical install and ~94k with all 24 tools available, which is far too
|
|
268
|
+
* much to leave out of a budget. `maxTokens` is multiplied by four because
|
|
269
|
+
* the budget is in characters.
|
|
270
|
+
*/
|
|
271
|
+
const fixedChars = system.length + (systemVolatile?.length ?? 0) + tools.reduce((n, t) => n + t.description.length + JSON.stringify(t.inputSchema).length, 0);
|
|
272
|
+
/** The reserve at the ratio believed RIGHT NOW — it moves as usage reports come in. */
|
|
273
|
+
const requestReserve = () => fixedChars + calibration.chars(maxTokens);
|
|
274
|
+
/**
|
|
275
|
+
* Caps the AGENT's wall-clock, not the user's. Time spent parked on an
|
|
276
|
+
* approval sheet is pushed onto the deadline as it is spent (see the
|
|
277
|
+
* `requestApproval` await below) — otherwise a user who takes three minutes
|
|
278
|
+
* to tap Approve gets the change applied and then "this is taking too long"
|
|
279
|
+
* for the turn that applied it.
|
|
280
|
+
*/
|
|
281
|
+
let deadline = Date.now() + DEFAULT_TURN_MS;
|
|
282
|
+
/** Retries spent this turn, per kind — the budget is per turn, not per step. */
|
|
283
|
+
const retries = {
|
|
284
|
+
throttled: 0,
|
|
285
|
+
overflow: 0
|
|
286
|
+
};
|
|
287
|
+
/** Tightened after an overflow; see budgetForRequest's `cap`. In tokens, so a calibration change re-scales it. */
|
|
288
|
+
let historyCapTokens = MAX_HISTORY_TOKENS;
|
|
116
289
|
|
|
117
|
-
// Every http(s) URL that appeared in a tool result
|
|
118
|
-
// tells the model image URLs must come from here; this Set is
|
|
119
|
-
// enforcement — a model-remembered or injected URL in an image block is
|
|
290
|
+
// Every http(s) URL that appeared in a tool result in this CONVERSATION.
|
|
291
|
+
// The prompt tells the model image URLs must come from here; this Set is
|
|
292
|
+
// the enforcement — a model-remembered or injected URL in an image block is
|
|
120
293
|
// both a fabrication vector and a data-exfil beacon (the GET carries
|
|
121
|
-
// whatever is encoded into the URL).
|
|
122
|
-
|
|
294
|
+
// whatever is encoded into the URL). Session-owned when the caller passes
|
|
295
|
+
// one; see RunTurnInput.seenUrls for why it is not per-turn.
|
|
296
|
+
const seenUrls = input.seenUrls ?? new Set();
|
|
123
297
|
const URL_RE = /https?:\/\/[^\s"'\\)\]}>]+/g;
|
|
298
|
+
|
|
299
|
+
/** Whether any text has gone out THIS TURN — see `breakBefore`. */
|
|
300
|
+
let answerStarted = false;
|
|
301
|
+
|
|
302
|
+
/**
|
|
303
|
+
* The last few rounds' call signatures. See LOOP_REPEATS — a run of
|
|
304
|
+
* identical rounds gets one warning appended to the results, once, so the
|
|
305
|
+
* model is told rather than silently capped.
|
|
306
|
+
*/
|
|
307
|
+
const recentSignatures = [];
|
|
308
|
+
let loopWarned = false;
|
|
309
|
+
|
|
310
|
+
/**
|
|
311
|
+
* Everything the model SAID this turn, in order. Read once, at the end, by
|
|
312
|
+
* the ask gate — a question asked in round one and left hanging is the same
|
|
313
|
+
* dead end as one asked in the last round. See askGate.ts.
|
|
314
|
+
*/
|
|
315
|
+
const spoken = [];
|
|
316
|
+
/**
|
|
317
|
+
* Whether the user already has something to tap. An `actions` or
|
|
318
|
+
* `suggestions` block is the model doing this job itself, and a second card
|
|
319
|
+
* asking the same thing underneath it is worse than none.
|
|
320
|
+
*
|
|
321
|
+
* Note this does NOT suppress the shortfall gate below — that one asks
|
|
322
|
+
* whether the user can tap the GAP, which unrelated chips do not answer.
|
|
323
|
+
*/
|
|
324
|
+
let offeredTap = false;
|
|
325
|
+
/**
|
|
326
|
+
* Whether this turn actually went and looked — navigated, described the
|
|
327
|
+
* screen, tapped something, refetched. The shortfall gate stays quiet when
|
|
328
|
+
* it did: an agent that tried and came up short has earned its answer.
|
|
329
|
+
*/
|
|
330
|
+
let attemptedAcquisition = false;
|
|
124
331
|
for (let step = 0; step < maxSteps; step++) {
|
|
125
332
|
if (signal?.aborted) {
|
|
126
333
|
yield {
|
|
127
|
-
type: "done"
|
|
334
|
+
type: "done",
|
|
335
|
+
stopReason: "stopped"
|
|
128
336
|
};
|
|
129
337
|
return messages;
|
|
130
338
|
}
|
|
131
339
|
if (Date.now() > deadline) {
|
|
340
|
+
// Same shape as the step cap below. This used to be an `error`, which
|
|
341
|
+
// drew the failed-turn card with Try again — and Try again re-sends the
|
|
342
|
+
// question, restarting the whole investigation that just ran out of
|
|
343
|
+
// time. Continue picks up with everything so far still in history.
|
|
132
344
|
yield {
|
|
133
|
-
type: "
|
|
134
|
-
|
|
345
|
+
type: "block",
|
|
346
|
+
block: {
|
|
347
|
+
id: `cap${Date.now()}`,
|
|
348
|
+
kind: "notice",
|
|
349
|
+
tone: "warning",
|
|
350
|
+
// Keeps the "This is taking too long" prefix: the eval classifier
|
|
351
|
+
// (evals/ask-buoy/lib/trajectory.ts) tells a time-out from a step
|
|
352
|
+
// cap by it.
|
|
353
|
+
text: `This is taking too long — stopped after ${Math.round(DEFAULT_TURN_MS / 1000)}s. What it found so far is above.`,
|
|
354
|
+
actions: [{
|
|
355
|
+
label: "Continue",
|
|
356
|
+
primary: true,
|
|
357
|
+
send: "Continue where you left off."
|
|
358
|
+
}]
|
|
359
|
+
}
|
|
135
360
|
};
|
|
136
361
|
yield {
|
|
137
|
-
type: "done"
|
|
362
|
+
type: "done",
|
|
363
|
+
stopReason: "time-cap"
|
|
138
364
|
};
|
|
139
365
|
return messages;
|
|
140
366
|
}
|
|
367
|
+
|
|
368
|
+
/**
|
|
369
|
+
* BUDGET, EVERY REQUEST — not once per turn.
|
|
370
|
+
*
|
|
371
|
+
* The session trims when a turn opens, and that used to be the only
|
|
372
|
+
* check. But a turn is not one request: each step appends an assistant
|
|
373
|
+
* message and a tool-results message, and a single result can be 24,000
|
|
374
|
+
* characters. A twelve-step investigation could therefore add six figures
|
|
375
|
+
* of history AFTER the only check had run, and the turn died on the
|
|
376
|
+
* provider's context limit with an opaque error — in exactly the long
|
|
377
|
+
* sessions the tool is for.
|
|
378
|
+
*
|
|
379
|
+
* The current round is never touched: it is the question being answered.
|
|
380
|
+
*/
|
|
381
|
+
const budgeted = budgetForRequest(messages, requestReserve(), calibration.chars(historyCapTokens), calibration.charsPerToken, ledger.liveCallIds());
|
|
382
|
+
if (budgeted.droppedRounds > 0) {
|
|
383
|
+
messages.length = 0;
|
|
384
|
+
messages.push(...budgeted.messages);
|
|
385
|
+
yield {
|
|
386
|
+
type: "history-trimmed",
|
|
387
|
+
droppedRounds: budgeted.droppedRounds,
|
|
388
|
+
droppedPinned: budgeted.droppedPinned
|
|
389
|
+
};
|
|
390
|
+
} else if (budgeted.messages !== messages) {
|
|
391
|
+
messages.length = 0;
|
|
392
|
+
messages.push(...budgeted.messages);
|
|
393
|
+
}
|
|
141
394
|
let text = "";
|
|
142
|
-
|
|
395
|
+
/** See `breakBefore`: set once the first text of THIS round has gone out. */
|
|
396
|
+
let textThisRound = false;
|
|
397
|
+
const textEvent = delta => {
|
|
398
|
+
// Only when this round opens a NEW paragraph of an answer that has
|
|
399
|
+
// already started. A turn whose first words arrive in round three has
|
|
400
|
+
// nothing to be separated from, and marking that would put the flag on
|
|
401
|
+
// events where it means nothing.
|
|
402
|
+
const opensNewRound = answerStarted && !textThisRound;
|
|
403
|
+
textThisRound = true;
|
|
404
|
+
answerStarted = true;
|
|
405
|
+
return opensNewRound ? {
|
|
406
|
+
type: "text",
|
|
407
|
+
delta,
|
|
408
|
+
breakBefore: true
|
|
409
|
+
} : {
|
|
410
|
+
type: "text",
|
|
411
|
+
delta
|
|
412
|
+
};
|
|
413
|
+
};
|
|
414
|
+
let calls = [];
|
|
415
|
+
/**
|
|
416
|
+
* Recovered from text rather than emitted as calls — see textToolCalls.ts.
|
|
417
|
+
* Tracked by id so each one's result can tell the model to stop doing that;
|
|
418
|
+
* they are otherwise ordinary calls and get every check the others get.
|
|
419
|
+
*/
|
|
420
|
+
const salvagedIds = new Set();
|
|
421
|
+
/**
|
|
422
|
+
* Text withheld from the screen while it could still be a tool-call
|
|
423
|
+
* envelope. Streamed text cannot be un-shown, so the choice has to be made
|
|
424
|
+
* before it leaves — and a model that writes its call as JSON must not
|
|
425
|
+
* have that JSON become the answer the user reads.
|
|
426
|
+
*/
|
|
427
|
+
let held = "";
|
|
428
|
+
let holding = true;
|
|
143
429
|
// Carried, never read: Anthropic requires the turn's thinking blocks back
|
|
144
430
|
// verbatim with its tool results. See providers/anthropic.ts note 5.
|
|
145
|
-
|
|
431
|
+
let thinking = [];
|
|
146
432
|
let failed = false;
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
433
|
+
/**
|
|
434
|
+
* What the provider said about how the stream ended.
|
|
435
|
+
*
|
|
436
|
+
* Undefined means it never said — a bare EOF, or a host-supplied provider
|
|
437
|
+
* whose generator simply returned. Both are treated as incomplete, so the
|
|
438
|
+
* fail-safe direction is the default rather than something each adapter
|
|
439
|
+
* has to remember to opt into.
|
|
440
|
+
*/
|
|
441
|
+
let outcome;
|
|
442
|
+
/** Set for the two kinds that are recoverable rather than a hard failure. */
|
|
443
|
+
let incomplete;
|
|
444
|
+
|
|
445
|
+
/**
|
|
446
|
+
* THE REQUEST, with its retries. One request per pass; a pass that fails
|
|
447
|
+
* before the model generated anything may be sent again (see
|
|
448
|
+
* THROTTLE_RETRIES). Everything the pass accumulated is reset first —
|
|
449
|
+
* there is nothing to keep, by the rule that made the retry safe.
|
|
450
|
+
*/
|
|
451
|
+
for (;;) {
|
|
452
|
+
text = "";
|
|
453
|
+
calls = [];
|
|
454
|
+
held = "";
|
|
455
|
+
holding = true;
|
|
456
|
+
thinking = [];
|
|
457
|
+
failed = false;
|
|
458
|
+
outcome = undefined;
|
|
459
|
+
incomplete = undefined;
|
|
460
|
+
/** The failure that ended this pass, when it is one the engine may retry. */
|
|
461
|
+
let retryable;
|
|
462
|
+
/** What this pass sends, in characters — the numerator of the calibration sample. */
|
|
463
|
+
const sentChars = fixedChars + messages.reduce((n, m) => n + messageSize(m), 0);
|
|
464
|
+
for await (const ev of provider.send({
|
|
465
|
+
messages,
|
|
466
|
+
system,
|
|
467
|
+
systemVolatile,
|
|
468
|
+
tools,
|
|
469
|
+
model,
|
|
470
|
+
maxTokens,
|
|
471
|
+
signal
|
|
472
|
+
})) {
|
|
473
|
+
if (ev.type === "text") {
|
|
474
|
+
text += ev.delta;
|
|
475
|
+
if (!holding) {
|
|
476
|
+
yield textEvent(ev.delta);
|
|
477
|
+
} else if (couldBeToolCallEnvelope(text)) {
|
|
478
|
+
held += ev.delta;
|
|
479
|
+
} else {
|
|
480
|
+
// Not an envelope after all. Release everything at once and stream
|
|
481
|
+
// the rest as usual — the reader loses nothing but a few characters
|
|
482
|
+
// of latency at the very start of the answer.
|
|
483
|
+
holding = false;
|
|
484
|
+
held = "";
|
|
485
|
+
yield textEvent(text);
|
|
486
|
+
}
|
|
487
|
+
} else if (ev.type === "tool-call") {
|
|
488
|
+
calls.push(ev.call);
|
|
489
|
+
} else if (ev.type === "thinking") {
|
|
490
|
+
thinking.push(ev.block);
|
|
491
|
+
// Surfaced as well as carried: the block goes back to the provider
|
|
492
|
+
// verbatim (note above), and the readable half goes to the UI so a
|
|
493
|
+
// tester can see WHY a turn did what it did. Redacted blocks have no
|
|
494
|
+
// readable half and are carried only.
|
|
495
|
+
if (ev.block.type === "thinking" && ev.block.thinking.trim()) {
|
|
496
|
+
yield {
|
|
497
|
+
type: "reasoning",
|
|
498
|
+
text: ev.block.thinking
|
|
499
|
+
};
|
|
500
|
+
}
|
|
501
|
+
} else if (ev.type === "error") {
|
|
502
|
+
if (ev.kind === "truncated" || ev.kind === "timeout") {
|
|
503
|
+
// Not a failure the user did anything about, and not one Try again
|
|
504
|
+
// can fix — re-sending the QUESTION restarts an investigation that
|
|
505
|
+
// may already have written things. Held back from the error card
|
|
506
|
+
// (which is what draws Try again) and handled below as a recoverable
|
|
507
|
+
// stop with a Continue button.
|
|
508
|
+
incomplete = {
|
|
509
|
+
message: ev.message
|
|
510
|
+
};
|
|
511
|
+
} else if (ev.problem && (ev.problem.kind === "throttled" || ev.problem.kind === "overflow") && text === "" && calls.length === 0 && thinking.length === 0) {
|
|
512
|
+
// Nothing generated, and a kind that a wait or a trim can fix.
|
|
513
|
+
// Decided after the stream closes; the adapter returns on error.
|
|
514
|
+
retryable = {
|
|
515
|
+
problem: ev.problem,
|
|
516
|
+
message: ev.message
|
|
517
|
+
};
|
|
518
|
+
} else {
|
|
519
|
+
// A mid-stream error can arrive on an already-committed 200.
|
|
520
|
+
yield {
|
|
521
|
+
type: "error",
|
|
522
|
+
message: ev.message
|
|
523
|
+
};
|
|
524
|
+
failed = true;
|
|
525
|
+
}
|
|
526
|
+
} else if (ev.type === "done") {
|
|
527
|
+
outcome = ev.outcome;
|
|
528
|
+
if (ev.usage) {
|
|
529
|
+
// The provider just counted this prompt. One sample per request
|
|
530
|
+
// keeps the chars↔tokens ratio honest for the NEXT budget.
|
|
531
|
+
calibration.observe(sentChars, promptTokensOf(ev.usage, provider.protocol));
|
|
532
|
+
// Surfaced so a host can meter cost per seat. Parsed for a long time;
|
|
533
|
+
// went nowhere anyone could see.
|
|
534
|
+
yield {
|
|
535
|
+
type: "usage",
|
|
536
|
+
...ev.usage,
|
|
537
|
+
model: ev.model
|
|
538
|
+
};
|
|
539
|
+
}
|
|
540
|
+
}
|
|
541
|
+
}
|
|
542
|
+
if (!retryable) break;
|
|
543
|
+
const kind = retryable.problem.kind;
|
|
544
|
+
const maxAttempts = 1 + (kind === "throttled" ? THROTTLE_RETRIES : OVERFLOW_RETRIES);
|
|
545
|
+
const spent = retries[kind];
|
|
546
|
+
let waitMs = kind === "throttled" ? Math.min(retryable.problem.retryAfterMs ?? RETRY_BASE_MS * 2 ** spent, RETRY_AFTER_CAP_MS) + Math.floor(Math.random() * RETRY_JITTER_MS) : 0;
|
|
547
|
+
let giveUp = spent >= maxAttempts - 1 || signal?.aborted === true || Date.now() + waitMs > deadline;
|
|
548
|
+
if (!giveUp && kind === "overflow") {
|
|
549
|
+
// Halve the ceiling and trim again. If that changes nothing — the
|
|
550
|
+
// current round alone is over the limit — a retry would only repeat
|
|
551
|
+
// the refusal, so report it instead.
|
|
552
|
+
historyCapTokens = Math.floor(historyCapTokens / 2);
|
|
553
|
+
const before = messages.reduce((n, m) => n + messageSize(m), 0);
|
|
554
|
+
const again = budgetForRequest(messages, requestReserve(), calibration.chars(historyCapTokens), calibration.charsPerToken, ledger.liveCallIds());
|
|
555
|
+
const after = again.messages.reduce((n, m) => n + messageSize(m), 0);
|
|
556
|
+
if (after >= before) {
|
|
557
|
+
giveUp = true;
|
|
558
|
+
} else {
|
|
559
|
+
messages.length = 0;
|
|
560
|
+
messages.push(...again.messages);
|
|
561
|
+
if (again.droppedRounds > 0) yield {
|
|
562
|
+
type: "history-trimmed",
|
|
563
|
+
droppedRounds: again.droppedRounds,
|
|
564
|
+
droppedPinned: again.droppedPinned
|
|
173
565
|
};
|
|
174
566
|
}
|
|
175
|
-
}
|
|
176
|
-
|
|
567
|
+
}
|
|
568
|
+
if (giveUp) {
|
|
569
|
+
// Plain words first; the endpoint's own text after, for whoever files
|
|
570
|
+
// the bug. A user reading "prompt is too long: 213000 tokens" has no
|
|
571
|
+
// move; "start a new conversation" is one.
|
|
572
|
+
const plain = kind === "throttled" ? spent > 0 ? `The AI endpoint is rate-limiting requests — tried ${spent + 1} times.` : "The AI endpoint is rate-limiting requests." : "This conversation is too large for the AI endpoint and there was nothing older to trim. Start a new conversation, or ask a shorter question.";
|
|
177
573
|
yield {
|
|
178
574
|
type: "error",
|
|
179
|
-
message:
|
|
575
|
+
message: `${plain} (${retryable.message})`
|
|
180
576
|
};
|
|
181
577
|
failed = true;
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
578
|
+
break;
|
|
579
|
+
}
|
|
580
|
+
retries[kind] += 1;
|
|
581
|
+
yield {
|
|
582
|
+
type: "retrying",
|
|
583
|
+
reason: kind,
|
|
584
|
+
attempt: retries[kind] + 1,
|
|
585
|
+
maxAttempts,
|
|
586
|
+
inMs: waitMs
|
|
587
|
+
};
|
|
588
|
+
if (waitMs > 0) await sleep(waitMs, signal);
|
|
589
|
+
if (signal?.aborted) {
|
|
185
590
|
yield {
|
|
186
|
-
type: "
|
|
187
|
-
|
|
188
|
-
output: ev.usage.output
|
|
591
|
+
type: "done",
|
|
592
|
+
stopReason: "stopped"
|
|
189
593
|
};
|
|
594
|
+
return messages;
|
|
190
595
|
}
|
|
191
596
|
}
|
|
192
597
|
if (failed) {
|
|
598
|
+
// Whatever was withheld is the model's own words on the way to an error.
|
|
599
|
+
// Show it rather than swallowing it.
|
|
600
|
+
if (held) yield textEvent(held);
|
|
601
|
+
yield {
|
|
602
|
+
type: "done",
|
|
603
|
+
stopReason: "error"
|
|
604
|
+
};
|
|
605
|
+
return messages;
|
|
606
|
+
}
|
|
607
|
+
|
|
608
|
+
/**
|
|
609
|
+
* THE COMPLETION GATE. Nothing below this runs a tool unless the provider
|
|
610
|
+
* said the response was whole.
|
|
611
|
+
*
|
|
612
|
+
* The failure it exists for is not theoretical and does not look like a
|
|
613
|
+
* failure: a connection cut after the model emitted
|
|
614
|
+
* `{"key":"cart","value":[]}` leaves valid JSON, a plausible-looking plan,
|
|
615
|
+
* and no error anywhere. The old code parsed those arguments, dispatched
|
|
616
|
+
* the write, sent the result back and carried on — a real mutation from
|
|
617
|
+
* half a sentence. Bare EOF is not consent.
|
|
618
|
+
*
|
|
619
|
+
* `output-limited` is the same shape from a different cause: the model hit
|
|
620
|
+
* max_tokens partway through planning. What it managed to SAY is real and
|
|
621
|
+
* is shown; what it was part-way through DOING is not run.
|
|
622
|
+
*/
|
|
623
|
+
if (!incomplete && outcome !== "completed") {
|
|
624
|
+
incomplete = {
|
|
625
|
+
message: outcome === "output-limited" ? calls.length ? "The model ran out of room mid-plan, so nothing was run. What it found so far is above." : "The model ran out of room before finishing this answer." : "The answer ended before the endpoint said it was finished, so nothing was run."
|
|
626
|
+
};
|
|
627
|
+
}
|
|
628
|
+
if (incomplete) {
|
|
629
|
+
// Held text is the model's own words on the way out. Show it: the user
|
|
630
|
+
// watched it arrive and hiding it now reads as the app losing work.
|
|
631
|
+
if (held) yield textEvent(held);
|
|
632
|
+
/**
|
|
633
|
+
* Text only — never the tool calls, and never the thinking.
|
|
634
|
+
*
|
|
635
|
+
* An assistant turn carrying `tool_use` with no matching `tool_result`
|
|
636
|
+
* is an instant 400 on the next request, and these calls are exactly the
|
|
637
|
+
* ones that must not be answered. Keeping the text preserves what the
|
|
638
|
+
* user can see on screen for whatever comes next.
|
|
639
|
+
*/
|
|
640
|
+
if (text.trim()) messages.push({
|
|
641
|
+
role: "assistant",
|
|
642
|
+
text
|
|
643
|
+
});
|
|
644
|
+
yield {
|
|
645
|
+
type: "block",
|
|
646
|
+
block: {
|
|
647
|
+
id: `cut${Date.now()}`,
|
|
648
|
+
kind: "notice",
|
|
649
|
+
tone: "warning",
|
|
650
|
+
text: incomplete.message,
|
|
651
|
+
actions: [{
|
|
652
|
+
label: "Continue",
|
|
653
|
+
primary: true,
|
|
654
|
+
send: "Continue where you left off."
|
|
655
|
+
}]
|
|
656
|
+
}
|
|
657
|
+
};
|
|
193
658
|
yield {
|
|
194
|
-
type: "done"
|
|
659
|
+
type: "done",
|
|
660
|
+
stopReason: "incomplete"
|
|
195
661
|
};
|
|
196
662
|
return messages;
|
|
197
663
|
}
|
|
664
|
+
if (held) {
|
|
665
|
+
// A model that wrote its call out as JSON instead of calling it. Run it,
|
|
666
|
+
// and show its `reply` — never the blob. See textToolCalls.ts.
|
|
667
|
+
const salvaged = calls.length === 0 ? parseTextToolCalls(held, catalog) : undefined;
|
|
668
|
+
if (salvaged) {
|
|
669
|
+
for (const call of salvaged.calls) {
|
|
670
|
+
calls.push(call);
|
|
671
|
+
salvagedIds.add(call.id);
|
|
672
|
+
}
|
|
673
|
+
text = salvaged.reply;
|
|
674
|
+
if (salvaged.reply) yield textEvent(salvaged.reply);
|
|
675
|
+
} else {
|
|
676
|
+
yield textEvent(held);
|
|
677
|
+
}
|
|
678
|
+
held = "";
|
|
679
|
+
}
|
|
198
680
|
messages.push({
|
|
199
681
|
role: "assistant",
|
|
200
682
|
text,
|
|
201
683
|
toolCalls: calls.length ? calls : undefined,
|
|
202
684
|
thinking: thinking.length ? thinking : undefined
|
|
203
685
|
});
|
|
686
|
+
if (text.trim()) spoken.push(text.trim());
|
|
204
687
|
if (calls.length === 0) {
|
|
688
|
+
// The turn is over and the model wrote its own words. Two ways those
|
|
689
|
+
// words end badly, and they are different failures — see askGate.ts.
|
|
690
|
+
const said = spoken.join("\n\n");
|
|
691
|
+
|
|
692
|
+
// 1. It answered a smaller question than it was asked and handed the
|
|
693
|
+
// rest back. Fires even when chips were offered: the question is
|
|
694
|
+
// whether the user can tap the GAP, not whether they can tap.
|
|
695
|
+
const gap = synthesizeShortfallBlock({
|
|
696
|
+
text: said,
|
|
697
|
+
attempted: attemptedAcquisition
|
|
698
|
+
});
|
|
699
|
+
if (gap) yield {
|
|
700
|
+
type: "block",
|
|
701
|
+
block: gap
|
|
702
|
+
};
|
|
703
|
+
|
|
704
|
+
// 2. It ended waiting on the user in prose. Suppressed once anything
|
|
705
|
+
// tappable is on screen, the gap button included.
|
|
706
|
+
const asked = offeredTap || gap ? null : synthesizeAskBlock(said);
|
|
707
|
+
if (asked) {
|
|
708
|
+
yield {
|
|
709
|
+
type: "block",
|
|
710
|
+
block: asked
|
|
711
|
+
};
|
|
712
|
+
yield {
|
|
713
|
+
type: "awaiting-user",
|
|
714
|
+
blockId: asked.id
|
|
715
|
+
};
|
|
716
|
+
}
|
|
205
717
|
yield {
|
|
206
|
-
type: "done"
|
|
718
|
+
type: "done",
|
|
719
|
+
stopReason: asked ? "awaiting-user" : "answered"
|
|
207
720
|
};
|
|
208
721
|
return messages;
|
|
209
722
|
}
|
|
210
723
|
const results = [];
|
|
211
724
|
let realmWillDie = false;
|
|
725
|
+
|
|
726
|
+
/**
|
|
727
|
+
* Reads, already in flight.
|
|
728
|
+
*
|
|
729
|
+
* The loop below stays strictly serial — every yield, every ledger entry
|
|
730
|
+
* and every approval keeps its order — but a round of independent READS
|
|
731
|
+
* used to pay each round trip end to end. The prompt tells the model to
|
|
732
|
+
* chain reads freely ("reading is cheap; do it"), and a three-read round
|
|
733
|
+
* cost three times the device latency for no reason.
|
|
734
|
+
*
|
|
735
|
+
* ONLY when the whole batch is safe to overlap: every call a catalog read,
|
|
736
|
+
* none needing approval, no `buoy_ui`, nothing snapshot-only or
|
|
737
|
+
* session-local. One write, one approval or one unknown action in the
|
|
738
|
+
* batch and nothing is prefetched — the mixed case is where ordering
|
|
739
|
+
* actually matters, and it is not worth the risk for the latency.
|
|
740
|
+
*/
|
|
741
|
+
const prefetch = new Map();
|
|
742
|
+
if (calls.length > 1) {
|
|
743
|
+
const safe = calls.every(c => {
|
|
744
|
+
if (c.name === UI_TOOL_NAME) return false;
|
|
745
|
+
const id = fromProviderToolName(c.name, catalog);
|
|
746
|
+
if (!id || id === "ask-buoy") return false;
|
|
747
|
+
const act = c.input?.action;
|
|
748
|
+
if (typeof act !== "string" || act === SNAPSHOT_ACTION) return false;
|
|
749
|
+
const d = findDescriptor(catalog, id, act);
|
|
750
|
+
if (!d || d.effect !== "read") return false;
|
|
751
|
+
return decide({
|
|
752
|
+
descriptor: d,
|
|
753
|
+
toolId: id,
|
|
754
|
+
policy,
|
|
755
|
+
isRelease
|
|
756
|
+
}).verdict === "allow";
|
|
757
|
+
});
|
|
758
|
+
if (safe) {
|
|
759
|
+
for (const c of calls) {
|
|
760
|
+
const id = fromProviderToolName(c.name, catalog);
|
|
761
|
+
const act = String(c.input.action);
|
|
762
|
+
const raw = {
|
|
763
|
+
...(c.input.params ?? {})
|
|
764
|
+
};
|
|
765
|
+
const d = findDescriptor(catalog, id, act);
|
|
766
|
+
const norm = normalizeParams(id, act, raw);
|
|
767
|
+
const withDefaults = applyDefaults(norm.params, d.params);
|
|
768
|
+
if (!validateParams(withDefaults, d.params).ok) continue;
|
|
769
|
+
// Rejections are swallowed here and re-awaited in the loop, where
|
|
770
|
+
// they are turned into the same tool_result they always were.
|
|
771
|
+
const p = Promise.resolve(dispatch(id, act, withDefaults)).catch(e => {
|
|
772
|
+
throw e;
|
|
773
|
+
});
|
|
774
|
+
p.catch(() => {});
|
|
775
|
+
prefetch.set(c.id, p);
|
|
776
|
+
}
|
|
777
|
+
}
|
|
778
|
+
}
|
|
212
779
|
let awaiting = null;
|
|
780
|
+
|
|
781
|
+
/**
|
|
782
|
+
* A call Stop got to first — as a row, so the transcript says which ones.
|
|
783
|
+
*
|
|
784
|
+
* Resolved leniently: the call has not been validated yet, and a label
|
|
785
|
+
* that falls back to `tool.action` is better than no row for a call the
|
|
786
|
+
* user needs to know did not run.
|
|
787
|
+
*/
|
|
788
|
+
const stoppedStep = call => {
|
|
789
|
+
const toolId = fromProviderToolName(call.name, catalog) ?? call.name;
|
|
790
|
+
const action = typeof call.input?.action === "string" ? call.input.action : "";
|
|
791
|
+
const descriptor = action ? findDescriptor(catalog, toolId, action) : undefined;
|
|
792
|
+
const params = call.input?.params ?? {};
|
|
793
|
+
return [{
|
|
794
|
+
type: "tool-start",
|
|
795
|
+
id: call.id,
|
|
796
|
+
toolId,
|
|
797
|
+
action,
|
|
798
|
+
label: descriptor ? describeCall(toolId, descriptor, params) : `${toolId}.${action}`,
|
|
799
|
+
description: descriptor?.summary ?? "",
|
|
800
|
+
params,
|
|
801
|
+
effect: descriptor?.effect ?? "read"
|
|
802
|
+
}, {
|
|
803
|
+
type: "tool-end",
|
|
804
|
+
id: call.id,
|
|
805
|
+
ok: false,
|
|
806
|
+
summary: "stopped",
|
|
807
|
+
durationMs: 0
|
|
808
|
+
}];
|
|
809
|
+
};
|
|
810
|
+
|
|
811
|
+
/**
|
|
812
|
+
* THE BATCH BARRIER — what the whole batch will do, decided before any of
|
|
813
|
+
* it does anything.
|
|
814
|
+
*
|
|
815
|
+
* A model answers with several calls at once, and they used to run one at
|
|
816
|
+
* a time with each approval asked when its own call came up. So
|
|
817
|
+
* `[set the flag, delete the cache]` wrote the flag, THEN asked about the
|
|
818
|
+
* delete — and a user who declined had already changed the app, with the
|
|
819
|
+
* bubble reading "changed the flag" directly above "left the cache alone".
|
|
820
|
+
* Declining is supposed to mean the plan does not happen.
|
|
821
|
+
*
|
|
822
|
+
* The fix is not to ask again later or to refuse afterwards, neither of
|
|
823
|
+
* which can unrun a write. It is to ask FIRST: every gated call in the
|
|
824
|
+
* batch is put to the user before the batch's first mutation dispatches,
|
|
825
|
+
* so consent is given with the whole plan visible.
|
|
826
|
+
*
|
|
827
|
+
* Resolution here is pure and duplicates the loop's own — deliberately.
|
|
828
|
+
* Anything it cannot resolve (an unknown tool, a bad shape) comes back as
|
|
829
|
+
* neither mutating nor gated, so the loop reports it exactly as it always
|
|
830
|
+
* did and the barrier simply does not fire. Failing back to today's
|
|
831
|
+
* behaviour is the right failure for a safety gate to have.
|
|
832
|
+
*/
|
|
833
|
+
const planned = calls.map(call => {
|
|
834
|
+
if (call.name === UI_TOOL_NAME) return undefined;
|
|
835
|
+
const toolId = fromProviderToolName(call.name, catalog);
|
|
836
|
+
const action = typeof call.input?.action === "string" ? call.input.action : undefined;
|
|
837
|
+
if (!toolId || !action) return undefined;
|
|
838
|
+
const descriptor = findDescriptor(catalog, toolId, action);
|
|
839
|
+
if (!descriptor) return undefined;
|
|
840
|
+
const raw = call.input.params ?? {};
|
|
841
|
+
const params = applyDefaults(normalizeParams(toolId, action, raw).params, descriptor.params);
|
|
842
|
+
if (!validateParams(params, descriptor.params).ok) return undefined;
|
|
843
|
+
const verdict = decide({
|
|
844
|
+
descriptor,
|
|
845
|
+
toolId,
|
|
846
|
+
policy,
|
|
847
|
+
isRelease
|
|
848
|
+
});
|
|
849
|
+
return {
|
|
850
|
+
call,
|
|
851
|
+
toolId,
|
|
852
|
+
action,
|
|
853
|
+
descriptor,
|
|
854
|
+
params,
|
|
855
|
+
label: describeCall(toolId, descriptor, params),
|
|
856
|
+
mutates: descriptor.effect !== "read",
|
|
857
|
+
gated: verdict.verdict === "needs-approval",
|
|
858
|
+
reason: verdict.verdict === "needs-approval" ? verdict.reason : ""
|
|
859
|
+
};
|
|
860
|
+
});
|
|
861
|
+
const gatedInBatch = planned.some(p => p?.gated);
|
|
862
|
+
/** Decisions the barrier has already taken, so no call is asked twice. */
|
|
863
|
+
const decided = new Map();
|
|
864
|
+
let barrierRun = false;
|
|
865
|
+
let batchDeclined = false;
|
|
866
|
+
|
|
867
|
+
/** Put every gated call in the batch to the user, in the order written. */
|
|
868
|
+
async function* runBarrier() {
|
|
869
|
+
barrierRun = true;
|
|
870
|
+
for (const p of planned) {
|
|
871
|
+
if (!p?.gated || decided.has(p.call.id)) continue;
|
|
872
|
+
if (trusted?.has(`${p.toolId}.${p.action}`)) {
|
|
873
|
+
// Waived for this conversation: runs like any allowed write, no card.
|
|
874
|
+
decided.set(p.call.id, true);
|
|
875
|
+
continue;
|
|
876
|
+
}
|
|
877
|
+
yield {
|
|
878
|
+
type: "approval-required",
|
|
879
|
+
id: p.call.id,
|
|
880
|
+
toolId: p.toolId,
|
|
881
|
+
action: p.action,
|
|
882
|
+
label: p.label,
|
|
883
|
+
description: p.descriptor.summary,
|
|
884
|
+
reason: p.reason
|
|
885
|
+
};
|
|
886
|
+
const askedAt = Date.now();
|
|
887
|
+
const targetDigest = await readTargetDigest({
|
|
888
|
+
toolId: p.toolId,
|
|
889
|
+
action: p.action,
|
|
890
|
+
params: p.params,
|
|
891
|
+
dispatch,
|
|
892
|
+
catalog
|
|
893
|
+
});
|
|
894
|
+
const answer = readAnswer(trusted?.has(`${p.toolId}.${p.action}`) ? true : requestApproval ? await requestApproval({
|
|
895
|
+
id: p.call.id,
|
|
896
|
+
toolId: p.toolId,
|
|
897
|
+
action: p.action,
|
|
898
|
+
label: p.label,
|
|
899
|
+
description: p.descriptor.summary,
|
|
900
|
+
reason: p.reason,
|
|
901
|
+
params: p.params,
|
|
902
|
+
targetDigest
|
|
903
|
+
}) : false);
|
|
904
|
+
// The user's deliberation is not the agent's runtime.
|
|
905
|
+
deadline += Date.now() - askedAt;
|
|
906
|
+
const approved = answer.approved;
|
|
907
|
+
decided.set(p.call.id, approved);
|
|
908
|
+
if (answer.trust) trusted?.add(`${p.toolId}.${p.action}`);
|
|
909
|
+
if (answer.reason) declineReasons.set(p.call.id, answer.reason);
|
|
910
|
+
if (!approved) {
|
|
911
|
+
// One refusal ends the PLAN, not just the call. The remaining
|
|
912
|
+
// changes were proposed together and nothing has run yet, so
|
|
913
|
+
// carrying on with the rest would apply half of something the user
|
|
914
|
+
// has just turned down.
|
|
915
|
+
batchDeclined = true;
|
|
916
|
+
return;
|
|
917
|
+
}
|
|
918
|
+
}
|
|
919
|
+
}
|
|
213
920
|
for (const call of calls) {
|
|
921
|
+
/**
|
|
922
|
+
* Stop was pressed while this batch was running.
|
|
923
|
+
*
|
|
924
|
+
* The abort signal used to be read only at the top of a ROUND, so a
|
|
925
|
+
* response carrying three calls ran all three after the tap — and the
|
|
926
|
+
* bubble then said "Stopped." over two writes the user believed it had
|
|
927
|
+
* prevented. Each call now checks before it starts. A dispatch already
|
|
928
|
+
* in flight is left to finish (nothing can pull a store write back out
|
|
929
|
+
* of an adapter) and keeps its receipt; everything after it is reported
|
|
930
|
+
* as not run, each as its own row, so the transcript says exactly which
|
|
931
|
+
* changes landed. A result is still pushed for every call — the provider
|
|
932
|
+
* requires one per tool_use or the next request is rejected.
|
|
933
|
+
*/
|
|
934
|
+
if (signal?.aborted) {
|
|
935
|
+
// A `buoy_ui` card that was never shown is not a step; no row for it.
|
|
936
|
+
if (call.name !== UI_TOOL_NAME) {
|
|
937
|
+
for (const ev of stoppedStep(call)) yield ev;
|
|
938
|
+
}
|
|
939
|
+
results.push({
|
|
940
|
+
toolCallId: call.id,
|
|
941
|
+
content: "Not run — the user pressed Stop before this call started.",
|
|
942
|
+
isError: true
|
|
943
|
+
});
|
|
944
|
+
continue;
|
|
945
|
+
}
|
|
946
|
+
|
|
214
947
|
// `buoy_ui`: show or ask. Validated like any action; an Ask block ends
|
|
215
948
|
// the turn after this batch — the answer comes back as a user message.
|
|
216
949
|
if (call.name === UI_TOOL_NAME) {
|
|
@@ -218,19 +951,35 @@ export async function* runAgentTurn(input) {
|
|
|
218
951
|
if (!built.ok || !built.block) {
|
|
219
952
|
results.push({
|
|
220
953
|
toolCallId: call.id,
|
|
221
|
-
|
|
954
|
+
/**
|
|
955
|
+
* The trailing sentence is the expensive half.
|
|
956
|
+
*
|
|
957
|
+
* A model that gets a block bounced rewrites its WHOLE answer on
|
|
958
|
+
* the retry, and one assistant bubble spans every round of a turn
|
|
959
|
+
* (see `breakBefore`), so the user reads the same paragraph again
|
|
960
|
+
* — measured at four near-identical closings from three schema
|
|
961
|
+
* misses in a row on one live turn. The block is wrong; the
|
|
962
|
+
* sentence under it was fine.
|
|
963
|
+
*/
|
|
964
|
+
content: `Invalid ${UI_TOOL_NAME} block:\n${built.errors.join("\n")}\n\nResend ONLY the corrected block. Do not rewrite your answer — whatever you already said this turn has been shown to the user, and saying it again repeats it on their screen.`,
|
|
222
965
|
isError: true
|
|
223
966
|
});
|
|
224
967
|
continue;
|
|
225
968
|
}
|
|
226
969
|
// Provenance for pictures, ENFORCED: an image URL the model did not
|
|
227
970
|
// read from a tool result this turn does not render. Live or nothing.
|
|
971
|
+
// Dropping pictures SILENTLY is its own bug: the model then writes
|
|
972
|
+
// "here they are, with artwork" over a card that has none, and the
|
|
973
|
+
// user is told something untrue about their own screen. Every drop
|
|
974
|
+
// comes back in the tool result so the next sentence can be honest.
|
|
975
|
+
let droppedImages = 0;
|
|
228
976
|
if (built.block.kind === "imageGrid") {
|
|
229
977
|
const kept = built.block.images.filter(img => seenUrls.has(img.url));
|
|
978
|
+
droppedImages = built.block.images.length - kept.length;
|
|
230
979
|
if (kept.length === 0) {
|
|
231
980
|
results.push({
|
|
232
981
|
toolCallId: call.id,
|
|
233
|
-
content:
|
|
982
|
+
content: URL_PROVENANCE_REJECTION,
|
|
234
983
|
isError: true
|
|
235
984
|
});
|
|
236
985
|
continue;
|
|
@@ -241,12 +990,17 @@ export async function* runAgentTurn(input) {
|
|
|
241
990
|
};
|
|
242
991
|
}
|
|
243
992
|
if (built.block.kind === "list") {
|
|
244
|
-
built.block
|
|
245
|
-
|
|
246
|
-
|
|
993
|
+
const items = built.block.items.map(item => {
|
|
994
|
+
if (!item.image || seenUrls.has(item.image)) return item;
|
|
995
|
+
droppedImages += 1;
|
|
996
|
+
return {
|
|
247
997
|
...item,
|
|
248
998
|
image: undefined
|
|
249
|
-
}
|
|
999
|
+
};
|
|
1000
|
+
});
|
|
1001
|
+
built.block = {
|
|
1002
|
+
...built.block,
|
|
1003
|
+
items
|
|
250
1004
|
};
|
|
251
1005
|
}
|
|
252
1006
|
yield {
|
|
@@ -254,6 +1008,7 @@ export async function* runAgentTurn(input) {
|
|
|
254
1008
|
block: built.block,
|
|
255
1009
|
forToolCallId: call.id
|
|
256
1010
|
};
|
|
1011
|
+
if (built.block.kind === "actions" || built.block.kind === "suggestions") offeredTap = true;
|
|
257
1012
|
if (isAskBlock(built.block)) {
|
|
258
1013
|
awaiting = built.block.id;
|
|
259
1014
|
results.push({
|
|
@@ -263,7 +1018,7 @@ export async function* runAgentTurn(input) {
|
|
|
263
1018
|
} else {
|
|
264
1019
|
results.push({
|
|
265
1020
|
toolCallId: call.id,
|
|
266
|
-
content: "Shown to the user. Don't repeat its contents in text."
|
|
1021
|
+
content: "Shown to the user. Don't repeat its contents in text." + (droppedImages > 0 ? droppedImageNote(droppedImages) : "")
|
|
267
1022
|
});
|
|
268
1023
|
}
|
|
269
1024
|
continue;
|
|
@@ -271,12 +1026,40 @@ export async function* runAgentTurn(input) {
|
|
|
271
1026
|
const toolId = fromProviderToolName(call.name, catalog);
|
|
272
1027
|
const action = call.input.action;
|
|
273
1028
|
|
|
1029
|
+
/**
|
|
1030
|
+
* A call that never reached a tool at all.
|
|
1031
|
+
*
|
|
1032
|
+
* Same rule as a refusal: the model made a move, the move went nowhere,
|
|
1033
|
+
* and a transcript that hides it leaves the reader watching the agent
|
|
1034
|
+
* change its mind for no visible reason. Caught on device — a model
|
|
1035
|
+
* guessed three navigation tool names that do not exist, spent 41
|
|
1036
|
+
* seconds doing it, and the only trace was a sentence it chose to write.
|
|
1037
|
+
* `read` because nothing was touched.
|
|
1038
|
+
*/
|
|
1039
|
+
const deadCall = (label, summary) => [{
|
|
1040
|
+
type: "tool-start",
|
|
1041
|
+
id: call.id,
|
|
1042
|
+
toolId: toolId ?? call.name,
|
|
1043
|
+
action: action ?? "",
|
|
1044
|
+
label,
|
|
1045
|
+
description: "",
|
|
1046
|
+
params: call.input.params ?? {},
|
|
1047
|
+
effect: "read"
|
|
1048
|
+
}, {
|
|
1049
|
+
type: "tool-end",
|
|
1050
|
+
id: call.id,
|
|
1051
|
+
ok: false,
|
|
1052
|
+
summary,
|
|
1053
|
+
durationMs: 0
|
|
1054
|
+
}];
|
|
1055
|
+
|
|
274
1056
|
// Two different mistakes, kept apart on purpose. Merging them tells a
|
|
275
1057
|
// model that forgot `action` that the TOOL does not exist, and it stops
|
|
276
1058
|
// reaching for a tool that was fine — the same distinction
|
|
277
1059
|
// `dispatchToolAction` draws between an unknown tool and an unknown
|
|
278
1060
|
// action, for the same reason.
|
|
279
1061
|
if (!toolId) {
|
|
1062
|
+
for (const ev of deadCall(call.name, "No such tool")) yield ev;
|
|
280
1063
|
results.push({
|
|
281
1064
|
toolCallId: call.id,
|
|
282
1065
|
content: `There is no tool called "${call.name}". Use one of the tools you were given.`,
|
|
@@ -285,6 +1068,7 @@ export async function* runAgentTurn(input) {
|
|
|
285
1068
|
continue;
|
|
286
1069
|
}
|
|
287
1070
|
if (!action) {
|
|
1071
|
+
for (const ev of deadCall(call.name, "No action given")) yield ev;
|
|
288
1072
|
results.push({
|
|
289
1073
|
toolCallId: call.id,
|
|
290
1074
|
content: `"${call.name}" needs an "action" — it is required, and it names which of this tool's actions to run. See the action list in the tool's description.`,
|
|
@@ -298,6 +1082,7 @@ export async function* runAgentTurn(input) {
|
|
|
298
1082
|
// has nowhere to go but guess again, and the next guess is no better
|
|
299
1083
|
// informed than the last.
|
|
300
1084
|
const known = catalog.find(t => t.toolId === toolId)?.actions.map(a => a.action).join(", ");
|
|
1085
|
+
for (const ev of deadCall(`${toolId}.${action}`, "No such action")) yield ev;
|
|
301
1086
|
results.push({
|
|
302
1087
|
toolCallId: call.id,
|
|
303
1088
|
content: `"${action}" is not an action on ${toolId}.${known ? ` Valid actions: ${known}.` : ""}`,
|
|
@@ -306,7 +1091,11 @@ export async function* runAgentTurn(input) {
|
|
|
306
1091
|
continue;
|
|
307
1092
|
}
|
|
308
1093
|
const raw = call.input.params ?? {};
|
|
309
|
-
|
|
1094
|
+
// Accept the parameter-name misses every model makes (queryKey for
|
|
1095
|
+
// queryHash, requestId for id, flat rule fields, …) before validation,
|
|
1096
|
+
// so a mechanical alias never costs a turn. See normalizeParams.ts.
|
|
1097
|
+
const normalized = normalizeParams(toolId, action, raw);
|
|
1098
|
+
const params = applyDefaults(normalized.params, descriptor.params);
|
|
310
1099
|
const check = validateParams(params, descriptor.params);
|
|
311
1100
|
if (!check.ok) {
|
|
312
1101
|
results.push({
|
|
@@ -316,6 +1105,7 @@ export async function* runAgentTurn(input) {
|
|
|
316
1105
|
});
|
|
317
1106
|
continue;
|
|
318
1107
|
}
|
|
1108
|
+
const aliasTrailer = aliasNote(normalized.notes);
|
|
319
1109
|
const label = describeCall(toolId, descriptor, params);
|
|
320
1110
|
const verdict = decide({
|
|
321
1111
|
descriptor,
|
|
@@ -323,14 +1113,29 @@ export async function* runAgentTurn(input) {
|
|
|
323
1113
|
policy,
|
|
324
1114
|
isRelease
|
|
325
1115
|
});
|
|
1116
|
+
|
|
1117
|
+
// A step the user can SEE, for a call that never ran. `tool-end` alone
|
|
1118
|
+
// updates a row that was never opened, so a refusal used to leave no
|
|
1119
|
+
// trace in the transcript at all — the model simply changed its mind
|
|
1120
|
+
// between one sentence and the next.
|
|
1121
|
+
const unrunStep = summary => [{
|
|
1122
|
+
type: "tool-start",
|
|
1123
|
+
id: call.id,
|
|
1124
|
+
toolId,
|
|
1125
|
+
action,
|
|
1126
|
+
label,
|
|
1127
|
+
description: descriptor.summary,
|
|
1128
|
+
params,
|
|
1129
|
+
effect: descriptor.effect
|
|
1130
|
+
}, {
|
|
1131
|
+
type: "tool-end",
|
|
1132
|
+
id: call.id,
|
|
1133
|
+
ok: false,
|
|
1134
|
+
summary,
|
|
1135
|
+
durationMs: 0
|
|
1136
|
+
}];
|
|
326
1137
|
if (verdict.verdict === "refuse") {
|
|
327
|
-
yield
|
|
328
|
-
type: "tool-end",
|
|
329
|
-
id: call.id,
|
|
330
|
-
ok: false,
|
|
331
|
-
summary: "refused",
|
|
332
|
-
durationMs: 0
|
|
333
|
-
};
|
|
1138
|
+
for (const ev of unrunStep("refused")) yield ev;
|
|
334
1139
|
results.push({
|
|
335
1140
|
toolCallId: call.id,
|
|
336
1141
|
content: verdict.reason,
|
|
@@ -338,33 +1143,93 @@ export async function* runAgentTurn(input) {
|
|
|
338
1143
|
});
|
|
339
1144
|
continue;
|
|
340
1145
|
}
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
})
|
|
1146
|
+
|
|
1147
|
+
/**
|
|
1148
|
+
* Nothing in this batch changes anything until every card in it has been
|
|
1149
|
+
* answered. See `runBarrier` — this is the line that makes a decline
|
|
1150
|
+
* mean "the plan does not happen" rather than "the rest of the plan does
|
|
1151
|
+
* not happen".
|
|
1152
|
+
*/
|
|
1153
|
+
const gatedHere = verdict.verdict === "needs-approval";
|
|
1154
|
+
if (gatedInBatch && !barrierRun && (descriptor.effect !== "read" || gatedHere)) {
|
|
1155
|
+
yield* runBarrier();
|
|
1156
|
+
}
|
|
1157
|
+
if (batchDeclined && (descriptor.effect !== "read" || gatedHere)) {
|
|
1158
|
+
for (const ev of unrunStep("declined")) yield ev;
|
|
1159
|
+
results.push({
|
|
1160
|
+
toolCallId: call.id,
|
|
1161
|
+
content: decided.get(call.id) === false ? declinedResult(declineReasons.get(call.id)) : "Not run — the user declined another change in this same batch, so none of it was applied. Ask what they would prefer before proposing it again.",
|
|
1162
|
+
isError: true
|
|
1163
|
+
});
|
|
1164
|
+
continue;
|
|
1165
|
+
}
|
|
1166
|
+
if (gatedHere) {
|
|
1167
|
+
let approved;
|
|
1168
|
+
if (decided.has(call.id)) {
|
|
1169
|
+
// The barrier already put this card up, before the batch's first
|
|
1170
|
+
// mutation. Asking again would be the same question twice.
|
|
1171
|
+
approved = decided.get(call.id);
|
|
1172
|
+
} else if (trusted?.has(`${toolId}.${action}`)) {
|
|
1173
|
+
// Waived for this conversation — see RunTurnInput.trusted.
|
|
1174
|
+
approved = true;
|
|
1175
|
+
decided.set(call.id, true);
|
|
1176
|
+
} else {
|
|
1177
|
+
// The barrier could not resolve this call (see `planned`), so it was
|
|
1178
|
+
// never offered. Ask here, exactly as this always did — a gate that
|
|
1179
|
+
// fails closed by silently declining would be worse than one that
|
|
1180
|
+
// asks late.
|
|
1181
|
+
yield {
|
|
1182
|
+
type: "approval-required",
|
|
1183
|
+
id: call.id,
|
|
1184
|
+
toolId,
|
|
1185
|
+
action,
|
|
1186
|
+
label,
|
|
1187
|
+
description: descriptor.summary,
|
|
1188
|
+
reason: verdict.reason
|
|
1189
|
+
};
|
|
1190
|
+
const askedAt = Date.now();
|
|
1191
|
+
// What the write is about to replace, as of NOW. Only a restored
|
|
1192
|
+
// card ever reads it back — see readTargetDigest.
|
|
1193
|
+
const targetDigest = await readTargetDigest({
|
|
1194
|
+
toolId,
|
|
1195
|
+
action,
|
|
1196
|
+
params,
|
|
1197
|
+
dispatch,
|
|
1198
|
+
catalog
|
|
1199
|
+
});
|
|
1200
|
+
const answer = readAnswer(requestApproval ? await requestApproval({
|
|
1201
|
+
id: call.id,
|
|
1202
|
+
toolId,
|
|
1203
|
+
action,
|
|
1204
|
+
label,
|
|
1205
|
+
description: descriptor.summary,
|
|
1206
|
+
reason: verdict.reason,
|
|
1207
|
+
params,
|
|
1208
|
+
targetDigest
|
|
1209
|
+
}) : false);
|
|
1210
|
+
// The user's deliberation is not the agent's runtime. Give the clock
|
|
1211
|
+
// back before anything else can trip the deadline check.
|
|
1212
|
+
deadline += Date.now() - askedAt;
|
|
1213
|
+
approved = answer.approved;
|
|
1214
|
+
decided.set(call.id, approved);
|
|
1215
|
+
if (answer.trust) trusted?.add(`${toolId}.${action}`);
|
|
1216
|
+
if (answer.reason) declineReasons.set(call.id, answer.reason);
|
|
1217
|
+
}
|
|
359
1218
|
if (!approved) {
|
|
1219
|
+
// The row is the whole point here. Without it the bubble read as a
|
|
1220
|
+
// contradiction — "Bumped the line from 5 to 9." straight into "I
|
|
1221
|
+
// left it as it was" — with nothing on screen to say a change had
|
|
1222
|
+
// been proposed and turned down.
|
|
1223
|
+
for (const ev of unrunStep("declined")) yield ev;
|
|
360
1224
|
results.push({
|
|
361
1225
|
toolCallId: call.id,
|
|
362
|
-
content:
|
|
1226
|
+
content: declinedResult(declineReasons.get(call.id)),
|
|
363
1227
|
isError: true
|
|
364
1228
|
});
|
|
365
1229
|
continue;
|
|
366
1230
|
}
|
|
367
1231
|
}
|
|
1232
|
+
if (ACQUISITION_CALLS.has(`${toolId}.${action}`)) attemptedAcquisition = true;
|
|
368
1233
|
yield {
|
|
369
1234
|
type: "tool-start",
|
|
370
1235
|
id: call.id,
|
|
@@ -381,7 +1246,50 @@ export async function* runAgentTurn(input) {
|
|
|
381
1246
|
// exactly. This is the one thing that makes storage reversible here when
|
|
382
1247
|
// the Scenarios engine has to treat it as permanent.
|
|
383
1248
|
const before = await captureBefore(descriptor, toolId, params, dispatch);
|
|
384
|
-
const stateBefore = await
|
|
1249
|
+
const stateBefore = await captureStoreState(descriptor, toolId, params, dispatch);
|
|
1250
|
+
|
|
1251
|
+
/**
|
|
1252
|
+
* THE FINAL GUARD. Everything above this line happened in the past.
|
|
1253
|
+
*
|
|
1254
|
+
* `decide()` ran before the approval card went up, and between there and
|
|
1255
|
+
* here are three awaits: the target read, the person deciding, and two
|
|
1256
|
+
* pre-reads. Each one yields the thread, and a person can do a lot in
|
|
1257
|
+
* that gap — press Stop, or flip the read-only toggle in the settings
|
|
1258
|
+
* sheet while the card is still on screen. Both were reproduced: the
|
|
1259
|
+
* write went ahead on the verdict it captured before they touched
|
|
1260
|
+
* anything, which is precisely the moment the product promises it will
|
|
1261
|
+
* not. `live` is mutated in place by the session (see `refreshLive`), so
|
|
1262
|
+
* asking again reads the toggles as they are NOW.
|
|
1263
|
+
*
|
|
1264
|
+
* Below this there is no `await` before `dispatch` is CALLED. That is
|
|
1265
|
+
* load-bearing: an await here would reopen the same gap one line lower.
|
|
1266
|
+
*/
|
|
1267
|
+
if (signal?.aborted) {
|
|
1268
|
+
for (const ev of unrunStep("stopped")) yield ev;
|
|
1269
|
+
results.push({
|
|
1270
|
+
toolCallId: call.id,
|
|
1271
|
+
content: "Not run — the user pressed Stop before this call started.",
|
|
1272
|
+
isError: true
|
|
1273
|
+
});
|
|
1274
|
+
continue;
|
|
1275
|
+
}
|
|
1276
|
+
const stillAllowed = decide({
|
|
1277
|
+
descriptor,
|
|
1278
|
+
toolId,
|
|
1279
|
+
policy,
|
|
1280
|
+
isRelease
|
|
1281
|
+
});
|
|
1282
|
+
if (stillAllowed.verdict === "refuse") {
|
|
1283
|
+
// `needs-approval` is NOT re-refused: reaching here means the tap
|
|
1284
|
+
// already happened, and asking twice for one call is its own bug.
|
|
1285
|
+
for (const ev of unrunStep("refused")) yield ev;
|
|
1286
|
+
results.push({
|
|
1287
|
+
toolCallId: call.id,
|
|
1288
|
+
content: stillAllowed.reason,
|
|
1289
|
+
isError: true
|
|
1290
|
+
});
|
|
1291
|
+
continue;
|
|
1292
|
+
}
|
|
385
1293
|
try {
|
|
386
1294
|
// `getSnapshot` is reserved: it is not an adapter action, so it goes to
|
|
387
1295
|
// the snapshot reader and is trimmed to the fields worth sending. See
|
|
@@ -389,34 +1297,69 @@ export async function* runAgentTurn(input) {
|
|
|
389
1297
|
// ledger the model asks about lives HERE, not behind an adapter, so
|
|
390
1298
|
// routing it through dispatch would only work on a device and would
|
|
391
1299
|
// record the undo as a fresh effect.
|
|
392
|
-
const result = action === SNAPSHOT_ACTION ? projectSnapshot(toolId, await requireSnapshot(readSnapshot, toolId), params) : toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch) : await dispatch(toolId, action, params);
|
|
1300
|
+
const result = action === SNAPSHOT_ACTION ? projectSnapshot(toolId, await requireSnapshot(readSnapshot, toolId), params) : toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch, evidence, params, procedures) : await (prefetch.get(call.id) ?? dispatch(toolId, action, params));
|
|
393
1301
|
const cleaned = redact(stripSelfTraffic(result));
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
1302
|
+
const refused = reportsFailure(result);
|
|
1303
|
+
|
|
1304
|
+
// Kept in full — after redaction, never before — so the model can go
|
|
1305
|
+
// back to it. Not a retrieve's own result (a retrieve of a retrieve is
|
|
1306
|
+
// a loop), and not a refusal (nothing to go back to).
|
|
1307
|
+
const text = fullText(cleaned);
|
|
1308
|
+
const ref = evidence && !refused && !(toolId === "ask-buoy" && action === "retrieve") ? evidence.stash({
|
|
1309
|
+
toolId,
|
|
1310
|
+
action,
|
|
1311
|
+
params,
|
|
1312
|
+
capturedAt: Date.now(),
|
|
1313
|
+
text
|
|
1314
|
+
}) : undefined;
|
|
1315
|
+
const encodedResult = encodeResult(text, ref);
|
|
1316
|
+
const traceResult = encodedResult.length > MAX_TRACE_RESULT_CHARS ? `${encodedResult.slice(0, MAX_TRACE_RESULT_CHARS)}…` : encodedResult;
|
|
403
1317
|
yield {
|
|
404
1318
|
type: "tool-end",
|
|
405
1319
|
id: call.id,
|
|
406
|
-
ok:
|
|
1320
|
+
ok: !refused,
|
|
407
1321
|
summary: summarise(result),
|
|
408
1322
|
durationMs: Date.now() - startedAt,
|
|
409
|
-
result:
|
|
1323
|
+
result: traceResult
|
|
410
1324
|
};
|
|
411
1325
|
|
|
412
1326
|
// The receipt: what the user sees of this result, built from the
|
|
413
1327
|
// same redacted data the model gets. See blocks/receipts.ts.
|
|
414
|
-
|
|
1328
|
+
//
|
|
1329
|
+
// The store is read back a second time so the diff shows what the
|
|
1330
|
+
// write ACTUALLY did rather than what it asked for — the two differ on
|
|
1331
|
+
// every `path` edit and every keyed list edit, and the request-shaped
|
|
1332
|
+
// version reported untouched fields as deleted.
|
|
1333
|
+
const stateAfter = stateBefore === undefined || refused ? undefined : await captureStoreState(descriptor, toolId, params, dispatch);
|
|
1334
|
+
if (descriptor.effect !== "read" && toolId !== "ask-buoy" && !refused) {
|
|
1335
|
+
// A delete of something we created IS the undo, whoever asked for
|
|
1336
|
+
// it — mark the matching entry undone instead of leaving the bar
|
|
1337
|
+
// promising an undo against a rule that no longer exists.
|
|
1338
|
+
ledger.noteExternalRevert(toolId, action, params);
|
|
1339
|
+
// Recorded AFTER the read-back so a cache edit's entry can carry both
|
|
1340
|
+
// halves: the data to restore and a fingerprint of what the write
|
|
1341
|
+
// left behind, which is how undo knows whether the app has since
|
|
1342
|
+
// replaced it. See effectFor / ledger "query-write".
|
|
1343
|
+
const fx = effectFor(toolId, descriptor, params, result, before, {
|
|
1344
|
+
before: stateBefore,
|
|
1345
|
+
after: stateAfter
|
|
1346
|
+
});
|
|
1347
|
+
if (fx) {
|
|
1348
|
+
const entry = ledger.record(toolId, action, fx, ledgerGeneration);
|
|
1349
|
+
// Ties the change to its round, so the budget keeps that round
|
|
1350
|
+
// while the change is live. See LedgerEntry.callId.
|
|
1351
|
+
if (entry) entry.callId = call.id;
|
|
1352
|
+
}
|
|
1353
|
+
}
|
|
1354
|
+
// No receipt for a call that changed nothing: a "what changed" card
|
|
1355
|
+
// under a refusal is the same lie as the ledger entry, drawn bigger.
|
|
1356
|
+
const receipt = refused ? undefined : projectReceipt({
|
|
415
1357
|
toolId,
|
|
416
1358
|
action,
|
|
417
1359
|
params,
|
|
418
1360
|
result: cleaned,
|
|
419
1361
|
before: stateBefore ?? before,
|
|
1362
|
+
after: stateAfter,
|
|
420
1363
|
effect: descriptor.effect
|
|
421
1364
|
});
|
|
422
1365
|
if (receipt) yield {
|
|
@@ -424,10 +1367,36 @@ export async function* runAgentTurn(input) {
|
|
|
424
1367
|
block: receipt,
|
|
425
1368
|
forToolCallId: call.id
|
|
426
1369
|
};
|
|
1370
|
+
/**
|
|
1371
|
+
* THE OUTCOME CHECK. A write's own `{ok:true}` says the adapter
|
|
1372
|
+
* accepted it; this reads the app back and says whether the requested
|
|
1373
|
+
* state is actually there. Only for actions that have a verifier, only
|
|
1374
|
+
* for writes that were not refused, and never itself a write. Its
|
|
1375
|
+
* verdict rides in the tool result as a trailer the model reads, and
|
|
1376
|
+
* on a second tool-end so the row says "verified" / "unverified" /
|
|
1377
|
+
* "check failed". See engine/verify.ts for the three meanings.
|
|
1378
|
+
*/
|
|
1379
|
+
const verification = descriptor.effect !== "read" && !refused ? await verifyOutcome({
|
|
1380
|
+
toolId,
|
|
1381
|
+
action,
|
|
1382
|
+
params,
|
|
1383
|
+
result,
|
|
1384
|
+
dispatch,
|
|
1385
|
+
signal,
|
|
1386
|
+
after: stateAfter
|
|
1387
|
+
}) : undefined;
|
|
1388
|
+
if (verification) yield {
|
|
1389
|
+
type: "tool-verified",
|
|
1390
|
+
id: call.id,
|
|
1391
|
+
verification
|
|
1392
|
+
};
|
|
427
1393
|
for (const url of encodedResult.match(URL_RE) ?? []) seenUrls.add(url);
|
|
428
1394
|
results.push({
|
|
429
1395
|
toolCallId: call.id,
|
|
430
|
-
content: encodedResult + ownedKeyNote(toolId, descriptor, params, storeKeyOwners)
|
|
1396
|
+
content: encodedResult + ownedKeyNote(toolId, descriptor, params, storeKeyOwners) + aliasTrailer + (verification ? verificationTrailer(verification) : "") + (salvagedIds.has(call.id) ? SALVAGE_NOTE : ""),
|
|
1397
|
+
...(ref ? {
|
|
1398
|
+
ref
|
|
1399
|
+
} : {})
|
|
431
1400
|
});
|
|
432
1401
|
if (toolId === "app" && (action === "reloadApp" || action === "reload")) {
|
|
433
1402
|
realmWillDie = true;
|
|
@@ -448,6 +1417,20 @@ export async function* runAgentTurn(input) {
|
|
|
448
1417
|
});
|
|
449
1418
|
}
|
|
450
1419
|
}
|
|
1420
|
+
|
|
1421
|
+
// Looping? Say so, once, on the LAST result of this round — so it arrives
|
|
1422
|
+
// attached to the thing being repeated rather than as a free-floating
|
|
1423
|
+
// user turn, and the model reads it before deciding what to do next.
|
|
1424
|
+
if (calls.length) {
|
|
1425
|
+
recentSignatures.push(callSignature(calls));
|
|
1426
|
+
if (recentSignatures.length > LOOP_REPEATS) recentSignatures.shift();
|
|
1427
|
+
const looping = !loopWarned && recentSignatures.length === LOOP_REPEATS && recentSignatures.every(sig => sig === recentSignatures[0]);
|
|
1428
|
+
if (looping) {
|
|
1429
|
+
loopWarned = true;
|
|
1430
|
+
const last = results[results.length - 1];
|
|
1431
|
+
if (last) last.content = `${last.content}\n\n${LOOP_WARNING}`;
|
|
1432
|
+
}
|
|
1433
|
+
}
|
|
451
1434
|
messages.push({
|
|
452
1435
|
role: "tool-results",
|
|
453
1436
|
results
|
|
@@ -458,7 +1441,8 @@ export async function* runAgentTurn(input) {
|
|
|
458
1441
|
blockId: awaiting
|
|
459
1442
|
};
|
|
460
1443
|
yield {
|
|
461
|
-
type: "done"
|
|
1444
|
+
type: "done",
|
|
1445
|
+
stopReason: "awaiting-user"
|
|
462
1446
|
};
|
|
463
1447
|
return messages;
|
|
464
1448
|
}
|
|
@@ -466,7 +1450,8 @@ export async function* runAgentTurn(input) {
|
|
|
466
1450
|
// Anything after this dies with the JS realm. Stop cleanly instead of
|
|
467
1451
|
// sending a request whose answer can never arrive.
|
|
468
1452
|
yield {
|
|
469
|
-
type: "done"
|
|
1453
|
+
type: "done",
|
|
1454
|
+
stopReason: "realm-died"
|
|
470
1455
|
};
|
|
471
1456
|
return messages;
|
|
472
1457
|
}
|
|
@@ -486,7 +1471,8 @@ export async function* runAgentTurn(input) {
|
|
|
486
1471
|
}
|
|
487
1472
|
};
|
|
488
1473
|
yield {
|
|
489
|
-
type: "done"
|
|
1474
|
+
type: "done",
|
|
1475
|
+
stopReason: "step-cap"
|
|
490
1476
|
};
|
|
491
1477
|
return messages;
|
|
492
1478
|
}
|
|
@@ -512,6 +1498,29 @@ function ownedKeyNote(toolId, descriptor, params, storeKeyOwners) {
|
|
|
512
1498
|
if (!storeName) return "";
|
|
513
1499
|
return `\n\n[Buoy] This wrote to disk, but "${params.key}" is the saved copy of the live "${storeName}" store, so nothing on screen has changed — the app read that key once at startup and has held the state in memory since. Call zustand.rehydrate({"storeName":"${storeName}"}) to make the app pick this up, or use zustand.setState next time to change the store directly. If you meant to set the value for the app's next launch, this is already done and no further action is needed.`;
|
|
514
1500
|
}
|
|
1501
|
+
export { digestOf };
|
|
1502
|
+
|
|
1503
|
+
/**
|
|
1504
|
+
* Fingerprint the value a write is about to replace.
|
|
1505
|
+
*
|
|
1506
|
+
* Built for the approval card that outlives its turn: the app crashed with
|
|
1507
|
+
* "Set bag line L-01 qty to 14" still asking, relaunched, and the card came
|
|
1508
|
+
* back — and Allow wrote qty 14 against whatever the bag was NOW, maybe a
|
|
1509
|
+
* different account, maybe a line that no longer existed. A hash of the
|
|
1510
|
+
* params alone would not catch that (same label, same params, different
|
|
1511
|
+
* world). So the card carries a digest of the TARGET at ask time, and a
|
|
1512
|
+
* restored Allow re-reads and compares before it writes. Undefined when the
|
|
1513
|
+
* target cannot be read — then the card can only warn, not check.
|
|
1514
|
+
*/
|
|
1515
|
+
export async function readTargetDigest(input) {
|
|
1516
|
+
const descriptor = findDescriptor(input.catalog ?? CATALOG, input.toolId, input.action);
|
|
1517
|
+
if (!descriptor || descriptor.effect === "read") return undefined;
|
|
1518
|
+
const store = await captureStoreState(descriptor, input.toolId, input.params, input.dispatch);
|
|
1519
|
+
if (store !== undefined) return digestOf(store);
|
|
1520
|
+
const before = await captureBefore(descriptor, input.toolId, input.params, input.dispatch);
|
|
1521
|
+
if (before !== undefined) return digestOf(before);
|
|
1522
|
+
return undefined;
|
|
1523
|
+
}
|
|
515
1524
|
/**
|
|
516
1525
|
* One human-initiated action through EVERY gate the model's calls go through:
|
|
517
1526
|
* catalog lookup, param validation, policy, pre-read, effect ledger.
|
|
@@ -544,7 +1553,8 @@ export async function runGatedAction(input) {
|
|
|
544
1553
|
error: `${toolId}.${action} is not in Buoy's catalog.`
|
|
545
1554
|
};
|
|
546
1555
|
}
|
|
547
|
-
const
|
|
1556
|
+
const normalized = normalizeParams(toolId, action, input.params ?? {});
|
|
1557
|
+
const params = applyDefaults(normalized.params, descriptor.params);
|
|
548
1558
|
const check = validateParams(params, descriptor.params);
|
|
549
1559
|
if (!check.ok) {
|
|
550
1560
|
return {
|
|
@@ -565,11 +1575,51 @@ export async function runGatedAction(input) {
|
|
|
565
1575
|
};
|
|
566
1576
|
}
|
|
567
1577
|
const before = await captureBefore(descriptor, toolId, params, dispatch);
|
|
1578
|
+
/**
|
|
1579
|
+
* The same read-back the model's path takes. It was missing here, and the
|
|
1580
|
+
* gap was invisible: a `setQueryData` through a block button or a restored
|
|
1581
|
+
* approval card recorded a ledger entry with no prior value, so Undo had
|
|
1582
|
+
* nothing to put back and the concurrent-writer check had no fingerprint to
|
|
1583
|
+
* compare. The bar said "1 undoable" over a change that could not be undone.
|
|
1584
|
+
*/
|
|
1585
|
+
const stateBefore = await captureStoreState(descriptor, toolId, params, dispatch);
|
|
1586
|
+
|
|
1587
|
+
// Asked again after the pre-reads, for the same reason the model's path asks
|
|
1588
|
+
// again: those are awaits, and read-only can be switched on inside one.
|
|
1589
|
+
const stillAllowed = decide({
|
|
1590
|
+
descriptor,
|
|
1591
|
+
toolId,
|
|
1592
|
+
policy,
|
|
1593
|
+
isRelease
|
|
1594
|
+
});
|
|
1595
|
+
if (stillAllowed.verdict === "refuse") {
|
|
1596
|
+
return {
|
|
1597
|
+
ok: false,
|
|
1598
|
+
error: stillAllowed.reason
|
|
1599
|
+
};
|
|
1600
|
+
}
|
|
568
1601
|
try {
|
|
569
|
-
const result = toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch) : await dispatch(toolId, action, params);
|
|
1602
|
+
const result = toolId === "ask-buoy" ? await runAskBuoyAction(action, ledger, dispatch, undefined, params) : await dispatch(toolId, action, params);
|
|
1603
|
+
|
|
1604
|
+
// Same rule as the model's path: a resolved `{ok:false}` is a refusal, not
|
|
1605
|
+
// a change. Without this a block button (or an approval answered after a
|
|
1606
|
+
// reload) put a phantom entry in the undo bar too.
|
|
1607
|
+
if (reportsFailure(result)) {
|
|
1608
|
+
const err = result.error;
|
|
1609
|
+
return {
|
|
1610
|
+
ok: false,
|
|
1611
|
+
error: typeof err === "string" ? err : "The tool refused this."
|
|
1612
|
+
};
|
|
1613
|
+
}
|
|
570
1614
|
if (descriptor.effect !== "read" && toolId !== "ask-buoy") {
|
|
571
1615
|
ledger.noteExternalRevert(toolId, action, params);
|
|
572
|
-
|
|
1616
|
+
// Read back AFTER the write, exactly as the model's path does — the two
|
|
1617
|
+
// halves together are what make a cache edit undoable.
|
|
1618
|
+
const stateAfter = await captureStoreState(descriptor, toolId, params, dispatch);
|
|
1619
|
+
const fx = effectFor(toolId, descriptor, params, result, before, {
|
|
1620
|
+
before: stateBefore,
|
|
1621
|
+
after: stateAfter
|
|
1622
|
+
});
|
|
573
1623
|
if (fx) ledger.record(toolId, action, fx);
|
|
574
1624
|
}
|
|
575
1625
|
return {
|
|
@@ -591,7 +1641,33 @@ export async function runGatedAction(input) {
|
|
|
591
1641
|
* adapter's same-named actions remain for desktop/MCP, where the ledger is
|
|
592
1642
|
* reached over the broker instead.
|
|
593
1643
|
*/
|
|
594
|
-
async function runAskBuoyAction(action, ledger, dispatch) {
|
|
1644
|
+
async function runAskBuoyAction(action, ledger, dispatch, evidence, params, procedures) {
|
|
1645
|
+
if (action === "retrieve") {
|
|
1646
|
+
return retrieveEvidence(evidence, params ?? {});
|
|
1647
|
+
}
|
|
1648
|
+
if (action === "openProcedure") {
|
|
1649
|
+
const id = typeof params?.id === "string" ? params.id : "";
|
|
1650
|
+
const found = procedures?.find(p => p.id === id);
|
|
1651
|
+
if (!found) {
|
|
1652
|
+
const ids = (procedures ?? []).map(p => p.id);
|
|
1653
|
+
return {
|
|
1654
|
+
ok: false,
|
|
1655
|
+
error: ids.length ? `No procedure "${id}". This app's procedures: ${ids.join(", ")}.` : "This app has no procedures written for it."
|
|
1656
|
+
};
|
|
1657
|
+
}
|
|
1658
|
+
return {
|
|
1659
|
+
id: found.id,
|
|
1660
|
+
title: found.title,
|
|
1661
|
+
...(found.version ? {
|
|
1662
|
+
version: found.version
|
|
1663
|
+
} : {}),
|
|
1664
|
+
...(found.requires?.length ? {
|
|
1665
|
+
requires: found.requires
|
|
1666
|
+
} : {}),
|
|
1667
|
+
body: found.body,
|
|
1668
|
+
note: "Follow these steps with the ordinary tools. Every write still goes through the same policy, approval and undo as any other call; a step that is refused stays refused."
|
|
1669
|
+
};
|
|
1670
|
+
}
|
|
595
1671
|
if (action === "listChanges") {
|
|
596
1672
|
const changes = ledger.list().filter(e => !e.undoneAt).map(e => ({
|
|
597
1673
|
toolId: e.toolId,
|
|
@@ -641,16 +1717,46 @@ async function requireSnapshot(readSnapshot, toolId) {
|
|
|
641
1717
|
* Zustand only for now: `getStoreState` is cheap and the write params carry
|
|
642
1718
|
* the after-state, so the diff is exact without the store's cooperation.
|
|
643
1719
|
*/
|
|
644
|
-
|
|
645
|
-
|
|
646
|
-
|
|
647
|
-
|
|
648
|
-
|
|
649
|
-
|
|
650
|
-
|
|
651
|
-
|
|
652
|
-
|
|
1720
|
+
/**
|
|
1721
|
+
* Read a zustand store whole, for the before/after halves of a write receipt.
|
|
1722
|
+
* Called on both sides of the dispatch — see the receipt above.
|
|
1723
|
+
*/
|
|
1724
|
+
async function captureStoreState(descriptor, toolId, params, dispatch) {
|
|
1725
|
+
if (toolId === "zustand" && descriptor.action === "setState") {
|
|
1726
|
+
if (typeof params.storeName !== "string") return undefined;
|
|
1727
|
+
try {
|
|
1728
|
+
return await dispatch(toolId, "getStoreState", {
|
|
1729
|
+
storeName: params.storeName
|
|
1730
|
+
});
|
|
1731
|
+
} catch {
|
|
1732
|
+
return undefined;
|
|
1733
|
+
}
|
|
1734
|
+
}
|
|
1735
|
+
/**
|
|
1736
|
+
* The same read-both-sides trick for the CACHE.
|
|
1737
|
+
*
|
|
1738
|
+
* A zustand edit produced a "what changed" card and the identical edit to a
|
|
1739
|
+
* query produced nothing — for the write the prompt calls "THE way to change
|
|
1740
|
+
* what a server-backed screen shows". The QA tester's confirmation that the
|
|
1741
|
+
* right field moved was missing from the most-used tool in the product.
|
|
1742
|
+
*/
|
|
1743
|
+
if (toolId === "query" && descriptor.action === "setQueryData") {
|
|
1744
|
+
// hashQueryKey, not JSON.stringify: React Query SORTS object keys when it
|
|
1745
|
+
// hashes, so a key carrying `{query:"",category:null}` hashes as
|
|
1746
|
+
// `{"category":null,"query":""}`. Stringifying in insertion order produced
|
|
1747
|
+
// a hash that matched nothing, the read came back empty, and the card
|
|
1748
|
+
// silently did not draw.
|
|
1749
|
+
const queryHash = typeof params.queryHash === "string" ? params.queryHash : Array.isArray(params.queryKey) ? hashQueryKey(params.queryKey) : undefined;
|
|
1750
|
+
if (!queryHash) return undefined;
|
|
1751
|
+
try {
|
|
1752
|
+
return await dispatch(toolId, "getQueryData", {
|
|
1753
|
+
queryHash
|
|
1754
|
+
});
|
|
1755
|
+
} catch {
|
|
1756
|
+
return undefined;
|
|
1757
|
+
}
|
|
653
1758
|
}
|
|
1759
|
+
return undefined;
|
|
654
1760
|
}
|
|
655
1761
|
async function captureBefore(descriptor, toolId, params, dispatch) {
|
|
656
1762
|
if (descriptor.effect === "read") return undefined;
|