@cohortapp/agent-sdk 2.5.1 → 2.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/bin/maestro.mjs +185 -88
  2. package/bin/maestro.test.mjs +175 -48
  3. package/docs/runbooks/backup-restore.md +65 -33
  4. package/framework-features.json +4 -4
  5. package/lib/backup/policy.mjs +710 -0
  6. package/lib/backup/policy.test.mjs +305 -0
  7. package/lib/budget-escalate.mjs +133 -0
  8. package/lib/budget-escalate.test.mjs +232 -0
  9. package/lib/budget-guard.envelope.test.mjs +476 -0
  10. package/lib/budget-guard.mjs +853 -75
  11. package/lib/budget-guard.test.mjs +91 -42
  12. package/lib/cadences.mjs +33 -0
  13. package/lib/channels/orgmail/adapter.mjs +88 -3
  14. package/lib/channels/orgmail/adapter.test.mjs +137 -0
  15. package/lib/channels/repeat-suppressor.mjs +198 -0
  16. package/lib/channels/repeat-suppressor.test.mjs +134 -0
  17. package/lib/comms/receipts.mjs +297 -0
  18. package/lib/cost/ledger-row.mjs +333 -0
  19. package/lib/cost/ledger-row.test.mjs +183 -0
  20. package/lib/execution/drive.mjs +28 -1
  21. package/lib/execution/effects.mjs +191 -12
  22. package/lib/execution/effects.test.mjs +50 -11
  23. package/lib/goals/admission.mjs +13 -1
  24. package/lib/goals/admission.test.mjs +26 -1
  25. package/lib/goals/loop.mjs +13 -0
  26. package/lib/kpi-sensors.test.mjs +3 -0
  27. package/lib/mandate/cache.mjs +13 -5
  28. package/lib/mandate/derive.mjs +146 -21
  29. package/lib/mandate/derive.test.mjs +50 -6
  30. package/lib/mandate/model.mjs +32 -4
  31. package/lib/mandate/refresh.test.mjs +16 -2
  32. package/lib/mcp/server.test.mjs +12 -3
  33. package/lib/model-router/economics.mjs +107 -76
  34. package/lib/model-router/economics.test.mjs +64 -46
  35. package/lib/model-router/integration-coverage.test.mjs +39 -37
  36. package/lib/model-router/ledger.mjs +75 -22
  37. package/lib/model-router/ledger.test.mjs +35 -2
  38. package/lib/org/client.mjs +14 -0
  39. package/lib/org/cost-sync.mjs +16 -2
  40. package/lib/org/doctor.mjs +62 -1
  41. package/lib/org/doctor.test.mjs +36 -3
  42. package/lib/org/email-remedy.mjs +49 -0
  43. package/lib/org/engagement-ledger.mjs +376 -0
  44. package/lib/org/engagement-ledger.test.mjs +112 -0
  45. package/lib/org/engagement.mjs +1056 -0
  46. package/lib/org/engagement.test.mjs +739 -0
  47. package/lib/org/messaging.mjs +230 -3
  48. package/lib/org/messaging.test.mjs +110 -1
  49. package/lib/org/param-contract.mjs +56 -2
  50. package/lib/org/param-contract.test.mjs +26 -0
  51. package/lib/org/protocol.checksum +1 -1
  52. package/lib/org/protocol.mjs +5 -0
  53. package/lib/org/protocol.test.mjs +7 -1
  54. package/lib/org/tool-surface.mjs +506 -10
  55. package/lib/org/tool-surface.test.mjs +191 -7
  56. package/lib/org/ui-parity.mjs +333 -6
  57. package/lib/org/ui-parity.test.mjs +96 -3
  58. package/lib/org/work-ledger.mjs +241 -0
  59. package/lib/org/work-ledger.test.mjs +237 -0
  60. package/lib/plan/adoption-e2e.test.mjs +366 -0
  61. package/lib/plan/budget-enforcement.test.mjs +400 -0
  62. package/lib/plan/budget-runtime.mjs +215 -0
  63. package/lib/plan/compile.mjs +201 -5
  64. package/lib/plan/compile.test.mjs +19 -5
  65. package/lib/plan/emit.mjs +8 -0
  66. package/lib/plan/emit.test.mjs +18 -0
  67. package/lib/resource-governor.mjs +58 -12
  68. package/lib/resource-governor.test.mjs +41 -1
  69. package/lib/security/audit-engine.mjs +45 -8
  70. package/lib/security/audit-engine.test.mjs +35 -0
  71. package/lib/setup/enroll-from-cohort.mjs +14 -1
  72. package/lib/setup/sections/mandate.mjs +48 -7
  73. package/lib/setup/sections/mandate.test.mjs +17 -2
  74. package/lib/setup/sections/orgmail.mjs +10 -2
  75. package/lib/setup/state.mjs +83 -2
  76. package/lib/telemetry/collect.mjs +360 -20
  77. package/lib/telemetry/collect.test.mjs +266 -0
  78. package/package.json +1 -1
  79. package/scripts/cost/track-claude-usage.mjs +207 -48
  80. package/scripts/cost/track-claude-usage.test.mjs +148 -0
  81. package/scripts/daemon/agent-daemon.mjs +315 -17
  82. package/scripts/daemon/assurance-e2e.test.mjs +421 -0
  83. package/scripts/daemon/assurance.mjs +944 -0
  84. package/scripts/daemon/assurance.test.mjs +668 -0
  85. package/scripts/daemon/cadence-consumer-governance.test.mjs +56 -0
  86. package/scripts/daemon/cadence-consumer.mjs +147 -9
  87. package/scripts/daemon/cadence-consumer.test.mjs +6 -0
  88. package/scripts/daemon/cadence-handlers.mjs +158 -0
  89. package/scripts/daemon/cadence-handlers.test.mjs +64 -0
  90. package/scripts/daemon/deliver.mjs +314 -0
  91. package/scripts/daemon/dispatcher-governance.test.mjs +10 -0
  92. package/scripts/daemon/dispatcher.mjs +64 -6
  93. package/scripts/daemon/responder-cost.test.mjs +68 -0
  94. package/scripts/daemon/responder.mjs +351 -298
  95. package/scripts/local-triggers/generate-plists.test.mjs +7 -4
  96. package/scripts/maintenance/backup-run.mjs +415 -0
  97. package/scripts/maintenance/backup-to-cloud.sh +16 -116
  98. package/scripts/org/send-orgmail.mjs +16 -0
  99. package/scripts/record-receipt.sh +63 -0
  100. package/scripts/restore-from-backup.sh +14 -3
  101. package/scripts/restore-from-backup.test.mjs +8 -5
  102. package/scripts/send-email-threaded.py +47 -0
  103. package/scripts/send-sms.sh +4 -0
  104. package/scripts/send-whatsapp.sh +4 -0
  105. package/scripts/setup/init-backup.mjs +93 -38
  106. package/scripts/slack-send.sh +12 -0
@@ -0,0 +1,314 @@
1
+ /**
2
+ * deliver.mjs — put a specific string in front of a specific human. No model.
3
+ *
4
+ * This module exists because the acknowledgement was an LLM call.
5
+ *
6
+ * The old holding-message path spawned a full `claude --print` child, under a
7
+ * 60s hard cap, on a machine sitting at 95-99% memory, purely to write "let me
8
+ * look into it". It lost that race roughly two times in three (20 of 32
9
+ * observed attempts died on `claude CLI timed out after 60000ms`), and when it
10
+ * lost, the failure was caught, logged, and the human was told NOTHING while a
11
+ * 15-45 minute session ran behind a typing indicator.
12
+ *
13
+ * Everything on this path is therefore synchronous string work plus one HTTP
14
+ * call. There is no generation step that can time out, no model to be starved
15
+ * of memory, and no budget that can refuse it. It is the floor beneath every
16
+ * "never silent" guarantee in assurance.mjs: whatever else fails, THIS can run.
17
+ *
18
+ * The three per-service senders were previously private to responder.mjs, so
19
+ * nothing but a generated reply could reach a human. They are now here, and
20
+ * responder.mjs imports them — one delivery path, used by generated replies,
21
+ * acknowledgements, progress updates and failure notices alike.
22
+ *
23
+ * @module scripts/daemon/deliver
24
+ */
25
+
26
+ "use strict";
27
+
28
+ import { execFileSync } from "child_process";
29
+ import { mkdirSync, writeFileSync } from "fs";
30
+ import { join } from "path";
31
+ import { screenOutbound } from "../../lib/comms/send-gate.mjs";
32
+ import { getHookBus } from "../../lib/hooks/bus.mjs";
33
+ import { recordOutbound } from "../../lib/comms/receipts.mjs";
34
+
35
+ const AGENT_REPO_DIR = process.env.AGENT_DIR || join(new URL(".", import.meta.url).pathname, "../..");
36
+
37
+ function today() {
38
+ return new Date().toISOString().split("T")[0];
39
+ }
40
+
41
+ function getSlackToken() {
42
+ return process.env.SLACK_USER_TOKEN || process.env.SLACK_BOT_TOKEN;
43
+ }
44
+
45
+ // ---------------------------------------------------------------------------
46
+ // Channel resolution
47
+ // ---------------------------------------------------------------------------
48
+
49
+ /** Resolve a Slack channel id from a daemon inbox item. */
50
+ export function resolveSlackChannel(item) {
51
+ if (!item) return null;
52
+ // Direct channel ID (starts with D for DM, C for channel)
53
+ if (item.channel && /^[DC][A-Z0-9]{8,}$/.test(item.channel)) return item.channel;
54
+ if (item.channel_id) return item.channel_id;
55
+
56
+ // Extract from raw_ref format: "slack:D099N1JGKRQ:1775331885.690669"
57
+ if (item.raw_ref) {
58
+ const match = item.raw_ref.match(/slack:([DC][A-Z0-9]+):/);
59
+ if (match) return match[1];
60
+ const bare = item.raw_ref.match(/^([DC][A-Z0-9]{8,})$/);
61
+ if (bare) return bare[1];
62
+ }
63
+ return null;
64
+ }
65
+
66
+ /**
67
+ * The channel a reply to this item belongs in, whatever the service — the key
68
+ * both the receipt ledger and the obligation ledger are keyed on.
69
+ *
70
+ * Routing is by ROOM, not by person: the room the inbound arrived in is the
71
+ * room the answer belongs in.
72
+ */
73
+ export function replyTargetOf(item) {
74
+ if (!item) return null;
75
+ if (item.service === "slack") return resolveSlackChannel(item);
76
+ if (item.service === "gmail") return item.sender_email || item.sender || null;
77
+ return item.channel_id || item.channel || null;
78
+ }
79
+
80
+ /**
81
+ * The services this module can actually put a string in front of a human on.
82
+ *
83
+ * The daemon polls more than this: telegram, whatsapp, orgmail, voice, calendar
84
+ * and a second Gmail account all produce inbox items. `deliver` returns
85
+ * "unsupported service X" for those — a fact that used to be discovered only at
86
+ * send time, once per sweep tick, forever.
87
+ *
88
+ * Naming the set explicitly lets the caller ask BEFORE it promises anything,
89
+ * which is the difference between "I can't reach you here, so an operator has
90
+ * been told" and an apology retried 1,440 times a day into a void.
91
+ */
92
+ export const DELIVERABLE_SERVICES = Object.freeze(["slack", "gmail", "cohort"]);
93
+
94
+ /**
95
+ * Can `deliver` reach the human behind this item at all? Requires both a
96
+ * transport for the service AND a resolvable room to write into.
97
+ */
98
+ export function canDeliverTo(item) {
99
+ if (!item || !item.service) return false;
100
+ if (!DELIVERABLE_SERVICES.includes(item.service)) return false;
101
+ return Boolean(replyTargetOf(item));
102
+ }
103
+
104
+ // ---------------------------------------------------------------------------
105
+ // Slack
106
+ // ---------------------------------------------------------------------------
107
+
108
+ async function sendSlackMessage(channel, text, threadTs = null, o = {}) {
109
+ // Unified send-gate (P0-3): the daemon path passes the same chokepoint as the
110
+ // shell senders + BaseAdapter — banned-phrase, AI disclosure, and
111
+ // information-barrier screening at one place. Block on deny.
112
+ try {
113
+ const gate = await screenOutbound({ channel: "slack", recipient: channel, text, agentRoot: AGENT_REPO_DIR });
114
+ if (gate && gate.allow === false) {
115
+ throw new Error(`send-gate blocked Slack reply: ${gate.reason}`);
116
+ }
117
+ } catch (err) {
118
+ if (/send-gate blocked/.test(err.message)) throw err; // a real block propagates
119
+ // gate infra error (module/policy unreadable): fail-open for internal Slack
120
+ // channels (matches send-gate's internal posture) — never silently drop.
121
+ console.warn(`[deliver] send-gate check errored (allowing internal Slack): ${err.message}`);
122
+ try {
123
+ Promise.resolve(getHookBus().emit("onGuardFail", { guard: "send_gate", channel: "slack", recipient: channel, reason: err.message, failed_open: true }))
124
+ .catch(() => { /* bus emit is isolated; never propagate */ });
125
+ } catch { /* never let telemetry break the send path */ }
126
+ }
127
+
128
+ // Always use user token — bot tokens can't access DM channels
129
+ const token = getSlackToken();
130
+ if (!token) throw new Error("No Slack token available (set SLACK_USER_TOKEN in .env)");
131
+
132
+ const body = { channel, text, ...(threadTs ? { thread_ts: threadTs } : {}) };
133
+ const fetchImpl = o.fetchImpl || fetch;
134
+ const res = await fetchImpl("https://slack.com/api/chat.postMessage", {
135
+ method: "POST",
136
+ headers: { "Authorization": `Bearer ${token}`, "Content-Type": "application/json" },
137
+ body: JSON.stringify(body),
138
+ });
139
+ const data = await res.json();
140
+ if (!data.ok) throw new Error(`Slack API error: ${data.error}`);
141
+ return data;
142
+ }
143
+
144
+ // ---------------------------------------------------------------------------
145
+ // Gmail
146
+ // ---------------------------------------------------------------------------
147
+
148
+ async function sendGmailResponse(item, text) {
149
+ const to = item.sender_email || item.sender;
150
+ const subject = `Re: ${item.subject || "(no subject)"}`;
151
+ const sendScript = join(AGENT_REPO_DIR, "scripts", "send-email-threaded.py");
152
+
153
+ try {
154
+ const args = [sendScript, to, subject, text];
155
+ if (item.subject) args.push("--reply-to-subject", item.subject);
156
+ execFileSync("python3", args, { cwd: AGENT_REPO_DIR, timeout: 30000, encoding: "utf-8", env: { ...process.env } });
157
+ return { sent: true, via: "smtp", to };
158
+ } catch (err) {
159
+ console.error(`[deliver] Gmail send failed for ${to}: ${err.message}`);
160
+ // Fall back to draft file so the response is not lost
161
+ const draftPath = join(AGENT_REPO_DIR, "outputs", "drafts",
162
+ `${today()}-quick-reply-${String(item.sender || "unknown").replace(/[^a-z0-9]/gi, "-")}.md`);
163
+ mkdirSync(join(AGENT_REPO_DIR, "outputs", "drafts"), { recursive: true });
164
+ const content = `# Quick Reply Draft (SEND FAILED)\n\nTo: ${to}\nSubject: ${subject}\nGenerated: ${new Date().toISOString()}\nError: ${err.message}\n\n---\n\n${text}\n`;
165
+ writeFileSync(draftPath, content);
166
+ return { sent: false, via: "draft_fallback", draft_path: draftPath, error: err.message };
167
+ }
168
+ }
169
+
170
+ // ---------------------------------------------------------------------------
171
+ // Cohort (the org's own messaging app)
172
+ // ---------------------------------------------------------------------------
173
+
174
+ /**
175
+ * Send onto Cohort. Goes through lib/org/messaging.sendMessage rather than the
176
+ * raw RPC so the outbound send-gate, the afterSend hook and cost attribution
177
+ * all still run.
178
+ *
179
+ * `idempotencySuffix` distinguishes the several distinct things we may say
180
+ * about ONE item — the acknowledgement, the FIRST progress update, the SECOND
181
+ * one, the answer, a failure notice. Without it they share a client message id
182
+ * and hq's server-side dedup swallows every message after the first, which
183
+ * would turn "never silent" into "acknowledged once and then silent forever".
184
+ *
185
+ * It must be distinct per MESSAGE, not per kind: two progress updates that both
186
+ * key on "progress" are one message as far as hq is concerned. Callers that can
187
+ * emit a kind more than once pass an ordinal (`progress-1`, `progress-2`). A
188
+ * genuine transport RETRY deliberately reuses the same suffix — that is the
189
+ * dedup doing its job.
190
+ */
191
+ async function sendCohortReply(item, text, o = {}) {
192
+ try {
193
+ const channel = item.channel_id || item.channel || "";
194
+ if (!channel) return { sent: false, via: null, error: "no channel_id on item" };
195
+ const { sendMessage } = await import("../../lib/org/messaging.mjs");
196
+ const { loadOrgConfig } = await import("../../lib/org/client.mjs");
197
+ const agentRoot = process.env.AGENT_ROOT || process.env.AGENT_DIR || process.cwd();
198
+ const base = item.message_id || item.id || item.raw_ref || Date.now();
199
+ const suffix = o.idempotencySuffix ? `-${o.idempotencySuffix}` : "";
200
+ const frame = await sendMessage(
201
+ {
202
+ channel,
203
+ body: text,
204
+ idempotencyId: `reply-${base}${suffix}`,
205
+ ...(item.thread_id ? { threadId: item.thread_id } : {}),
206
+ ...(Array.isArray(o.mentions) && o.mentions.length ? { mentions: o.mentions } : {}),
207
+ },
208
+ { cfg: loadOrgConfig(agentRoot), agentRoot },
209
+ );
210
+ if (frame && frame.ok) return { sent: true, via: "cohort", channel };
211
+ return { sent: false, via: null, channel, error: (frame && frame.error && frame.error.message) || "send failed" };
212
+ } catch (err) {
213
+ return { sent: false, via: null, error: err && err.message };
214
+ }
215
+ }
216
+
217
+ // ---------------------------------------------------------------------------
218
+ // The one public entry point
219
+ // ---------------------------------------------------------------------------
220
+
221
+ /**
222
+ * Deliver `text` to the human behind `item`. No generation, no model, no
223
+ * budget — only transport.
224
+ *
225
+ * A successful delivery writes a RECEIPT (lib/comms/receipts). That receipt is
226
+ * how the daemon later answers "did this session actually say anything to the
227
+ * requester", which is the question it previously could not ask and therefore
228
+ * always answered "yes".
229
+ *
230
+ * @param {object} item daemon inbox item
231
+ * @param {string} text exactly what the human will read
232
+ * @param {object} [o]
233
+ * @param {string} [o.kind] receipt kind: "ack" | "progress" | "failure" | "reply"
234
+ * @param {string[]} [o.mentions] org member ids to @-tag (cohort only)
235
+ * @param {function} [o.fetchImpl] test seam
236
+ * @returns {Promise<{sent:boolean, via:string|null, channel?:string, error?:string}>}
237
+ * NEVER throws — a transport failure is returned, so the caller can
238
+ * escalate rather than lose the message to an exception.
239
+ */
240
+ export async function deliver(item, text, o = {}) {
241
+ const kind = o.kind || "reply";
242
+ if (!item || !text || !String(text).trim()) {
243
+ return { sent: false, via: null, error: "nothing to deliver" };
244
+ }
245
+ // `permanent` says retrying changes nothing. A caller that cannot tell the
246
+ // difference between "hq blipped" and "there is no transport for telegram"
247
+ // retries both at the same cadence, which is how an impossible send becomes an
248
+ // infinite loop.
249
+ let result = { sent: false, via: null, permanent: true, error: `unsupported service ${item && item.service}` };
250
+ try {
251
+ if (item.service === "slack") {
252
+ const channel = resolveSlackChannel(item);
253
+ if (!channel) {
254
+ result = { sent: false, via: null, permanent: true, error: "could not resolve slack channel" };
255
+ } else {
256
+ await sendSlackMessage(channel, text, item.thread_id || null, o);
257
+ result = { sent: true, via: "slack_api", channel };
258
+ }
259
+ } else if (item.service === "gmail") {
260
+ const r = await sendGmailResponse(item, text);
261
+ result = { sent: r.sent, via: r.via, channel: r.to, ...(r.draft_path ? { draft_path: r.draft_path } : {}), ...(r.error ? { error: r.error } : {}) };
262
+ } else if (item.service === "cohort") {
263
+ const r = await sendCohortReply(item, text, { ...o, idempotencySuffix: o.idempotencySuffix || (kind === "reply" ? "" : kind) });
264
+ result = r;
265
+ }
266
+ } catch (err) {
267
+ result = { sent: false, via: null, error: err && err.message ? err.message : String(err) };
268
+ }
269
+
270
+ if (result.sent) {
271
+ recordOutbound({
272
+ service: item.service,
273
+ channel: result.channel || replyTargetOf(item),
274
+ kind,
275
+ via: result.via,
276
+ chars: String(text).length,
277
+ agentRoot: AGENT_REPO_DIR,
278
+ });
279
+ }
280
+ return result;
281
+ }
282
+
283
+ /**
284
+ * Deliver with bounded retry. Transport failures are usually transient (a 429,
285
+ * a dropped socket, hq restarting); a courtesy message that gives up on the
286
+ * first refusal is how silence happens.
287
+ *
288
+ * Deliberately short and bounded: this runs in front of a waiting human, so the
289
+ * total budget here is ~3s, not a minute. If it still fails, the caller gets
290
+ * `{sent:false}` and the obligation ledger keeps the debt.
291
+ *
292
+ * @param {number} [o.attempts=3]
293
+ * @param {number} [o.baseDelayMs=250]
294
+ * @param {(ms:number)=>Promise<void>} [o.sleep] test seam
295
+ */
296
+ export async function deliverWithRetry(item, text, o = {}) {
297
+ const attempts = Number.isFinite(o.attempts) ? o.attempts : 3;
298
+ const base = Number.isFinite(o.baseDelayMs) ? o.baseDelayMs : 250;
299
+ const sleep = o.sleep || ((ms) => new Promise((r) => setTimeout(r, ms)));
300
+ let last = { sent: false, via: null, error: "not attempted" };
301
+ for (let i = 0; i < attempts; i++) {
302
+ last = await deliver(item, text, o);
303
+ if (last.sent) return { ...last, attempts: i + 1 };
304
+ // A policy block is a decision, not a fault — retrying re-runs the same
305
+ // deterministic screen and gets the same answer. Stop and surface it.
306
+ if (last.permanent || (last.error && /send-gate blocked|blocked by send-gate|FORBIDDEN/i.test(last.error))) {
307
+ return { ...last, attempts: i + 1, permanent: true };
308
+ }
309
+ if (i < attempts - 1) await sleep(base * Math.pow(2, i));
310
+ }
311
+ return { ...last, attempts };
312
+ }
313
+
314
+ export default { deliver, deliverWithRetry, resolveSlackChannel, replyTargetOf, canDeliverTo, DELIVERABLE_SERVICES };
@@ -504,6 +504,16 @@ test("0.3: a finished inbox session writes a nonzero cost-ledger row from its re
504
504
  new URL("../cost/track-claude-usage.mjs", import.meta.url),
505
505
  join(dir, "scripts/cost/track-claude-usage.mjs")
506
506
  );
507
+ // The tracker imports the shared billing contract (lib/cost/ledger-row.mjs)
508
+ // relative to itself, so a synthetic agent root needs it too — otherwise the
509
+ // detached `record` spawn dies on import and writes nothing. (In the wild
510
+ // that failure is silent, which is why doctor's tripwire now goes RED when
511
+ // the daemon logs sessions that never reach the ledger.)
512
+ await fsp.mkdir(join(dir, "lib/cost"), { recursive: true });
513
+ await fsp.copyFile(
514
+ new URL("../../lib/cost/ledger-row.mjs", import.meta.url),
515
+ join(dir, "lib/cost/ledger-row.mjs")
516
+ );
507
517
 
508
518
  const proc = controllableProc();
509
519
  const restore = mod.setSpawnForTests(() => proc);
@@ -64,7 +64,16 @@ function admitFor(source, priority) {
64
64
  try {
65
65
  let mode = null;
66
66
  try {
67
- if (budgetGuard.dailyStatus({ agentRoot: AGENT_REPO_DIR }).essentialOnly) mode = "essential-only";
67
+ // The full ladder, not just the top rung: `mode` is
68
+ // normal|degraded|suspended|refused, banded against the seat's hq-funded
69
+ // envelope. Reading `.essentialOnly` here (as this did) collapsed four
70
+ // rungs into one and threw away both the degrade rung and the spawn-level
71
+ // refuse.
72
+ const st = budgetGuard.dailyStatus({ agentRoot: AGENT_REPO_DIR });
73
+ if (st.mode && st.mode !== "normal") {
74
+ mode = st.mode;
75
+ console.warn(`[dispatcher] budget ${st.mode}: ${st.pct}% of $${st.capUSD}/day (${st.capSource}) — gating ${source} work`);
76
+ }
68
77
  } catch { /* budget read best-effort */ }
69
78
  return governor.admit({ source, priority, mode }, governor.defaultDeps({ agentRoot: AGENT_REPO_DIR }));
70
79
  } catch {
@@ -242,6 +251,12 @@ export function parseUsageFromText(text) {
242
251
  const out = { ok: true, inputTokens, outputTokens };
243
252
  const cacheRead = Number(usage.cache_read_input_tokens);
244
253
  if (Number.isFinite(cacheRead)) out.cacheReadTokens = cacheRead;
254
+ // Cache CREATION tokens bill at 1.25x input and were previously dropped on
255
+ // the floor — on a cache-heavy workload that is a large slice of the real
256
+ // cost, and its absence is most of the gap between the local estimate and
257
+ // the CLI's total_cost_usd.
258
+ const cacheWrite = Number(usage.cache_creation_input_tokens);
259
+ if (Number.isFinite(cacheWrite)) out.cacheWriteTokens = cacheWrite;
245
260
  const totalCost = Number(obj.total_cost_usd);
246
261
  if (Number.isFinite(totalCost) && totalCost >= 0) out.totalCostUsd = totalCost;
247
262
  if (typeof obj.model === "string" && obj.model) {
@@ -253,9 +268,17 @@ export function parseUsageFromText(text) {
253
268
  /**
254
269
  * Append a TRUTHFUL cost-ledger row for a finished dispatcher session via
255
270
  * scripts/cost/track-claude-usage.mjs (source "dispatcher"). Real token counts
256
- * are parsed from the run's --output-format json stdout; on a parse failure we
257
- * pass NO token flags (tracker records 0) rather than fabricate zeros, and the
258
- * caller logs the gap. Best-effort + detached; never blocks the close path.
271
+ * are parsed from the run's --output-format json stdout.
272
+ *
273
+ * On a parse failure we pass `--tokens-unknown <reason>`, which records
274
+ * measurement:"unknown" with NULL tokens. Previously we passed no token flags
275
+ * at all and the tracker defaulted them to 0 — a row that read as a free
276
+ * session. "Omit the flags rather than fabricate zeros" was the intent, but the
277
+ * tracker fabricated them anyway, so a systematic parse regression would drive
278
+ * the day's measured spend toward $0 and the budget governor would go quiet
279
+ * exactly when it should have been shouting. Fail-open is fine; silent is not.
280
+ *
281
+ * Best-effort + detached; never blocks the close path.
259
282
  */
260
283
  function recordDispatcherCost({ stdout, model, durationMs, exitCode, decisionId }) {
261
284
  const usage = parseUsageFromText(stdout);
@@ -278,7 +301,20 @@ function recordDispatcherCost({ stdout, model, durationMs, exitCode, decisionId
278
301
  trackerArgs.push("--input-tokens", String(usage.inputTokens));
279
302
  trackerArgs.push("--output-tokens", String(usage.outputTokens));
280
303
  if (usage.cacheReadTokens != null) trackerArgs.push("--cache-read-tokens", String(usage.cacheReadTokens));
304
+ if (usage.cacheWriteTokens != null) trackerArgs.push("--cache-creation-tokens", String(usage.cacheWriteTokens));
281
305
  if (usage.totalCostUsd != null) trackerArgs.push("--total-cost-usd", String(usage.totalCostUsd));
306
+ } else {
307
+ // Explicit "we ran a session and could not measure it" marker. NEVER a 0.
308
+ trackerArgs.push("--tokens-unknown", String(usage.reason || "usage-parse-failed"));
309
+ try {
310
+ logSession({
311
+ event: "cost_usage_parse_failed",
312
+ reason: usage.reason || "unknown",
313
+ model: model || null,
314
+ duration_ms: durationMs,
315
+ exit_code: exitCode,
316
+ });
317
+ } catch { /* logging must not break the close path */ }
282
318
  }
283
319
  spawn(process.execPath, trackerArgs, {
284
320
  stdio: "ignore",
@@ -732,7 +768,17 @@ export function canDispatchBacklog(item) {
732
768
  * which leave the durable on-disk item untouched for the next sweep anyway.
733
769
  */
734
770
  export function dispatch(prompt, item, classResult, source = "inbox", opts = {}) {
735
- const entry = { prompt, item, classResult, source, onClose: typeof opts.onClose === "function" ? opts.onClose : null };
771
+ // `obligationKey` names the debt this session was spawned to discharge. It is
772
+ // carried into the child's environment so the CLI send lanes (the ONLY way a
773
+ // session is allowed to reply) can stamp their delivery receipts with it.
774
+ // Without it a receipt says only "somebody spoke into that room", and a room
775
+ // is shared: two asks in one DM produce two debts, the first reply discharges
776
+ // both, and the unanswered one is closed as answered with nobody told.
777
+ const entry = {
778
+ prompt, item, classResult, source,
779
+ obligationKey: opts.obligationKey || null,
780
+ onClose: typeof opts.onClose === "function" ? opts.onClose : null,
781
+ };
736
782
  const priority = classResult.priority;
737
783
  const isPriorityInbox = source === "inbox" && (priority === "critical" || priority === "high");
738
784
 
@@ -892,7 +938,12 @@ export function _hasReDrainArmed() {
892
938
  function currentBudgetBand() {
893
939
  try {
894
940
  const st = budgetGuard.dailyStatus({ agentRoot: AGENT_REPO_DIR });
895
- return budgetLadder(st.spentUSD, st.capUSD).band;
941
+ // Pass the GOVERNOR'S band rather than letting budgetLadder re-derive one
942
+ // from (spent, cap). dailyStatus has already reconciled the day reading, the
943
+ // month reading, the cap latch and blindness; re-deriving here collapsed to
944
+ // band 0 whenever the monthly envelope was spent out (cap === 0), so the
945
+ // router stopped degrading exactly when the seat was furthest over.
946
+ return budgetLadder(st.spentUSD, st.capUSD, { band: st.band }).band;
896
947
  } catch {
897
948
  return 0;
898
949
  }
@@ -1088,6 +1139,13 @@ function spawnSession(entry) {
1088
1139
  ANTHROPIC_AUTH_TOKEN: "",
1089
1140
  ...(target.envForSpawn || {}),
1090
1141
  };
1142
+ // ATTRIBUTION for delivery receipts. Set after the merges above so a router
1143
+ // target's envForSpawn can never drop it — a retargeted session replies
1144
+ // through exactly the same CLI lanes and owes exactly the same receipt.
1145
+ // These are plain identifiers, never credentials, so the §7.3 allowlist scrub
1146
+ // in buildChildEnv has nothing to object to.
1147
+ spawnEnv.MAESTRO_SESSION_ID = sessionId;
1148
+ if (entry.obligationKey) spawnEnv.MAESTRO_OBLIGATION_KEY = String(entry.obligationKey);
1091
1149
 
1092
1150
  const proc = _spawn(CLAUDE_BIN, args, {
1093
1151
  cwd: AGENT_REPO_DIR,
@@ -0,0 +1,68 @@
1
+ /**
2
+ * responder-cost.test.mjs — every responder spawn is billed, including the ones
3
+ * that FAIL.
4
+ *
5
+ * The defect this guards: `recordResponderCost` was reached only at the bottom
6
+ * of the `close` handler, after `if (settled) return;` and after
7
+ * `if (code !== 0) { reject; return; }`. So the two most expensive outcomes the
8
+ * daemon has — a 60s CLI timeout and a non-zero exit — wrote NO ledger row at
9
+ * all, not even `--tokens-unknown`. On the live seat that was 15 timeouts on
10
+ * 2026-08-11 and 10 on 2026-08-12: a full minute of paid model work each,
11
+ * invisible to the governor and to doctor.
12
+ *
13
+ * Run: `node --test scripts/daemon/responder-cost.test.mjs`
14
+ */
15
+
16
+ import { test } from "node:test";
17
+ import assert from "node:assert/strict";
18
+ import { readFileSync } from "node:fs";
19
+ import { join, dirname } from "node:path";
20
+ import { fileURLToPath } from "node:url";
21
+
22
+ import { unmeasuredReasonFor } from "./responder.mjs";
23
+
24
+ const HERE = dirname(fileURLToPath(import.meta.url));
25
+ const SRC = readFileSync(join(HERE, "responder.mjs"), "utf-8");
26
+
27
+ test("a TIMED-OUT session is billed as unmeasured, naming the timeout", () => {
28
+ const r = unmeasuredReasonFor({ parsed: null, timedOut: true, exitCode: null, parseError: null, timeoutMs: 60_000 });
29
+ assert.equal(r, "cli-timeout-60000ms");
30
+ });
31
+
32
+ test("a NON-ZERO exit is billed as unmeasured, naming the code", () => {
33
+ const r = unmeasuredReasonFor({ parsed: null, timedOut: false, exitCode: 1, parseError: null, timeoutMs: 60_000 });
34
+ assert.equal(r, "cli-exit-1");
35
+ });
36
+
37
+ test("an unparseable envelope on a clean exit is billed as unmeasured", () => {
38
+ const r = unmeasuredReasonFor({ parsed: null, timedOut: false, exitCode: 0, parseError: new Error("bad json"), timeoutMs: 60_000 });
39
+ assert.equal(r, "json-parse-failed");
40
+ });
41
+
42
+ test("a parsed envelope is measured — no reason at all", () => {
43
+ const r = unmeasuredReasonFor({ parsed: { usage: {} }, timedOut: true, exitCode: 1, parseError: new Error("x"), timeoutMs: 60_000 });
44
+ assert.equal(r, null, "a real envelope wins over every failure signal");
45
+ });
46
+
47
+ test("the timeout path never masks itself as a plain non-zero exit", () => {
48
+ // SIGTERM'd processes close with a code, and the operator needs to know WHICH
49
+ // failure burned the minute — the fixes are different.
50
+ const r = unmeasuredReasonFor({ parsed: null, timedOut: true, exitCode: 143, parseError: null, timeoutMs: 60_000 });
51
+ assert.equal(r, "cli-timeout-60000ms");
52
+ });
53
+
54
+ test("STRUCTURAL: the cost record happens BEFORE the close handler's early returns", () => {
55
+ // This is an ordering bug in code that cannot be unit-tested without a spawn
56
+ // harness, so the ordering itself is the assertion. If someone moves the
57
+ // recording back below `if (settled) return;`, every timeout goes unbilled
58
+ // again — silently, because the reply still works.
59
+ const close = SRC.slice(SRC.indexOf('proc.on("close"'));
60
+ const record = close.indexOf("recordResponderCost({");
61
+ const settledGuard = close.indexOf("if (settled) return;");
62
+ const exitGuard = close.indexOf("if (code !== 0) {");
63
+
64
+ assert.ok(record > 0 && settledGuard > 0 && exitGuard > 0, "all three markers present");
65
+ assert.ok(record < settledGuard, "cost is recorded before the settled early-return (the TIMEOUT path)");
66
+ assert.ok(record < exitGuard, "cost is recorded before the non-zero-exit early-return");
67
+ assert.match(close.slice(0, record), /costRecorded/, "and exactly once per spawn");
68
+ });