@cohortapp/agent-sdk 2.5.1 → 2.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/maestro.mjs +185 -88
- package/bin/maestro.test.mjs +175 -48
- package/docs/runbooks/backup-restore.md +65 -33
- package/framework-features.json +4 -4
- package/lib/backup/policy.mjs +710 -0
- package/lib/backup/policy.test.mjs +305 -0
- package/lib/budget-escalate.mjs +133 -0
- package/lib/budget-escalate.test.mjs +232 -0
- package/lib/budget-guard.envelope.test.mjs +476 -0
- package/lib/budget-guard.mjs +853 -75
- package/lib/budget-guard.test.mjs +91 -42
- package/lib/cadences.mjs +33 -0
- package/lib/channels/orgmail/adapter.mjs +88 -3
- package/lib/channels/orgmail/adapter.test.mjs +137 -0
- package/lib/channels/repeat-suppressor.mjs +198 -0
- package/lib/channels/repeat-suppressor.test.mjs +134 -0
- package/lib/comms/receipts.mjs +297 -0
- package/lib/cost/ledger-row.mjs +333 -0
- package/lib/cost/ledger-row.test.mjs +183 -0
- package/lib/execution/drive.mjs +28 -1
- package/lib/execution/effects.mjs +191 -12
- package/lib/execution/effects.test.mjs +50 -11
- package/lib/goals/admission.mjs +13 -1
- package/lib/goals/admission.test.mjs +26 -1
- package/lib/goals/loop.mjs +13 -0
- package/lib/kpi-sensors.test.mjs +3 -0
- package/lib/mandate/cache.mjs +13 -5
- package/lib/mandate/derive.mjs +146 -21
- package/lib/mandate/derive.test.mjs +50 -6
- package/lib/mandate/model.mjs +32 -4
- package/lib/mandate/refresh.test.mjs +16 -2
- package/lib/mcp/server.test.mjs +12 -3
- package/lib/model-router/economics.mjs +107 -76
- package/lib/model-router/economics.test.mjs +64 -46
- package/lib/model-router/integration-coverage.test.mjs +39 -37
- package/lib/model-router/ledger.mjs +75 -22
- package/lib/model-router/ledger.test.mjs +35 -2
- package/lib/org/client.mjs +14 -0
- package/lib/org/cost-sync.mjs +16 -2
- package/lib/org/doctor.mjs +62 -1
- package/lib/org/doctor.test.mjs +36 -3
- package/lib/org/email-remedy.mjs +49 -0
- package/lib/org/engagement-ledger.mjs +376 -0
- package/lib/org/engagement-ledger.test.mjs +112 -0
- package/lib/org/engagement.mjs +1056 -0
- package/lib/org/engagement.test.mjs +739 -0
- package/lib/org/messaging.mjs +230 -3
- package/lib/org/messaging.test.mjs +110 -1
- package/lib/org/param-contract.mjs +56 -2
- package/lib/org/param-contract.test.mjs +26 -0
- package/lib/org/protocol.checksum +1 -1
- package/lib/org/protocol.mjs +5 -0
- package/lib/org/protocol.test.mjs +7 -1
- package/lib/org/tool-surface.mjs +506 -10
- package/lib/org/tool-surface.test.mjs +191 -7
- package/lib/org/ui-parity.mjs +333 -6
- package/lib/org/ui-parity.test.mjs +96 -3
- package/lib/org/work-ledger.mjs +241 -0
- package/lib/org/work-ledger.test.mjs +237 -0
- package/lib/plan/adoption-e2e.test.mjs +366 -0
- package/lib/plan/budget-enforcement.test.mjs +400 -0
- package/lib/plan/budget-runtime.mjs +215 -0
- package/lib/plan/compile.mjs +201 -5
- package/lib/plan/compile.test.mjs +19 -5
- package/lib/plan/emit.mjs +8 -0
- package/lib/plan/emit.test.mjs +18 -0
- package/lib/resource-governor.mjs +58 -12
- package/lib/resource-governor.test.mjs +41 -1
- package/lib/security/audit-engine.mjs +45 -8
- package/lib/security/audit-engine.test.mjs +35 -0
- package/lib/setup/enroll-from-cohort.mjs +14 -1
- package/lib/setup/sections/mandate.mjs +48 -7
- package/lib/setup/sections/mandate.test.mjs +17 -2
- package/lib/setup/sections/orgmail.mjs +10 -2
- package/lib/setup/state.mjs +83 -2
- package/lib/telemetry/collect.mjs +360 -20
- package/lib/telemetry/collect.test.mjs +266 -0
- package/package.json +1 -1
- package/scripts/cost/track-claude-usage.mjs +207 -48
- package/scripts/cost/track-claude-usage.test.mjs +148 -0
- package/scripts/daemon/agent-daemon.mjs +315 -17
- package/scripts/daemon/assurance-e2e.test.mjs +421 -0
- package/scripts/daemon/assurance.mjs +944 -0
- package/scripts/daemon/assurance.test.mjs +668 -0
- package/scripts/daemon/cadence-consumer-governance.test.mjs +56 -0
- package/scripts/daemon/cadence-consumer.mjs +147 -9
- package/scripts/daemon/cadence-consumer.test.mjs +6 -0
- package/scripts/daemon/cadence-handlers.mjs +158 -0
- package/scripts/daemon/cadence-handlers.test.mjs +64 -0
- package/scripts/daemon/deliver.mjs +314 -0
- package/scripts/daemon/dispatcher-governance.test.mjs +10 -0
- package/scripts/daemon/dispatcher.mjs +64 -6
- package/scripts/daemon/responder-cost.test.mjs +68 -0
- package/scripts/daemon/responder.mjs +351 -298
- package/scripts/local-triggers/generate-plists.test.mjs +7 -4
- package/scripts/maintenance/backup-run.mjs +415 -0
- package/scripts/maintenance/backup-to-cloud.sh +16 -116
- package/scripts/org/send-orgmail.mjs +16 -0
- package/scripts/record-receipt.sh +63 -0
- package/scripts/restore-from-backup.sh +14 -3
- package/scripts/restore-from-backup.test.mjs +8 -5
- package/scripts/send-email-threaded.py +47 -0
- package/scripts/send-sms.sh +4 -0
- package/scripts/send-whatsapp.sh +4 -0
- package/scripts/setup/init-backup.mjs +93 -38
- package/scripts/slack-send.sh +12 -0
|
@@ -0,0 +1,314 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* deliver.mjs — put a specific string in front of a specific human. No model.
|
|
3
|
+
*
|
|
4
|
+
* This module exists because the acknowledgement was an LLM call.
|
|
5
|
+
*
|
|
6
|
+
* The old holding-message path spawned a full `claude --print` child, under a
|
|
7
|
+
* 60s hard cap, on a machine sitting at 95-99% memory, purely to write "let me
|
|
8
|
+
* look into it". It lost that race roughly two times in three (20 of 32
|
|
9
|
+
* observed attempts died on `claude CLI timed out after 60000ms`), and when it
|
|
10
|
+
* lost, the failure was caught, logged, and the human was told NOTHING while a
|
|
11
|
+
* 15-45 minute session ran behind a typing indicator.
|
|
12
|
+
*
|
|
13
|
+
* Everything on this path is therefore synchronous string work plus one HTTP
|
|
14
|
+
* call. There is no generation step that can time out, no model to be starved
|
|
15
|
+
* of memory, and no budget that can refuse it. It is the floor beneath every
|
|
16
|
+
* "never silent" guarantee in assurance.mjs: whatever else fails, THIS can run.
|
|
17
|
+
*
|
|
18
|
+
* The three per-service senders were previously private to responder.mjs, so
|
|
19
|
+
* nothing but a generated reply could reach a human. They are now here, and
|
|
20
|
+
* responder.mjs imports them — one delivery path, used by generated replies,
|
|
21
|
+
* acknowledgements, progress updates and failure notices alike.
|
|
22
|
+
*
|
|
23
|
+
* @module scripts/daemon/deliver
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
"use strict";
|
|
27
|
+
|
|
28
|
+
import { execFileSync } from "child_process";
|
|
29
|
+
import { mkdirSync, writeFileSync } from "fs";
|
|
30
|
+
import { join } from "path";
|
|
31
|
+
import { screenOutbound } from "../../lib/comms/send-gate.mjs";
|
|
32
|
+
import { getHookBus } from "../../lib/hooks/bus.mjs";
|
|
33
|
+
import { recordOutbound } from "../../lib/comms/receipts.mjs";
|
|
34
|
+
|
|
35
|
+
const AGENT_REPO_DIR = process.env.AGENT_DIR || join(new URL(".", import.meta.url).pathname, "../..");
|
|
36
|
+
|
|
37
|
+
function today() {
|
|
38
|
+
return new Date().toISOString().split("T")[0];
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
function getSlackToken() {
|
|
42
|
+
return process.env.SLACK_USER_TOKEN || process.env.SLACK_BOT_TOKEN;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
// ---------------------------------------------------------------------------
|
|
46
|
+
// Channel resolution
|
|
47
|
+
// ---------------------------------------------------------------------------
|
|
48
|
+
|
|
49
|
+
/** Resolve a Slack channel id from a daemon inbox item. */
|
|
50
|
+
export function resolveSlackChannel(item) {
|
|
51
|
+
if (!item) return null;
|
|
52
|
+
// Direct channel ID (starts with D for DM, C for channel)
|
|
53
|
+
if (item.channel && /^[DC][A-Z0-9]{8,}$/.test(item.channel)) return item.channel;
|
|
54
|
+
if (item.channel_id) return item.channel_id;
|
|
55
|
+
|
|
56
|
+
// Extract from raw_ref format: "slack:D099N1JGKRQ:1775331885.690669"
|
|
57
|
+
if (item.raw_ref) {
|
|
58
|
+
const match = item.raw_ref.match(/slack:([DC][A-Z0-9]+):/);
|
|
59
|
+
if (match) return match[1];
|
|
60
|
+
const bare = item.raw_ref.match(/^([DC][A-Z0-9]{8,})$/);
|
|
61
|
+
if (bare) return bare[1];
|
|
62
|
+
}
|
|
63
|
+
return null;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* The channel a reply to this item belongs in, whatever the service — the key
|
|
68
|
+
* both the receipt ledger and the obligation ledger are keyed on.
|
|
69
|
+
*
|
|
70
|
+
* Routing is by ROOM, not by person: the room the inbound arrived in is the
|
|
71
|
+
* room the answer belongs in.
|
|
72
|
+
*/
|
|
73
|
+
export function replyTargetOf(item) {
|
|
74
|
+
if (!item) return null;
|
|
75
|
+
if (item.service === "slack") return resolveSlackChannel(item);
|
|
76
|
+
if (item.service === "gmail") return item.sender_email || item.sender || null;
|
|
77
|
+
return item.channel_id || item.channel || null;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* The services this module can actually put a string in front of a human on.
|
|
82
|
+
*
|
|
83
|
+
* The daemon polls more than this: telegram, whatsapp, orgmail, voice, calendar
|
|
84
|
+
* and a second Gmail account all produce inbox items. `deliver` returns
|
|
85
|
+
* "unsupported service X" for those — a fact that used to be discovered only at
|
|
86
|
+
* send time, once per sweep tick, forever.
|
|
87
|
+
*
|
|
88
|
+
* Naming the set explicitly lets the caller ask BEFORE it promises anything,
|
|
89
|
+
* which is the difference between "I can't reach you here, so an operator has
|
|
90
|
+
* been told" and an apology retried 1,440 times a day into a void.
|
|
91
|
+
*/
|
|
92
|
+
export const DELIVERABLE_SERVICES = Object.freeze(["slack", "gmail", "cohort"]);
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Can `deliver` reach the human behind this item at all? Requires both a
|
|
96
|
+
* transport for the service AND a resolvable room to write into.
|
|
97
|
+
*/
|
|
98
|
+
export function canDeliverTo(item) {
|
|
99
|
+
if (!item || !item.service) return false;
|
|
100
|
+
if (!DELIVERABLE_SERVICES.includes(item.service)) return false;
|
|
101
|
+
return Boolean(replyTargetOf(item));
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
// ---------------------------------------------------------------------------
|
|
105
|
+
// Slack
|
|
106
|
+
// ---------------------------------------------------------------------------
|
|
107
|
+
|
|
108
|
+
async function sendSlackMessage(channel, text, threadTs = null, o = {}) {
|
|
109
|
+
// Unified send-gate (P0-3): the daemon path passes the same chokepoint as the
|
|
110
|
+
// shell senders + BaseAdapter — banned-phrase, AI disclosure, and
|
|
111
|
+
// information-barrier screening at one place. Block on deny.
|
|
112
|
+
try {
|
|
113
|
+
const gate = await screenOutbound({ channel: "slack", recipient: channel, text, agentRoot: AGENT_REPO_DIR });
|
|
114
|
+
if (gate && gate.allow === false) {
|
|
115
|
+
throw new Error(`send-gate blocked Slack reply: ${gate.reason}`);
|
|
116
|
+
}
|
|
117
|
+
} catch (err) {
|
|
118
|
+
if (/send-gate blocked/.test(err.message)) throw err; // a real block propagates
|
|
119
|
+
// gate infra error (module/policy unreadable): fail-open for internal Slack
|
|
120
|
+
// channels (matches send-gate's internal posture) — never silently drop.
|
|
121
|
+
console.warn(`[deliver] send-gate check errored (allowing internal Slack): ${err.message}`);
|
|
122
|
+
try {
|
|
123
|
+
Promise.resolve(getHookBus().emit("onGuardFail", { guard: "send_gate", channel: "slack", recipient: channel, reason: err.message, failed_open: true }))
|
|
124
|
+
.catch(() => { /* bus emit is isolated; never propagate */ });
|
|
125
|
+
} catch { /* never let telemetry break the send path */ }
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
// Always use user token — bot tokens can't access DM channels
|
|
129
|
+
const token = getSlackToken();
|
|
130
|
+
if (!token) throw new Error("No Slack token available (set SLACK_USER_TOKEN in .env)");
|
|
131
|
+
|
|
132
|
+
const body = { channel, text, ...(threadTs ? { thread_ts: threadTs } : {}) };
|
|
133
|
+
const fetchImpl = o.fetchImpl || fetch;
|
|
134
|
+
const res = await fetchImpl("https://slack.com/api/chat.postMessage", {
|
|
135
|
+
method: "POST",
|
|
136
|
+
headers: { "Authorization": `Bearer ${token}`, "Content-Type": "application/json" },
|
|
137
|
+
body: JSON.stringify(body),
|
|
138
|
+
});
|
|
139
|
+
const data = await res.json();
|
|
140
|
+
if (!data.ok) throw new Error(`Slack API error: ${data.error}`);
|
|
141
|
+
return data;
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
// ---------------------------------------------------------------------------
|
|
145
|
+
// Gmail
|
|
146
|
+
// ---------------------------------------------------------------------------
|
|
147
|
+
|
|
148
|
+
async function sendGmailResponse(item, text) {
|
|
149
|
+
const to = item.sender_email || item.sender;
|
|
150
|
+
const subject = `Re: ${item.subject || "(no subject)"}`;
|
|
151
|
+
const sendScript = join(AGENT_REPO_DIR, "scripts", "send-email-threaded.py");
|
|
152
|
+
|
|
153
|
+
try {
|
|
154
|
+
const args = [sendScript, to, subject, text];
|
|
155
|
+
if (item.subject) args.push("--reply-to-subject", item.subject);
|
|
156
|
+
execFileSync("python3", args, { cwd: AGENT_REPO_DIR, timeout: 30000, encoding: "utf-8", env: { ...process.env } });
|
|
157
|
+
return { sent: true, via: "smtp", to };
|
|
158
|
+
} catch (err) {
|
|
159
|
+
console.error(`[deliver] Gmail send failed for ${to}: ${err.message}`);
|
|
160
|
+
// Fall back to draft file so the response is not lost
|
|
161
|
+
const draftPath = join(AGENT_REPO_DIR, "outputs", "drafts",
|
|
162
|
+
`${today()}-quick-reply-${String(item.sender || "unknown").replace(/[^a-z0-9]/gi, "-")}.md`);
|
|
163
|
+
mkdirSync(join(AGENT_REPO_DIR, "outputs", "drafts"), { recursive: true });
|
|
164
|
+
const content = `# Quick Reply Draft (SEND FAILED)\n\nTo: ${to}\nSubject: ${subject}\nGenerated: ${new Date().toISOString()}\nError: ${err.message}\n\n---\n\n${text}\n`;
|
|
165
|
+
writeFileSync(draftPath, content);
|
|
166
|
+
return { sent: false, via: "draft_fallback", draft_path: draftPath, error: err.message };
|
|
167
|
+
}
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
// ---------------------------------------------------------------------------
|
|
171
|
+
// Cohort (the org's own messaging app)
|
|
172
|
+
// ---------------------------------------------------------------------------
|
|
173
|
+
|
|
174
|
+
/**
|
|
175
|
+
* Send onto Cohort. Goes through lib/org/messaging.sendMessage rather than the
|
|
176
|
+
* raw RPC so the outbound send-gate, the afterSend hook and cost attribution
|
|
177
|
+
* all still run.
|
|
178
|
+
*
|
|
179
|
+
* `idempotencySuffix` distinguishes the several distinct things we may say
|
|
180
|
+
* about ONE item — the acknowledgement, the FIRST progress update, the SECOND
|
|
181
|
+
* one, the answer, a failure notice. Without it they share a client message id
|
|
182
|
+
* and hq's server-side dedup swallows every message after the first, which
|
|
183
|
+
* would turn "never silent" into "acknowledged once and then silent forever".
|
|
184
|
+
*
|
|
185
|
+
* It must be distinct per MESSAGE, not per kind: two progress updates that both
|
|
186
|
+
* key on "progress" are one message as far as hq is concerned. Callers that can
|
|
187
|
+
* emit a kind more than once pass an ordinal (`progress-1`, `progress-2`). A
|
|
188
|
+
* genuine transport RETRY deliberately reuses the same suffix — that is the
|
|
189
|
+
* dedup doing its job.
|
|
190
|
+
*/
|
|
191
|
+
async function sendCohortReply(item, text, o = {}) {
|
|
192
|
+
try {
|
|
193
|
+
const channel = item.channel_id || item.channel || "";
|
|
194
|
+
if (!channel) return { sent: false, via: null, error: "no channel_id on item" };
|
|
195
|
+
const { sendMessage } = await import("../../lib/org/messaging.mjs");
|
|
196
|
+
const { loadOrgConfig } = await import("../../lib/org/client.mjs");
|
|
197
|
+
const agentRoot = process.env.AGENT_ROOT || process.env.AGENT_DIR || process.cwd();
|
|
198
|
+
const base = item.message_id || item.id || item.raw_ref || Date.now();
|
|
199
|
+
const suffix = o.idempotencySuffix ? `-${o.idempotencySuffix}` : "";
|
|
200
|
+
const frame = await sendMessage(
|
|
201
|
+
{
|
|
202
|
+
channel,
|
|
203
|
+
body: text,
|
|
204
|
+
idempotencyId: `reply-${base}${suffix}`,
|
|
205
|
+
...(item.thread_id ? { threadId: item.thread_id } : {}),
|
|
206
|
+
...(Array.isArray(o.mentions) && o.mentions.length ? { mentions: o.mentions } : {}),
|
|
207
|
+
},
|
|
208
|
+
{ cfg: loadOrgConfig(agentRoot), agentRoot },
|
|
209
|
+
);
|
|
210
|
+
if (frame && frame.ok) return { sent: true, via: "cohort", channel };
|
|
211
|
+
return { sent: false, via: null, channel, error: (frame && frame.error && frame.error.message) || "send failed" };
|
|
212
|
+
} catch (err) {
|
|
213
|
+
return { sent: false, via: null, error: err && err.message };
|
|
214
|
+
}
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
// ---------------------------------------------------------------------------
|
|
218
|
+
// The one public entry point
|
|
219
|
+
// ---------------------------------------------------------------------------
|
|
220
|
+
|
|
221
|
+
/**
|
|
222
|
+
* Deliver `text` to the human behind `item`. No generation, no model, no
|
|
223
|
+
* budget — only transport.
|
|
224
|
+
*
|
|
225
|
+
* A successful delivery writes a RECEIPT (lib/comms/receipts). That receipt is
|
|
226
|
+
* how the daemon later answers "did this session actually say anything to the
|
|
227
|
+
* requester", which is the question it previously could not ask and therefore
|
|
228
|
+
* always answered "yes".
|
|
229
|
+
*
|
|
230
|
+
* @param {object} item daemon inbox item
|
|
231
|
+
* @param {string} text exactly what the human will read
|
|
232
|
+
* @param {object} [o]
|
|
233
|
+
* @param {string} [o.kind] receipt kind: "ack" | "progress" | "failure" | "reply"
|
|
234
|
+
* @param {string[]} [o.mentions] org member ids to @-tag (cohort only)
|
|
235
|
+
* @param {function} [o.fetchImpl] test seam
|
|
236
|
+
* @returns {Promise<{sent:boolean, via:string|null, channel?:string, error?:string}>}
|
|
237
|
+
* NEVER throws — a transport failure is returned, so the caller can
|
|
238
|
+
* escalate rather than lose the message to an exception.
|
|
239
|
+
*/
|
|
240
|
+
export async function deliver(item, text, o = {}) {
|
|
241
|
+
const kind = o.kind || "reply";
|
|
242
|
+
if (!item || !text || !String(text).trim()) {
|
|
243
|
+
return { sent: false, via: null, error: "nothing to deliver" };
|
|
244
|
+
}
|
|
245
|
+
// `permanent` says retrying changes nothing. A caller that cannot tell the
|
|
246
|
+
// difference between "hq blipped" and "there is no transport for telegram"
|
|
247
|
+
// retries both at the same cadence, which is how an impossible send becomes an
|
|
248
|
+
// infinite loop.
|
|
249
|
+
let result = { sent: false, via: null, permanent: true, error: `unsupported service ${item && item.service}` };
|
|
250
|
+
try {
|
|
251
|
+
if (item.service === "slack") {
|
|
252
|
+
const channel = resolveSlackChannel(item);
|
|
253
|
+
if (!channel) {
|
|
254
|
+
result = { sent: false, via: null, permanent: true, error: "could not resolve slack channel" };
|
|
255
|
+
} else {
|
|
256
|
+
await sendSlackMessage(channel, text, item.thread_id || null, o);
|
|
257
|
+
result = { sent: true, via: "slack_api", channel };
|
|
258
|
+
}
|
|
259
|
+
} else if (item.service === "gmail") {
|
|
260
|
+
const r = await sendGmailResponse(item, text);
|
|
261
|
+
result = { sent: r.sent, via: r.via, channel: r.to, ...(r.draft_path ? { draft_path: r.draft_path } : {}), ...(r.error ? { error: r.error } : {}) };
|
|
262
|
+
} else if (item.service === "cohort") {
|
|
263
|
+
const r = await sendCohortReply(item, text, { ...o, idempotencySuffix: o.idempotencySuffix || (kind === "reply" ? "" : kind) });
|
|
264
|
+
result = r;
|
|
265
|
+
}
|
|
266
|
+
} catch (err) {
|
|
267
|
+
result = { sent: false, via: null, error: err && err.message ? err.message : String(err) };
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
if (result.sent) {
|
|
271
|
+
recordOutbound({
|
|
272
|
+
service: item.service,
|
|
273
|
+
channel: result.channel || replyTargetOf(item),
|
|
274
|
+
kind,
|
|
275
|
+
via: result.via,
|
|
276
|
+
chars: String(text).length,
|
|
277
|
+
agentRoot: AGENT_REPO_DIR,
|
|
278
|
+
});
|
|
279
|
+
}
|
|
280
|
+
return result;
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
/**
|
|
284
|
+
* Deliver with bounded retry. Transport failures are usually transient (a 429,
|
|
285
|
+
* a dropped socket, hq restarting); a courtesy message that gives up on the
|
|
286
|
+
* first refusal is how silence happens.
|
|
287
|
+
*
|
|
288
|
+
* Deliberately short and bounded: this runs in front of a waiting human, so the
|
|
289
|
+
* total budget here is ~3s, not a minute. If it still fails, the caller gets
|
|
290
|
+
* `{sent:false}` and the obligation ledger keeps the debt.
|
|
291
|
+
*
|
|
292
|
+
* @param {number} [o.attempts=3]
|
|
293
|
+
* @param {number} [o.baseDelayMs=250]
|
|
294
|
+
* @param {(ms:number)=>Promise<void>} [o.sleep] test seam
|
|
295
|
+
*/
|
|
296
|
+
export async function deliverWithRetry(item, text, o = {}) {
|
|
297
|
+
const attempts = Number.isFinite(o.attempts) ? o.attempts : 3;
|
|
298
|
+
const base = Number.isFinite(o.baseDelayMs) ? o.baseDelayMs : 250;
|
|
299
|
+
const sleep = o.sleep || ((ms) => new Promise((r) => setTimeout(r, ms)));
|
|
300
|
+
let last = { sent: false, via: null, error: "not attempted" };
|
|
301
|
+
for (let i = 0; i < attempts; i++) {
|
|
302
|
+
last = await deliver(item, text, o);
|
|
303
|
+
if (last.sent) return { ...last, attempts: i + 1 };
|
|
304
|
+
// A policy block is a decision, not a fault — retrying re-runs the same
|
|
305
|
+
// deterministic screen and gets the same answer. Stop and surface it.
|
|
306
|
+
if (last.permanent || (last.error && /send-gate blocked|blocked by send-gate|FORBIDDEN/i.test(last.error))) {
|
|
307
|
+
return { ...last, attempts: i + 1, permanent: true };
|
|
308
|
+
}
|
|
309
|
+
if (i < attempts - 1) await sleep(base * Math.pow(2, i));
|
|
310
|
+
}
|
|
311
|
+
return { ...last, attempts };
|
|
312
|
+
}
|
|
313
|
+
|
|
314
|
+
export default { deliver, deliverWithRetry, resolveSlackChannel, replyTargetOf, canDeliverTo, DELIVERABLE_SERVICES };
|
|
@@ -504,6 +504,16 @@ test("0.3: a finished inbox session writes a nonzero cost-ledger row from its re
|
|
|
504
504
|
new URL("../cost/track-claude-usage.mjs", import.meta.url),
|
|
505
505
|
join(dir, "scripts/cost/track-claude-usage.mjs")
|
|
506
506
|
);
|
|
507
|
+
// The tracker imports the shared billing contract (lib/cost/ledger-row.mjs)
|
|
508
|
+
// relative to itself, so a synthetic agent root needs it too — otherwise the
|
|
509
|
+
// detached `record` spawn dies on import and writes nothing. (In the wild
|
|
510
|
+
// that failure is silent, which is why doctor's tripwire now goes RED when
|
|
511
|
+
// the daemon logs sessions that never reach the ledger.)
|
|
512
|
+
await fsp.mkdir(join(dir, "lib/cost"), { recursive: true });
|
|
513
|
+
await fsp.copyFile(
|
|
514
|
+
new URL("../../lib/cost/ledger-row.mjs", import.meta.url),
|
|
515
|
+
join(dir, "lib/cost/ledger-row.mjs")
|
|
516
|
+
);
|
|
507
517
|
|
|
508
518
|
const proc = controllableProc();
|
|
509
519
|
const restore = mod.setSpawnForTests(() => proc);
|
|
@@ -64,7 +64,16 @@ function admitFor(source, priority) {
|
|
|
64
64
|
try {
|
|
65
65
|
let mode = null;
|
|
66
66
|
try {
|
|
67
|
-
|
|
67
|
+
// The full ladder, not just the top rung: `mode` is
|
|
68
|
+
// normal|degraded|suspended|refused, banded against the seat's hq-funded
|
|
69
|
+
// envelope. Reading `.essentialOnly` here (as this did) collapsed four
|
|
70
|
+
// rungs into one and threw away both the degrade rung and the spawn-level
|
|
71
|
+
// refuse.
|
|
72
|
+
const st = budgetGuard.dailyStatus({ agentRoot: AGENT_REPO_DIR });
|
|
73
|
+
if (st.mode && st.mode !== "normal") {
|
|
74
|
+
mode = st.mode;
|
|
75
|
+
console.warn(`[dispatcher] budget ${st.mode}: ${st.pct}% of $${st.capUSD}/day (${st.capSource}) — gating ${source} work`);
|
|
76
|
+
}
|
|
68
77
|
} catch { /* budget read best-effort */ }
|
|
69
78
|
return governor.admit({ source, priority, mode }, governor.defaultDeps({ agentRoot: AGENT_REPO_DIR }));
|
|
70
79
|
} catch {
|
|
@@ -242,6 +251,12 @@ export function parseUsageFromText(text) {
|
|
|
242
251
|
const out = { ok: true, inputTokens, outputTokens };
|
|
243
252
|
const cacheRead = Number(usage.cache_read_input_tokens);
|
|
244
253
|
if (Number.isFinite(cacheRead)) out.cacheReadTokens = cacheRead;
|
|
254
|
+
// Cache CREATION tokens bill at 1.25x input and were previously dropped on
|
|
255
|
+
// the floor — on a cache-heavy workload that is a large slice of the real
|
|
256
|
+
// cost, and its absence is most of the gap between the local estimate and
|
|
257
|
+
// the CLI's total_cost_usd.
|
|
258
|
+
const cacheWrite = Number(usage.cache_creation_input_tokens);
|
|
259
|
+
if (Number.isFinite(cacheWrite)) out.cacheWriteTokens = cacheWrite;
|
|
245
260
|
const totalCost = Number(obj.total_cost_usd);
|
|
246
261
|
if (Number.isFinite(totalCost) && totalCost >= 0) out.totalCostUsd = totalCost;
|
|
247
262
|
if (typeof obj.model === "string" && obj.model) {
|
|
@@ -253,9 +268,17 @@ export function parseUsageFromText(text) {
|
|
|
253
268
|
/**
|
|
254
269
|
* Append a TRUTHFUL cost-ledger row for a finished dispatcher session via
|
|
255
270
|
* scripts/cost/track-claude-usage.mjs (source "dispatcher"). Real token counts
|
|
256
|
-
* are parsed from the run's --output-format json stdout
|
|
257
|
-
*
|
|
258
|
-
*
|
|
271
|
+
* are parsed from the run's --output-format json stdout.
|
|
272
|
+
*
|
|
273
|
+
* On a parse failure we pass `--tokens-unknown <reason>`, which records
|
|
274
|
+
* measurement:"unknown" with NULL tokens. Previously we passed no token flags
|
|
275
|
+
* at all and the tracker defaulted them to 0 — a row that read as a free
|
|
276
|
+
* session. "Omit the flags rather than fabricate zeros" was the intent, but the
|
|
277
|
+
* tracker fabricated them anyway, so a systematic parse regression would drive
|
|
278
|
+
* the day's measured spend toward $0 and the budget governor would go quiet
|
|
279
|
+
* exactly when it should have been shouting. Fail-open is fine; silent is not.
|
|
280
|
+
*
|
|
281
|
+
* Best-effort + detached; never blocks the close path.
|
|
259
282
|
*/
|
|
260
283
|
function recordDispatcherCost({ stdout, model, durationMs, exitCode, decisionId }) {
|
|
261
284
|
const usage = parseUsageFromText(stdout);
|
|
@@ -278,7 +301,20 @@ function recordDispatcherCost({ stdout, model, durationMs, exitCode, decisionId
|
|
|
278
301
|
trackerArgs.push("--input-tokens", String(usage.inputTokens));
|
|
279
302
|
trackerArgs.push("--output-tokens", String(usage.outputTokens));
|
|
280
303
|
if (usage.cacheReadTokens != null) trackerArgs.push("--cache-read-tokens", String(usage.cacheReadTokens));
|
|
304
|
+
if (usage.cacheWriteTokens != null) trackerArgs.push("--cache-creation-tokens", String(usage.cacheWriteTokens));
|
|
281
305
|
if (usage.totalCostUsd != null) trackerArgs.push("--total-cost-usd", String(usage.totalCostUsd));
|
|
306
|
+
} else {
|
|
307
|
+
// Explicit "we ran a session and could not measure it" marker. NEVER a 0.
|
|
308
|
+
trackerArgs.push("--tokens-unknown", String(usage.reason || "usage-parse-failed"));
|
|
309
|
+
try {
|
|
310
|
+
logSession({
|
|
311
|
+
event: "cost_usage_parse_failed",
|
|
312
|
+
reason: usage.reason || "unknown",
|
|
313
|
+
model: model || null,
|
|
314
|
+
duration_ms: durationMs,
|
|
315
|
+
exit_code: exitCode,
|
|
316
|
+
});
|
|
317
|
+
} catch { /* logging must not break the close path */ }
|
|
282
318
|
}
|
|
283
319
|
spawn(process.execPath, trackerArgs, {
|
|
284
320
|
stdio: "ignore",
|
|
@@ -732,7 +768,17 @@ export function canDispatchBacklog(item) {
|
|
|
732
768
|
* which leave the durable on-disk item untouched for the next sweep anyway.
|
|
733
769
|
*/
|
|
734
770
|
export function dispatch(prompt, item, classResult, source = "inbox", opts = {}) {
|
|
735
|
-
|
|
771
|
+
// `obligationKey` names the debt this session was spawned to discharge. It is
|
|
772
|
+
// carried into the child's environment so the CLI send lanes (the ONLY way a
|
|
773
|
+
// session is allowed to reply) can stamp their delivery receipts with it.
|
|
774
|
+
// Without it a receipt says only "somebody spoke into that room", and a room
|
|
775
|
+
// is shared: two asks in one DM produce two debts, the first reply discharges
|
|
776
|
+
// both, and the unanswered one is closed as answered with nobody told.
|
|
777
|
+
const entry = {
|
|
778
|
+
prompt, item, classResult, source,
|
|
779
|
+
obligationKey: opts.obligationKey || null,
|
|
780
|
+
onClose: typeof opts.onClose === "function" ? opts.onClose : null,
|
|
781
|
+
};
|
|
736
782
|
const priority = classResult.priority;
|
|
737
783
|
const isPriorityInbox = source === "inbox" && (priority === "critical" || priority === "high");
|
|
738
784
|
|
|
@@ -892,7 +938,12 @@ export function _hasReDrainArmed() {
|
|
|
892
938
|
function currentBudgetBand() {
|
|
893
939
|
try {
|
|
894
940
|
const st = budgetGuard.dailyStatus({ agentRoot: AGENT_REPO_DIR });
|
|
895
|
-
|
|
941
|
+
// Pass the GOVERNOR'S band rather than letting budgetLadder re-derive one
|
|
942
|
+
// from (spent, cap). dailyStatus has already reconciled the day reading, the
|
|
943
|
+
// month reading, the cap latch and blindness; re-deriving here collapsed to
|
|
944
|
+
// band 0 whenever the monthly envelope was spent out (cap === 0), so the
|
|
945
|
+
// router stopped degrading exactly when the seat was furthest over.
|
|
946
|
+
return budgetLadder(st.spentUSD, st.capUSD, { band: st.band }).band;
|
|
896
947
|
} catch {
|
|
897
948
|
return 0;
|
|
898
949
|
}
|
|
@@ -1088,6 +1139,13 @@ function spawnSession(entry) {
|
|
|
1088
1139
|
ANTHROPIC_AUTH_TOKEN: "",
|
|
1089
1140
|
...(target.envForSpawn || {}),
|
|
1090
1141
|
};
|
|
1142
|
+
// ATTRIBUTION for delivery receipts. Set after the merges above so a router
|
|
1143
|
+
// target's envForSpawn can never drop it — a retargeted session replies
|
|
1144
|
+
// through exactly the same CLI lanes and owes exactly the same receipt.
|
|
1145
|
+
// These are plain identifiers, never credentials, so the §7.3 allowlist scrub
|
|
1146
|
+
// in buildChildEnv has nothing to object to.
|
|
1147
|
+
spawnEnv.MAESTRO_SESSION_ID = sessionId;
|
|
1148
|
+
if (entry.obligationKey) spawnEnv.MAESTRO_OBLIGATION_KEY = String(entry.obligationKey);
|
|
1091
1149
|
|
|
1092
1150
|
const proc = _spawn(CLAUDE_BIN, args, {
|
|
1093
1151
|
cwd: AGENT_REPO_DIR,
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* responder-cost.test.mjs — every responder spawn is billed, including the ones
|
|
3
|
+
* that FAIL.
|
|
4
|
+
*
|
|
5
|
+
* The defect this guards: `recordResponderCost` was reached only at the bottom
|
|
6
|
+
* of the `close` handler, after `if (settled) return;` and after
|
|
7
|
+
* `if (code !== 0) { reject; return; }`. So the two most expensive outcomes the
|
|
8
|
+
* daemon has — a 60s CLI timeout and a non-zero exit — wrote NO ledger row at
|
|
9
|
+
* all, not even `--tokens-unknown`. On the live seat that was 15 timeouts on
|
|
10
|
+
* 2026-08-11 and 10 on 2026-08-12: a full minute of paid model work each,
|
|
11
|
+
* invisible to the governor and to doctor.
|
|
12
|
+
*
|
|
13
|
+
* Run: `node --test scripts/daemon/responder-cost.test.mjs`
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
import { test } from "node:test";
|
|
17
|
+
import assert from "node:assert/strict";
|
|
18
|
+
import { readFileSync } from "node:fs";
|
|
19
|
+
import { join, dirname } from "node:path";
|
|
20
|
+
import { fileURLToPath } from "node:url";
|
|
21
|
+
|
|
22
|
+
import { unmeasuredReasonFor } from "./responder.mjs";
|
|
23
|
+
|
|
24
|
+
const HERE = dirname(fileURLToPath(import.meta.url));
|
|
25
|
+
const SRC = readFileSync(join(HERE, "responder.mjs"), "utf-8");
|
|
26
|
+
|
|
27
|
+
test("a TIMED-OUT session is billed as unmeasured, naming the timeout", () => {
|
|
28
|
+
const r = unmeasuredReasonFor({ parsed: null, timedOut: true, exitCode: null, parseError: null, timeoutMs: 60_000 });
|
|
29
|
+
assert.equal(r, "cli-timeout-60000ms");
|
|
30
|
+
});
|
|
31
|
+
|
|
32
|
+
test("a NON-ZERO exit is billed as unmeasured, naming the code", () => {
|
|
33
|
+
const r = unmeasuredReasonFor({ parsed: null, timedOut: false, exitCode: 1, parseError: null, timeoutMs: 60_000 });
|
|
34
|
+
assert.equal(r, "cli-exit-1");
|
|
35
|
+
});
|
|
36
|
+
|
|
37
|
+
test("an unparseable envelope on a clean exit is billed as unmeasured", () => {
|
|
38
|
+
const r = unmeasuredReasonFor({ parsed: null, timedOut: false, exitCode: 0, parseError: new Error("bad json"), timeoutMs: 60_000 });
|
|
39
|
+
assert.equal(r, "json-parse-failed");
|
|
40
|
+
});
|
|
41
|
+
|
|
42
|
+
test("a parsed envelope is measured — no reason at all", () => {
|
|
43
|
+
const r = unmeasuredReasonFor({ parsed: { usage: {} }, timedOut: true, exitCode: 1, parseError: new Error("x"), timeoutMs: 60_000 });
|
|
44
|
+
assert.equal(r, null, "a real envelope wins over every failure signal");
|
|
45
|
+
});
|
|
46
|
+
|
|
47
|
+
test("the timeout path never masks itself as a plain non-zero exit", () => {
|
|
48
|
+
// SIGTERM'd processes close with a code, and the operator needs to know WHICH
|
|
49
|
+
// failure burned the minute — the fixes are different.
|
|
50
|
+
const r = unmeasuredReasonFor({ parsed: null, timedOut: true, exitCode: 143, parseError: null, timeoutMs: 60_000 });
|
|
51
|
+
assert.equal(r, "cli-timeout-60000ms");
|
|
52
|
+
});
|
|
53
|
+
|
|
54
|
+
test("STRUCTURAL: the cost record happens BEFORE the close handler's early returns", () => {
|
|
55
|
+
// This is an ordering bug in code that cannot be unit-tested without a spawn
|
|
56
|
+
// harness, so the ordering itself is the assertion. If someone moves the
|
|
57
|
+
// recording back below `if (settled) return;`, every timeout goes unbilled
|
|
58
|
+
// again — silently, because the reply still works.
|
|
59
|
+
const close = SRC.slice(SRC.indexOf('proc.on("close"'));
|
|
60
|
+
const record = close.indexOf("recordResponderCost({");
|
|
61
|
+
const settledGuard = close.indexOf("if (settled) return;");
|
|
62
|
+
const exitGuard = close.indexOf("if (code !== 0) {");
|
|
63
|
+
|
|
64
|
+
assert.ok(record > 0 && settledGuard > 0 && exitGuard > 0, "all three markers present");
|
|
65
|
+
assert.ok(record < settledGuard, "cost is recorded before the settled early-return (the TIMEOUT path)");
|
|
66
|
+
assert.ok(record < exitGuard, "cost is recorded before the non-zero-exit early-return");
|
|
67
|
+
assert.match(close.slice(0, record), /costRecorded/, "and exactly once per spawn");
|
|
68
|
+
});
|