@cohortapp/agent-sdk 2.5.1 → 2.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (107) hide show
  1. package/bin/maestro.mjs +305 -89
  2. package/bin/maestro.test.mjs +357 -48
  3. package/docs/runbooks/backup-restore.md +65 -33
  4. package/framework-features.json +4 -4
  5. package/lib/backup/policy.mjs +710 -0
  6. package/lib/backup/policy.test.mjs +305 -0
  7. package/lib/budget-escalate.mjs +133 -0
  8. package/lib/budget-escalate.test.mjs +232 -0
  9. package/lib/budget-guard.envelope.test.mjs +476 -0
  10. package/lib/budget-guard.mjs +853 -75
  11. package/lib/budget-guard.test.mjs +91 -42
  12. package/lib/cadences.mjs +33 -0
  13. package/lib/channels/orgmail/adapter.mjs +88 -3
  14. package/lib/channels/orgmail/adapter.test.mjs +137 -0
  15. package/lib/channels/repeat-suppressor.mjs +198 -0
  16. package/lib/channels/repeat-suppressor.test.mjs +134 -0
  17. package/lib/comms/receipts.mjs +297 -0
  18. package/lib/cost/ledger-row.mjs +333 -0
  19. package/lib/cost/ledger-row.test.mjs +183 -0
  20. package/lib/execution/drive.mjs +28 -1
  21. package/lib/execution/effects.mjs +191 -12
  22. package/lib/execution/effects.test.mjs +50 -11
  23. package/lib/goals/admission.mjs +13 -1
  24. package/lib/goals/admission.test.mjs +26 -1
  25. package/lib/goals/loop.mjs +13 -0
  26. package/lib/kpi-sensors.test.mjs +3 -0
  27. package/lib/mandate/cache.mjs +13 -5
  28. package/lib/mandate/derive.mjs +146 -21
  29. package/lib/mandate/derive.test.mjs +50 -6
  30. package/lib/mandate/model.mjs +32 -4
  31. package/lib/mandate/refresh.test.mjs +16 -2
  32. package/lib/mcp/server.test.mjs +12 -3
  33. package/lib/model-router/economics.mjs +107 -76
  34. package/lib/model-router/economics.test.mjs +64 -46
  35. package/lib/model-router/integration-coverage.test.mjs +39 -37
  36. package/lib/model-router/ledger.mjs +75 -22
  37. package/lib/model-router/ledger.test.mjs +35 -2
  38. package/lib/org/client.mjs +14 -0
  39. package/lib/org/cost-sync.mjs +16 -2
  40. package/lib/org/doctor.mjs +62 -1
  41. package/lib/org/doctor.test.mjs +36 -3
  42. package/lib/org/email-remedy.mjs +49 -0
  43. package/lib/org/engagement-ledger.mjs +376 -0
  44. package/lib/org/engagement-ledger.test.mjs +112 -0
  45. package/lib/org/engagement.mjs +1056 -0
  46. package/lib/org/engagement.test.mjs +739 -0
  47. package/lib/org/messaging.mjs +230 -3
  48. package/lib/org/messaging.test.mjs +110 -1
  49. package/lib/org/param-contract.mjs +56 -2
  50. package/lib/org/param-contract.test.mjs +26 -0
  51. package/lib/org/protocol.checksum +1 -1
  52. package/lib/org/protocol.mjs +5 -0
  53. package/lib/org/protocol.test.mjs +7 -1
  54. package/lib/org/tool-surface.mjs +506 -10
  55. package/lib/org/tool-surface.test.mjs +191 -7
  56. package/lib/org/ui-parity.mjs +333 -6
  57. package/lib/org/ui-parity.test.mjs +96 -3
  58. package/lib/org/work-ledger.mjs +241 -0
  59. package/lib/org/work-ledger.test.mjs +237 -0
  60. package/lib/plan/adoption-e2e.test.mjs +366 -0
  61. package/lib/plan/budget-enforcement.test.mjs +400 -0
  62. package/lib/plan/budget-runtime.mjs +215 -0
  63. package/lib/plan/compile.mjs +201 -5
  64. package/lib/plan/compile.test.mjs +19 -5
  65. package/lib/plan/emit.mjs +8 -0
  66. package/lib/plan/emit.test.mjs +18 -0
  67. package/lib/resource-governor.mjs +58 -12
  68. package/lib/resource-governor.test.mjs +41 -1
  69. package/lib/security/audit-engine.mjs +45 -8
  70. package/lib/security/audit-engine.test.mjs +35 -0
  71. package/lib/setup/enroll-from-cohort.mjs +14 -1
  72. package/lib/setup/sections/mandate.mjs +48 -7
  73. package/lib/setup/sections/mandate.test.mjs +17 -2
  74. package/lib/setup/sections/orgmail.mjs +10 -2
  75. package/lib/setup/state.mjs +83 -2
  76. package/lib/telemetry/collect.mjs +360 -20
  77. package/lib/telemetry/collect.test.mjs +266 -0
  78. package/package.json +1 -1
  79. package/scripts/cost/track-claude-usage.mjs +207 -48
  80. package/scripts/cost/track-claude-usage.test.mjs +148 -0
  81. package/scripts/daemon/agent-daemon.mjs +315 -17
  82. package/scripts/daemon/assurance-e2e.test.mjs +421 -0
  83. package/scripts/daemon/assurance.mjs +944 -0
  84. package/scripts/daemon/assurance.test.mjs +668 -0
  85. package/scripts/daemon/cadence-consumer-governance.test.mjs +56 -0
  86. package/scripts/daemon/cadence-consumer.mjs +147 -9
  87. package/scripts/daemon/cadence-consumer.test.mjs +6 -0
  88. package/scripts/daemon/cadence-handlers.mjs +158 -0
  89. package/scripts/daemon/cadence-handlers.test.mjs +64 -0
  90. package/scripts/daemon/classifier.test.mjs +18 -9
  91. package/scripts/daemon/deliver.mjs +314 -0
  92. package/scripts/daemon/dispatcher-governance.test.mjs +10 -0
  93. package/scripts/daemon/dispatcher.mjs +64 -6
  94. package/scripts/daemon/responder-cost.test.mjs +68 -0
  95. package/scripts/daemon/responder.mjs +351 -298
  96. package/scripts/local-triggers/generate-plists.test.mjs +7 -4
  97. package/scripts/maintenance/backup-run.mjs +415 -0
  98. package/scripts/maintenance/backup-to-cloud.sh +16 -116
  99. package/scripts/org/send-orgmail.mjs +16 -0
  100. package/scripts/record-receipt.sh +63 -0
  101. package/scripts/restore-from-backup.sh +14 -3
  102. package/scripts/restore-from-backup.test.mjs +8 -5
  103. package/scripts/send-email-threaded.py +47 -0
  104. package/scripts/send-sms.sh +4 -0
  105. package/scripts/send-whatsapp.sh +4 -0
  106. package/scripts/setup/init-backup.mjs +93 -38
  107. package/scripts/slack-send.sh +12 -0
@@ -0,0 +1,668 @@
1
+ /**
2
+ * assurance.test.mjs — the four ways an ask used to end in silence, and the
3
+ * message the human now gets instead of each of them.
4
+ *
5
+ * Run: node --test scripts/daemon/assurance.test.mjs
6
+ *
7
+ * Every test here uses an INJECTED FAKE for transport. Nothing in this file
8
+ * touches Slack, Gmail, hq or the live workspace, and no test spawns a model.
9
+ * That is not only hygiene — it is the point being proven: the whole
10
+ * acknowledgement path is now deterministic string work, so a test can assert
11
+ * the exact words a human would read.
12
+ */
13
+
14
+ import { test, describe, before, beforeEach, after } from "node:test";
15
+ import assert from "node:assert/strict";
16
+ import { mkdtempSync, rmSync, mkdirSync } from "node:fs";
17
+ import { tmpdir } from "node:os";
18
+ import { join } from "node:path";
19
+
20
+ // Point the whole module tree at a scratch root BEFORE importing anything that
21
+ // resolves AGENT_DIR at module load.
22
+ const ROOT = mkdtempSync(join(tmpdir(), "assurance-test-"));
23
+ process.env.AGENT_DIR = ROOT;
24
+ process.env.AGENT_ROOT = ROOT;
25
+ mkdirSync(join(ROOT, "state"), { recursive: true });
26
+
27
+ const assurance = await import("./assurance.mjs");
28
+ const receipts = await import("../../lib/comms/receipts.mjs");
29
+
30
+ const {
31
+ shouldAcknowledge,
32
+ composeAck,
33
+ composeProgress,
34
+ composeFailure,
35
+ composeSilentSuccess,
36
+ classifyFailure,
37
+ openAndAcknowledge,
38
+ settleSession,
39
+ sweepObligations,
40
+ readObligation,
41
+ openObligations,
42
+ _resetObligations,
43
+ ACK_GRACE_MS,
44
+ PROGRESS_AFTER_MS,
45
+ PROGRESS_MAX,
46
+ STALE_AFTER_MS,
47
+ INTERRUPT_GRACE_MS,
48
+ ACK_MAX_ATTEMPTS,
49
+ RETRY_MAX,
50
+ } = assurance;
51
+
52
+ after(() => { try { rmSync(ROOT, { recursive: true, force: true }); } catch { /* */ } });
53
+
54
+ // ---------------------------------------------------------------------------
55
+ // Fakes
56
+ // ---------------------------------------------------------------------------
57
+
58
+ /** A transport that records what a human would have read, and can be told to fail. */
59
+ function fakeTransport(opts = {}) {
60
+ const sent = [];
61
+ let failures = opts.failTimes || 0;
62
+ const impl = async (item, text, o = {}) => {
63
+ if (failures > 0) { failures--; return { sent: false, via: null, error: opts.error || "network unreachable" }; }
64
+ sent.push({ to: item.sender, channel: item.channel_id || item.channel, kind: o.kind || "reply", text });
65
+ return { sent: true, via: "fake", channel: item.channel_id || item.channel };
66
+ };
67
+ return { sent, impl, ackSender: async (item, cr) => {
68
+ const text = composeAck(item, cr);
69
+ const r = await impl(item, text, { kind: "ack" });
70
+ return { sent: r.sent, holdingText: text, error: r.error };
71
+ } };
72
+ }
73
+
74
+ const ITEM = {
75
+ id: "msg-1001",
76
+ raw_ref: "cohort:cmql18026:1001",
77
+ message_id: "1001",
78
+ service: "cohort",
79
+ channel_id: "cmql1802601iihhzpf774vwbk",
80
+ sender: "Dana Whitfield",
81
+ content: "with that in mind, fix these issues you just noted please and push their fixes to git",
82
+ };
83
+
84
+ const CLASS_COMPLEX = { action: "respond", priority: "critical", model: "opus", summary: "Fix the noted issues and push the fixes to git" };
85
+
86
+ beforeEach(() => { _resetObligations(); });
87
+
88
+ // ---------------------------------------------------------------------------
89
+ // 1. THE JUDGEMENT — simple asks must NOT get "I'll look into it"
90
+ // ---------------------------------------------------------------------------
91
+
92
+ describe("shouldAcknowledge — the simple/complex distinction", () => {
93
+ test("a fast ask answered in this turn gets NO acknowledgement", () => {
94
+ // The quick path already sent the answer. An ack here is pure noise.
95
+ const v = shouldAcknowledge({ willSpawnSession: false, item: ITEM, source: "inbox" });
96
+ assert.equal(v.ack, false);
97
+ assert.equal(v.reason, "answered-in-turn");
98
+ });
99
+
100
+ test("a slow ask that spawns a session DOES get one", () => {
101
+ const v = shouldAcknowledge({ willSpawnSession: true, item: ITEM, source: "inbox" });
102
+ assert.equal(v.ack, true);
103
+ });
104
+
105
+ test("REGRESSION: a `queue` action still gets one", () => {
106
+ // classifier.mjs's heuristic fallback (used when the LLM classifier itself
107
+ // fails) emits action:"queue". The old gate was
108
+ // `respond || draft || research`, so exactly when classification degraded, a
109
+ // directed DM got a 45-minute session and total silence. The gate no longer
110
+ // consults the classifier at all.
111
+ const v = shouldAcknowledge({ willSpawnSession: true, item: ITEM, source: "inbox", classResult: { action: "queue" } });
112
+ assert.equal(v.ack, true, "a degraded classification must not cost the human their acknowledgement");
113
+ });
114
+
115
+ test("backlog work, which nobody is waiting on, gets none", () => {
116
+ assert.equal(shouldAcknowledge({ willSpawnSession: true, item: ITEM, source: "backlog" }).ack, false);
117
+ });
118
+
119
+ test("an item with no resolvable reply target gets none (nowhere to send it)", () => {
120
+ const orphan = { ...ITEM, channel_id: null, channel: null };
121
+ assert.equal(shouldAcknowledge({ willSpawnSession: true, item: orphan, source: "inbox" }).ack, false);
122
+ });
123
+ });
124
+
125
+ // ---------------------------------------------------------------------------
126
+ // 2. THE ACKNOWLEDGEMENT — composed, not generated
127
+ // ---------------------------------------------------------------------------
128
+
129
+ describe("composeAck", () => {
130
+ test("is instant and deterministic — no model, no I/O, no timeout to lose", () => {
131
+ const t0 = process.hrtime.bigint();
132
+ const a = composeAck(ITEM, CLASS_COMPLEX);
133
+ const elapsedMs = Number(process.hrtime.bigint() - t0) / 1e6;
134
+ assert.ok(elapsedMs < 5, `ack composition took ${elapsedMs}ms; it must be effectively free`);
135
+ assert.equal(a, composeAck(ITEM, CLASS_COMPLEX), "same input must give the same words");
136
+ });
137
+
138
+ test("names the actual ask and promises to come back either way", () => {
139
+ const a = composeAck(ITEM, CLASS_COMPLEX);
140
+ assert.match(a, /fix the noted issues and push the fixes to git/i);
141
+ assert.match(a, /either way/i, "the promise the rest of the system exists to keep");
142
+ assert.match(a, /10-20 minutes/, "an opus/critical item is honestly framed as slow");
143
+ });
144
+
145
+ test("degrades gracefully when the classifier gave nothing to go on", () => {
146
+ const a = composeAck(ITEM, {});
147
+ assert.ok(a.length > 40);
148
+ assert.match(a, /either way/i);
149
+ });
150
+ });
151
+
152
+ // ---------------------------------------------------------------------------
153
+ // 3. FAILURE CLASSIFICATION — retry the transient, stop on the permanent
154
+ // ---------------------------------------------------------------------------
155
+
156
+ describe("classifyFailure", () => {
157
+ const cases = [
158
+ [{ code: 143, error: "SIGTERM" }, true, "timeout"],
159
+ [{ code: 137 }, true, "killed"],
160
+ [{ code: 1, error: "rate limit exceeded (429)" }, true, "rate_limited"],
161
+ [{ code: null, error: "claude CLI spawn error: ENOENT" }, false, "spawn_failed"],
162
+ [{ code: 1, error: "essential-only: daily budget cap reached" }, false, "budget_refused"],
163
+ [{ code: 1, error: "low free memory (need >=3686MB)" }, true, "resource_starved"],
164
+ [{ code: 0 }, false, "silent_success"],
165
+ ];
166
+ for (const [input, transient, label] of cases) {
167
+ test(`${label} → ${transient ? "retry" : "stop"}`, () => {
168
+ const f = classifyFailure(input);
169
+ assert.equal(f.label, label);
170
+ assert.equal(f.transient, transient);
171
+ assert.ok(f.human.length > 10, "every cause must have plain-English wording for the human");
172
+ assert.doesNotMatch(f.human, /error occurred|something went wrong/i, "no content-free apologies");
173
+ });
174
+ }
175
+ });
176
+
177
+ // ---------------------------------------------------------------------------
178
+ // 4. THE LEDGER — a debt survives the send failing
179
+ // ---------------------------------------------------------------------------
180
+
181
+ describe("openAndAcknowledge", () => {
182
+ test("acknowledges and records the debt as acknowledged", async () => {
183
+ const t = fakeTransport();
184
+ const r = await openAndAcknowledge({ item: ITEM, classResult: CLASS_COMPLEX, service: "cohort", deps: { ackSender: t.ackSender } });
185
+ assert.equal(r.acked, true);
186
+ assert.equal(t.sent.length, 1);
187
+ assert.equal(t.sent[0].kind, "ack");
188
+ const rec = readObligation(r.key);
189
+ assert.equal(rec.acknowledged, true);
190
+ assert.equal(rec.state, "open", "the debt is acknowledged, not discharged — an ack is not an answer");
191
+ });
192
+
193
+ test("REGRESSION: a FAILED acknowledgement leaves an open debt instead of silence", async () => {
194
+ // This is the exact production case: 20 of 32 acknowledgements died and the
195
+ // old code did `logResponse(...); return {sent:false}` — the human was never
196
+ // told, and nothing ever retried.
197
+ const t = fakeTransport({ failTimes: 99, error: "claude CLI timed out after 60000ms" });
198
+ const r = await openAndAcknowledge({ item: ITEM, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
199
+ assert.equal(r.acked, false);
200
+ const rec = readObligation(r.key);
201
+ assert.equal(rec.acknowledged, false);
202
+ assert.equal(rec.state, "open");
203
+ assert.equal(openObligations().length, 1, "the debt must be visible to the sweep");
204
+ });
205
+
206
+ test("acknowledges ONCE — a re-delivered item does not re-announce itself", async () => {
207
+ const t = fakeTransport();
208
+ await openAndAcknowledge({ item: ITEM, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
209
+ await openAndAcknowledge({ item: ITEM, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
210
+ assert.equal(t.sent.length, 1, "do not spam");
211
+ });
212
+ });
213
+
214
+ // ---------------------------------------------------------------------------
215
+ // 5. SESSION SETTLEMENT — the four terminal states
216
+ // ---------------------------------------------------------------------------
217
+
218
+ describe("settleSession", () => {
219
+ async function openOne(t) {
220
+ const r = await openAndAcknowledge({ item: ITEM, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
221
+ return r.key;
222
+ }
223
+
224
+ test("exit 0 AND the session spoke → discharge quietly, add nothing", async () => {
225
+ const t = fakeTransport();
226
+ const key = await openOne(t);
227
+ const before = t.sent.length;
228
+ const r = await settleSession({ key, ok: true, code: 0, deps: { deliverImpl: t.impl, spokeSinceImpl: () => true } });
229
+ assert.equal(r.verdict, "answered-by-session");
230
+ assert.equal(t.sent.length, before, "the human already has their answer — saying more is spam");
231
+ assert.equal(readObligation(key).state, "answered");
232
+ });
233
+
234
+ test("REGRESSION: exit 0 and the session said NOTHING → the daemon speaks", async () => {
235
+ // Observed live: session s-1786600074079-52, exit 0, 241s, $2.36, its own
236
+ // result text ending "Nothing was sent." Marked processed forever, emitted
237
+ // as `sent`, never re-delivered. This is that case.
238
+ const t = fakeTransport();
239
+ const key = await openOne(t);
240
+ const stdout = JSON.stringify({ type: "result", result: "I reviewed the three failing checks and pushed fixes for two. The third needs a decision from you. Nothing was sent." });
241
+ const r = await settleSession({ key, ok: true, code: 0, stdout, deps: { deliverImpl: t.impl, spokeSinceImpl: () => false } });
242
+ assert.equal(r.verdict, "silent-success");
243
+ assert.equal(r.spoke, true);
244
+ const msg = t.sent.at(-1).text;
245
+ assert.match(msg, /didn't get a reply out to you/i);
246
+ assert.match(msg, /pushed fixes for two/, "the work the session actually did must reach the human");
247
+ assert.equal(readObligation(key).state, "answered");
248
+ });
249
+
250
+ test("an acknowledgement does NOT count as the session having answered", async () => {
251
+ const t = fakeTransport();
252
+ const key = await openOne(t);
253
+ // A truthful receipt store that only ever saw the ack.
254
+ const spokeSinceImpl = ({ excludeKinds }) => !(excludeKinds || []).includes("ack") ;
255
+ const r = await settleSession({ key, ok: true, code: 0, stdout: "", deps: { deliverImpl: t.impl, spokeSinceImpl } });
256
+ assert.equal(r.verdict, "silent-success", "'I'm on it' is not an answer");
257
+ });
258
+
259
+ test("transient failure with retries left → tell them, keep the debt, retry", async () => {
260
+ const t = fakeTransport();
261
+ const key = await openOne(t);
262
+ const r = await settleSession({ key, ok: false, code: 143, error: "SIGTERM", deps: { deliverImpl: t.impl, spokeSinceImpl: () => false } });
263
+ assert.equal(r.willRetry, true);
264
+ assert.match(r.verdict, /^retrying:timeout$/);
265
+ const msg = t.sent.at(-1).text;
266
+ assert.match(msg, /ran past its time limit/i);
267
+ assert.match(msg, /retrying it now/i);
268
+ assert.equal(readObligation(key).state, "open", "still owed — the retry has to be covered too");
269
+ });
270
+
271
+ test("permanent failure → tell them, stop, escalate durably", async () => {
272
+ const t = fakeTransport();
273
+ const key = await openOne(t);
274
+ const r = await settleSession({ key, ok: false, code: 1, error: "claude CLI spawn error: ENOENT", deps: { deliverImpl: t.impl, spokeSinceImpl: () => false } });
275
+ assert.equal(r.willRetry, false);
276
+ assert.match(r.verdict, /^failed:spawn_failed$/);
277
+ const msg = t.sent.at(-1).text;
278
+ assert.match(msg, /couldn't get this done/i);
279
+ assert.match(msg, /stopped retrying/i);
280
+ assert.match(msg, /\?$/, "a dead end must end in a question, not a shrug");
281
+ assert.equal(readObligation(key).state, "failed");
282
+ });
283
+
284
+ test("retries are BOUNDED — the second timeout does not promise a third attempt", async () => {
285
+ const t = fakeTransport();
286
+ const key = await openOne(t);
287
+ // Simulate having already burned the allowed attempts.
288
+ const rec = readObligation(key);
289
+ rec.attempts = RETRY_MAX + 1;
290
+ const { writeFileSync } = await import("node:fs");
291
+ writeFileSync(join(ROOT, "state", "obligations", `${assurance.sanitiseKey(key)}.json`), JSON.stringify(rec));
292
+ const r = await settleSession({ key, ok: false, code: 143, error: "SIGTERM", deps: { deliverImpl: t.impl, spokeSinceImpl: () => false } });
293
+ assert.equal(r.willRetry, false, "an unbounded silent retry loop is itself the defect");
294
+ assert.match(t.sent.at(-1).text, /stopped retrying/i);
295
+ });
296
+ });
297
+
298
+ // ---------------------------------------------------------------------------
299
+ // 6. THE SWEEP — the safety net under everything above
300
+ // ---------------------------------------------------------------------------
301
+
302
+ describe("sweepObligations", () => {
303
+ test("retries an acknowledgement whose send failed", async () => {
304
+ const failing = fakeTransport({ failTimes: 99 });
305
+ const r = await openAndAcknowledge({ item: ITEM, classResult: CLASS_COMPLEX, deps: { ackSender: failing.ackSender } });
306
+ assert.equal(r.acked, false);
307
+
308
+ const working = fakeTransport();
309
+ const now = Date.now() + ACK_GRACE_MS + 1000;
310
+ const stats = await sweepObligations({ now, deps: { deliverImpl: working.impl, spokeSinceImpl: () => false } });
311
+ assert.equal(stats.acked, 1);
312
+ assert.match(working.sent[0].text, /come back to you/i);
313
+ assert.equal(readObligation(r.key).acknowledged, true);
314
+ });
315
+
316
+ test("long work gets an interim update, not silence", async () => {
317
+ const t = fakeTransport();
318
+ const r = await openAndAcknowledge({ item: ITEM, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
319
+ const now = Date.now() + PROGRESS_AFTER_MS + 1000;
320
+ const stats = await sweepObligations({ now, deps: { deliverImpl: t.impl, spokeSinceImpl: () => false } });
321
+ assert.equal(stats.progressed, 1);
322
+ const msg = t.sent.at(-1).text;
323
+ assert.match(msg, /still working/i);
324
+ assert.match(msg, /minutes in/i, "an interim update that omits how long it has been is not much of an update");
325
+ });
326
+
327
+ test("interim updates are CAPPED — attentive, not noisy", async () => {
328
+ const t = fakeTransport();
329
+ await openAndAcknowledge({ item: ITEM, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
330
+ const deps = { deliverImpl: t.impl, spokeSinceImpl: () => false };
331
+ let now = Date.now();
332
+ for (let i = 0; i < 10; i++) {
333
+ now += PROGRESS_AFTER_MS + 60_000 * 11;
334
+ if (now - Date.now() >= STALE_AFTER_MS) break;
335
+ await sweepObligations({ now, deps });
336
+ }
337
+ const progressCount = t.sent.filter((m) => m.kind === "progress").length;
338
+ assert.ok(progressCount <= PROGRESS_MAX, `sent ${progressCount} progress updates; cap is ${PROGRESS_MAX}`);
339
+ });
340
+
341
+ test("work that outruns every session timeout is declared dead, not left hanging", async () => {
342
+ const t = fakeTransport();
343
+ const r = await openAndAcknowledge({ item: ITEM, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
344
+ const now = Date.now() + STALE_AFTER_MS + 1000;
345
+ const stats = await sweepObligations({ now, deps: { deliverImpl: t.impl, spokeSinceImpl: () => false } });
346
+ assert.equal(stats.staled, 1);
347
+ assert.match(t.sent.at(-1).text, /outrun its time limit/i);
348
+ assert.equal(readObligation(r.key).state, "failed");
349
+ });
350
+
351
+ test("a debt the session has since answered is discharged silently", async () => {
352
+ const t = fakeTransport();
353
+ const r = await openAndAcknowledge({ item: ITEM, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
354
+ const before = t.sent.length;
355
+ const now = Date.now() + STALE_AFTER_MS + 1000;
356
+ const stats = await sweepObligations({ now, deps: { deliverImpl: t.impl, spokeSinceImpl: () => true } });
357
+ assert.equal(stats.closed, 1);
358
+ assert.equal(t.sent.length, before, "no message when the human already has their answer");
359
+ assert.equal(readObligation(r.key).state, "answered");
360
+ });
361
+
362
+ test("a debt orphaned by a DEAD daemon is spoken to", async () => {
363
+ const t = fakeTransport();
364
+ const r = await openAndAcknowledge({ item: ITEM, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
365
+ // Rewrite the record as if a now-dead process had opened it.
366
+ const rec = readObligation(r.key);
367
+ rec.daemonPid = process.pid + 99999;
368
+ const { writeFileSync } = await import("node:fs");
369
+ writeFileSync(join(ROOT, "state", "obligations", `${assurance.sanitiseKey(r.key)}.json`), JSON.stringify(rec));
370
+
371
+ const stats = await sweepObligations({
372
+ now: Date.now() + INTERRUPT_GRACE_MS + 1000,
373
+ deps: { deliverImpl: t.impl, spokeSinceImpl: () => false, isPidAliveImpl: () => false },
374
+ });
375
+ assert.equal(stats.interrupted, 1);
376
+ assert.match(t.sent.at(-1).text, /was interrupted before it finished/i);
377
+ });
378
+
379
+ test("REGRESSION: a LIVE daemon's obligation is never declared interrupted", async () => {
380
+ // Two daemons on one AGENT_DIR is what an operator creates the moment they
381
+ // run one by hand next to the launchd one. Each sees the other's healthy,
382
+ // actively-running debts as foreign. With only a pid-mismatch test — which
383
+ // is what shipped — both then tell those requesters "my session was
384
+ // interrupted" while the work runs to completion behind the apology.
385
+ const t = fakeTransport();
386
+ const r = await openAndAcknowledge({ item: ITEM, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
387
+ const rec = readObligation(r.key);
388
+ rec.daemonPid = process.pid + 99999;
389
+ const { writeFileSync } = await import("node:fs");
390
+ writeFileSync(join(ROOT, "state", "obligations", `${assurance.sanitiseKey(r.key)}.json`), JSON.stringify(rec));
391
+
392
+ const before = t.sent.length;
393
+ const stats = await sweepObligations({
394
+ now: Date.now() + INTERRUPT_GRACE_MS + 1000,
395
+ deps: { deliverImpl: t.impl, spokeSinceImpl: () => false, isPidAliveImpl: () => true },
396
+ });
397
+ assert.equal(stats.interrupted, 0, "the owning daemon is alive — its work is not interrupted");
398
+ assert.equal(t.sent.length, before, "and nobody is told a lie about it");
399
+ });
400
+
401
+ test("REGRESSION: a FRESH obligation is never declared interrupted, dead pid or not", async () => {
402
+ // The age guard is the backstop for a recycled pid. One second after the ask
403
+ // arrived, no reading of the evidence supports "your session was interrupted".
404
+ const t = fakeTransport();
405
+ const r = await openAndAcknowledge({ item: ITEM, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
406
+ const rec = readObligation(r.key);
407
+ rec.daemonPid = process.pid + 99999;
408
+ const { writeFileSync } = await import("node:fs");
409
+ writeFileSync(join(ROOT, "state", "obligations", `${assurance.sanitiseKey(r.key)}.json`), JSON.stringify(rec));
410
+
411
+ const stats = await sweepObligations({
412
+ now: Date.now() + 1000,
413
+ deps: { deliverImpl: t.impl, spokeSinceImpl: () => false, isPidAliveImpl: () => false },
414
+ });
415
+ assert.equal(stats.interrupted, 0);
416
+ });
417
+
418
+ test("each distinct message carries a distinct idempotency suffix", async () => {
419
+ // hq de-dups server-side on the client message id. Two progress updates that
420
+ // both key on "progress" are ONE message as far as hq is concerned, so the
421
+ // second would be silently swallowed — "never silent" degrading into
422
+ // "acknowledged once, then silent forever". This is invisible in any test
423
+ // that stubs transport without inspecting the options, hence this one.
424
+ const seen = [];
425
+ const impl = async (item, text, o = {}) => { seen.push(o.idempotencySuffix); return { sent: true, via: "fake", channel: "c" }; };
426
+ const t = fakeTransport();
427
+ await openAndAcknowledge({ item: ITEM, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
428
+
429
+ const deps = { deliverImpl: impl, spokeSinceImpl: () => false };
430
+ await sweepObligations({ now: Date.now() + PROGRESS_AFTER_MS + 1000, deps });
431
+ await sweepObligations({ now: Date.now() + PROGRESS_AFTER_MS + 1000 + assurance.PROGRESS_EVERY_MS + 1000, deps });
432
+
433
+ assert.ok(seen.length >= 2, `expected two progress sends, saw ${seen.length}`);
434
+ assert.equal(new Set(seen).size, seen.length, `suffixes must be unique per message, got ${JSON.stringify(seen)}`);
435
+ });
436
+
437
+ test("emits at most ONE message per obligation per tick", async () => {
438
+ const t = fakeTransport({ failTimes: 1 }); // the ack fails
439
+ const r = await openAndAcknowledge({ item: ITEM, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
440
+ assert.equal(r.acked, false);
441
+ const before = t.sent.length;
442
+ // Old enough to qualify for BOTH the ack retry and a progress update.
443
+ const now = Date.now() + PROGRESS_AFTER_MS + 1000;
444
+ await sweepObligations({ now, deps: { deliverImpl: t.impl, spokeSinceImpl: () => false } });
445
+ assert.equal(t.sent.length - before, 1, "two messages in one tick reads as a malfunction");
446
+ });
447
+ });
448
+
449
+ // ---------------------------------------------------------------------------
450
+ // 7. RECEIPTS — the evidence the whole guarantee rests on
451
+ // ---------------------------------------------------------------------------
452
+
453
+ describe("delivery receipts", () => {
454
+ test("a recorded send is visible to a later spokeSince query", () => {
455
+ const now = Date.now();
456
+ receipts.recordOutbound({ service: "cohort", channel: "C-RECEIPT-1", kind: "reply", chars: 12, agentRoot: ROOT, now });
457
+ assert.equal(receipts.spokeSince({ channel: "C-RECEIPT-1", sinceMs: now - 1000, agentRoot: ROOT }), true);
458
+ });
459
+
460
+ test("acks and progress pings are excludable — they are not answers", () => {
461
+ const now = Date.now();
462
+ receipts.recordOutbound({ service: "cohort", channel: "C-RECEIPT-2", kind: "ack", agentRoot: ROOT, now });
463
+ receipts.recordOutbound({ service: "cohort", channel: "C-RECEIPT-2", kind: "progress", agentRoot: ROOT, now });
464
+ assert.equal(
465
+ receipts.spokeSince({ channel: "C-RECEIPT-2", sinceMs: now - 1000, excludeKinds: ["ack", "progress", "failure"], agentRoot: ROOT }),
466
+ false,
467
+ );
468
+ });
469
+
470
+ test("an unreadable ledger reads as 'the human heard nothing', never as 'fine'", () => {
471
+ // The fail-open direction is deliberate: the cost of guessing wrong this way
472
+ // is one redundant message; guessing the other way is the bug being fixed.
473
+ assert.equal(receipts.spokeSince({ channel: "never-used-channel", sinceMs: Date.now(), agentRoot: ROOT }), false);
474
+ });
475
+ });
476
+
477
+ // ---------------------------------------------------------------------------
478
+ // 7. ATTRIBUTION — a room is shared; a debt is not
479
+ // ---------------------------------------------------------------------------
480
+
481
+ describe("attribution: which debt did that message actually discharge", () => {
482
+ const mk = (id, content) => ({
483
+ id, raw_ref: `cohort:C-SHARED:${id}`, message_id: id, service: "cohort",
484
+ channel: "C-SHARED", channel_id: "C-SHARED", sender: "Dana Whitfield", content,
485
+ });
486
+
487
+ test("REGRESSION: one session's answer does NOT discharge a different ask in the same room", async () => {
488
+ // The owner sends two things a minute apart in one DM. Two debts, one room.
489
+ // Keyed on the room alone — which is what shipped — the first session's
490
+ // reply closed BOTH, and the second ask, which nobody answered, was recorded
491
+ // as "answered" with not one word sent about it. This is that scenario.
492
+ const t = fakeTransport();
493
+ const a = mk("ASK-A", "what's our runway?");
494
+ const b = mk("ASK-B", "also please fix the flaky poller test and push it");
495
+ const ra = await openAndAcknowledge({ item: a, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
496
+ const rb = await openAndAcknowledge({ item: b, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
497
+ assurance.noteSession(ra.key, "s-A");
498
+ assurance.noteSession(rb.key, "s-B");
499
+
500
+ // Session A answers, stamping the receipt with the debt it was spawned for.
501
+ receipts.recordOutbound({
502
+ service: "cohort", channel: "C-SHARED", kind: "session",
503
+ obligationKey: ra.key, sessionId: "s-A", agentRoot: ROOT,
504
+ });
505
+
506
+ const before = t.sent.length;
507
+ const settledA = await settleSession({ key: ra.key, ok: true, code: 0, deps: { deliverImpl: t.impl } });
508
+ assert.equal(settledA.verdict, "answered-by-session");
509
+ assert.equal(settledA.basis, "obligation-keyed");
510
+ assert.equal(t.sent.length, before, "A was answered — saying more would be spam");
511
+
512
+ // B exited 0 having said nothing. It must NOT ride on A's receipt.
513
+ const settledB = await settleSession({
514
+ key: rb.key, ok: true, code: 0,
515
+ stdout: JSON.stringify({ type: "result", result: "Could not reproduce the flake." }),
516
+ deps: { deliverImpl: t.impl },
517
+ });
518
+ assert.equal(settledB.verdict, "silent-success", "B was never answered and must be rescued");
519
+ assert.match(t.sent.at(-1).text, /didn't get a reply out to you/i);
520
+ assert.match(t.sent.at(-1).text, /Could not reproduce the flake/, "and B's actual work reaches the human");
521
+ });
522
+
523
+ test("an UNATTRIBUTED message in a room holding two open debts discharges neither", async () => {
524
+ const t = fakeTransport();
525
+ const a = mk("AMB-A", "first ask");
526
+ const b = mk("AMB-B", "second ask");
527
+ const ra = await openAndAcknowledge({ item: a, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
528
+ await openAndAcknowledge({ item: b, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
529
+ assurance.noteSession(ra.key, "s-amb");
530
+ receipts.recordOutbound({ service: "cohort", channel: "C-SHARED", kind: "session", agentRoot: ROOT });
531
+
532
+ const r = await settleSession({ key: ra.key, ok: true, code: 0, stdout: "", deps: { deliverImpl: t.impl } });
533
+ assert.equal(r.verdict, "silent-success", "ambiguous evidence is not evidence");
534
+ });
535
+
536
+ test("an unattributed message DOES discharge the only open debt in a quiet room", async () => {
537
+ // The tolerant rung. A send lane that loses the dispatcher's env still leaves
538
+ // a receipt; treating that as silence would make every such reply produce a
539
+ // redundant apology. Safe only because the room holds one debt and a session
540
+ // of ours has actually run.
541
+ const t = fakeTransport();
542
+ const solo = { ...ITEM, id: "SOLO-1", raw_ref: "cohort:C-QUIET:SOLO-1", channel: "C-QUIET", channel_id: "C-QUIET" };
543
+ const r = await openAndAcknowledge({ item: solo, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
544
+ assurance.noteSession(r.key, "s-solo");
545
+ receipts.recordOutbound({ service: "cohort", channel: "C-QUIET", kind: "session", agentRoot: ROOT });
546
+
547
+ const before = t.sent.length;
548
+ const settled = await settleSession({ key: r.key, ok: true, code: 0, deps: { deliverImpl: t.impl } });
549
+ assert.equal(settled.verdict, "answered-by-session");
550
+ assert.equal(settled.basis, "sole-debt-in-room");
551
+ assert.equal(t.sent.length, before);
552
+ });
553
+
554
+ test("REGRESSION: a message that arrives before any session of ours ran is not our answer", async () => {
555
+ // A debt whose acknowledgement failed, and something unrelated posts into the
556
+ // room. Closing on that leaves a human never acknowledged, never answered,
557
+ // and the debt recorded as discharged.
558
+ const t = fakeTransport({ failTimes: 1 });
559
+ const solo = { ...ITEM, id: "PRE-1", raw_ref: "cohort:C-PRE:PRE-1", channel: "C-PRE", channel_id: "C-PRE" };
560
+ const r = await openAndAcknowledge({ item: solo, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
561
+ assert.equal(readObligation(r.key).acknowledged, false);
562
+ receipts.recordOutbound({ service: "cohort", channel: "C-PRE", kind: "reply", agentRoot: ROOT });
563
+
564
+ const stats = await sweepObligations({ now: Date.now() + ACK_GRACE_MS + 1000, deps: { deliverImpl: t.impl } });
565
+ assert.equal(stats.closed, 0, "no session of ours had run — that message was not our answer");
566
+ assert.equal(stats.acked, 1, "instead the missing acknowledgement is finally sent");
567
+ assert.match(t.sent.at(-1).text, /let me look into this|I'm on this now|let me dig into this/i);
568
+ });
569
+ });
570
+
571
+ // ---------------------------------------------------------------------------
572
+ // 8. UNREACHABLE CHANNELS — an impossible send must not become a loop
573
+ // ---------------------------------------------------------------------------
574
+
575
+ describe("channels with no transport", () => {
576
+ const TELEGRAM = {
577
+ id: "TG-1", raw_ref: "telegram:12345:9", message_id: "TG-1",
578
+ service: "telegram", channel_id: "12345", sender: "Dana Whitfield",
579
+ content: "please redo the deck and send it over",
580
+ };
581
+
582
+ test("no acknowledgement is promised on a service we cannot write to", () => {
583
+ const v = shouldAcknowledge({ willSpawnSession: true, item: TELEGRAM, source: "inbox" });
584
+ assert.equal(v.ack, false);
585
+ assert.equal(v.reason, "no-transport");
586
+ });
587
+
588
+ test("REGRESSION: the debt is still opened, and the sweep escalates it ONCE instead of retrying forever", async () => {
589
+ // What shipped: the ack branch's `continue` sat above the stale branch, so an
590
+ // obligation whose send could never succeed was retried once per tick —
591
+ // 1,440 impossible sends a day, an immortal debt, and a human told nothing.
592
+ const t = fakeTransport();
593
+ const opened = await openAndAcknowledge({ item: TELEGRAM, classResult: CLASS_COMPLEX, ack: false });
594
+ assert.equal(opened.opened, true, "the debt exists even though we cannot speak to it");
595
+ assert.equal(readObligation(opened.key).deliverable, false);
596
+
597
+ let attempts = 0;
598
+ const counting = async (...a) => { attempts++; return t.impl(...a); };
599
+ const t0 = Date.now();
600
+ for (let m = 1; m <= 240; m++) {
601
+ await sweepObligations({ now: t0 + m * 60_000, deps: { deliverImpl: counting } });
602
+ }
603
+ assert.equal(attempts, 0, "not one impossible send is attempted");
604
+ const rec = readObligation(opened.key);
605
+ assert.equal(rec.state, "undeliverable", "the debt is resolved, not immortal");
606
+ assert.equal(openObligations().length, 0);
607
+ });
608
+
609
+ test("an acknowledgement that keeps failing is counted out and escalated", async () => {
610
+ const t = fakeTransport({ failTimes: 999 });
611
+ const r = await openAndAcknowledge({ item: ITEM, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
612
+ const t0 = Date.now();
613
+ let ticks = 0;
614
+ for (let m = 1; m <= 60; m++) {
615
+ const s = await sweepObligations({ now: t0 + m * 60_000, deps: { deliverImpl: t.impl } });
616
+ ticks++;
617
+ if (s.unreachable) break;
618
+ }
619
+ assert.ok(ticks <= ACK_MAX_ATTEMPTS + 2, `gave up after ${ticks} ticks, not forever`);
620
+ assert.equal(readObligation(r.key).state, "undeliverable");
621
+ });
622
+ });
623
+
624
+ // ---------------------------------------------------------------------------
625
+ // 9. THE RETRY CLOCK — never contradict a message already sent
626
+ // ---------------------------------------------------------------------------
627
+
628
+ describe("a running retry is not declared dead", () => {
629
+ test("REGRESSION: 'I'm retrying it now' is not followed by 'I've stopped retrying' five minutes later", async () => {
630
+ // Attempt 1 is SIGTERMed at the dispatcher's 45-minute cap. STALE_AFTER_MS is
631
+ // 50 minutes and the clock used to run from the ORIGINAL arrival, so at
632
+ // minute 50 the sweep told the same human the exact opposite of what they had
633
+ // been told at minute 45 — while the retry was still running.
634
+ const t = fakeTransport();
635
+ const r = await openAndAcknowledge({ item: ITEM, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
636
+ assurance.noteSession(r.key, "s-1");
637
+ const t0 = readObligation(r.key).openedAt;
638
+
639
+ const settled = await settleSession({
640
+ key: r.key, ok: false, code: 143, error: "session_close exit 143",
641
+ now: t0 + 45 * 60_000, deps: { deliverImpl: t.impl, spokeSinceImpl: () => false },
642
+ });
643
+ assert.equal(settled.willRetry, true);
644
+ assert.match(t.sent.at(-1).text, /retrying it now/i);
645
+
646
+ const stats = await sweepObligations({ now: t0 + 51 * 60_000, deps: { deliverImpl: t.impl, spokeSinceImpl: () => false } });
647
+ assert.equal(stats.staled, 0, "the retry is minutes old, not fifty");
648
+ assert.equal(readObligation(r.key).state, "open");
649
+ assert.doesNotMatch(t.sent.at(-1).text, /stopped retrying/i);
650
+ });
651
+
652
+ test("but a retry that itself goes nowhere DOES eventually stale out", async () => {
653
+ const t = fakeTransport();
654
+ const r = await openAndAcknowledge({ item: ITEM, classResult: CLASS_COMPLEX, deps: { ackSender: t.ackSender } });
655
+ assurance.noteSession(r.key, "s-1");
656
+ const t0 = readObligation(r.key).openedAt;
657
+ await settleSession({
658
+ key: r.key, ok: false, code: 143, error: "session_close exit 143",
659
+ now: t0 + 45 * 60_000, deps: { deliverImpl: t.impl, spokeSinceImpl: () => false },
660
+ });
661
+ const stats = await sweepObligations({
662
+ now: t0 + 45 * 60_000 + STALE_AFTER_MS + 1000,
663
+ deps: { deliverImpl: t.impl, spokeSinceImpl: () => false },
664
+ });
665
+ assert.equal(stats.staled, 1, "the promise still has a floor under it");
666
+ assert.match(t.sent.at(-1).text, /outrun its time limit/i);
667
+ });
668
+ });