@cohortapp/agent-sdk 2.5.1 → 2.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (107) hide show
  1. package/bin/maestro.mjs +305 -89
  2. package/bin/maestro.test.mjs +357 -48
  3. package/docs/runbooks/backup-restore.md +65 -33
  4. package/framework-features.json +4 -4
  5. package/lib/backup/policy.mjs +710 -0
  6. package/lib/backup/policy.test.mjs +305 -0
  7. package/lib/budget-escalate.mjs +133 -0
  8. package/lib/budget-escalate.test.mjs +232 -0
  9. package/lib/budget-guard.envelope.test.mjs +476 -0
  10. package/lib/budget-guard.mjs +853 -75
  11. package/lib/budget-guard.test.mjs +91 -42
  12. package/lib/cadences.mjs +33 -0
  13. package/lib/channels/orgmail/adapter.mjs +88 -3
  14. package/lib/channels/orgmail/adapter.test.mjs +137 -0
  15. package/lib/channels/repeat-suppressor.mjs +198 -0
  16. package/lib/channels/repeat-suppressor.test.mjs +134 -0
  17. package/lib/comms/receipts.mjs +297 -0
  18. package/lib/cost/ledger-row.mjs +333 -0
  19. package/lib/cost/ledger-row.test.mjs +183 -0
  20. package/lib/execution/drive.mjs +28 -1
  21. package/lib/execution/effects.mjs +191 -12
  22. package/lib/execution/effects.test.mjs +50 -11
  23. package/lib/goals/admission.mjs +13 -1
  24. package/lib/goals/admission.test.mjs +26 -1
  25. package/lib/goals/loop.mjs +13 -0
  26. package/lib/kpi-sensors.test.mjs +3 -0
  27. package/lib/mandate/cache.mjs +13 -5
  28. package/lib/mandate/derive.mjs +146 -21
  29. package/lib/mandate/derive.test.mjs +50 -6
  30. package/lib/mandate/model.mjs +32 -4
  31. package/lib/mandate/refresh.test.mjs +16 -2
  32. package/lib/mcp/server.test.mjs +12 -3
  33. package/lib/model-router/economics.mjs +107 -76
  34. package/lib/model-router/economics.test.mjs +64 -46
  35. package/lib/model-router/integration-coverage.test.mjs +39 -37
  36. package/lib/model-router/ledger.mjs +75 -22
  37. package/lib/model-router/ledger.test.mjs +35 -2
  38. package/lib/org/client.mjs +14 -0
  39. package/lib/org/cost-sync.mjs +16 -2
  40. package/lib/org/doctor.mjs +62 -1
  41. package/lib/org/doctor.test.mjs +36 -3
  42. package/lib/org/email-remedy.mjs +49 -0
  43. package/lib/org/engagement-ledger.mjs +376 -0
  44. package/lib/org/engagement-ledger.test.mjs +112 -0
  45. package/lib/org/engagement.mjs +1056 -0
  46. package/lib/org/engagement.test.mjs +739 -0
  47. package/lib/org/messaging.mjs +230 -3
  48. package/lib/org/messaging.test.mjs +110 -1
  49. package/lib/org/param-contract.mjs +56 -2
  50. package/lib/org/param-contract.test.mjs +26 -0
  51. package/lib/org/protocol.checksum +1 -1
  52. package/lib/org/protocol.mjs +5 -0
  53. package/lib/org/protocol.test.mjs +7 -1
  54. package/lib/org/tool-surface.mjs +506 -10
  55. package/lib/org/tool-surface.test.mjs +191 -7
  56. package/lib/org/ui-parity.mjs +333 -6
  57. package/lib/org/ui-parity.test.mjs +96 -3
  58. package/lib/org/work-ledger.mjs +241 -0
  59. package/lib/org/work-ledger.test.mjs +237 -0
  60. package/lib/plan/adoption-e2e.test.mjs +366 -0
  61. package/lib/plan/budget-enforcement.test.mjs +400 -0
  62. package/lib/plan/budget-runtime.mjs +215 -0
  63. package/lib/plan/compile.mjs +201 -5
  64. package/lib/plan/compile.test.mjs +19 -5
  65. package/lib/plan/emit.mjs +8 -0
  66. package/lib/plan/emit.test.mjs +18 -0
  67. package/lib/resource-governor.mjs +58 -12
  68. package/lib/resource-governor.test.mjs +41 -1
  69. package/lib/security/audit-engine.mjs +45 -8
  70. package/lib/security/audit-engine.test.mjs +35 -0
  71. package/lib/setup/enroll-from-cohort.mjs +14 -1
  72. package/lib/setup/sections/mandate.mjs +48 -7
  73. package/lib/setup/sections/mandate.test.mjs +17 -2
  74. package/lib/setup/sections/orgmail.mjs +10 -2
  75. package/lib/setup/state.mjs +83 -2
  76. package/lib/telemetry/collect.mjs +360 -20
  77. package/lib/telemetry/collect.test.mjs +266 -0
  78. package/package.json +1 -1
  79. package/scripts/cost/track-claude-usage.mjs +207 -48
  80. package/scripts/cost/track-claude-usage.test.mjs +148 -0
  81. package/scripts/daemon/agent-daemon.mjs +315 -17
  82. package/scripts/daemon/assurance-e2e.test.mjs +421 -0
  83. package/scripts/daemon/assurance.mjs +944 -0
  84. package/scripts/daemon/assurance.test.mjs +668 -0
  85. package/scripts/daemon/cadence-consumer-governance.test.mjs +56 -0
  86. package/scripts/daemon/cadence-consumer.mjs +147 -9
  87. package/scripts/daemon/cadence-consumer.test.mjs +6 -0
  88. package/scripts/daemon/cadence-handlers.mjs +158 -0
  89. package/scripts/daemon/cadence-handlers.test.mjs +64 -0
  90. package/scripts/daemon/classifier.test.mjs +18 -9
  91. package/scripts/daemon/deliver.mjs +314 -0
  92. package/scripts/daemon/dispatcher-governance.test.mjs +10 -0
  93. package/scripts/daemon/dispatcher.mjs +64 -6
  94. package/scripts/daemon/responder-cost.test.mjs +68 -0
  95. package/scripts/daemon/responder.mjs +351 -298
  96. package/scripts/local-triggers/generate-plists.test.mjs +7 -4
  97. package/scripts/maintenance/backup-run.mjs +415 -0
  98. package/scripts/maintenance/backup-to-cloud.sh +16 -116
  99. package/scripts/org/send-orgmail.mjs +16 -0
  100. package/scripts/record-receipt.sh +63 -0
  101. package/scripts/restore-from-backup.sh +14 -3
  102. package/scripts/restore-from-backup.test.mjs +8 -5
  103. package/scripts/send-email-threaded.py +47 -0
  104. package/scripts/send-sms.sh +4 -0
  105. package/scripts/send-whatsapp.sh +4 -0
  106. package/scripts/setup/init-backup.mjs +93 -38
  107. package/scripts/slack-send.sh +12 -0
@@ -0,0 +1,148 @@
1
+ /**
2
+ * scripts/cost/track-claude-usage.test.mjs — the record contract every spawn
3
+ * path depends on.
4
+ *
5
+ * The defect these lock down: `Number(flags["input-tokens"] || 0)` turned an
6
+ * ABSENT token flag into a literal 0, so "we could not measure this session"
7
+ * and "this session was free" wrote identical rows. Every reader summed both as
8
+ * $0, and the budget governor could never trip.
9
+ */
10
+
11
+ import test from "node:test";
12
+ import assert from "node:assert/strict";
13
+ import { spawnSync } from "node:child_process";
14
+ import { mkdtempSync, rmSync, readFileSync, existsSync } from "node:fs";
15
+ import { tmpdir } from "node:os";
16
+ import { join } from "node:path";
17
+ import { fileURLToPath } from "node:url";
18
+
19
+ const TRACKER = fileURLToPath(new URL("./track-claude-usage.mjs", import.meta.url));
20
+
21
+ function withRoot(fn) {
22
+ const root = mkdtempSync(join(tmpdir(), "track-usage-"));
23
+ try { return fn(root); } finally { rmSync(root, { recursive: true, force: true }); }
24
+ }
25
+
26
+ function record(root, args) {
27
+ const r = spawnSync(process.execPath, [TRACKER, "record", ...args], {
28
+ encoding: "utf-8",
29
+ env: { ...process.env, AGENT_ROOT: root, AGENT_DIR: root },
30
+ });
31
+ assert.equal(r.status, 0, `tracker exited ${r.status}: ${r.stderr}`);
32
+ return r;
33
+ }
34
+
35
+ function rows(root) {
36
+ const date = new Date().toISOString().slice(0, 10);
37
+ const file = join(root, "state/cost-tracking", `${date}.jsonl`);
38
+ if (!existsSync(file)) return [];
39
+ return readFileSync(file, "utf-8").split("\n").filter(Boolean).map((l) => JSON.parse(l));
40
+ }
41
+
42
+ test("record: real usage is priced cache-aware and keeps the authoritative cost", () => {
43
+ withRoot((root) => {
44
+ // The live token profile: almost all prompt tokens are cache reads. The old
45
+ // per-1K, cache-blind table priced this session at $0.075 against a real
46
+ // $0.533 — a 7x under-read on a single row.
47
+ record(root, [
48
+ "--cadence", "inbox", "--source", "dispatcher", "--model", "sonnet",
49
+ "--input-tokens", "16", "--output-tokens", "5000",
50
+ "--cache-read-tokens", "553181", "--cache-creation-tokens", "78000",
51
+ "--total-cost-usd", "0.533412", "--exit", "0",
52
+ ]);
53
+ const [row] = rows(root);
54
+ assert.equal(row.measurement, "measured");
55
+ assert.equal(row.input_tokens, 16);
56
+ assert.equal(row.output_tokens, 5000);
57
+ assert.equal(row.cache_read_input_tokens, 553181);
58
+ assert.equal(row.cache_creation_input_tokens, 78000);
59
+ assert.equal(row.total_cost_usd, 0.533412);
60
+ // The estimate must now land in the same ballpark as the CLI's own figure.
61
+ // (Pre-fix it was 0.075048 — the cache tiers were simply not priced.)
62
+ const drift = Math.abs(row.estimated_usd - row.total_cost_usd) / row.total_cost_usd;
63
+ assert.ok(drift < 0.05, `cache-aware estimate should track the authoritative cost, drift=${drift}`);
64
+ });
65
+ });
66
+
67
+ test("record: an unmeasured session writes NULL tokens, never a zero", () => {
68
+ withRoot((root) => {
69
+ const r = record(root, [
70
+ "--cadence", "inbox", "--source", "dispatcher",
71
+ "--tokens-unknown", "no-usage-field",
72
+ ]);
73
+ const [row] = rows(root);
74
+ assert.equal(row.measurement, "unknown");
75
+ assert.equal(row.input_tokens, null, "NOT 0 — a zero here is what blinded the governor");
76
+ assert.equal(row.output_tokens, null);
77
+ assert.equal(row.estimated_usd, null, "no cost claim at all");
78
+ assert.equal(row.total_cost_usd, null);
79
+ assert.equal(row.unmeasured_reason, "no-usage-field");
80
+ // And it must be loud: this is the only place that observes the gap at
81
+ // write time, and every bug in this system has been a quiet catch.
82
+ assert.match(r.stderr, /UNMEASURED session recorded/);
83
+ assert.match(r.stderr, /unknown, not zero/);
84
+ });
85
+ });
86
+
87
+ test("record: a caller that passes NO token flags degrades to unknown, not $0", () => {
88
+ // An un-migrated caller must fail loudly rather than silently claim a free
89
+ // session — this is the exact shape of the original bug.
90
+ withRoot((root) => {
91
+ record(root, ["--cadence", "inbox", "--source", "dispatcher", "--model", "sonnet"]);
92
+ const [row] = rows(root);
93
+ assert.equal(row.measurement, "unknown");
94
+ assert.equal(row.input_tokens, null);
95
+ assert.equal(row.unmeasured_reason, "caller-passed-no-token-counts");
96
+ });
97
+ });
98
+
99
+ test("record: a genuine zero-token session stays MEASURED and free", () => {
100
+ withRoot((root) => {
101
+ record(root, [
102
+ "--cadence", "inbox", "--source", "dispatcher", "--model", "sonnet",
103
+ "--input-tokens", "0", "--output-tokens", "0", "--total-cost-usd", "0",
104
+ ]);
105
+ const [row] = rows(root);
106
+ assert.equal(row.measurement, "measured", "we looked; the answer was zero");
107
+ assert.equal(row.input_tokens, 0);
108
+ assert.equal(row.estimated_usd, 0);
109
+ });
110
+ });
111
+
112
+ test("record: --non-llm marks attribution rows that never invoked a model", () => {
113
+ withRoot((root) => {
114
+ record(root, ["--cadence", "messaging", "--source", "messaging", "--non-llm"]);
115
+ const [row] = rows(root);
116
+ assert.equal(row.measurement, "n/a");
117
+ assert.equal(row.model, null);
118
+ assert.equal(row.estimated_usd, 0, "$0 is a fact here, not a measurement gap");
119
+ });
120
+ });
121
+
122
+ test("summarise: reports unmeasured sessions instead of averaging them away", () => {
123
+ withRoot((root) => {
124
+ record(root, [
125
+ "--source", "dispatcher", "--model", "sonnet",
126
+ "--input-tokens", "16", "--output-tokens", "5000",
127
+ "--cache-read-tokens", "553181", "--total-cost-usd", "0.5",
128
+ ]);
129
+ record(root, ["--source", "dispatcher", "--tokens-unknown", "empty-stdout"]);
130
+ record(root, ["--source", "messaging", "--non-llm"]);
131
+
132
+ const r = spawnSync(process.execPath, [TRACKER, "summarise", "--days", "1"], {
133
+ encoding: "utf-8",
134
+ env: { ...process.env, AGENT_ROOT: root, AGENT_DIR: root },
135
+ });
136
+ assert.equal(r.status, 0, r.stderr);
137
+ const out = JSON.parse(r.stdout);
138
+ assert.equal(out.measurement.measured_sessions, 1);
139
+ assert.equal(out.measurement.unmeasured_sessions, 1);
140
+ assert.equal(out.measurement.non_llm_rows, 1, "a send is not a session");
141
+ assert.equal(out.measurement.authoritative_usd, 0.5);
142
+ assert.equal(out.totals.sessions, 2, "measured + unmeasured; the send is excluded");
143
+ assert.ok(
144
+ out.measurement.degradations.some((d) => /NO token counts/.test(d)),
145
+ "the gap is named, not swallowed"
146
+ );
147
+ });
148
+ });
@@ -54,6 +54,22 @@ import { classifyItem, isDirectedAtAgent } from "./classifier.mjs";
54
54
  import { dispatch, getStatus, availableSlots, canDispatchBacklog, resetActiveSessions } from "./dispatcher.mjs";
55
55
  import { buildPrompt } from "./prompt-builder.mjs";
56
56
  import { sendQuickResponse, sendHoldingMessage, isQuickReply } from "./responder.mjs";
57
+ // ANSWER ASSURANCE (scripts/daemon/assurance.mjs). The guarantee that an ask is
58
+ // always answered by SOMETHING: an instant model-free acknowledgement when a
59
+ // session is about to make the human wait, an interim update when the work
60
+ // outruns its welcome, a message on every terminal failure, and a durable
61
+ // on-disk obligation that a sweep discharges when this process cannot. It is
62
+ // imported here because this file is where every one of those moments happens —
63
+ // the dispatch decision, the session close, and the tick loop.
64
+ import {
65
+ shouldAcknowledge,
66
+ openAndAcknowledge,
67
+ noteSession,
68
+ settleSession,
69
+ sweepObligations,
70
+ openObligations,
71
+ escalate,
72
+ } from "./assurance.mjs";
57
73
  import { recordPoll, recordClassification, recordSession, writeHealthDashboard } from "./health.mjs";
58
74
  import { acquireLock, releaseLock, updateLock, scanStaleLocks, acquireThreadLock, claimRequest, hasActiveClaim, sweepStaleItemClaims, sanitiseItemId } from "./session-lock.mjs";
59
75
  import { markDeferred } from "./inbox-deferral.mjs";
@@ -69,7 +85,8 @@ const RUNG_MECHANISM = Object.fromEntries(RUNGS.map((r) => [r.id, r.mechanism]))
69
85
  // local SQLite/interaction logs remain the fail-open cache.
70
86
  import { isEnabled as orgEnabled, loadOrgConfig } from "../../lib/org/client.mjs";
71
87
  import { remember as orgRemember } from "../../lib/org/knowledge.mjs";
72
- import { sweepSessionOutcomes } from "./session-outcomes.mjs";
88
+ import { sweepSessionOutcomes, resultTextFromStdout } from "./session-outcomes.mjs";
89
+ import { recordWorkStep as orgRecordWorkStep } from "../../lib/org/work-ledger.mjs";
73
90
  // Observability spine (WS — diagnostics). mintTraceId + withTrace give each
74
91
  // inbound item ONE trace_id that deep callees inherit via AsyncLocalStorage;
75
92
  // emitEvent appends the canonical interaction hops (item_received → … → sent /
@@ -518,6 +535,27 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
518
535
  const result = await _sendQuickResponse(item, classResult, routed);
519
536
  if (result.sent) {
520
537
  markProcessed(item, service);
538
+ // WORK VISIBILITY on the FAST path too, with `deferred: false`. This is
539
+ // what stops the two planes drifting: an ask answered inside the turn
540
+ // gets a row ONLY if its text asks for something to be DONE, and that
541
+ // judgement is made by the server's gate — the same gate, the same
542
+ // answer, whether the agent is driven from here or from hq. Most
543
+ // quick replies leave no board row, which is correct: the answer IS the
544
+ // artifact, and a board full of answered questions is a board nobody
545
+ // reads.
546
+ // `accepted`, not `done`. The quick path posted a reply and nothing else,
547
+ // and the server gate only tracks a `deferred:false` ask when its text
548
+ // ASKED FOR WORK — exactly the case where a reply is not the work. A row
549
+ // created and closed in the same instant claims the ask was delivered by
550
+ // a message; leaving it open is the truthful reading, and the stable
551
+ // askKey means whatever really finishes it closes THIS row.
552
+ void trackWorkStep({
553
+ item,
554
+ stage: "accepted",
555
+ deferred: false,
556
+ note: "Answered directly in the conversation.",
557
+ source: { service, path: "quick_reply" },
558
+ });
521
559
  return { ok: true, path: "quick_reply", reason: classResult.model };
522
560
  }
523
561
  // If quick reply failed to send or was blocked by validation, fall through to dispatch a full session
@@ -525,20 +563,80 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
525
563
  console.warn(`[daemon] Quick reply not sent (${reason}), falling through to session dispatch`);
526
564
  }
527
565
 
528
- // COMPLEX WORK PATH: Send immediate holding message, then dispatch full session
566
+ // COMPLEX WORK PATH — reaching this line IS the decision that the answer is
567
+ // not arriving in this turn. Everything above either answered the human
568
+ // (quick reply) or declined to; from here a session spawns and the human
569
+ // waits minutes. So this is exactly where an acknowledgement is owed, and
570
+ // the only place it can be decided from fact rather than guess.
571
+ //
572
+ // THE OLD GATE ASKED THE CLASSIFIER: `action === respond|draft|research`.
573
+ // That silently excluded `queue` — which is what classifier.mjs's HEURISTIC
574
+ // FALLBACK emits when the LLM classifier itself fails. So precisely when the
575
+ // system was degraded, a directed DM got a 15-45 minute session and no
576
+ // acknowledgement whatsoever. `shouldAcknowledge` asks about control flow
577
+ // instead: a session is spawning and a human is on the other end of a
578
+ // resolvable channel, therefore say something now.
579
+ //
580
+ // The OBLIGATION is opened before the send and outlives this process. If the
581
+ // acknowledgement fails to go out, the debt is still on disk and the
582
+ // assurance sweep retries it — the old code caught that failure, logged it,
583
+ // and left the human staring at a typing indicator.
529
584
  let holdingText = null;
585
+ let obligationKeyForItem = null;
586
+ const ackVerdict = shouldAcknowledge({ willSpawnSession: true, item, source: "inbox" });
530
587
  try {
531
- if (classResult.action === "respond" || classResult.action === "draft" || classResult.action === "research") {
532
- const holdResult = await _sendHoldingMessage(item, classResult);
533
- holdingText = holdResult.sent ? holdResult.holdingText : null;
534
- if (holdResult.sent) {
535
- updateLock(itemId, { holdingSent: true });
536
- }
588
+ // The DEBT is opened whether or not we can speak. Those are two different
589
+ // questions and conflating them is what made a whole class of ask
590
+ // disappear: an item with no resolvable reply channel (calendar, voice, an
591
+ // inbound with no room) got no obligation, so when its session died the
592
+ // failure path had nothing to consult, marked the item processed on the
593
+ // FIRST failure, said nothing, and escalated nothing. The ask was gone.
594
+ // Now the record always exists — `ack:false` merely means the compensating
595
+ // action is an operator escalation instead of a message.
596
+ const opened = await openAndAcknowledge({
597
+ item,
598
+ classResult,
599
+ service,
600
+ traceId: trace_id,
601
+ ack: ackVerdict.ack,
602
+ deps: { ackSender: _sendHoldingMessage },
603
+ });
604
+ obligationKeyForItem = opened.key;
605
+ holdingText = opened.ackText;
606
+ if (opened.acked) {
607
+ updateLock(itemId, { holdingSent: true });
608
+ } else if (ackVerdict.ack) {
609
+ // NOT fatal, NOT silent, NOT forgotten. Loud here; retried by the sweep.
610
+ console.warn(`[daemon] acknowledgement not delivered for ${itemId} (${opened.error}) — obligation ${opened.key} left open for the assurance sweep`);
611
+ counters.bump("assurance.ack_deferred", { service });
612
+ } else {
613
+ console.log(`[daemon] no acknowledgement for ${itemId} (${ackVerdict.reason}) — debt ${opened.key} still tracked`);
537
614
  }
538
615
  } catch (err) {
539
- console.error(`[daemon] Holding message failed (non-fatal): ${err.message}`);
616
+ // openAndAcknowledge does not throw; if it somehow does, the debt may not
617
+ // exist, so this is the one case worth shouting about.
618
+ console.error(`[daemon] acknowledgement path threw for ${itemId}: ${err.message}`);
619
+ emitEvent({ type: EVENT_TYPES.ERROR, trace_id, attrs: { item_id: itemId, service, stage: "acknowledge", error: err.message } });
540
620
  }
541
621
 
622
+ // WORK VISIBILITY, step 1 of 3: the ask goes on the board BEFORE the work
623
+ // starts, not after it finishes. Reaching this line is the same fact that
624
+ // earned the acknowledgement above — the answer is not arriving this turn —
625
+ // so the board row and the "let me look into it" have ONE trigger between
626
+ // them. A human who is about to wait fifteen minutes gets something to look
627
+ // at for those fifteen minutes, with himself tagged on it.
628
+ //
629
+ // `deferred: true` is the whole signal the server's gate needs; everything
630
+ // else about whether this deserves a row is decided there, by the same code
631
+ // that decides it for hq's responder.
632
+ void trackWorkStep({
633
+ item,
634
+ stage: "accepted",
635
+ deferred: true,
636
+ note: classResult && classResult.summary ? `Picked this up. ${classResult.summary}` : "Picked this up — starting work now.",
637
+ source: { service, trace_id, classified: String(classResult && classResult.action) },
638
+ });
639
+
542
640
  // Build prompt with holding message context and dispatch
543
641
  const prompt = await buildPrompt(item, classResult, {
544
642
  type: "inbox",
@@ -553,8 +651,38 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
553
651
  // re-delivered on the next poll; a crash mid-flight leaves the admission for
554
652
  // recoverInFlight() to re-deliver on restart.
555
653
  recordInFlight(item, service);
654
+ if (obligationKeyForItem) noteSession(obligationKeyForItem, null);
556
655
  _dispatch(prompt, item, classResult, "inbox", {
656
+ // Carried into the session child's env so its CLI send lanes stamp their
657
+ // delivery receipts with the debt they discharge. This is what lets the
658
+ // close handler below distinguish "the session answered him" from "the
659
+ // session exited 0 in silence" — and, just as importantly, from "a
660
+ // DIFFERENT session answered a different ask in the same room".
661
+ obligationKey: obligationKeyForItem,
557
662
  onClose: ({ ok, code, sessionId, stdout }) => {
663
+ // EVERY terminal state now ends in a decision about what the human
664
+ // hears. `settleSession` owns that decision and is the only thing here
665
+ // that can speak; the branches below only handle bookkeeping.
666
+ //
667
+ // It cannot be awaited (onClose is sync by contract), so it is a
668
+ // fire-and-forget promise with its own catch — but unlike the fail-open
669
+ // catches around it, a throw HERE means the human may be owed a message,
670
+ // so it is logged at error level and the obligation stays open for the
671
+ // sweep to pick up.
672
+ const settle = obligationKeyForItem
673
+ ? Promise.resolve(settleSession({
674
+ key: obligationKeyForItem,
675
+ ok,
676
+ code,
677
+ sessionId,
678
+ stdout,
679
+ error: ok ? null : `session_close exit ${code}`,
680
+ })).catch((err) => {
681
+ console.error(`[daemon] settleSession threw for ${itemId}: ${err.message} — obligation left open for sweep`);
682
+ return { spoke: false, verdict: "settle_threw", willRetry: false };
683
+ })
684
+ : Promise.resolve({ spoke: false, verdict: "no-obligation", willRetry: false });
685
+
558
686
  if (ok) {
559
687
  markProcessed(item, service);
560
688
  clearInFlight(item);
@@ -570,20 +698,87 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
570
698
  // DAEMON_SESSION_OUTCOMES=1; idempotent + fail-open; fire-and-forget.
571
699
  Promise.resolve(sweepOutcomesForSession({ sessionId, stdout, item, classResult }))
572
700
  .catch(() => { /* fail-open */ });
701
+ // WORK VISIBILITY, step 2 of 3: the row moves to done and carries what
702
+ // the turn actually produced as a comment, so the board says what
703
+ // happened rather than merely that something did. The requester is
704
+ // tagged, so "it's finished" reaches him without him having to look.
705
+ void trackWorkStep({
706
+ item,
707
+ stage: "done",
708
+ deferred: true,
709
+ note: sessionCompletionNote(stdout),
710
+ source: { service, sessionId: String(sessionId || ""), exit_code: String(code) },
711
+ });
573
712
  // Finalize (interaction end): emit `sent` carrying the SAME trace_id
574
713
  // minted at item_received, so the full hop chain (item_received → … →
575
- // sent) groups by one stable id. The session itself posts the actual
576
- // user-facing reply via the Slack/Gmail APIs; this row marks the clean
577
- // close of the work the user's message kicked off.
578
- emitEvent({ type: EVENT_TYPES.SENT, trace_id, attrs: { item_id: itemId, service, exit_code: code } });
714
+ // sent) groups by one stable id.
715
+ //
716
+ // `exit_code: 0` is NOT evidence that the human was answered. The old
717
+ // comment here asserted "the session itself posts the actual reply",
718
+ // and nothing ever checked — so a session that exited 0 having sent
719
+ // nothing (observed: 241s, $2.36, final text "Nothing was sent.") was
720
+ // marked processed forever and emitted as `sent`. The `spoke` attr now
721
+ // carries what settleSession actually established.
722
+ settle.then((s) => {
723
+ emitEvent({
724
+ type: EVENT_TYPES.SENT,
725
+ trace_id,
726
+ attrs: { item_id: itemId, service, exit_code: code, verdict: s.verdict, daemon_spoke: !!s.spoke },
727
+ });
728
+ if (s.verdict === "silent-success") {
729
+ counters.bump("assurance.silent_success_rescued", { service });
730
+ }
731
+ });
579
732
  } else {
580
- // Leave the inbox file un-`.processed` (re-deliverable). Drop the
581
- // admission + release the per-item lock so the next poll re-delivers
582
- // immediately rather than waiting out the stale-lock TTL.
583
733
  clearInFlight(item);
584
734
  const key = inflightKey(item);
585
735
  if (key) releaseLock(key);
586
736
  emitEvent({ type: EVENT_TYPES.ERROR, trace_id, attrs: { item_id: itemId, service, exit_code: code, stage: "session_close" } });
737
+ // WORK VISIBILITY, step 3 of 3 — THE ONE THAT MATTERS. A session that
738
+ // dies used to leave exactly one observable: the typing dot stopping.
739
+ // The row now moves to BLOCKED (never quietly to done — the work is
740
+ // still owed to somebody), carries the exit code, and TAGS the
741
+ // requester. This is the difference between a process that failed and
742
+ // a process that failed invisibly.
743
+ void trackWorkStep({
744
+ item,
745
+ stage: "failed",
746
+ deferred: true,
747
+ note: `The session handling this ended without completing (exit ${code}). It is back on the board, blocked, rather than quietly dropped.`,
748
+ source: { service, sessionId: String(sessionId || ""), exit_code: String(code) },
749
+ });
750
+ // SELF-HEAL, BOUNDED. A transient failure with retries left leaves the
751
+ // inbox file un-`.processed` so the next poll re-delivers it — that was
752
+ // already true, but it was UNBOUNDED and SILENT: the same item could
753
+ // cycle through 45-minute sessions forever with the human hearing
754
+ // nothing. Now the requester is told a retry is running, and when the
755
+ // retries are spent the item is marked processed so the loop STOPS,
756
+ // the human is told it failed, and a durable needs-attention record is
757
+ // written. A dropped ask must leave a trace; it must not leave a loop.
758
+ settle.then((s) => {
759
+ if (!s.willRetry) {
760
+ // `no-obligation` reaches here only if the item had no stable key
761
+ // at all, so nothing durable is tracking it. Marking it processed
762
+ // would delete the ask on its FIRST failure with no message and no
763
+ // trace — the exact silent drop this whole unit exists to end. Write
764
+ // the escalation ourselves before the item goes.
765
+ if (s.verdict === "no-obligation") {
766
+ escalate(
767
+ { key: itemId, sender: item.sender, service, channel: item.channel_id || item.channel || null,
768
+ summary: classResult && classResult.summary, openedAt: Date.now(),
769
+ attempts: 1, sessionId: String(sessionId || ""), traceId: trace_id,
770
+ item: { content: item.content } },
771
+ { failure: { label: `exit_${code}` }, told: false },
772
+ );
773
+ }
774
+ markProcessed(item, service);
775
+ counters.bump("assurance.gave_up", { service, verdict: s.verdict });
776
+ console.error(`[daemon] ${itemId} failed terminally (${s.verdict}); requester told: ${!!s.spoke}`);
777
+ } else {
778
+ counters.bump("assurance.retrying", { service, verdict: s.verdict });
779
+ console.warn(`[daemon] ${itemId} failed (${s.verdict}); retrying — requester told: ${!!s.spoke}`);
780
+ }
781
+ });
587
782
  }
588
783
  },
589
784
  });
@@ -1094,6 +1289,76 @@ export function _resetDaemonOrgCfgCache() {
1094
1289
  _cachedDaemonOrgCfg = undefined;
1095
1290
  }
1096
1291
 
1292
+ /** Longest completion note we will put on a board comment. */
1293
+ const COMPLETION_NOTE_CHARS = 1200;
1294
+
1295
+ /**
1296
+ * What the finished session actually produced, as a board comment.
1297
+ *
1298
+ * Reuses `session-outcomes.mjs#resultTextFromStdout` — the same tolerant
1299
+ * extractor the outcome sweep uses — rather than re-parsing the `claude --print`
1300
+ * envelope a second way. A turn that produced nothing readable still gets an
1301
+ * honest line: "finished" with no evidence is worse than nothing, because it is
1302
+ * the exact claim `exit 0` already made falsely.
1303
+ *
1304
+ * @param {string} stdout the session's raw stdout
1305
+ * @returns {string}
1306
+ */
1307
+ export function sessionCompletionNote(stdout) {
1308
+ let text = "";
1309
+ try {
1310
+ text = String(resultTextFromStdout(stdout) || "").trim();
1311
+ } catch {
1312
+ text = "";
1313
+ }
1314
+ if (!text) return "The session finished but produced no readable result text.";
1315
+ const clipped = text.length > COMPLETION_NOTE_CHARS ? `${text.slice(0, COMPLETION_NOTE_CHARS)}…` : text;
1316
+ return `Done. What the work produced:\n\n${clipped}`;
1317
+ }
1318
+
1319
+ /**
1320
+ * WORK VISIBILITY. Record one step of the work this item kicked off against the
1321
+ * org board (`board.track` → hq's `server/work/ledger.ts`).
1322
+ *
1323
+ * DRIVEN, not left to the model. The three call sites below are the three
1324
+ * moments a human's experience changes — the ask is taken on, the work lands,
1325
+ * the work dies — and none of them may depend on a session remembering to write
1326
+ * something down. `DAEMON_SESSION_OUTCOMES` is the cautionary tale: a whole
1327
+ * outcome-discipline module, correct and tested, that has never run in
1328
+ * production because the flag is absent from the launchd plist. This has no
1329
+ * flag. The gate is on the SERVER, where it is one implementation shared with
1330
+ * hq's responder, and where it can be changed without redeploying a laptop.
1331
+ *
1332
+ * Fail-open and fire-and-forget: never awaited on the dispatch path, never able
1333
+ * to throw into it. But never silent either — the reason is always logged, which
1334
+ * is the rule this entire workstream exists to enforce.
1335
+ *
1336
+ * @param {object} a - { item, stage, deferred?, note?, title?, source? } plus
1337
+ * test seams ({ cfg?, trackImpl?, fetchImpl? })
1338
+ * @returns {Promise<object>} the step result (or a benign refusal)
1339
+ */
1340
+ export async function trackWorkStep(a = {}) {
1341
+ try {
1342
+ const cfg = a.cfg !== undefined ? a.cfg : daemonOrgCfg();
1343
+ const res = await orgRecordWorkStep({ ...a, cfg });
1344
+ if (res && res.tracked) {
1345
+ console.log(
1346
+ `[board] ${a.stage} → task ${res.taskId} (${res.col})` +
1347
+ `${res.created ? " created" : ""}${res.moved ? " moved" : ""}` +
1348
+ `${res.commented ? " commented" : ""}${res.tagged ? ` tagged=${res.tagged}` : ""}`,
1349
+ );
1350
+ } else {
1351
+ // A refusal is INFORMATION, not noise: "no-substance" is the gate working,
1352
+ // "org-disabled" is configuration, and an error frame is a real fault.
1353
+ console.log(`[board] ${a.stage} not tracked: ${(res && res.reason) || "unknown"}`);
1354
+ }
1355
+ return res;
1356
+ } catch (err) {
1357
+ console.warn(`[board] track failed (${a.stage}): ${err && err.message}`);
1358
+ return { tracked: false, reason: "error" };
1359
+ }
1360
+ }
1361
+
1097
1362
  // A distillation longer than this is truncated before it lands in the store —
1098
1363
  // the shared memory holds a salient one-liner, never a transcript.
1099
1364
  const SESSION_DISTILL_MAX_CHARS = 400;
@@ -1528,6 +1793,23 @@ async function main() {
1528
1793
  // if the work had completed). Runs BEFORE the first poll so recovery wins the
1529
1794
  // race against re-acquisition.
1530
1795
  recoverInFlight();
1796
+ // ANSWER ASSURANCE on startup. `recoverInFlight` above re-delivers the WORK;
1797
+ // this speaks to the PEOPLE. Any obligation still open was opened by a daemon
1798
+ // process that no longer exists — meaning a human asked for something, was
1799
+ // told "I'll come back to you", and then this process died mid-session. They
1800
+ // are owed a word before anything else happens, and the sweep's interrupted
1801
+ // branch is what gives it to them. Awaited (unlike the tick-loop call) so the
1802
+ // first poll cannot spawn new work ahead of settling old debts.
1803
+ const owed = openObligations().length;
1804
+ if (owed > 0) {
1805
+ console.log(`[daemon] ${owed} unanswered obligation(s) survived the last run — settling before first poll`);
1806
+ try {
1807
+ const s = await sweepObligations();
1808
+ console.log(`[daemon] startup assurance sweep — acked:${s.acked} interrupted:${s.interrupted} stale:${s.staled} closed:${s.closed}`);
1809
+ } catch (err) {
1810
+ console.error("[daemon] startup assurance sweep failed:", err.message);
1811
+ }
1812
+ }
1531
1813
  // WS4: on startup, reconcile in-flight session resume markers — re-dispatch
1532
1814
  // `claude --print --resume` for any work that was mid-flight when the box
1533
1815
  // rebooted/lost power (3-strike cap → status:blocked). Shutdown paths below
@@ -1558,7 +1840,14 @@ async function main() {
1558
1840
  }
1559
1841
  }, BACKLOG_INTERVAL);
1560
1842
 
1561
- // Health dashboard + stale claim sweep
1843
+ // Health dashboard + stale claim sweep + the ANSWER-ASSURANCE sweep.
1844
+ //
1845
+ // The assurance sweep is the safety net under every other path: it is the one
1846
+ // mechanism with no in-memory state, so it works identically on the first tick
1847
+ // after a restart and the ten-thousandth tick of a healthy process. It is what
1848
+ // makes the promise hold when an acknowledgement's send failed, when a
1849
+ // dispatcher never called back, when a promise rejected unhandled, and when
1850
+ // the daemon itself died mid-session.
1562
1851
  setInterval(() => {
1563
1852
  try {
1564
1853
  writeHealthDashboard();
@@ -1570,6 +1859,15 @@ async function main() {
1570
1859
  } catch (err) {
1571
1860
  console.error("[daemon] Health write error:", err.message);
1572
1861
  }
1862
+ // Separate try: a health-write failure must not skip the sweep, because the
1863
+ // sweep is the thing that speaks to humans.
1864
+ Promise.resolve(sweepObligations())
1865
+ .then((s) => {
1866
+ if (s.acked || s.progressed || s.staled || s.interrupted) {
1867
+ console.log(`[daemon] assurance sweep — acked:${s.acked} progress:${s.progressed} stale:${s.staled} interrupted:${s.interrupted} closed:${s.closed} (open:${s.swept})`);
1868
+ }
1869
+ })
1870
+ .catch((err) => console.error("[daemon] assurance sweep error:", err.message));
1573
1871
  }, HEALTH_INTERVAL);
1574
1872
 
1575
1873
  // Graceful shutdown — clean up active.json so next startup doesn't see stale sessions