@cohortapp/agent-sdk 2.5.1 → 2.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/bin/maestro.mjs +185 -88
  2. package/bin/maestro.test.mjs +175 -48
  3. package/docs/runbooks/backup-restore.md +65 -33
  4. package/framework-features.json +4 -4
  5. package/lib/backup/policy.mjs +710 -0
  6. package/lib/backup/policy.test.mjs +305 -0
  7. package/lib/budget-escalate.mjs +133 -0
  8. package/lib/budget-escalate.test.mjs +232 -0
  9. package/lib/budget-guard.envelope.test.mjs +476 -0
  10. package/lib/budget-guard.mjs +853 -75
  11. package/lib/budget-guard.test.mjs +91 -42
  12. package/lib/cadences.mjs +33 -0
  13. package/lib/channels/orgmail/adapter.mjs +88 -3
  14. package/lib/channels/orgmail/adapter.test.mjs +137 -0
  15. package/lib/channels/repeat-suppressor.mjs +198 -0
  16. package/lib/channels/repeat-suppressor.test.mjs +134 -0
  17. package/lib/comms/receipts.mjs +297 -0
  18. package/lib/cost/ledger-row.mjs +333 -0
  19. package/lib/cost/ledger-row.test.mjs +183 -0
  20. package/lib/execution/drive.mjs +28 -1
  21. package/lib/execution/effects.mjs +191 -12
  22. package/lib/execution/effects.test.mjs +50 -11
  23. package/lib/goals/admission.mjs +13 -1
  24. package/lib/goals/admission.test.mjs +26 -1
  25. package/lib/goals/loop.mjs +13 -0
  26. package/lib/kpi-sensors.test.mjs +3 -0
  27. package/lib/mandate/cache.mjs +13 -5
  28. package/lib/mandate/derive.mjs +146 -21
  29. package/lib/mandate/derive.test.mjs +50 -6
  30. package/lib/mandate/model.mjs +32 -4
  31. package/lib/mandate/refresh.test.mjs +16 -2
  32. package/lib/mcp/server.test.mjs +12 -3
  33. package/lib/model-router/economics.mjs +107 -76
  34. package/lib/model-router/economics.test.mjs +64 -46
  35. package/lib/model-router/integration-coverage.test.mjs +39 -37
  36. package/lib/model-router/ledger.mjs +75 -22
  37. package/lib/model-router/ledger.test.mjs +35 -2
  38. package/lib/org/client.mjs +14 -0
  39. package/lib/org/cost-sync.mjs +16 -2
  40. package/lib/org/doctor.mjs +62 -1
  41. package/lib/org/doctor.test.mjs +36 -3
  42. package/lib/org/email-remedy.mjs +49 -0
  43. package/lib/org/engagement-ledger.mjs +376 -0
  44. package/lib/org/engagement-ledger.test.mjs +112 -0
  45. package/lib/org/engagement.mjs +1056 -0
  46. package/lib/org/engagement.test.mjs +739 -0
  47. package/lib/org/messaging.mjs +230 -3
  48. package/lib/org/messaging.test.mjs +110 -1
  49. package/lib/org/param-contract.mjs +56 -2
  50. package/lib/org/param-contract.test.mjs +26 -0
  51. package/lib/org/protocol.checksum +1 -1
  52. package/lib/org/protocol.mjs +5 -0
  53. package/lib/org/protocol.test.mjs +7 -1
  54. package/lib/org/tool-surface.mjs +506 -10
  55. package/lib/org/tool-surface.test.mjs +191 -7
  56. package/lib/org/ui-parity.mjs +333 -6
  57. package/lib/org/ui-parity.test.mjs +96 -3
  58. package/lib/org/work-ledger.mjs +241 -0
  59. package/lib/org/work-ledger.test.mjs +237 -0
  60. package/lib/plan/adoption-e2e.test.mjs +366 -0
  61. package/lib/plan/budget-enforcement.test.mjs +400 -0
  62. package/lib/plan/budget-runtime.mjs +215 -0
  63. package/lib/plan/compile.mjs +201 -5
  64. package/lib/plan/compile.test.mjs +19 -5
  65. package/lib/plan/emit.mjs +8 -0
  66. package/lib/plan/emit.test.mjs +18 -0
  67. package/lib/resource-governor.mjs +58 -12
  68. package/lib/resource-governor.test.mjs +41 -1
  69. package/lib/security/audit-engine.mjs +45 -8
  70. package/lib/security/audit-engine.test.mjs +35 -0
  71. package/lib/setup/enroll-from-cohort.mjs +14 -1
  72. package/lib/setup/sections/mandate.mjs +48 -7
  73. package/lib/setup/sections/mandate.test.mjs +17 -2
  74. package/lib/setup/sections/orgmail.mjs +10 -2
  75. package/lib/setup/state.mjs +83 -2
  76. package/lib/telemetry/collect.mjs +360 -20
  77. package/lib/telemetry/collect.test.mjs +266 -0
  78. package/package.json +1 -1
  79. package/scripts/cost/track-claude-usage.mjs +207 -48
  80. package/scripts/cost/track-claude-usage.test.mjs +148 -0
  81. package/scripts/daemon/agent-daemon.mjs +315 -17
  82. package/scripts/daemon/assurance-e2e.test.mjs +421 -0
  83. package/scripts/daemon/assurance.mjs +944 -0
  84. package/scripts/daemon/assurance.test.mjs +668 -0
  85. package/scripts/daemon/cadence-consumer-governance.test.mjs +56 -0
  86. package/scripts/daemon/cadence-consumer.mjs +147 -9
  87. package/scripts/daemon/cadence-consumer.test.mjs +6 -0
  88. package/scripts/daemon/cadence-handlers.mjs +158 -0
  89. package/scripts/daemon/cadence-handlers.test.mjs +64 -0
  90. package/scripts/daemon/deliver.mjs +314 -0
  91. package/scripts/daemon/dispatcher-governance.test.mjs +10 -0
  92. package/scripts/daemon/dispatcher.mjs +64 -6
  93. package/scripts/daemon/responder-cost.test.mjs +68 -0
  94. package/scripts/daemon/responder.mjs +351 -298
  95. package/scripts/local-triggers/generate-plists.test.mjs +7 -4
  96. package/scripts/maintenance/backup-run.mjs +415 -0
  97. package/scripts/maintenance/backup-to-cloud.sh +16 -116
  98. package/scripts/org/send-orgmail.mjs +16 -0
  99. package/scripts/record-receipt.sh +63 -0
  100. package/scripts/restore-from-backup.sh +14 -3
  101. package/scripts/restore-from-backup.test.mjs +8 -5
  102. package/scripts/send-email-threaded.py +47 -0
  103. package/scripts/send-sms.sh +4 -0
  104. package/scripts/send-whatsapp.sh +4 -0
  105. package/scripts/setup/init-backup.mjs +93 -38
  106. package/scripts/slack-send.sh +12 -0
@@ -1,14 +1,23 @@
1
1
  /**
2
2
  * responder.mjs — Quick response layer
3
3
  *
4
- * Handles two scenarios with a single short-lived `claude --print` call:
4
+ * Handles two scenarios, and they no longer cost the same thing:
5
5
  *
6
- * 1. SIMPLE REPLIES: Invokes the Claude Code CLI to generate a response,
7
- * then posts directly via Slack/Gmail API.
6
+ * 1. SIMPLE REPLIES: a short-lived `claude --print` call generates an answer,
7
+ * which is then delivered via ./deliver.mjs. This is the only path here that
8
+ * spends a model call, because it is the only one producing an ANSWER.
8
9
  *
9
- * 2. HOLDING MESSAGES: For complex items that need a full session,
10
- * generates and sends an immediate acknowledgment so the sender
11
- * knows it's being worked on.
10
+ * 2. ACKNOWLEDGEMENTS: for complex items that need a full session, an immediate
11
+ * holding message so the sender knows the work has started.
12
+ *
13
+ * THIS ONE USED TO SPAWN A MODEL TOO, AND THAT WAS THE BUG. A cold
14
+ * `claude --print` child under a 60-second cap, on a machine at 95-99%
15
+ * memory, to compose two sentences of courtesy: it lost that race 20 times
16
+ * in 32, and each loss was caught, logged as `holding_message_error`, and
17
+ * told the requester nothing while a 15-45 minute session ran behind a
18
+ * typing indicator. It is now composed by assurance.composeAck — a string
19
+ * concat and one HTTP call, which cannot time out, cannot be starved of
20
+ * memory, and cannot be refused by a spend cap.
12
21
  *
13
22
  * Migrated off `@anthropic-ai/sdk` per CEO directive (Slack DM
14
23
  * D099N1JGKRQ, 2026-04-27 09:38Z + 11:33Z): all agent daemon model
@@ -16,8 +25,8 @@
16
25
  * subscription), not the Anthropic API.
17
26
  */
18
27
 
19
- import { readFileSync, writeFileSync, readdirSync, appendFileSync, mkdirSync } from "fs";
20
- import { execFileSync, spawn } from "child_process";
28
+ import { readFileSync, readdirSync, appendFileSync, mkdirSync, existsSync } from "fs";
29
+ import { spawn } from "child_process";
21
30
  import { join } from "path";
22
31
  import { randomUUID } from "crypto";
23
32
  import { checkRecentlySent, registerSent } from "./session-lock.mjs";
@@ -27,21 +36,29 @@ import { routingKey as deriveRoutingKey, createRouter } from "./lib/session-rout
27
36
  // scopes tools when the operator opts in (MAESTRO_SCOPED_PERMISSIONS=1), so the
28
37
  // quick-reply / holding spawn path honours the same gate as every other spawn.
29
38
  import { sessionPermissionArgs } from "../../lib/session-permissions.mjs";
30
- import { screenOutbound } from "../../lib/comms/send-gate.mjs";
31
39
  // Observability spine (WS — diagnostics). emitEvent("sent") marks a successful
32
- // quick-reply / holding send (interaction-end on the quick path); counters.bump
33
- // ("send.blocked") tallies a guard / send-gate / dedup block. When the send-gate
34
- // itself fails OPEN (infra error on an internal Slack channel) we emit
35
- // "onGuardFail" on the lifecycle bus so the daemon's real subscriber counts it —
36
- // proving the bus is live with a real consumer. All fail-open (never throw).
40
+ // quick-reply / acknowledgement send; counters.bump("send.blocked") tallies a
41
+ // guard / send-gate / dedup block. Both fail-open (never throw).
42
+ //
43
+ // The send-gate screen and its fail-open `onGuardFail` emit moved WITH the
44
+ // transport into ./deliver.mjs, so every outbound — answer, acknowledgement,
45
+ // progress update, failure notice — passes the one chokepoint rather than only
46
+ // the two that happened to live in this file.
37
47
  import { emitEvent, EVENT_TYPES } from "../../lib/diagnostics/events.mjs";
38
48
  import * as counters from "../../lib/diagnostics/counters.mjs";
39
- import { getHookBus } from "../../lib/hooks/bus.mjs";
40
49
  // The execution ladder's rung vocabulary. This responder IS rungs 0-1
41
50
  // (`function_call` / `skill`); `isQuickReply` consults the ladder's routing
42
51
  // decision so a reply that needs a session, a plugin, a workflow or a team is
43
52
  // not answered here just because the classifier called it simple.
44
53
  import { rungById } from "../../lib/execution/route.mjs";
54
+ // Transport, shared with the assurance layer. Sending is not generating: these
55
+ // put an exact string in front of a human with no model in the loop, which is
56
+ // the property the acknowledgement path needed and did not have.
57
+ import { deliver, deliverWithRetry, resolveSlackChannel } from "./deliver.mjs";
58
+ // The acknowledgement is now COMPOSED, not generated. See assurance.mjs for why
59
+ // (in short: a 60s `claude --print` spawn to write "let me look into it" lost
60
+ // its own race 2 times in 3, and lost it silently).
61
+ import { composeAck } from "./assurance.mjs";
45
62
 
46
63
  const AGENT_REPO_DIR = process.env.AGENT_DIR || join(new URL(".", import.meta.url).pathname, "../..");
47
64
  const SONNET_MODEL = "claude-sonnet-4-6";
@@ -124,6 +141,97 @@ function getSlackToken() {
124
141
 
125
142
  let cachedPreamble = null;
126
143
 
144
+ /**
145
+ * Why this session could not be measured, or null when it could.
146
+ *
147
+ * Exported and pure so the FAILING paths are testable without a spawn harness —
148
+ * they are the ones that used to record nothing at all, and they are the
149
+ * expensive ones (a 60s timeout is a full minute of paid model work).
150
+ *
151
+ * @param {{parsed:object|null, timedOut:boolean, exitCode:number|null,
152
+ * parseError:Error|null, timeoutMs:number}} o
153
+ * @returns {string|null}
154
+ */
155
+ export function unmeasuredReasonFor({ parsed, timedOut, exitCode, parseError, timeoutMs }) {
156
+ if (parsed) return null;
157
+ if (timedOut) return `cli-timeout-${timeoutMs}ms`;
158
+ if (exitCode !== 0 && exitCode != null) return `cli-exit-${exitCode}`;
159
+ if (parseError) return "json-parse-failed";
160
+ return "no-json-envelope";
161
+ }
162
+
163
+ /**
164
+ * Append a cost-ledger row for a responder session (source "responder").
165
+ *
166
+ * The responder spawns `claude --print` for every quick reply and holding
167
+ * message and, until now, recorded NOTHING — that spend existed only on the
168
+ * Anthropic bill. A budget governor cannot cap what it cannot see, so an
169
+ * entire class of the agent's paid work was structurally exempt.
170
+ *
171
+ * Token counts come from the run's `--output-format json` envelope. When the
172
+ * envelope is missing or unparseable we record measurement:"unknown" via
173
+ * --tokens-unknown, NEVER a zero: a zero that means "unmeasured" is what made
174
+ * the governor blind in the first place.
175
+ *
176
+ * EVERY EXIT PATH RECORDS, INCLUDING THE FAILING ONES. A timeout and a non-zero
177
+ * exit are the two most expensive outcomes this daemon has — the model ran, the
178
+ * tokens were generated and billed, and only the delivery failed. Recording only
179
+ * the happy path is the same "unmeasured looks free" bug one level up: on a
180
+ * single observed day 15 sessions hit `CLAUDE_CLI_TIMEOUT_MS` (60s of paid model
181
+ * work each) and wrote no ledger row at all, so neither the governor nor doctor
182
+ * could see them. The caller now records exactly once per spawn, on `close`,
183
+ * BEFORE the settled/exit-code returns.
184
+ *
185
+ * Best-effort, detached, never throws — cost accounting must not be able to
186
+ * break a reply.
187
+ *
188
+ * @param {{json:object|null, model:string, durationMs:number, exitCode:number, unmeasuredReason:string|null}} o
189
+ */
190
+ function recordResponderCost({ json, model, durationMs, exitCode, unmeasuredReason }) {
191
+ try {
192
+ const trackerPath = join(AGENT_REPO_DIR, "scripts/cost/track-claude-usage.mjs");
193
+ if (!existsSync(trackerPath)) return;
194
+
195
+ const usage = json && typeof json.usage === "object" && json.usage ? json.usage : null;
196
+ const inputTokens = usage ? Number(usage.input_tokens) : NaN;
197
+ const outputTokens = usage ? Number(usage.output_tokens) : NaN;
198
+ const measured = Number.isFinite(inputTokens) && Number.isFinite(outputTokens);
199
+
200
+ // Map the CLI-resolved model id onto the tracker's coarse pricing class.
201
+ const resolved = json && typeof json.model === "string" ? json.model : model || "";
202
+ const modelClass = /opus/i.test(resolved) ? "opus" : /haiku/i.test(resolved) ? "haiku" : "sonnet";
203
+
204
+ const args = [
205
+ trackerPath, "record",
206
+ "--cadence", "responder",
207
+ "--source", "responder",
208
+ "--model", modelClass,
209
+ "--duration-ms", String(durationMs),
210
+ "--exit", String(exitCode),
211
+ ];
212
+
213
+ if (measured) {
214
+ args.push("--input-tokens", String(inputTokens));
215
+ args.push("--output-tokens", String(outputTokens));
216
+ const cacheRead = Number(usage.cache_read_input_tokens);
217
+ if (Number.isFinite(cacheRead)) args.push("--cache-read-tokens", String(cacheRead));
218
+ const cacheWrite = Number(usage.cache_creation_input_tokens);
219
+ if (Number.isFinite(cacheWrite)) args.push("--cache-creation-tokens", String(cacheWrite));
220
+ const totalCost = Number(json.total_cost_usd);
221
+ if (Number.isFinite(totalCost) && totalCost >= 0) args.push("--total-cost-usd", String(totalCost));
222
+ } else {
223
+ const reason = unmeasuredReason || (json ? "no-usage-field" : "no-json-envelope");
224
+ args.push("--tokens-unknown", reason);
225
+ console.warn(`[responder] session usage unmeasured (${reason}) — recorded as unknown, not zero`);
226
+ }
227
+
228
+ spawn(process.execPath, args, {
229
+ stdio: "ignore",
230
+ env: { ...process.env, AGENT_ROOT: AGENT_REPO_DIR, AGENT_DIR: AGENT_REPO_DIR },
231
+ }).unref();
232
+ } catch { /* cost tracking is best-effort */ }
233
+ }
234
+
127
235
  // Spawn `claude --print` with the supplied system + user prompts and model.
128
236
  // Mirrors the pattern used in classifier.mjs:
129
237
  // • child_process.spawn (not exec) — avoids shell-escape injection on
@@ -144,6 +252,7 @@ let cachedPreamble = null;
144
252
  // @returns {Promise<{ text: string, jsonResult: object|null, exitCode: number }>}
145
253
  function runClaudeCLI(systemPrompt, userPrompt, model, opts = {}) {
146
254
  const { sessionId = null, router = null, routingKey = null } = opts;
255
+ const startedAt = Date.now();
147
256
 
148
257
  return new Promise((resolvePromise, rejectPromise) => {
149
258
  const args = [
@@ -152,11 +261,18 @@ function runClaudeCLI(systemPrompt, userPrompt, model, opts = {}) {
152
261
  ...daemonClaudeArgs(),
153
262
  "--model", model,
154
263
  "--append-system-prompt", systemPrompt,
264
+ // --output-format json is only valid in combination with --print (per b1
265
+ // report). We always pass --print above, so this is safe.
266
+ //
267
+ // This used to be requested ONLY when a sessionId was supplied, which
268
+ // meant every non-routed reply ran with no usage envelope and therefore
269
+ // no way to price it. Every reply this daemon sends costs real money;
270
+ // asking for the envelope unconditionally is what makes it measurable.
271
+ // The parse below already has a raw-text fallback for older CLIs.
272
+ "--output-format", "json",
155
273
  ];
156
274
  if (sessionId) {
157
- // --output-format json is only valid in combination with --print (per
158
- // b1 report). We always pass --print above, so this is safe.
159
- args.push("--session-id", sessionId, "--output-format", "json");
275
+ args.push("--session-id", sessionId);
160
276
  }
161
277
 
162
278
  const proc = spawn(CLAUDE_BIN, args, {
@@ -170,10 +286,17 @@ function runClaudeCLI(systemPrompt, userPrompt, model, opts = {}) {
170
286
  let stdout = "";
171
287
  let stderr = "";
172
288
  let settled = false;
289
+ // A spawn is billed exactly once, on its first `close`, whatever the exit
290
+ // path was. `settled` tracks the PROMISE; this tracks the LEDGER, and the
291
+ // two must not be conflated — that conflation is what made every timeout
292
+ // and every non-zero exit invisible spend.
293
+ let costRecorded = false;
294
+ let timedOut = false;
173
295
 
174
296
  const timer = setTimeout(() => {
175
297
  if (settled) return;
176
298
  settled = true;
299
+ timedOut = true;
177
300
  try { proc.kill("SIGTERM"); } catch (_) { /* noop */ }
178
301
  setTimeout(() => { try { if (!proc.killed) proc.kill("SIGKILL"); } catch (_) { /* noop */ } }, 2000);
179
302
  rejectPromise(new Error(`claude CLI timed out after ${CLAUDE_CLI_TIMEOUT_MS}ms`));
@@ -205,6 +328,43 @@ function runClaudeCLI(systemPrompt, userPrompt, model, opts = {}) {
205
328
  });
206
329
  }
207
330
 
331
+ // (b3) We always ask for JSON now, so always try to parse it. Parsed HERE,
332
+ // above every early return, because the cost record needs it on the
333
+ // failing paths too: a CLI that timed out or exited non-zero has often
334
+ // already emitted a usable usage envelope, and even when it has not, the
335
+ // session must be billed as UNMEASURED rather than not billed at all.
336
+ const trimmed = (stdout || "").trim();
337
+ let parsed = null;
338
+ let parseError = null;
339
+ try {
340
+ // Per b1 report: top-level `session_id` (snake_case UUID), `result`
341
+ // (text), `is_error` (bool). Top-level `uuid` is the message UUID,
342
+ // NOT the session id — do NOT use it.
343
+ parsed = trimmed ? JSON.parse(trimmed) : null;
344
+ if (!parsed) parseError = new Error("no stdout");
345
+ } catch (err) {
346
+ // Legacy fallback (rollout-safety): older CLIs or unexpected output
347
+ // shapes shouldn't crash the daemon. Log a warning, surface the raw
348
+ // text, and let the caller decide whether to call router.touch.
349
+ parseError = err;
350
+ console.warn(`[responder] claude CLI JSON parse failed (sessionId=${sessionId}): ${err.message} — falling back to raw stdout`);
351
+ }
352
+
353
+ // Every reply this daemon sends is a paid Claude session — INCLUDING the
354
+ // ones that time out and the ones that exit non-zero, which are the two
355
+ // most expensive outcomes there are (the model ran; only the delivery
356
+ // failed). Record before any early return, exactly once per spawn.
357
+ if (!costRecorded) {
358
+ costRecorded = true;
359
+ recordResponderCost({
360
+ json: parsed,
361
+ model,
362
+ durationMs: Date.now() - startedAt,
363
+ exitCode: code,
364
+ unmeasuredReason: unmeasuredReasonFor({ parsed, timedOut, exitCode: code, parseError, timeoutMs: CLAUDE_CLI_TIMEOUT_MS }),
365
+ });
366
+ }
367
+
208
368
  if (settled) return;
209
369
  settled = true;
210
370
  clearTimeout(timer);
@@ -214,24 +374,10 @@ function runClaudeCLI(systemPrompt, userPrompt, model, opts = {}) {
214
374
  return;
215
375
  }
216
376
 
217
- // (b3) If we asked for JSON, parse it. Otherwise return raw text.
218
- if (sessionId) {
219
- const trimmed = (stdout || "").trim();
220
- try {
221
- const parsed = JSON.parse(trimmed);
222
- // Per b1 report: top-level `session_id` (snake_case UUID), `result`
223
- // (text), `is_error` (bool). Top-level `uuid` is the message UUID,
224
- // NOT the session id — do NOT use it.
225
- resolvePromise({ text: parsed.result ?? "", jsonResult: parsed, exitCode: code });
226
- } catch (parseErr) {
227
- // Legacy fallback (rollout-safety): older CLIs or unexpected output
228
- // shapes shouldn't crash the daemon. Log a warning, surface the raw
229
- // text, and let the caller decide whether to call router.touch.
230
- console.warn(`[responder] claude CLI JSON parse failed (sessionId=${sessionId}): ${parseErr.message} — falling back to raw stdout`);
231
- resolvePromise({ text: trimmed, jsonResult: null, exitCode: code });
232
- }
377
+ if (parsed) {
378
+ resolvePromise({ text: parsed.result ?? "", jsonResult: parsed, exitCode: code });
233
379
  } else {
234
- resolvePromise({ text: stdout, jsonResult: null, exitCode: code });
380
+ resolvePromise({ text: trimmed, jsonResult: null, exitCode: code });
235
381
  }
236
382
  });
237
383
 
@@ -264,16 +410,31 @@ function logResponse(entry) {
264
410
  /**
265
411
  * Write an interaction record so future sessions can see this exchange.
266
412
  * Both the incoming message and the agent's reply are logged.
413
+ *
414
+ * THIS USED TO REFUSE EVERY SERVICE BUT SLACK. The guard read
415
+ * `item.service !== "slack"`, so a Cohort conversation — the agent's own
416
+ * workspace, and the only surface the owner actually uses — was never written
417
+ * down. `loadConversationHistory` then read from a directory tree that was
418
+ * therefore never created, returned null for every item on every service, and
419
+ * the generator answered a bare "?" chase message with "What's on your mind?".
420
+ * The owner's next line was "Why can't you see the context of our ongoing
421
+ * conversation?". That question had a one-word answer, and it was this guard.
422
+ *
423
+ * Records are now filed under the item's own service, so each service's history
424
+ * reads back through the matching lookup in `loadConversationHistory`.
267
425
  */
268
- function logInteraction(item, responseText) {
269
- if (!item || !item.sender || item.service !== "slack") return;
426
+ function logInteraction(item, responseText, o = {}) {
427
+ if (!item || !item.sender || !item.service) return;
428
+ const service = String(item.service);
270
429
  const senderSlug = item.sender.replace(/\s+/g, "-").toLowerCase();
271
- const channelId = resolveSlackChannel(item);
430
+ const channelId = service === "slack"
431
+ ? resolveSlackChannel(item)
432
+ : (item.channel_id || item.channel || null);
272
433
 
273
434
  // Write to both channel-ID and sender-slug directories
274
435
  const dirs = [];
275
- if (channelId) dirs.push(join(AGENT_REPO_DIR, "memory", "interactions", "slack", channelId));
276
- dirs.push(join(AGENT_REPO_DIR, "memory", "interactions", "slack", `dm-${senderSlug}`));
436
+ if (channelId) dirs.push(join(AGENT_REPO_DIR, "memory", "interactions", service, String(channelId)));
437
+ dirs.push(join(AGENT_REPO_DIR, "memory", "interactions", service, `dm-${senderSlug}`));
277
438
 
278
439
  const incomingEntry = {
279
440
  ts: item.ts || item.timestamp || new Date().toISOString(),
@@ -282,7 +443,7 @@ function logInteraction(item, responseText) {
282
443
  type: item.is_reply ? "thread_reply" : "dm",
283
444
  content: item.content || "",
284
445
  received_at: new Date().toISOString(),
285
- classification: "quick_reply",
446
+ classification: o.classification || "quick_reply",
286
447
  response_sent: true,
287
448
  };
288
449
 
@@ -293,7 +454,7 @@ function logInteraction(item, responseText) {
293
454
  type: "reply",
294
455
  content: responseText,
295
456
  sent_at: new Date().toISOString(),
296
- via: "quick_reply_responder",
457
+ via: o.classification === "holding" ? "holding_ack" : "quick_reply_responder",
297
458
  };
298
459
 
299
460
  for (const dir of dirs) {
@@ -362,15 +523,23 @@ function loadUserProfile(sender) {
362
523
  // Load recent conversation history for context
363
524
  // ---------------------------------------------------------------------------
364
525
 
526
+ /**
527
+ * The reader half of `logInteraction`. It looked ONLY under
528
+ * `memory/interactions/slack/…`, so even once a Cohort exchange was written it
529
+ * would not have been found. Both halves are now keyed on the item's service.
530
+ */
365
531
  function loadConversationHistory(item) {
366
- if (!item || !item.sender) return null;
532
+ if (!item || !item.sender || !item.service) return null;
367
533
 
534
+ const service = String(item.service);
368
535
  const senderSlug = item.sender.replace(/\s+/g, "-").toLowerCase();
369
- const channelId = resolveSlackChannel(item);
536
+ const channelId = service === "slack"
537
+ ? resolveSlackChannel(item)
538
+ : (item.channel_id || item.channel || null);
370
539
 
371
540
  const candidateDirs = [];
372
- if (channelId) candidateDirs.push(join(AGENT_REPO_DIR, "memory", "interactions", "slack", channelId));
373
- candidateDirs.push(join(AGENT_REPO_DIR, "memory", "interactions", "slack", `dm-${senderSlug}`));
541
+ if (channelId) candidateDirs.push(join(AGENT_REPO_DIR, "memory", "interactions", service, String(channelId)));
542
+ candidateDirs.push(join(AGENT_REPO_DIR, "memory", "interactions", service, `dm-${senderSlug}`));
374
543
 
375
544
  const entries = [];
376
545
  for (const dir of candidateDirs) {
@@ -412,9 +581,13 @@ function loadConversationHistory(item) {
412
581
  // generateResponse() through this slot.
413
582
  let _generateResponse;
414
583
  /**
415
- * Test seam: replace the response generator. Pass a function `(item,
416
- * classResult, isHolding) => Promise<string>` to force reply text; pass nothing
417
- * to restore the real CLI-backed generator. Returns a restore fn.
584
+ * Test seam: replace the response generator. Pass a function
585
+ * `(item, classResult) => Promise<string>` to force reply text; pass nothing to
586
+ * restore the real CLI-backed generator. Returns a restore fn.
587
+ *
588
+ * The third `isHolding` parameter is gone — this seam now covers ANSWERS only.
589
+ * The acknowledgement path no longer generates anything, so there is nothing
590
+ * there to stub.
418
591
  */
419
592
  export function setGenerateResponseForTests(fn) {
420
593
  const prev = _generateResponse;
@@ -422,28 +595,23 @@ export function setGenerateResponseForTests(fn) {
422
595
  return () => { _generateResponse = prev; };
423
596
  }
424
597
 
425
- async function realGenerateResponse(item, classResult, isHolding = false) {
598
+ /**
599
+ * Generate an ANSWER. Only an answer.
600
+ *
601
+ * This used to take an `isHolding` flag and, when set, spend a model call
602
+ * writing an acknowledgement. That branch is gone: acknowledgements are
603
+ * composed (assurance.composeAck), and the flag is deliberately not kept "just
604
+ * in case", because keeping it is how it comes back. The one thing a model was
605
+ * adding to a courtesy sentence was phrasing variety, at a cost of 13-46
606
+ * seconds when it worked and total silence when it did not.
607
+ */
608
+ async function realGenerateResponse(item, classResult) {
426
609
  const preamble = loadPreamble();
427
610
  const profile = loadUserProfile(item.sender);
428
- // Use Sonnet for holding messages (speed matters), but the classifier's
429
- // recommended model for actual replies (Opus for CEO, Sonnet for routine)
430
- const model = isHolding ? SONNET_MODEL : (classResult.model === "opus" ? "claude-opus-4-6" : SONNET_MODEL);
431
-
432
- let systemPrompt;
433
- if (isHolding) {
434
- systemPrompt = `${preamble}
435
-
436
- You are generating a brief acknowledgment/holding message. The sender has made a request that requires complex work. You need to:
437
- 1. Acknowledge receipt of their message
438
- 2. Briefly confirm what you understand they need
439
- 3. Let them know you're working on it
440
- 4. Give a rough timeframe if appropriate (e.g., "I'll have this ready within the hour")
441
-
442
- Keep it to 2-3 sentences max. Be warm but efficient. Match the sender's communication style.
443
- Do NOT use phrases like "I'm on it!" or overly casual language. Be professional.
444
- ${profile ? `\nSender profile:\n${profile}` : ""}`;
445
- } else {
446
- systemPrompt = `${preamble}
611
+ // The classifier's recommended model (Opus for CEO, Sonnet for routine).
612
+ const model = classResult.model === "opus" ? "claude-opus-4-6" : SONNET_MODEL;
613
+
614
+ const systemPrompt = `${preamble}
447
615
 
448
616
  You are generating a direct response to this message. Be concise and actionable.
449
617
  If it's a question, answer it. If it's a request, confirm and describe what you'll do or have done.
@@ -452,7 +620,6 @@ If it's informational, acknowledge appropriately.
452
620
  Keep responses focused — 1-4 sentences for simple items, up to a short paragraph for more nuanced ones.
453
621
  Match the sender's tone and urgency level.
454
622
  ${profile ? `\nSender profile:\n${profile}` : ""}`;
455
- }
456
623
 
457
624
  const conversationHistory = loadConversationHistory(item);
458
625
 
@@ -569,171 +736,17 @@ function validateQuickReply(text) {
569
736
  }
570
737
 
571
738
  // ---------------------------------------------------------------------------
572
- // Send via Slack API
573
- // ---------------------------------------------------------------------------
574
-
575
- async function sendSlackMessage(channel, text, threadTs = null) {
576
- // Unified send-gate (P0-3): the daemon quick-reply/holding path must pass the
577
- // same chokepoint as the shell senders + BaseAdapter — banned-phrase, AI
578
- // disclosure, and information-barrier screening at one place. Block on deny.
579
- try {
580
- const gate = await screenOutbound({ channel: "slack", recipient: channel, text, agentRoot: AGENT_REPO_DIR });
581
- if (gate && gate.allow === false) {
582
- throw new Error(`send-gate blocked Slack reply: ${gate.reason}`);
583
- }
584
- } catch (err) {
585
- if (/send-gate blocked/.test(err.message)) throw err; // a real block propagates
586
- // gate infra error (module/policy unreadable): fail-open for internal Slack
587
- // channels (matches send-gate's internal posture) — never silently drop.
588
- console.warn(`[responder] send-gate check errored (allowing internal Slack): ${err.message}`);
589
- // Observability: the send-gate just FAILED OPEN — surface it on the lifecycle
590
- // bus so the daemon's onGuardFail subscriber counts it (the bus's one real
591
- // consumer). Fire-and-forget; emit is fully isolated and never throws.
592
- try {
593
- Promise.resolve(getHookBus().emit("onGuardFail", { guard: "send_gate", channel: "slack", recipient: channel, reason: err.message, failed_open: true }))
594
- .catch(() => { /* bus emit is isolated; never propagate */ });
595
- } catch { /* never let telemetry break the send path */ }
596
- }
597
-
598
- // Always use user token — bot tokens can't access DM channels
599
- const token = getSlackToken();
600
- if (!token) throw new Error("No Slack token available (set SLACK_USER_TOKEN in .env)");
601
-
602
- const body = {
603
- channel,
604
- text,
605
- ...(threadTs ? { thread_ts: threadTs } : {}),
606
- };
607
-
608
- const res = await fetch("https://slack.com/api/chat.postMessage", {
609
- method: "POST",
610
- headers: {
611
- "Authorization": `Bearer ${token}`,
612
- "Content-Type": "application/json",
613
- },
614
- body: JSON.stringify(body),
615
- });
616
-
617
- const data = await res.json();
618
- if (!data.ok) {
619
- throw new Error(`Slack API error: ${data.error}`);
620
- }
621
-
622
- return data;
623
- }
624
-
625
- // ---------------------------------------------------------------------------
626
- // Send via Gmail API (draft or send)
627
- // Currently logs the response — full Gmail send integration can be added
628
- // ---------------------------------------------------------------------------
629
-
630
- async function sendGmailResponse(item, text) {
631
- const to = item.sender_email || item.sender;
632
- const subject = `Re: ${item.subject || "(no subject)"}`;
633
- const sendScript = join(AGENT_REPO_DIR, "scripts", "send-email-threaded.py");
634
-
635
- try {
636
- const args = [sendScript, to, subject, text];
637
- if (item.subject) {
638
- args.push("--reply-to-subject", item.subject);
639
- }
640
-
641
- execFileSync("python3", args, {
642
- cwd: AGENT_REPO_DIR,
643
- timeout: 30000,
644
- encoding: "utf-8",
645
- env: { ...process.env },
646
- });
647
-
648
- return { sent: true, via: "smtp", to };
649
- } catch (err) {
650
- console.error(`[responder] Gmail send failed for ${to}: ${err.message}`);
651
- // Fall back to draft file so the response is not lost
652
- const draftPath = join(AGENT_REPO_DIR, "outputs", "drafts",
653
- `${today()}-quick-reply-${item.sender.replace(/[^a-z0-9]/gi, "-")}.md`);
654
- mkdirSync(join(AGENT_REPO_DIR, "outputs", "drafts"), { recursive: true });
655
- const content = `# Quick Reply Draft (SEND FAILED)\n\nTo: ${to}\nSubject: ${subject}\nGenerated: ${new Date().toISOString()}\nError: ${err.message}\n\n---\n\n${text}\n`;
656
- writeFileSync(draftPath, content);
657
- return { sent: false, via: "draft_fallback", draft_path: draftPath, error: err.message };
658
- }
659
- }
660
-
661
- // ---------------------------------------------------------------------------
662
- // Resolve Slack channel ID from item
663
- // ---------------------------------------------------------------------------
664
-
665
- function resolveSlackChannel(item) {
666
- // Direct channel ID (starts with D for DM, C for channel)
667
- if (item.channel && /^[DC][A-Z0-9]{8,}$/.test(item.channel)) return item.channel;
668
- if (item.channel_id) return item.channel_id;
669
-
670
- // Extract from raw_ref format: "slack:D099N1JGKRQ:1775331885.690669"
671
- if (item.raw_ref) {
672
- const match = item.raw_ref.match(/slack:([DC][A-Z0-9]+):/);
673
- if (match) return match[1];
674
- // Also try bare channel ID in raw_ref
675
- const bare = item.raw_ref.match(/^([DC][A-Z0-9]{8,})$/);
676
- if (bare) return bare[1];
677
- }
678
-
679
- return null;
680
- }
681
-
682
- // ---------------------------------------------------------------------------
683
- // Public API
739
+ // Transport
684
740
  // ---------------------------------------------------------------------------
685
-
686
- /**
687
- * Send a quick response to a simple item. No claude --print session needed.
688
- * Uses Sonnet API to generate text, then posts directly via Slack/Gmail.
689
- *
690
- * @returns {{ sent: boolean, text: string, channel: string }} result
691
- */
692
-
693
- /**
694
- * Send a reply back onto Cohort (the org's own messaging app).
695
- *
696
- * The responder dispatched on `item.service` and only ever knew "slack" and
697
- * "gmail", so a `cohort` item fell through to `{ sent: false }`: the agent
698
- * COMPOSED a reply and dropped it on the floor, logging "not_sent". Observed in
699
- * production — an agent quietly did that to a human's DMs three times in a row.
700
- *
701
- * Routing is by CHANNEL, not by person: `channel_id` is the room the inbound
702
- * arrived in, which is also where the answer belongs (a DM room for a DM, the
703
- * space for a space). Threading is preserved when the inbound was threaded.
704
- *
705
- * Goes through lib/org/messaging.sendMessage rather than the raw RPC so the
706
- * outbound send-gate, the afterSend hook and cost attribution all still run —
707
- * the same chokepoint every other channel uses.
708
- */
709
- async function sendCohortReply(item, text) {
710
- try {
711
- const channel = item.channel_id || item.channel || "";
712
- if (!channel) return { sent: false, via: null, error: "no channel_id on item" };
713
- const { sendMessage } = await import("../../lib/org/messaging.mjs");
714
- const { loadOrgConfig } = await import("../../lib/org/client.mjs");
715
- const agentRoot = process.env.AGENT_ROOT || process.env.AGENT_DIR || process.cwd();
716
- const frame = await sendMessage(
717
- {
718
- channel,
719
- body: text,
720
- // Stable per (message, kind) so a retry dedupes server-side instead of
721
- // double-posting into a real room.
722
- idempotencyId: `reply-${item.message_id || item.id || Date.now()}`,
723
- ...(item.thread_id ? { threadId: item.thread_id } : {}),
724
- },
725
- { cfg: loadOrgConfig(agentRoot), agentRoot },
726
- );
727
- if (frame && frame.ok) return { sent: true, via: "cohort", channel };
728
- return {
729
- sent: false,
730
- via: null,
731
- error: (frame && frame.error && frame.error.message) || "send failed",
732
- };
733
- } catch (err) {
734
- return { sent: false, via: null, error: err && err.message };
735
- }
736
- }
741
+ //
742
+ // The three per-service senders used to live here, private to this module, so
743
+ // the ONLY thing that could ever reach a human was a generated reply. That is
744
+ // why an acknowledgement had to be generated: there was no other way to speak.
745
+ //
746
+ // They now live in ./deliver.mjs and are shared with the assurance layer, so a
747
+ // courtesy message, a progress update and a failure notice all travel the same
748
+ // road as an answer — and a successful send writes a delivery RECEIPT, which is
749
+ // how the daemon can later tell a session that answered from one that did not.
737
750
 
738
751
  export async function sendQuickResponse(item, classResult, routed = null) {
739
752
  const startTime = Date.now();
@@ -754,7 +767,7 @@ export async function sendQuickResponse(item, classResult, routed = null) {
754
767
  : null;
755
768
 
756
769
  try {
757
- const text = await _generateResponse(item, classResult, false);
770
+ const text = await _generateResponse(item, classResult);
758
771
 
759
772
  // Validate before sending — block replies that violate critical rules
760
773
  const validationIssues = validateQuickReply(text);
@@ -776,48 +789,48 @@ export async function sendQuickResponse(item, classResult, routed = null) {
776
789
 
777
790
  let sendResult = { sent: false };
778
791
 
792
+ // Slack keeps its sent-registry dedup check ahead of the send (the registry
793
+ // is Slack-keyed); every service then goes through the one transport.
779
794
  if (item.service === "slack") {
780
795
  const channel = resolveSlackChannel(item);
781
- if (channel) {
782
- // Check sent-message registry for duplicates
783
- const threadTs = item.thread_id || null;
784
- const dupCheck = checkRecentlySent(channel, threadTs, "quick_reply");
785
- if (!dupCheck.allowed) {
786
- console.log(`[responder] Quick reply blocked by sent-registry: ${dupCheck.reason}`);
787
- logResponse({ type: "quick_response_dedup_blocked", sender: item.sender, reason: dupCheck.reason });
788
- // Observability: a send-gate (dedup registry) block — count it.
789
- counters.bump("send.blocked", { stage: "dedup", service: "slack", reason: dupCheck.reason });
790
- return { sent: false, text: null, blocked: true, reason: dupCheck.reason };
791
- }
792
-
793
- await sendSlackMessage(channel, text, threadTs);
794
- registerSent(channel, threadTs, "quick_reply", "quick-responder", text.substring(0, 100));
795
- sendResult = { sent: true, channel, via: "slack_api" };
796
- // Observability: a quick reply actually left the agent — interaction-end
797
- // on the quick path. Ambient trace_id (set by the daemon at item_received)
798
- // correlates it; item.trace_id is carried explicitly as a belt-and-braces
799
- // hand-off. Never throws.
800
- emitEvent({ type: EVENT_TYPES.SENT, trace_id: item.trace_id || undefined, attrs: { service: "slack", channel, via: "quick_reply" } });
801
- // Log interaction to memory so future sessions have context
802
- logInteraction(item, text);
803
- } else {
796
+ if (!channel) {
804
797
  console.warn(`[responder] Could not resolve Slack channel for ${item.sender} (channel: ${item.channel}, raw_ref: ${item.raw_ref})`);
805
- sendResult = { sent: false, reason: "no_channel_id" };
798
+ return { sent: false, text: null, reason: "no_channel_id" };
806
799
  }
807
- } else if (item.service === "gmail") {
808
- const gmailResult = await sendGmailResponse(item, text);
809
- sendResult = { sent: gmailResult.sent, via: gmailResult.via, ...(gmailResult.draft_path ? { draft_path: gmailResult.draft_path } : {}), ...(gmailResult.to ? { to: gmailResult.to } : {}) };
810
- if (gmailResult.sent) {
811
- emitEvent({ type: EVENT_TYPES.SENT, trace_id: item.trace_id || undefined, attrs: { service: "gmail", via: gmailResult.via } });
812
- logInteraction(item, text);
800
+ const dupCheck = checkRecentlySent(channel, item.thread_id || null, "quick_reply");
801
+ if (!dupCheck.allowed) {
802
+ console.log(`[responder] Quick reply blocked by sent-registry: ${dupCheck.reason}`);
803
+ logResponse({ type: "quick_response_dedup_blocked", sender: item.sender, reason: dupCheck.reason });
804
+ // Observability: a send-gate (dedup registry) block — count it.
805
+ counters.bump("send.blocked", { stage: "dedup", service: "slack", reason: dupCheck.reason });
806
+ return { sent: false, text: null, blocked: true, reason: dupCheck.reason };
813
807
  }
814
- } else if (item.service === "cohort") {
815
- const orgResult = await sendCohortReply(item, text);
816
- sendResult = { sent: orgResult.sent, via: orgResult.via, ...(orgResult.channel ? { channel: orgResult.channel } : {}), ...(orgResult.error ? { error: orgResult.error } : {}) };
817
- if (orgResult.sent) {
818
- emitEvent({ type: EVENT_TYPES.SENT, trace_id: item.trace_id || undefined, attrs: { service: "cohort", channel: orgResult.channel, via: "quick_reply" } });
819
- logInteraction(item, text);
808
+ }
809
+
810
+ const delivered = await deliver(item, text, { kind: "reply" });
811
+ sendResult = {
812
+ sent: delivered.sent,
813
+ via: delivered.via,
814
+ ...(delivered.channel ? { channel: delivered.channel } : {}),
815
+ ...(delivered.draft_path ? { draft_path: delivered.draft_path } : {}),
816
+ ...(delivered.error ? { error: delivered.error } : {}),
817
+ };
818
+
819
+ if (delivered.sent) {
820
+ if (item.service === "slack") {
821
+ registerSent(resolveSlackChannel(item), item.thread_id || null, "quick_reply", "quick-responder", text.substring(0, 100));
820
822
  }
823
+ // Observability: a quick reply actually left the agent — interaction-end
824
+ // on the quick path. Ambient trace_id (set by the daemon at item_received)
825
+ // correlates it; item.trace_id is carried explicitly as a belt-and-braces
826
+ // hand-off. Never throws.
827
+ emitEvent({
828
+ type: EVENT_TYPES.SENT,
829
+ trace_id: item.trace_id || undefined,
830
+ attrs: { service: item.service, channel: delivered.channel || null, via: "quick_reply" },
831
+ });
832
+ // Log interaction to memory so future sessions have context.
833
+ logInteraction(item, text);
821
834
  }
822
835
 
823
836
  const duration = Date.now() - startTime;
@@ -845,52 +858,89 @@ export async function sendQuickResponse(item, classResult, routed = null) {
845
858
  }
846
859
 
847
860
  /**
848
- * Send a holding message for a complex item, then return the message text
849
- * so it can be included in the dispatched session's prompt.
861
+ * Acknowledge an ask that is about to become a long-running session, and return
862
+ * the text so the session's own prompt can see what the human was already told.
863
+ *
864
+ * THIS NO LONGER CALLS A MODEL, AND THAT IS THE WHOLE POINT.
865
+ *
866
+ * It used to. `_generateResponse(item, classResult, true)` spawned a cold
867
+ * `claude --print` child under a 60-second hard cap to compose two sentences of
868
+ * courtesy. On a box at 95-99% memory that spawn lost its race 20 times in 32;
869
+ * every loss threw, was caught below, logged as `holding_message_error`, and
870
+ * told the requester nothing at all — while a session ran for a median of 14.7
871
+ * minutes behind a typing indicator the human reads as "she's replying".
850
872
  *
851
- * @returns {{ sent: boolean, holdingText: string }} result
873
+ * Generation was never needed here. An acknowledgement has a fixed shape, and
874
+ * the one genuinely useful variable in it — what the ask is about — was already
875
+ * computed by the classifier and sitting in `classResult.summary`. So the text
876
+ * is composed (assurance.composeAck) and handed straight to transport. The path
877
+ * is now a string concat plus one HTTP call: it cannot time out, cannot be
878
+ * starved of memory, and cannot be refused by a spend cap.
879
+ *
880
+ * Failure is still possible (the network exists), and it is still not fatal —
881
+ * but it is no longer FORGOTTEN. The caller opens an obligation before calling
882
+ * this, and the assurance sweep retries any acknowledgement still undelivered
883
+ * after ASSURANCE_ACK_GRACE_MS.
884
+ *
885
+ * @returns {{ sent: boolean, holdingText: string|null }} result
852
886
  */
853
887
  export async function sendHoldingMessage(item, classResult) {
854
888
  const startTime = Date.now();
855
889
 
856
890
  try {
857
- const text = await _generateResponse(item, classResult, true);
858
- let sendResult = { sent: false };
891
+ const text = composeAck(item, classResult);
859
892
 
860
893
  if (item.service === "slack") {
861
894
  const channel = resolveSlackChannel(item);
862
- if (channel) {
863
- // Check sent-message registry for duplicates
864
- const threadTs = item.thread_id || null;
865
- const dupCheck = checkRecentlySent(channel, threadTs, "holding");
866
- if (!dupCheck.allowed) {
867
- console.log(`[responder] Holding message blocked by sent-registry: ${dupCheck.reason}`);
868
- logResponse({ type: "holding_dedup_blocked", sender: item.sender, reason: dupCheck.reason });
869
- counters.bump("send.blocked", { stage: "dedup", service: "slack", kind: "holding", reason: dupCheck.reason });
870
- return { sent: false, holdingText: null, blocked: true, reason: dupCheck.reason };
871
- }
872
-
873
- await sendSlackMessage(channel, text, threadTs);
874
- registerSent(channel, threadTs, "holding", "quick-responder", text.substring(0, 100));
875
- sendResult = { sent: true, channel, via: "slack_api" };
876
- } else {
895
+ if (!channel) {
877
896
  console.warn(`[responder] Could not resolve Slack channel for holding msg to ${item.sender} (channel: ${item.channel}, raw_ref: ${item.raw_ref})`);
897
+ logResponse({ type: "holding_message_error", sender: item.sender, error: "no_channel_id" });
898
+ return { sent: false, holdingText: null, error: "no_channel_id" };
878
899
  }
879
- } else if (item.service === "gmail") {
880
- const gmailResult = await sendGmailResponse(item, text);
881
- sendResult = { sent: gmailResult.sent, via: gmailResult.via, ...(gmailResult.draft_path ? { draft_path: gmailResult.draft_path } : {}), ...(gmailResult.to ? { to: gmailResult.to } : {}) };
882
- } else if (item.service === "cohort") {
883
- const orgResult = await sendCohortReply(item, text);
884
- sendResult = { sent: orgResult.sent, via: orgResult.via, ...(orgResult.error ? { error: orgResult.error } : {}) };
900
+ const dupCheck = checkRecentlySent(channel, item.thread_id || null, "holding");
901
+ if (!dupCheck.allowed) {
902
+ console.log(`[responder] Holding message blocked by sent-registry: ${dupCheck.reason}`);
903
+ logResponse({ type: "holding_dedup_blocked", sender: item.sender, reason: dupCheck.reason });
904
+ counters.bump("send.blocked", { stage: "dedup", service: "slack", kind: "holding", reason: dupCheck.reason });
905
+ return { sent: false, holdingText: null, blocked: true, reason: dupCheck.reason };
906
+ }
907
+ }
908
+
909
+ // Bounded retry on the SEND (not on any generation — there isn't any). A
910
+ // courtesy message that gives up on one dropped socket is how silence
911
+ // happens; ~3 seconds of retry in front of a waiting human is proportionate.
912
+ const delivered = await deliverWithRetry(item, text, { kind: "ack" });
913
+ const sendResult = {
914
+ sent: delivered.sent,
915
+ via: delivered.via,
916
+ ...(delivered.channel ? { channel: delivered.channel } : {}),
917
+ ...(delivered.error ? { error: delivered.error } : {}),
918
+ };
919
+
920
+ if (delivered.sent && item.service === "slack") {
921
+ registerSent(resolveSlackChannel(item), item.thread_id || null, "holding", "quick-responder", text.substring(0, 100));
922
+ }
923
+ if (delivered.sent) {
924
+ // The holding message never emitted `sent` and never logged an interaction,
925
+ // so an acknowledgement was invisible to the trace chain AND to the
926
+ // session's own memory of the conversation. Both now happen.
927
+ emitEvent({
928
+ type: EVENT_TYPES.SENT,
929
+ trace_id: item.trace_id || undefined,
930
+ attrs: { service: item.service, channel: delivered.channel || null, via: "holding" },
931
+ });
932
+ logInteraction(item, text, { classification: "holding" });
885
933
  }
886
934
 
887
935
  const duration = Date.now() - startTime;
888
936
  logResponse({
889
- type: "holding_message",
937
+ type: delivered.sent ? "holding_message" : "holding_message_error",
890
938
  sender: item.sender,
891
939
  service: item.service,
892
940
  duration_ms: duration,
893
941
  text_length: text.length,
942
+ generated: false,
943
+ attempts: delivered.attempts || 1,
894
944
  ...sendResult,
895
945
  });
896
946
 
@@ -898,6 +948,9 @@ export async function sendHoldingMessage(item, classResult) {
898
948
  return { sent: sendResult.sent, holdingText: text, ...sendResult };
899
949
 
900
950
  } catch (err) {
951
+ // Unreachable in practice — deliverWithRetry does not throw — but a silent
952
+ // catch here is exactly the bug being fixed, so it stays loud and the text
953
+ // is still returned so the session prompt knows what was (not) said.
901
954
  console.error(`[responder] Holding message failed for ${item.sender}:`, err.message);
902
955
  logResponse({ type: "holding_message_error", sender: item.sender, error: err.message });
903
956
  return { sent: false, holdingText: null, error: err.message };