@cohortapp/agent-sdk 2.5.1 → 2.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/bin/maestro.mjs +185 -88
  2. package/bin/maestro.test.mjs +175 -48
  3. package/docs/runbooks/backup-restore.md +65 -33
  4. package/framework-features.json +4 -4
  5. package/lib/backup/policy.mjs +710 -0
  6. package/lib/backup/policy.test.mjs +305 -0
  7. package/lib/budget-escalate.mjs +133 -0
  8. package/lib/budget-escalate.test.mjs +232 -0
  9. package/lib/budget-guard.envelope.test.mjs +476 -0
  10. package/lib/budget-guard.mjs +853 -75
  11. package/lib/budget-guard.test.mjs +91 -42
  12. package/lib/cadences.mjs +33 -0
  13. package/lib/channels/orgmail/adapter.mjs +88 -3
  14. package/lib/channels/orgmail/adapter.test.mjs +137 -0
  15. package/lib/channels/repeat-suppressor.mjs +198 -0
  16. package/lib/channels/repeat-suppressor.test.mjs +134 -0
  17. package/lib/comms/receipts.mjs +297 -0
  18. package/lib/cost/ledger-row.mjs +333 -0
  19. package/lib/cost/ledger-row.test.mjs +183 -0
  20. package/lib/execution/drive.mjs +28 -1
  21. package/lib/execution/effects.mjs +191 -12
  22. package/lib/execution/effects.test.mjs +50 -11
  23. package/lib/goals/admission.mjs +13 -1
  24. package/lib/goals/admission.test.mjs +26 -1
  25. package/lib/goals/loop.mjs +13 -0
  26. package/lib/kpi-sensors.test.mjs +3 -0
  27. package/lib/mandate/cache.mjs +13 -5
  28. package/lib/mandate/derive.mjs +146 -21
  29. package/lib/mandate/derive.test.mjs +50 -6
  30. package/lib/mandate/model.mjs +32 -4
  31. package/lib/mandate/refresh.test.mjs +16 -2
  32. package/lib/mcp/server.test.mjs +12 -3
  33. package/lib/model-router/economics.mjs +107 -76
  34. package/lib/model-router/economics.test.mjs +64 -46
  35. package/lib/model-router/integration-coverage.test.mjs +39 -37
  36. package/lib/model-router/ledger.mjs +75 -22
  37. package/lib/model-router/ledger.test.mjs +35 -2
  38. package/lib/org/client.mjs +14 -0
  39. package/lib/org/cost-sync.mjs +16 -2
  40. package/lib/org/doctor.mjs +62 -1
  41. package/lib/org/doctor.test.mjs +36 -3
  42. package/lib/org/email-remedy.mjs +49 -0
  43. package/lib/org/engagement-ledger.mjs +376 -0
  44. package/lib/org/engagement-ledger.test.mjs +112 -0
  45. package/lib/org/engagement.mjs +1056 -0
  46. package/lib/org/engagement.test.mjs +739 -0
  47. package/lib/org/messaging.mjs +230 -3
  48. package/lib/org/messaging.test.mjs +110 -1
  49. package/lib/org/param-contract.mjs +56 -2
  50. package/lib/org/param-contract.test.mjs +26 -0
  51. package/lib/org/protocol.checksum +1 -1
  52. package/lib/org/protocol.mjs +5 -0
  53. package/lib/org/protocol.test.mjs +7 -1
  54. package/lib/org/tool-surface.mjs +506 -10
  55. package/lib/org/tool-surface.test.mjs +191 -7
  56. package/lib/org/ui-parity.mjs +333 -6
  57. package/lib/org/ui-parity.test.mjs +96 -3
  58. package/lib/org/work-ledger.mjs +241 -0
  59. package/lib/org/work-ledger.test.mjs +237 -0
  60. package/lib/plan/adoption-e2e.test.mjs +366 -0
  61. package/lib/plan/budget-enforcement.test.mjs +400 -0
  62. package/lib/plan/budget-runtime.mjs +215 -0
  63. package/lib/plan/compile.mjs +201 -5
  64. package/lib/plan/compile.test.mjs +19 -5
  65. package/lib/plan/emit.mjs +8 -0
  66. package/lib/plan/emit.test.mjs +18 -0
  67. package/lib/resource-governor.mjs +58 -12
  68. package/lib/resource-governor.test.mjs +41 -1
  69. package/lib/security/audit-engine.mjs +45 -8
  70. package/lib/security/audit-engine.test.mjs +35 -0
  71. package/lib/setup/enroll-from-cohort.mjs +14 -1
  72. package/lib/setup/sections/mandate.mjs +48 -7
  73. package/lib/setup/sections/mandate.test.mjs +17 -2
  74. package/lib/setup/sections/orgmail.mjs +10 -2
  75. package/lib/setup/state.mjs +83 -2
  76. package/lib/telemetry/collect.mjs +360 -20
  77. package/lib/telemetry/collect.test.mjs +266 -0
  78. package/package.json +1 -1
  79. package/scripts/cost/track-claude-usage.mjs +207 -48
  80. package/scripts/cost/track-claude-usage.test.mjs +148 -0
  81. package/scripts/daemon/agent-daemon.mjs +315 -17
  82. package/scripts/daemon/assurance-e2e.test.mjs +421 -0
  83. package/scripts/daemon/assurance.mjs +944 -0
  84. package/scripts/daemon/assurance.test.mjs +668 -0
  85. package/scripts/daemon/cadence-consumer-governance.test.mjs +56 -0
  86. package/scripts/daemon/cadence-consumer.mjs +147 -9
  87. package/scripts/daemon/cadence-consumer.test.mjs +6 -0
  88. package/scripts/daemon/cadence-handlers.mjs +158 -0
  89. package/scripts/daemon/cadence-handlers.test.mjs +64 -0
  90. package/scripts/daemon/deliver.mjs +314 -0
  91. package/scripts/daemon/dispatcher-governance.test.mjs +10 -0
  92. package/scripts/daemon/dispatcher.mjs +64 -6
  93. package/scripts/daemon/responder-cost.test.mjs +68 -0
  94. package/scripts/daemon/responder.mjs +351 -298
  95. package/scripts/local-triggers/generate-plists.test.mjs +7 -4
  96. package/scripts/maintenance/backup-run.mjs +415 -0
  97. package/scripts/maintenance/backup-to-cloud.sh +16 -116
  98. package/scripts/org/send-orgmail.mjs +16 -0
  99. package/scripts/record-receipt.sh +63 -0
  100. package/scripts/restore-from-backup.sh +14 -3
  101. package/scripts/restore-from-backup.test.mjs +8 -5
  102. package/scripts/send-email-threaded.py +47 -0
  103. package/scripts/send-sms.sh +4 -0
  104. package/scripts/send-whatsapp.sh +4 -0
  105. package/scripts/setup/init-backup.mjs +93 -38
  106. package/scripts/slack-send.sh +12 -0
@@ -41,6 +41,7 @@ import { join, resolve } from "node:path";
41
41
 
42
42
  import { lookupModel } from "./catalog.mjs";
43
43
  import { appendJsonl } from "../fs-atomic.mjs";
44
+ import { billableUsd, summariseRows as summariseCostRows } from "../cost/ledger-row.mjs";
44
45
 
45
46
  const MTOK = 1e6;
46
47
 
@@ -288,6 +289,15 @@ export function buildLedgerRow(catalog, entry = {}, deps = {}) {
288
289
  const cacheReadTokens = nonNeg(entry.cacheReadTokens);
289
290
  const cacheWriteTokens = nonNeg(entry.cacheWriteTokens);
290
291
 
292
+ // DID THE CALLER ACTUALLY SUPPLY TOKEN COUNTS? `nonNeg` maps undefined/NaN to
293
+ // 0, so without this the row cannot tell "0 tokens" from "nobody told me", and
294
+ // an LLM session whose usage was unreadable is written as a measured, billable
295
+ // $0 — the exact defect lib/cost/ledger-row.mjs exists to end, re-created by
296
+ // its own v2 writer. `llm-task.mjs#writeRow` passes `usage.inputTokens`
297
+ // straight through, so absent usage lands here regularly.
298
+ const suppliedTokens =
299
+ Number.isFinite(Number(entry.inputTokens)) || Number.isFinite(Number(entry.outputTokens));
300
+
291
301
  const hasAuthoritative =
292
302
  typeof entry.totalCostUSD === "number" && Number.isFinite(entry.totalCostUSD);
293
303
 
@@ -304,10 +314,34 @@ export function buildLedgerRow(catalog, entry = {}, deps = {}) {
304
314
 
305
315
  const totalCostUSD = hasAuthoritative ? round6(entry.totalCostUSD) : est.usd;
306
316
  const estimated = !hasAuthoritative;
307
- // Load-bearing: budget-guard sums `estimated_usd`. Use the authoritative cost
308
- // when present, else the estimate. Never null here (default 0) so a row always
309
- // counts as a session with a finite spend.
310
- const estimated_usd = totalCostUSD == null ? 0 : totalCostUSD;
317
+ // Load-bearing: readers bill through lib/cost/ledger-row.mjs. `null` — NOT 0
318
+ // — when nothing could price this row (`estimateCost` returns
319
+ // `{usd:null, basis:"no-row"}` for a model that is not in the catalog). A 0
320
+ // here used to read downstream as "this session was free", which is how an
321
+ // unpriced model or an unreadable usage envelope quietly shrank the day's
322
+ // spend. Null classifies as `unknown` and gets IMPUTED instead.
323
+ const estimated_usd = totalCostUSD == null ? null : totalCostUSD;
324
+
325
+ // The explicit measurement class. Three cases, and the middle one is the fix:
326
+ // n/a no model ran (message-send attribution). $0 is a fact.
327
+ // unknown a model DID run but we have neither token counts nor a price —
328
+ // the cost is unknown, not zero. Readers impute it and doctor
329
+ // goes red; nothing here may launder it into a measured $0.
330
+ // measured real usage and/or a real price.
331
+ const measurement =
332
+ !ref && !suppliedTokens && inputTokens === 0 && outputTokens === 0
333
+ ? "n/a"
334
+ : !suppliedTokens && totalCostUSD == null
335
+ ? "unknown"
336
+ : totalCostUSD == null && inputTokens === 0 && outputTokens === 0
337
+ ? "unknown"
338
+ : "measured";
339
+ const unmeasuredReason =
340
+ measurement !== "unknown"
341
+ ? null
342
+ : !suppliedTokens
343
+ ? `no usage was supplied for ${ref || "an unknown model"} and no authoritative cost accompanied it`
344
+ : `usage was supplied as 0/0 and ${ref || "the model"} could not be priced (${est.basis})`;
311
345
 
312
346
  const now = (deps && typeof deps.now === "function" ? deps.now : Date.now)();
313
347
 
@@ -322,14 +356,21 @@ export function buildLedgerRow(catalog, entry = {}, deps = {}) {
322
356
  harness: entry.harness ?? null,
323
357
  backend: provider || null,
324
358
  model,
325
- input_tokens: inputTokens,
326
- output_tokens: outputTokens,
359
+ // Null, not 0, when nobody supplied a count — see `suppliedTokens`.
360
+ input_tokens: suppliedTokens ? inputTokens : null,
361
+ output_tokens: suppliedTokens ? outputTokens : null,
327
362
  cache_read_tokens: cacheReadTokens,
328
363
  cache_creation_tokens: cacheWriteTokens,
329
364
  total_cost_usd: totalCostUSD,
330
365
  // --- budget-guard compatibility (DO NOT REMOVE) ---
331
366
  estimated_usd,
332
367
  // ---------------------------------------------------
368
+ // Explicit measurement class so readers never have to infer it from the row
369
+ // shape. A row with no model and no tokens is zero-LLM attribution (a
370
+ // message send): $0 is a fact there, and it must not be counted as a session
371
+ // or arm the "sessions ran but spend is $0" tripwire.
372
+ measurement,
373
+ unmeasured_reason: unmeasuredReason,
333
374
  estimated,
334
375
  cost_basis: est.basis,
335
376
  volatile: est.volatile,
@@ -361,15 +402,26 @@ export function writeLedgerRow(catalog, entry = {}, deps = {}) {
361
402
  }
362
403
 
363
404
  /**
364
- * Sum today's ledger the SAME way budget-guard does (reads `estimated_usd`),
365
- * for tests / `maestro cost report`. NEVER throws.
405
+ * Sum today's ledger the SAME way budget-guard does, for tests /
406
+ * `maestro cost report`. NEVER throws.
407
+ *
408
+ * "The same way" now means via lib/cost/ledger-row.mjs: bill the CLI's
409
+ * authoritative `total_cost_usd` where present, fall back to the estimate, and
410
+ * treat an unmeasured session as UNKNOWN rather than free. It previously summed
411
+ * `estimated_usd` unconditionally, which both under-priced cache reads and
412
+ * silently reported unmeasured sessions as $0 — the two halves of the bug that
413
+ * kept the budget governor from ever tripping.
366
414
  *
367
415
  * @param {object} [deps] { now, ledgerDir, agentRoot }
368
- * @returns {{ spentUSD:number, sessions:number, byModel:Record<string,number> }}
416
+ * @returns {{ spentUSD:number, sessions:number, byModel:Record<string,number>,
417
+ * unmeasuredSessions:number, nonLlmRows:number, degradations:string[] }}
369
418
  */
370
419
  export function sumLedgerToday(deps = {}) {
371
420
  const file = ledgerFile(deps);
372
- const out = { spentUSD: 0, sessions: 0, byModel: {} };
421
+ const out = {
422
+ spentUSD: 0, sessions: 0, byModel: {},
423
+ unmeasuredSessions: 0, nonLlmRows: 0, degradations: [],
424
+ };
373
425
  if (!existsSync(file)) return out;
374
426
  let body;
375
427
  try {
@@ -377,22 +429,23 @@ export function sumLedgerToday(deps = {}) {
377
429
  } catch {
378
430
  return out;
379
431
  }
432
+ const rows = [];
380
433
  for (const line of body.split("\n")) {
381
434
  if (!line.trim()) continue;
382
- let row;
383
- try {
384
- row = JSON.parse(line);
385
- } catch {
386
- continue;
387
- }
388
- const usd = Number(row.estimated_usd);
389
- if (Number.isFinite(usd)) {
390
- out.spentUSD += usd;
391
- const key = row.model || row.backend || "unknown";
392
- out.byModel[key] = round6((out.byModel[key] || 0) + usd);
393
- }
435
+ try { rows.push(JSON.parse(line)); } catch { continue; }
436
+ }
437
+ for (const row of rows) {
438
+ const { usd, basis } = billableUsd(row);
439
+ if (basis === "non_llm" || basis === "invalid") continue;
394
440
  out.sessions += 1;
441
+ if (basis === "unmeasured") { out.unmeasuredSessions += 1; continue; }
442
+ out.spentUSD += usd;
443
+ const key = row.model || row.backend || "unknown";
444
+ out.byModel[key] = round6((out.byModel[key] || 0) + usd);
395
445
  }
446
+ const summary = summariseCostRows(rows);
447
+ out.nonLlmRows = summary.nonLlmRows;
448
+ out.degradations = summary.degradations;
396
449
  out.spentUSD = round6(out.spentUSD);
397
450
  return out;
398
451
  }
@@ -11,6 +11,7 @@ import { tmpdir } from "node:os";
11
11
  import { join } from "node:path";
12
12
 
13
13
  import { loadCatalog } from "./catalog.mjs";
14
+ import { billableUsd } from "../cost/ledger-row.mjs";
14
15
  import {
15
16
  estimateCost,
16
17
  buildLedgerRow,
@@ -313,11 +314,43 @@ test("buildLedgerRow uses the authoritative totalCostUSD when present (estimated
313
314
  assert.equal(row.estimated, false);
314
315
  });
315
316
 
316
- test("buildLedgerRow estimated_usd is never null (budget-guard counts every row)", () => {
317
+ test("buildLedgerRow: an UNPRICEABLE row is null, not 0 — a $0 here reads as a free session", () => {
318
+ // This asserted `estimated_usd === 0` for a model with no catalog row, on the
319
+ // theory that budget-guard needed a finite number on every line. It does not:
320
+ // readers bill through lib/cost/ledger-row.mjs, which treats a 0 on a row that
321
+ // NAMED a model as "could not price", imputes it, and turns doctor red. A
322
+ // literal 0 was the "a zero that means unmeasured" defect, re-created by the
323
+ // v2 writer the module was written to replace.
317
324
  const c = cat();
318
325
  const row = buildLedgerRow(c, { agent: "x", provider: "ghost", model: "model", inputTokens: 10 });
319
- assert.equal(row.estimated_usd, 0);
326
+ assert.equal(row.estimated_usd, null);
320
327
  assert.equal(row.total_cost_usd, null);
328
+ assert.equal(row.measurement, "measured", "usage WAS supplied; only the price is missing");
329
+ assert.equal(billableUsd(row).basis, "unpriced");
330
+ assert.equal(billableUsd(row).usd, null);
331
+ });
332
+
333
+ test("buildLedgerRow: usage nobody supplied is `unknown`, never a measured zero", () => {
334
+ // llm-task.mjs#writeRow passes `usage.inputTokens` straight through, so an
335
+ // unreadable usage envelope arrives here as undefined. `nonNeg` mapped that to
336
+ // 0 and the row was stamped `measurement:"measured"` billing $0 — no
337
+ // degradation, no unmeasured count, doctor green.
338
+ const c = cat();
339
+ const row = buildLedgerRow(c, { agent: "x", ref: "anthropic/claude-opus-4-6", provider: "anthropic", model: "claude-opus-4-6" });
340
+ assert.equal(row.measurement, "unknown");
341
+ assert.equal(row.input_tokens, null);
342
+ assert.equal(row.output_tokens, null);
343
+ assert.equal(row.estimated_usd, null);
344
+ assert.match(row.unmeasured_reason, /no usage was supplied/);
345
+ assert.equal(billableUsd(row).usd, null, "unknown cost, not free");
346
+ });
347
+
348
+ test("buildLedgerRow: a zero-model attribution row is still a real $0", () => {
349
+ // The one case where 0 is a FACT: a message send, no model, no tokens.
350
+ const row = buildLedgerRow(cat(), { agent: "x", source: "messaging", task_class: "messaging.send" });
351
+ assert.equal(row.measurement, "n/a");
352
+ assert.equal(billableUsd(row).usd, 0);
353
+ assert.equal(billableUsd(row).basis, "non_llm");
321
354
  });
322
355
 
323
356
  test("buildLedgerRow projects volatile spend at steady-state for budgeting", () => {
@@ -643,6 +643,20 @@ export function createItem(item, o = {}) {
643
643
  return call("board.create", item || {}, o);
644
644
  }
645
645
 
646
+ /**
647
+ * Record ONE step of this agent's work against the board (board.track).
648
+ *
649
+ * The server holds all the policy — whether the ask deserves a row at all, the
650
+ * find-or-open, the column ladder, comment de-duplication and who gets tagged —
651
+ * because the SAME code serves hq's in-process responder. This wrapper is
652
+ * deliberately dumb so the two planes cannot drift.
653
+ *
654
+ * Idempotent on (askKey, stage): pass a stable idempotencyKey for safe retries.
655
+ */
656
+ export function trackWork(params, o = {}) {
657
+ return call("board.track", params || {}, o);
658
+ }
659
+
646
660
  /**
647
661
  * Claim a board item (board.claim). The server resolves the race atomically; the
648
662
  * loser gets a CONFLICT error frame. Caller MUST pass a stable idempotencyKey
@@ -40,6 +40,7 @@ import { existsSync, readFileSync } from "node:fs";
40
40
  import { join, resolve } from "node:path";
41
41
 
42
42
  import { isEnabled, configFromAgent, costReport } from "./client.mjs";
43
+ import { summariseRows as summariseCostRows, imputeUnmeasured, billableUsd } from "../cost/ledger-row.mjs";
43
44
 
44
45
  const REPORT_BATCH = 100;
45
46
 
@@ -140,7 +141,13 @@ export function toReportRow(row, deps = {}) {
140
141
  output_tokens: numOr0(row.output_tokens),
141
142
  cache_read_tokens: numOr0(row.cache_read_tokens),
142
143
  cache_creation_tokens: numOr0(row.cache_creation_tokens),
143
- estimated_usd: numOr0(row.estimated_usd),
144
+ // The BILLABLE figure (authoritative `total_cost_usd` first, cache-aware
145
+ // estimate second) rather than the raw `estimated_usd` column, so the org
146
+ // rollup and the seat's own governor read the same file the same way. A row
147
+ // we could not measure reports 0 here on purpose: a per-row imputation would
148
+ // be a fabricated number on a permanent record. The day-level imputation
149
+ // lives in `localUSD`, which is the reconciliation figure.
150
+ estimated_usd: billableUsd(row).usd ?? 0,
144
151
  total_cost_usd: row.total_cost_usd ?? null,
145
152
  decision_id: row.decision_id ?? null,
146
153
  session_id: row.session_id ?? null,
@@ -183,7 +190,14 @@ function numOr0(v) {
183
190
  */
184
191
  export async function reportLedger(o = {}) {
185
192
  const rows = readLedgerRows(o);
186
- const localUSD = round6(rows.reduce((s, r) => s + numOr0(r.estimated_usd), 0));
193
+ // Bill through the ONE definition (lib/cost/ledger-row.mjs), not `estimated_usd`:
194
+ // the authoritative `total_cost_usd` where present, the cache-aware estimate as
195
+ // a fallback, and unmeasured sessions IMPUTED rather than summed as free. The
196
+ // old `numOr0(row.estimated_usd)` under-read the same days budget-guard did
197
+ // (2026-08-11: $90.79 reported to hq against $235.62 actually spent), so the
198
+ // org rollup and the seat's own governor disagreed about the same file.
199
+ const summary = summariseCostRows(rows);
200
+ const localUSD = round6(summary.measuredUsd + imputeUnmeasured(summary).imputedUsd);
187
201
 
188
202
  if (!isEnabled(o.cfg)) {
189
203
  return { enabled: false, rowsRead: rows.length, reported: 0, failed: 0, localUSD, batches: [] };
@@ -40,6 +40,7 @@ import { existsSync, readFileSync } from "node:fs";
40
40
  import { join } from "node:path";
41
41
  import { PROTOCOL_VERSION, methodDef } from "./protocol.mjs";
42
42
  import { loadOrgConfig, configFromAgent, call, fetchSelfProfile } from "./client.mjs";
43
+ import { remedyFor } from "./email-remedy.mjs";
43
44
 
44
45
  const PROBE_TIMEOUT_MS = 8000;
45
46
 
@@ -139,6 +140,66 @@ export function checkReactiveLaneWiring(agentRoot, deps = {}) {
139
140
  }
140
141
  }
141
142
 
143
+ /**
144
+ * Turn an `email.inbox` error frame into an ACTIONABLE doctor line.
145
+ *
146
+ * The old line was `workspace mailbox probe: NOT_FOUND — no active mailbox is
147
+ * assigned to this agent (admin: Cohort → Settings → Email)`. Every word of
148
+ * that is true and none of it tells an operator what to actually do, so on a
149
+ * live seat it sat unresolved while the daemon logged the same sentence 478
150
+ * times in a day. A doctor line that describes a state without naming the actor
151
+ * and the act is a line people learn to scroll past.
152
+ *
153
+ * So: name WHO does WHAT, and distinguish the three genuinely different causes,
154
+ * because they have three different owners.
155
+ *
156
+ * NOT_FOUND — the seat has no mailbox (or the domain isn't verified).
157
+ * FAIL: the email lane is dead, not degraded. Since 2026-08
158
+ * `pairing.approve` provisions a mailbox at enrolment, so
159
+ * reaching this on a paired agent means either the org has
160
+ * no VERIFIED domain or this seat was paired before that
161
+ * shipped — hence both remedies, in order.
162
+ * FORBIDDEN_SCOPE — the key isn't paired, or lacks the email scope. FAIL.
163
+ * anything else — transient/unknown: WARN, fail-open.
164
+ *
165
+ * Pure; exported for tests.
166
+ *
167
+ * @param {{code?: string, message?: string}} [err]
168
+ * @returns {{level:"warn"|"fail", msg:string}}
169
+ */
170
+ export function mailboxProbeVerdict(err) {
171
+ const code = (err && err.code) || "?";
172
+ const detail = (err && err.message) || "no detail";
173
+
174
+ if (code === "NOT_FOUND") {
175
+ return {
176
+ level: "fail",
177
+ msg:
178
+ "Workspace mailbox: NONE. This agent cannot receive email at all — the orgmail lane is dead, " +
179
+ "not degraded. TWO possible causes, check in this order: " +
180
+ "(1) the workspace has no VERIFIED email domain — a workspace ADMIN adds it at " +
181
+ "Cohort → Settings → Email → Domains, publishes the DNS records, and clicks Verify; " +
182
+ "(2) the domain IS verified but this seat predates automatic provisioning — the same ADMIN " +
183
+ "opens Cohort → Settings → Email → Mailboxes, picks this agent's member row, and assigns a " +
184
+ "local part (seats paired after 2026-08 get one automatically at `pairing.approve`). " +
185
+ `Server said: ${detail}`,
186
+ };
187
+ }
188
+ if (code === "FORBIDDEN_SCOPE" || code === "FORBIDDEN" || code === "UNAUTHORIZED") {
189
+ // Same sentence the adapter logs and `maestro setup` prints — one
190
+ // classifier, so an operator is told the same thing by whichever surface
191
+ // they happen to hit first (lib/org/email-remedy.mjs).
192
+ return {
193
+ level: "fail",
194
+ msg: `Workspace mailbox: ${remedyFor(code)}. Server said: ${detail}`,
195
+ };
196
+ }
197
+ return {
198
+ level: "warn",
199
+ msg: `workspace mailbox probe inconclusive (${code}: ${detail}) — retry; if it persists, a workspace ADMIN checks Cohort → Settings → Email`,
200
+ };
201
+ }
202
+
142
203
  /**
143
204
  * Run the Cohort connectivity probes.
144
205
  * @param {object} o
@@ -331,7 +392,7 @@ export async function checkOrgConnectivity(o = {}) {
331
392
  const addr = inbox.result?.mailbox?.address || "";
332
393
  results.push({ level: "ok", msg: `workspace mailbox reachable${addr ? ` (${addr})` : ""}` });
333
394
  } else {
334
- results.push({ level: "warn", msg: `workspace mailbox probe: ${inbox.error?.code || "?"} — ${inbox.error?.message || "no detail"} (admin: Cohort → Settings → Email)` });
395
+ results.push(mailboxProbeVerdict(inbox.error));
335
396
  }
336
397
  }
337
398
  }
@@ -197,7 +197,7 @@ test("probe 4: orgmail gate + family vendored → mailbox probe surfaces the add
197
197
  rmSync(r, { recursive: true, force: true });
198
198
  });
199
199
 
200
- test("probe 4: mailbox probe failure warns with the admin remedy (never a fail)", async () => {
200
+ test("probe 4: NO mailbox FAILS and names who does what (it is a dead lane, not a degraded one)", async () => {
201
201
  const r = root({ orgmail: true });
202
202
  const f = stubFetch({
203
203
  "/api/v1/directory": DIR_OK,
@@ -205,9 +205,42 @@ test("probe 4: mailbox probe failure warns with the admin remedy (never a fail)"
205
205
  "/api/v1/email.inbox": { status: 404, body: { ok: false, error: { code: "NOT_FOUND", message: "no mailbox for member" } } },
206
206
  });
207
207
  const rs = await checkOrgConnectivity({ agentRoot: r, fetchImpl: f, env: {} });
208
- const mail = rs.find((x) => /workspace mailbox probe/.test(x.msg));
208
+ const mail = rs.find((x) => /Workspace mailbox/.test(x.msg));
209
+ assert.equal(mail.level, "fail");
210
+ // The whole point: an ACTOR and an ACT, not just a state.
211
+ assert.match(mail.msg, /workspace ADMIN/);
212
+ assert.match(mail.msg, /Settings → Email → Mailboxes/);
213
+ assert.match(mail.msg, /Settings → Email → Domains/);
214
+ // The server's own words survive, so the operator can tell the two apart.
215
+ assert.match(mail.msg, /no mailbox for member/);
216
+ rmSync(r, { recursive: true, force: true });
217
+ });
218
+
219
+ test("probe 4: an unpaired/unscoped key is a DIFFERENT fail with a different remedy", async () => {
220
+ const r = root({ orgmail: true });
221
+ const f = stubFetch({
222
+ "/api/v1/directory": DIR_OK,
223
+ "/api/v1/messaging.channels": CHANNELS_OK,
224
+ "/api/v1/email.inbox": { status: 403, body: { ok: false, error: { code: "FORBIDDEN_SCOPE", message: "not paired" } } },
225
+ });
226
+ const rs = await checkOrgConnectivity({ agentRoot: r, fetchImpl: f, env: {} });
227
+ const mail = rs.find((x) => /Workspace mailbox/.test(x.msg));
228
+ assert.equal(mail.level, "fail");
229
+ assert.match(mail.msg, /Settings → API keys/);
230
+ assert.ok(!/Settings → Email → Mailboxes/.test(mail.msg), "wrong remedy for this cause");
231
+ rmSync(r, { recursive: true, force: true });
232
+ });
233
+
234
+ test("probe 4: an unknown error stays a WARN (fail-open on the unclassified)", async () => {
235
+ const r = root({ orgmail: true });
236
+ const f = stubFetch({
237
+ "/api/v1/directory": DIR_OK,
238
+ "/api/v1/messaging.channels": CHANNELS_OK,
239
+ "/api/v1/email.inbox": { status: 500, body: { ok: false, error: { code: "INTERNAL", message: "boom" } } },
240
+ });
241
+ const rs = await checkOrgConnectivity({ agentRoot: r, fetchImpl: f, env: {} });
242
+ const mail = rs.find((x) => /mailbox probe inconclusive/.test(x.msg));
209
243
  assert.equal(mail.level, "warn");
210
- assert.match(mail.msg, /Settings → Email/);
211
244
  rmSync(r, { recursive: true, force: true });
212
245
  });
213
246
 
@@ -0,0 +1,49 @@
1
+ /**
2
+ * lib/org/email-remedy.mjs — one classifier for `email.*` failures, so every
3
+ * surface tells the operator the same thing.
4
+ *
5
+ * Three places learn that a seat's mailbox is broken — the orgmail poll loop,
6
+ * `maestro doctor`, and the `maestro setup` orgmail section — and before this
7
+ * they each said something different, all of which described a STATE and none
8
+ * of which named an ACTOR and an ACT. The live consequence: a seat sat with no
9
+ * mailbox while its daemon wrote `NOT_FOUND no active mailbox is assigned to
10
+ * this agent` 478 times in one day. Every word true; nobody told what to do.
11
+ *
12
+ * So the remedy text lives here, once, and is imported by all three. Pure, no
13
+ * side effects, no imports — it is deliberately safe for the setup path to pull
14
+ * in (importing the adapter would run its `definePlatform` self-registration).
15
+ *
16
+ * @module lib/org/email-remedy
17
+ */
18
+
19
+ "use strict";
20
+
21
+ /**
22
+ * The remedy for a terminal `email.*` failure, in one sentence naming WHO does
23
+ * WHAT. Returns "" for codes that are transient (retry is the remedy).
24
+ *
25
+ * @param {string} code RPC error code
26
+ * @returns {string}
27
+ */
28
+ export function remedyFor(code) {
29
+ switch (code) {
30
+ case "NOT_FOUND":
31
+ return (
32
+ "this seat has NO workspace mailbox, so it can receive no email at all. " +
33
+ "A workspace ADMIN assigns one at Cohort → Settings → Email → Mailboxes " +
34
+ "(and first verifies the domain under → Domains if it is not verified yet). " +
35
+ "Seats paired after 2026-08 get one automatically at pairing approval"
36
+ );
37
+ case "FORBIDDEN_SCOPE":
38
+ case "FORBIDDEN":
39
+ case "UNAUTHORIZED":
40
+ return (
41
+ "this key cannot read email — it is unpaired or lacks the email scope. " +
42
+ "Run `cohort pair`, or a workspace ADMIN re-mints it at Cohort → Settings → API keys"
43
+ );
44
+ default:
45
+ return "";
46
+ }
47
+ }
48
+
49
+ export default { remedyFor };