@cohortapp/agent-sdk 2.10.0 → 2.11.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/.claude/commands/init-maestro.md +16 -9
  2. package/docs/guides/mac-mini.md +11 -1
  3. package/docs/runbooks/cohort-cutover.md +16 -0
  4. package/lib/channels/inbox-item.mjs +12 -0
  5. package/lib/channels/inbox-item.test.mjs +33 -0
  6. package/lib/execution/disposition.mjs +13 -2
  7. package/lib/execution/disposition.test.mjs +19 -2
  8. package/lib/execution/pipeline.test.mjs +4 -1
  9. package/lib/mcp/server.test.mjs +16 -4
  10. package/lib/org/client.mjs +58 -1
  11. package/lib/org/messaging.mjs +5 -0
  12. package/lib/org/messaging.test.mjs +7 -0
  13. package/lib/org/protocol.checksum +1 -1
  14. package/lib/org/protocol.mjs +98 -0
  15. package/lib/org/protocol.test.mjs +19 -2
  16. package/lib/org/resource-tools.mjs +317 -0
  17. package/lib/org/resource-tools.test.mjs +361 -0
  18. package/lib/org/tool-access.mjs +176 -0
  19. package/lib/org/tool-access.test.mjs +144 -0
  20. package/lib/org/tool-surface.mjs +431 -5
  21. package/lib/org/tool-surface.test.mjs +385 -8
  22. package/lib/org/ui-parity.mjs +196 -3
  23. package/lib/org/ui-parity.test.mjs +126 -7
  24. package/lib/tool-definitions.js +23 -2
  25. package/package.json +2 -2
  26. package/plugins/maestro-skills/.claude-plugin/marketplace.json +1 -1
  27. package/plugins/maestro-skills/plugin.json +4 -0
  28. package/plugins/maestro-skills/skills/venture-deliverables.md +176 -0
  29. package/policies/information-barriers.yaml +34 -7
  30. package/scripts/ci/check-no-residual-identity.mjs +281 -9
  31. package/scripts/ci/check-no-residual-identity.test.mjs +115 -2
  32. package/scripts/cloud-relay/voice/relay-identity.test.mjs +96 -0
  33. package/scripts/cloud-relay/voice/server.mjs +42 -2
  34. package/scripts/cost/track-claude-usage-pricing.test.mjs +183 -0
  35. package/scripts/cost/track-claude-usage.mjs +113 -4
  36. package/scripts/daemon/agent-daemon.mjs +212 -5
  37. package/scripts/daemon/agent-daemon.test.mjs +307 -0
  38. package/scripts/daemon/assurance.mjs +38 -15
  39. package/scripts/daemon/assurance.test.mjs +39 -1
  40. package/scripts/daemon/cadence-handlers.mjs +48 -5
  41. package/scripts/daemon/cadence-handlers.test.mjs +57 -2
  42. package/scripts/daemon/classifier-identity.test.mjs +137 -0
  43. package/scripts/daemon/classifier.mjs +98 -17
  44. package/scripts/daemon/inbox-deferral.mjs +49 -24
  45. package/scripts/daemon/inbox-deferral.test.mjs +39 -1
  46. package/scripts/daemon/prompt-builder-preamble.test.mjs +210 -0
  47. package/scripts/daemon/prompt-builder.mjs +264 -41
  48. package/scripts/daemon/prompt-builder.test.mjs +5 -5
  49. package/scripts/daemon/responder.mjs +9 -0
  50. package/scripts/disclosure_boundaries.py +56 -5
  51. package/scripts/huddle/huddle-prompt.test.mjs +176 -0
  52. package/scripts/huddle/huddle-server.mjs +128 -13
  53. package/scripts/local-triggers/autoupdate.sh +83 -0
  54. package/scripts/local-triggers/generate-plists.sh +9 -0
  55. package/scripts/local-triggers/generate-plists.test.mjs +12 -10
  56. package/scripts/media-generation/brand-clause.test.mjs +135 -0
  57. package/scripts/media-generation/gemini-image-client.mjs +27 -9
  58. package/scripts/media-generation/generate-assets.mjs +102 -7
  59. package/scripts/poller/inbox-scan-poller.mjs +7 -0
  60. package/scripts/poller/utils.mjs +11 -0
  61. package/scripts/pre-draft-context.py +91 -15
  62. package/scripts/spawn-session.sh +36 -6
  63. package/scripts/test-employer-grounding.py +348 -0
  64. package/scripts/validate_outbound.py +190 -26
@@ -0,0 +1,183 @@
1
+ /**
2
+ * track-claude-usage-pricing.test.mjs — an unknown model must not be priced.
3
+ *
4
+ * WHAT BROKE
5
+ * `estimateUsd()` ended `loadPrices()[model] || DEFAULT_PRICES.sonnet`. An
6
+ * unknown model — a third-party backend, a retired catalog id, a typo in a
7
+ * spawn flag — was silently priced at Sonnet's rate and written as a real
8
+ * `estimated_usd`. lib/cost/ledger-row.mjs#billableUsd then reported
9
+ * basis:"estimated", a MEASURED tier, and that laundered number flowed into
10
+ * budget-guard's degrade/refuse band, fleet-digest, and the presence beat's
11
+ * `machine.spend24h` — which peers and humans read as this seat's REAL spend.
12
+ * A guessed price is a fabricated measurement of money.
13
+ *
14
+ * WHAT THESE TESTS PIN (end-to-end, through the real CLI)
15
+ * 1. GROUNDED — a model priced by a REAL source (the local tier table, or a
16
+ * lib/model-router/catalog row with its cost_provenance stamp) still bills.
17
+ * 2. EMPTY — a model no real source prices writes `estimated_usd: null`, an
18
+ * `unpriced_reason`, and lands in the UNKNOWN class as basis:"unpriced" —
19
+ * the vocabulary lib/cost/ledger-row.mjs already has for exactly this case
20
+ * — so it is imputed, not counted as a free session.
21
+ *
22
+ * Run: node --test scripts/cost/track-claude-usage-pricing.test.mjs
23
+ */
24
+
25
+ "use strict";
26
+
27
+ import { test } from "node:test";
28
+ import assert from "node:assert/strict";
29
+ import { execFileSync } from "node:child_process";
30
+ import { mkdtempSync, mkdirSync, copyFileSync, readFileSync, readdirSync, rmSync } from "node:fs";
31
+ import { tmpdir } from "node:os";
32
+ import { join } from "node:path";
33
+
34
+ import { billableUsd, classifyRow, summariseRows, UNKNOWN } from "../../lib/cost/ledger-row.mjs";
35
+
36
+ const CLI = new URL("./track-claude-usage.mjs", import.meta.url).pathname;
37
+
38
+ /** Run `record` against a throwaway ledger dir and return {stdout, row}. */
39
+ function record(root, args) {
40
+ const out = execFileSync(process.execPath, [CLI, "record", ...args], {
41
+ env: { ...process.env, AGENT_ROOT: root },
42
+ encoding: "utf8",
43
+ stdio: ["ignore", "pipe", "pipe"],
44
+ });
45
+ const dir = join(root, "state/cost-tracking");
46
+ const files = readdirSync(dir).filter((f) => f.endsWith(".jsonl"));
47
+ const lines = readFileSync(join(dir, files[0]), "utf8").trim().split("\n");
48
+ return { stdout: JSON.parse(out), row: JSON.parse(lines[lines.length - 1]) };
49
+ }
50
+
51
+ // ── 1. GROUNDED ─────────────────────────────────────────────────────────────
52
+
53
+ test("grounded: a tier alias in the local price table still bills", () => {
54
+ const root = mkdtempSync(join(tmpdir(), "cost-"));
55
+ const { stdout, row } = record(root, [
56
+ "--cadence", "inbox", "--model", "sonnet",
57
+ "--input-tokens", "1000", "--output-tokens", "500", "--exit", "0",
58
+ ]);
59
+ assert.equal(stdout.basis, "estimated");
60
+ assert.ok(row.estimated_usd > 0, "a real price produces a real figure");
61
+ assert.equal(row.unpriced_reason, undefined);
62
+ rmSync(root, { recursive: true, force: true });
63
+ });
64
+
65
+ test("grounded: a concrete catalog id is priced from lib/model-router/catalog", () => {
66
+ const root = mkdtempSync(join(tmpdir(), "cost-"));
67
+ const { stdout, row } = record(root, [
68
+ "--cadence", "inbox", "--model", "claude-opus-4-8",
69
+ "--input-tokens", "1000000", "--output-tokens", "0", "--exit", "0",
70
+ ]);
71
+ assert.equal(stdout.basis, "estimated");
72
+ // anthropic.yaml: claude-opus-4-8 input = $5.00/MTok. This is the catalog's
73
+ // own figure, not a family guess.
74
+ assert.equal(row.estimated_usd, 5);
75
+ rmSync(root, { recursive: true, force: true });
76
+ });
77
+
78
+ test("grounded: the CLI's authoritative total_cost_usd always wins", () => {
79
+ const root = mkdtempSync(join(tmpdir(), "cost-"));
80
+ const { stdout } = record(root, [
81
+ "--cadence", "inbox", "--model", "sonnet",
82
+ "--input-tokens", "16", "--output-tokens", "10", "--cache-read-tokens", "553181",
83
+ "--total-cost-usd", "0.533412", "--exit", "0",
84
+ ]);
85
+ assert.equal(stdout.basis, "authoritative");
86
+ assert.equal(stdout.billable_usd, 0.533412);
87
+ rmSync(root, { recursive: true, force: true });
88
+ });
89
+
90
+ // ── 2. EMPTY — no real price, no number ─────────────────────────────────────
91
+
92
+ test("empty: an unknown model is UNPRICED, not silently billed at the Sonnet rate", () => {
93
+ const root = mkdtempSync(join(tmpdir(), "cost-"));
94
+ const { stdout, row } = record(root, [
95
+ "--cadence", "inbox", "--model", "some-vendor-model-v9",
96
+ "--input-tokens", "1000", "--output-tokens", "500", "--exit", "0",
97
+ ]);
98
+
99
+ assert.equal(row.estimated_usd, null, "no guessed figure is written");
100
+ assert.equal(row.total_cost_usd, undefined, "and none is invented");
101
+ assert.match(row.unpriced_reason, /no price for model "some-vendor-model-v9"/);
102
+
103
+ // The reader's verdict is the thing that matters downstream.
104
+ const { usd, basis } = billableUsd(row);
105
+ assert.equal(usd, null, "spend is UNKNOWN, not zero");
106
+ assert.equal(basis, "unpriced");
107
+ assert.equal(classifyRow(row), "measured");
108
+ assert.equal(billableUsd(row).cls, UNKNOWN, "it falls into the UNKNOWN class");
109
+ assert.equal(stdout.billable_usd, null);
110
+ assert.equal(stdout.basis, "unpriced");
111
+ rmSync(root, { recursive: true, force: true });
112
+ });
113
+
114
+ test("empty: an unpriced row is counted in `unmeasured` and flagged as a degradation", () => {
115
+ const root = mkdtempSync(join(tmpdir(), "cost-"));
116
+ const { row } = record(root, [
117
+ "--cadence", "inbox", "--model", "some-vendor-model-v9",
118
+ "--input-tokens", "1000", "--output-tokens", "500", "--exit", "0",
119
+ ]);
120
+ const s = summariseRows([row]);
121
+ assert.equal(s.unmeasured, 1, "imputeUnmeasured's base, not a measured $0");
122
+ assert.equal(s.measured, 0);
123
+ assert.equal(s.measuredUsd, 0);
124
+ assert.ok(s.degradations.some((d) => /NO price/.test(d)), "the reader says pricing is broken");
125
+ rmSync(root, { recursive: true, force: true });
126
+ });
127
+
128
+ test("empty: the unpriced row does NOT sneak into a spend total via summarise", () => {
129
+ const root = mkdtempSync(join(tmpdir(), "cost-"));
130
+ record(root, ["--cadence", "inbox", "--model", "some-vendor-model-v9", "--input-tokens", "1000", "--output-tokens", "500", "--exit", "0"]);
131
+ const out = JSON.parse(
132
+ execFileSync(process.execPath, [CLI, "summarise", "--days", "7"], {
133
+ env: { ...process.env, AGENT_ROOT: root },
134
+ encoding: "utf8",
135
+ stdio: ["ignore", "pipe", "pipe"],
136
+ }),
137
+ );
138
+ assert.equal(out.totals.sessions, 1);
139
+ assert.equal(out.totals.unmeasured_sessions, 1, "counted as unmeasured");
140
+ assert.equal(out.totals.unpriced_sessions, 1, "and specifically as unpriced");
141
+ assert.equal(out.totals.billable_usd, 0, "no null-arithmetic leaked a number in");
142
+ assert.equal(out.measurement.unmeasured_sessions, 1);
143
+ rmSync(root, { recursive: true, force: true });
144
+ });
145
+
146
+ test("the tracker still RECORDS when lib/model-router is absent (a lost row is worse)", () => {
147
+ // The tracker runs as a detached child from an agent repo whose lib/ tree may
148
+ // not carry the model router. A hard dependency on it would kill `record` on
149
+ // import and write NOTHING — invisible spend, which is strictly worse than an
150
+ // unpriced row. Stage a synthetic root with ONLY the billing contract present.
151
+ const root = mkdtempSync(join(tmpdir(), "cost-norouter-"));
152
+ mkdirSync(join(root, "scripts/cost"), { recursive: true });
153
+ mkdirSync(join(root, "lib/cost"), { recursive: true });
154
+ copyFileSync(new URL("./track-claude-usage.mjs", import.meta.url), join(root, "scripts/cost/track-claude-usage.mjs"));
155
+ copyFileSync(new URL("../../lib/cost/ledger-row.mjs", import.meta.url), join(root, "lib/cost/ledger-row.mjs"));
156
+
157
+ const out = execFileSync(
158
+ process.execPath,
159
+ [join(root, "scripts/cost/track-claude-usage.mjs"), "record",
160
+ "--cadence", "inbox", "--model", "sonnet",
161
+ "--input-tokens", "1000", "--output-tokens", "500", "--exit", "0"],
162
+ { env: { ...process.env, AGENT_ROOT: root }, encoding: "utf8", stdio: ["ignore", "pipe", "pipe"] },
163
+ );
164
+ const res = JSON.parse(out);
165
+ assert.equal(res.ok, true, "the row is still written with no model router present");
166
+ assert.equal(res.basis, "estimated", "the local tier table still prices the tier aliases");
167
+
168
+ // …and an unknown model there is still UNPRICED, not guessed.
169
+ const out2 = execFileSync(
170
+ process.execPath,
171
+ [join(root, "scripts/cost/track-claude-usage.mjs"), "record",
172
+ "--cadence", "inbox", "--model", "claude-opus-4-8",
173
+ "--input-tokens", "1000", "--output-tokens", "500", "--exit", "0"],
174
+ { env: { ...process.env, AGENT_ROOT: root }, encoding: "utf8", stdio: ["ignore", "pipe", "pipe"] },
175
+ );
176
+ assert.equal(JSON.parse(out2).basis, "unpriced", "no catalog ⇒ a catalog-only id is unpriced, not guessed");
177
+ rmSync(root, { recursive: true, force: true });
178
+ });
179
+
180
+ test("the source no longer defaults an unknown model to a known tier's price", () => {
181
+ const src = readFileSync(new URL("./track-claude-usage.mjs", import.meta.url), "utf8");
182
+ assert.doesNotMatch(src, /loadPrices\(\)\[model\]\s*\|\|\s*DEFAULT_PRICES/);
183
+ });
@@ -93,19 +93,112 @@ function loadPrices() {
93
93
  } catch { return DEFAULT_PRICES; }
94
94
  }
95
95
 
96
+ /**
97
+ * Catalog rows, loaded ONCE at startup — OPTIONALLY.
98
+ *
99
+ * lib/model-router/catalog.mjs is imported dynamically, not statically, because
100
+ * this tracker is invoked as a detached child from an agent repo whose lib/ tree
101
+ * may not carry the model router. A static import would make the whole `record`
102
+ * command die on import and write NOTHING — and a session that never reaches the
103
+ * ledger is invisible spend, which is worse than an unpriced row. Degrade to
104
+ * "fewer models are priceable" instead; an unpriceable model is recorded as
105
+ * UNPRICED, never guessed.
106
+ *
107
+ * @type {object[]}
108
+ */
109
+ let _catalogRows = [];
110
+ try {
111
+ const cat = await import("../../lib/model-router/catalog.mjs");
112
+ const loaded = cat.loadCatalogCached(AGENT_DIR);
113
+ _catalogRows = cat.listModels(loaded) || [];
114
+ } catch {
115
+ _catalogRows = [];
116
+ }
117
+
118
+ /**
119
+ * Resolve real prices for a model id, or null when nothing can price it.
120
+ *
121
+ * WHY NULL AND NOT A DEFAULT
122
+ * This used to end in `|| DEFAULT_PRICES.sonnet`: an unknown model — a
123
+ * third-party backend, a model retired from the catalog, a typo in a spawn
124
+ * flag — was silently priced at Sonnet's rate and written as a real
125
+ * `estimated_usd`. lib/cost/ledger-row.mjs#billableUsd then reported
126
+ * basis:"estimated", a MEASURED tier, and that laundered number flowed into
127
+ * budget-guard's degrade/refuse band, fleet-digest, and the presence beat's
128
+ * `machine.spend24h` — which peers and humans read as this seat's real spend.
129
+ * A guessed price is a fabricated measurement.
130
+ *
131
+ * Sources, in order, all real:
132
+ * 1. the local table (scripts/cost/pricing.json merged over DEFAULT_PRICES) —
133
+ * the coarse tier aliases the spawn paths pass ("opus"/"sonnet"/"haiku");
134
+ * 2. lib/model-router/catalog/*.yaml — per-row `cost` blocks carrying a
135
+ * `cost_provenance` stamp, which is where concrete ids
136
+ * ("claude-opus-4-8", "provider/model") are priced.
137
+ *
138
+ * Nothing else. No family-prefix guessing: "sonnet-ish id, therefore Sonnet
139
+ * pricing" is the same fabrication with extra steps.
140
+ *
141
+ * @param {string} model
142
+ * @returns {{input:number, output:number, cacheRead?:number, cacheWrite?:number, source:string}|null}
143
+ */
144
+ function resolvePrices(model) {
145
+ const key = typeof model === "string" ? model.trim().toLowerCase() : "";
146
+ if (!key) return null;
147
+
148
+ const local = loadPrices()[key];
149
+ if (local && Number.isFinite(local.input) && Number.isFinite(local.output)) {
150
+ return { input: local.input, output: local.output, source: "local-table" };
151
+ }
152
+
153
+ // The catalog is the real per-model price source. An absent/unloadable catalog
154
+ // is not an error here — it just means fewer models can be priced, and an
155
+ // unpriceable model is recorded as UNPRICED rather than guessed.
156
+ const row = _catalogRows.find(
157
+ (r) => String(r.ref || "").toLowerCase() === key || String(r.id || "").toLowerCase() === key,
158
+ );
159
+ const cost = row && row.cost && typeof row.cost === "object" ? row.cost : null;
160
+ if (cost && Number.isFinite(Number(cost.input)) && Number.isFinite(Number(cost.output))) {
161
+ return {
162
+ input: Number(cost.input),
163
+ output: Number(cost.output),
164
+ cacheRead: Number.isFinite(Number(cost.cache_read)) ? Number(cost.cache_read) : undefined,
165
+ cacheWrite: Number.isFinite(Number(cost.cache_write)) ? Number(cost.cache_write) : undefined,
166
+ source: "catalog",
167
+ };
168
+ }
169
+
170
+ return null;
171
+ }
172
+
96
173
  /**
97
174
  * Price a session from token counts. FALLBACK ONLY — readers prefer the CLI's
98
175
  * authoritative total_cost_usd (see lib/cost/ledger-row.mjs). Token semantics
99
176
  * follow the Anthropic usage envelope: `input_tokens` is the UNCACHED remainder,
100
177
  * with cache reads and cache writes counted separately (not a subset of it).
178
+ *
179
+ * Returns NULL when the model cannot be priced from a real source. The caller
180
+ * writes `estimated_usd: null`, which billableUsd() reports as basis:"unpriced"
181
+ * / class UNKNOWN — so the row is counted in `unmeasured` and imputed by
182
+ * lib/cost/ledger-row.mjs#imputeUnmeasured instead of being billed as a guess.
183
+ *
184
+ * @returns {number|null}
101
185
  */
102
186
  function estimateUsd(model, { inputTokens, outputTokens, cacheReadTokens, cacheWriteTokens }) {
103
- const prices = loadPrices()[model] || DEFAULT_PRICES.sonnet;
187
+ const prices = resolvePrices(model);
188
+ if (!prices) return null;
189
+ // Catalog rows carry real cache rates; the local table does not, so it keeps
190
+ // the documented multipliers (read 0.1x input, 5m write 1.25x input).
191
+ const cacheReadRate = Number.isFinite(prices.cacheRead)
192
+ ? prices.cacheRead
193
+ : prices.input * CACHE_READ_MULTIPLIER;
194
+ const cacheWriteRate = Number.isFinite(prices.cacheWrite)
195
+ ? prices.cacheWrite
196
+ : prices.input * CACHE_WRITE_MULTIPLIER;
104
197
  const usd =
105
198
  (inputTokens / MTOK) * prices.input +
106
199
  (outputTokens / MTOK) * prices.output +
107
- (cacheReadTokens / MTOK) * prices.input * CACHE_READ_MULTIPLIER +
108
- (cacheWriteTokens / MTOK) * prices.input * CACHE_WRITE_MULTIPLIER;
200
+ (cacheReadTokens / MTOK) * cacheReadRate +
201
+ (cacheWriteTokens / MTOK) * cacheWriteRate;
109
202
  return +usd.toFixed(6);
110
203
  }
111
204
 
@@ -181,12 +274,16 @@ function recordCmd(flags) {
181
274
  const cacheWrite = numFlag(flags, "cache-creation-tokens") ?? 0;
182
275
  row.cache_read_input_tokens = cacheRead;
183
276
  row.cache_creation_input_tokens = cacheWrite;
184
- row.estimated_usd = estimateUsd(row.model, {
277
+ // null when no real source prices this model — the row then lands in the
278
+ // UNPRICED/UNKNOWN class rather than carrying a guessed figure.
279
+ const est = estimateUsd(row.model, {
185
280
  inputTokens: row.input_tokens,
186
281
  outputTokens: row.output_tokens,
187
282
  cacheReadTokens: cacheRead,
188
283
  cacheWriteTokens: cacheWrite,
189
284
  });
285
+ row.estimated_usd = est;
286
+ if (est === null) row.unpriced_reason = `no price for model "${row.model}" in pricing.json or the model-router catalog`;
190
287
  // The CLI's own figure. Readers bill against THIS when present: it accounts
191
288
  // for cache tiers, resolved model and account pricing that no local table
192
289
  // tracks. estimated_usd stays alongside it as the reconciliation fallback.
@@ -211,6 +308,13 @@ function recordCmd(flags) {
211
308
  `[track-claude-usage] UNMEASURED session recorded (source=${row.source} cadence=${row.cadence} reason=${unmeasuredReason}) — spend for this session is unknown, not zero\n`
212
309
  );
213
310
  }
311
+ if (row.unpriced_reason) {
312
+ // Equally loud, different failure: telemetry worked, PRICING did not. The
313
+ // row is not free — it is unpriceable, and doctor/budget-guard must see it.
314
+ process.stderr.write(
315
+ `[track-claude-usage] UNPRICED session recorded (source=${row.source} cadence=${row.cadence}) — ${row.unpriced_reason}; spend for this session is unknown, not zero\n`
316
+ );
317
+ }
214
318
  process.stdout.write(JSON.stringify({ ok: true, file, measurement, billable_usd: usd, basis }) + "\n");
215
319
  }
216
320
 
@@ -246,6 +350,10 @@ function summariseCmd(flags) {
246
350
  byCadence[cad] = byCadence[cad] || empty();
247
351
  for (const bucket of [totals, byModel[key], byCadence[cad]]) {
248
352
  bucket.sessions += 1;
353
+ // "unpriced" is a session whose cost is UNKNOWN (usage read, nothing
354
+ // could price it). It must not add `null` into a spend total and be
355
+ // counted as a measured $0 — that is the same laundering one layer up.
356
+ if (basis === "unpriced") { bucket.unmeasured_sessions += 1; bucket.unpriced_sessions += 1; continue; }
249
357
  if (basis === "unmeasured") { bucket.unmeasured_sessions += 1; continue; }
250
358
  bucket.input_tokens += row.input_tokens || 0;
251
359
  bucket.output_tokens += row.output_tokens || 0;
@@ -283,6 +391,7 @@ function empty() {
283
391
  return {
284
392
  sessions: 0,
285
393
  unmeasured_sessions: 0,
394
+ unpriced_sessions: 0,
286
395
  estimated_fallback_sessions: 0,
287
396
  input_tokens: 0,
288
397
  output_tokens: 0,
@@ -69,6 +69,7 @@ import {
69
69
  sweepObligations,
70
70
  openObligations,
71
71
  escalate,
72
+ closeObligation,
72
73
  } from "./assurance.mjs";
73
74
  import { recordPoll, recordClassification, recordSession, writeHealthDashboard } from "./health.mjs";
74
75
  import { acquireLock, releaseLock, updateLock, scanStaleLocks, acquireThreadLock, claimRequest, hasActiveClaim, sweepStaleItemClaims, sanitiseItemId } from "./session-lock.mjs";
@@ -83,10 +84,14 @@ const RUNG_MECHANISM = Object.fromEntries(RUNGS.map((r) => [r.id, r.mechanism]))
83
84
  // work and land it in the org's shared, ACL'd store so the fleet's memory
84
85
  // compounds. Fail-open + best-effort — never blocks or throws into the hot path;
85
86
  // local SQLite/interaction logs remain the fail-open cache.
86
- import { isEnabled as orgEnabled, loadOrgConfig } from "../../lib/org/client.mjs";
87
+ import { isEnabled as orgEnabled, loadOrgConfig, configFromAgent, electResponder } from "../../lib/org/client.mjs";
87
88
  import { remember as orgRemember } from "../../lib/org/knowledge.mjs";
88
89
  import { sweepSessionOutcomes, resultTextFromStdout } from "./session-outcomes.mjs";
89
- import { recordWorkStep as orgRecordWorkStep } from "../../lib/org/work-ledger.mjs";
90
+ import { recordWorkStep as orgRecordWorkStep, cohortChannelId, cohortMessageId } from "../../lib/org/work-ledger.mjs";
91
+ // Claude CLI liveness (lib/claude-bin.mjs). The ack-before-work fail-safe reads
92
+ // it: a holding "let me look into it" must not go out when the reply itself can
93
+ // never be generated (the storm's other half — a promise the seat cannot keep).
94
+ import { resolveClaudeBin, isSafeExecutable } from "../../lib/claude-bin.mjs";
90
95
  import { boardPriority } from "./board-mirror.mjs";
91
96
  // Observability spine (WS — diagnostics). mintTraceId + withTrace give each
92
97
  // inbound item ONE trace_id that deep callees inherit via AsyncLocalStorage;
@@ -322,7 +327,29 @@ async function processItemTraced(item, service, itemId, trace_id, deps = {}) {
322
327
  export function enrichItem(item, service) {
323
328
  const channelStr = (item.channel || "").toLowerCase();
324
329
  const channelId = item.channel_id || "";
325
- const isDm = channelStr.startsWith("dm/") || channelId.startsWith("D");
330
+ // Cohort carries the channel kind explicitly (DM/GROUP_DM/…); its channel ids
331
+ // are cuids, so the Slack-style "dm/"/"D…" prefix heuristics never fire. When
332
+ // the kind is known it is AUTHORITATIVE:
333
+ // - DM → a 1:1 DM: is_dm true, so it skips the group election gate.
334
+ // - GROUP_DM → a multi-participant channel: is_dm FALSE, so an UNDIRECTED
335
+ // message goes through the election exactly like any other
336
+ // channel (only ONE seat should answer an ambient group DM).
337
+ // - any other explicit kind (PUBLIC/PRIVATE/…) → not a 1:1 DM.
338
+ // When the kind is ABSENT we do NOT immediately reach for the prefix
339
+ // heuristics: the DEFAULT (wide) inbound reader — `lib/org/inbound/project.mjs`
340
+ // toMessageEvent — projects a Cohort 1:1 DM with `is_dm:true`/`channel_type:'dm'`
341
+ // but does NOT emit `channel_kind`. That boolean survives the whole
342
+ // eventToInboxItem → writeInboxItem → parseInboxItemYaml chain, so HONOR it
343
+ // before falling back. Overwriting it with the cuid-blind prefix heuristics
344
+ // (which never match a Cohort id) is exactly what sent a genuine 1:1 DM through
345
+ // the group election gate and suppressed it. Precedence: explicit kind →
346
+ // pre-existing boolean is_dm → Slack-style id/label prefix heuristics.
347
+ const channelKind = String(item.channel_kind || "").toUpperCase();
348
+ const isDm = channelKind
349
+ ? channelKind === "DM"
350
+ : typeof item.is_dm === "boolean"
351
+ ? item.is_dm
352
+ : channelStr.startsWith("dm/") || channelId.startsWith("D");
326
353
  const myFirstName = loadAgent().firstName || "Agent";
327
354
  const agentThreadRegex = new RegExp(`^${myFirstName}:`, "m");
328
355
  const agentInThread = !!(item.thread_context && agentThreadRegex.test(item.thread_context));
@@ -332,6 +359,95 @@ export function enrichItem(item, service) {
332
359
  return item;
333
360
  }
334
361
 
362
+ /**
363
+ * Is the Claude CLI actually present on this machine, so a spawned reply session
364
+ * has any chance of running?
365
+ *
366
+ * `resolveClaudeBin()` returns an absolute path when the binary is found and the
367
+ * bare string "claude" as its last-resort fallback when nothing on the search
368
+ * path exists — so a non-absolute result IS the "not installed" signal. When it
369
+ * does resolve, the same owner-only safety check the spawn path trusts (M2)
370
+ * decides availability.
371
+ *
372
+ * This is a NECESSARY, not sufficient, liveness test: it catches a missing/
373
+ * unsafe binary (which would ENOENT every spawn) before we promise a reply, but
374
+ * it cannot see an API outage or a rate-limit — those still surface as an
375
+ * onClose failure, where the assurance sweep already owns the compensation.
376
+ *
377
+ * @returns {boolean}
378
+ */
379
+ function claudeAvailable() {
380
+ try {
381
+ const bin = resolveClaudeBin();
382
+ if (!bin || bin === "claude") return false; // bare fallback = nothing found
383
+ return isSafeExecutable(bin);
384
+ } catch {
385
+ return false;
386
+ }
387
+ }
388
+
389
+ /**
390
+ * Ask hq's election whether THIS seat should respond to an UNDIRECTED message in
391
+ * a channel. This is the multi-participant deferral: the should-I-respond
392
+ * decision moves off the daemon and onto hq's `routeResponders`, which elects ONE
393
+ * responder or NONE for a group and fails safe (no model → quiet). Computed once
394
+ * per messageId and cached in a durable ledger event server-side, so concurrent
395
+ * daemons read the same verdict.
396
+ *
397
+ * FAIL SAFE — THE INVERSION. Unlike every other org call in this file, whose
398
+ * failure means "proceed anyway", this one's failure means STAY SILENT: a
399
+ * disabled org, a missing message/channel id, an unreachable server, a timeout,
400
+ * a malformed frame, or a verdict that does not name this seat ALL resolve to
401
+ * `electedMe:false`. A degraded election must never fall back to respond-to-all —
402
+ * that is the storm this whole change exists to end.
403
+ *
404
+ * @param {object} item the (cohort) inbox item
405
+ * @param {object} [deps] { electResponder, orgCfg } injectable seams for tests
406
+ * @returns {Promise<{electedMe:boolean, reason:string, election?:string, mode?:string}>}
407
+ */
408
+ async function electResponderVerdict(item, deps = {}) {
409
+ const impl = deps.electResponder || electResponder;
410
+ try {
411
+ const cfg = deps.orgCfg !== undefined ? deps.orgCfg : daemonOrgCfg();
412
+ if (!orgEnabled(cfg)) return { electedMe: false, reason: "org-disabled" };
413
+ const channelId = cohortChannelId(item);
414
+ const messageId = cohortMessageId(item);
415
+ if (!channelId || !messageId) return { electedMe: false, reason: "no-message-id" };
416
+ const conn = configFromAgent(cfg);
417
+ const res = await impl(
418
+ { messageId, channelId },
419
+ { base: conn.base, token: conn.token, orgId: conn.orgId },
420
+ );
421
+ if (!res || res.error) {
422
+ const reason = res && res.error
423
+ ? `${res.error.code || "error"}: ${res.error.message || ""}`
424
+ : "no-response";
425
+ return { electedMe: false, reason: reason.slice(0, 200) };
426
+ }
427
+ const out = res.result !== undefined ? res.result : res;
428
+ const responders = Array.isArray(out && out.responders) ? out.responders : [];
429
+ // THIS seat's org member slug — the same `me` every directedness join is
430
+ // relative to (COHORT_AGENT_ID / config org.cohort.agentId). No slug on
431
+ // record ⇒ cannot match ⇒ stay silent (fail safe), never respond-to-all.
432
+ const mySlug = String(
433
+ process.env.COHORT_AGENT_ID ||
434
+ (cfg && cfg.org && cfg.org.cohort && cfg.org.cohort.agentId) ||
435
+ "",
436
+ ).trim().toLowerCase();
437
+ const electedMe =
438
+ !!mySlug &&
439
+ responders.some((r) => String((r && r.slug) || "").trim().toLowerCase() === mySlug);
440
+ return {
441
+ electedMe,
442
+ reason: electedMe ? "elected" : "not-elected",
443
+ election: out && out.election,
444
+ mode: out && out.mode,
445
+ };
446
+ } catch (err) {
447
+ return { electedMe: false, reason: `error: ${err && err.message ? err.message : String(err)}`.slice(0, 200) };
448
+ }
449
+ }
450
+
335
451
  /**
336
452
  * The classifier → quick-reply-or-session path. This is UNCHANGED behaviour:
337
453
  * it is the body `processItem` used to run inline. It is now a named mechanism
@@ -356,6 +472,13 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
356
472
  const _sendQuickResponse = deps.sendQuickResponse || sendQuickResponse;
357
473
  const _sendHoldingMessage = deps.sendHoldingMessage || sendHoldingMessage;
358
474
  const _isQuickReply = deps.isQuickReply || isQuickReply;
475
+ // The multi-participant election + the claude-liveness fail-safe. Both default
476
+ // to the real implementations; tests stub the hq election call and the
477
+ // liveness probe through these seams. `deps.electResponder` (the raw client
478
+ // helper) is threaded into the verdict so a test stubs one hq call.
479
+ const _electResponderVerdict = deps.electResponderVerdict
480
+ || ((item2) => electResponderVerdict(item2, { electResponder: deps.electResponder, orgCfg: deps.orgCfg }));
481
+ const _claudeAvailable = deps.claudeAvailable || claudeAvailable;
359
482
 
360
483
  try {
361
484
  const isDm = item.is_dm === true;
@@ -398,6 +521,44 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
398
521
  return { ok: true, path: "ignored", reason: "classifier_ignore" };
399
522
  }
400
523
 
524
+ // ── ELECTION GATE (multi-participant cohort channels) ────────────────────
525
+ // The SHOULD-I-RESPOND decision for an UNDIRECTED message in a cohort channel
526
+ // does not belong to the daemon — it belongs to hq's `routeResponders`, which
527
+ // elects exactly ONE responder or NONE for a group and fails safe (no model →
528
+ // quiet). This is the fix for the multi-AI storm: without it, every paired
529
+ // seat in a cohort channel independently classified the same ambient message
530
+ // and every one of them answered.
531
+ //
532
+ // A message that is UNAMBIGUOUSLY for THIS seat — a 1:1 DM, an @mention, a
533
+ // named address — skips the election entirely (`isDirectedAtAgent` already
534
+ // decided; an @mention is not a thing to hold an election over) and keeps the
535
+ // local fast path. Only a cohort CHANNEL message that is NOT directed defers.
536
+ //
537
+ // FAIL SAFE, INVERTED. For this ambient-channel path a failed / timed-out /
538
+ // empty / not-me election means STAY SILENT — never fall back to
539
+ // respond-to-all. The directed fast path is untouched, so an org outage can
540
+ // never drop a genuinely-directed DM. Slack keeps its own local gate below,
541
+ // where no org roster exists to elect against.
542
+ if (service === "cohort" && !isDm && !isDirectedAtAgent(item)) {
543
+ const verdict = await _electResponderVerdict(item);
544
+ if (!verdict.electedMe) {
545
+ console.log(`[daemon] Election: not elected to respond (${verdict.reason}) — staying silent for ${item.sender} in ${item.channel}`);
546
+ logEvent("classifications", {
547
+ item_id: itemId,
548
+ sender: item.sender,
549
+ service,
550
+ skipped: true,
551
+ reason: `not_elected: ${verdict.reason}`,
552
+ election: verdict.election || null,
553
+ mode: verdict.mode || null,
554
+ summary: classResult.summary,
555
+ });
556
+ markProcessed(item, service);
557
+ return { ok: true, path: "filtered", reason: "not_elected" };
558
+ }
559
+ console.log(`[daemon] Election: elected to respond (${verdict.election || "elected"}, mode=${verdict.mode || "?"}) for ${item.sender} in ${item.channel}`);
560
+ }
561
+
401
562
  // DIRECTED-MESSAGE GATE: In channels and group chats, only respond to
402
563
  // messages that are clearly directed at the agent. DMs always pass.
403
564
  // This prevents the agent from inserting itself into every conversation.
@@ -559,7 +720,44 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
559
720
  });
560
721
  return { ok: true, path: "quick_reply", reason: classResult.model };
561
722
  }
562
- // If quick reply failed to send or was blocked by validation, fall through to dispatch a full session
723
+ // PERMANENT SEND FAILURE — NEVER FALL THROUGH. A permanent failure
724
+ // (FORBIDDEN_SCOPE from the send-gate, NOT_FOUND/BAD_REQUEST from hq) means
725
+ // retrying changes nothing AND that a spawned session would fail to post the
726
+ // very same way. Worse, that session runs AUTONOMOUSLY: denied its intended
727
+ // reply, it improvises and posts UNRELATED work into a channel this seat was
728
+ // just told it may not write to. So we stop here — open the durable debt so
729
+ // the ask is not lost, write a needs-attention escalation an operator can
730
+ // sweep, close the obligation as failed so the assurance sweep does not loop
731
+ // on an impossible send, and retire the item so the poll lane stops
732
+ // re-delivering it. The requester is not told (the only channel we had was
733
+ // the forbidden one); the escalation is the trace that a human must act.
734
+ if (result.permanent) {
735
+ const cause = result.code || result.error || "permanent send failure";
736
+ console.error(`[daemon] Quick reply PERMANENTLY failed for ${item.sender} (${cause}) — NOT spawning a session; opening obligation + escalating`);
737
+ let obligationKey = itemId;
738
+ try {
739
+ const opened = await openAndAcknowledge({
740
+ item, classResult, service, traceId: trace_id, ack: false,
741
+ });
742
+ if (opened && opened.key) obligationKey = opened.key;
743
+ } catch (err) {
744
+ console.error(`[daemon] openAndAcknowledge threw for ${itemId} on permanent send failure: ${err.message}`);
745
+ }
746
+ escalate(
747
+ { key: obligationKey, sender: item.sender, service,
748
+ channel: item.channel_id || item.channel || null,
749
+ summary: classResult && classResult.summary, openedAt: Date.now(),
750
+ attempts: 1, sessionId: null, traceId: trace_id, lastError: cause,
751
+ item: { content: item.content } },
752
+ { failure: { label: cause }, told: false },
753
+ );
754
+ try { closeObligation(obligationKey, { outcome: "failed", note: cause }); } catch { /* best-effort */ }
755
+ counters.bump("send.permanent_failure", { service, code: result.code || "unknown" });
756
+ emitEvent({ type: EVENT_TYPES.ERROR, trace_id, attrs: { item_id: itemId, service, stage: "quick_reply_permanent", error: cause } });
757
+ markProcessed(item, service);
758
+ return { ok: true, path: "quick_reply_permanent", reason: cause };
759
+ }
760
+ // If quick reply failed transiently or was blocked by validation, fall through to dispatch a full session
563
761
  const reason = result.blocked ? `validation blocked: ${result.issues?.map(i => i.rule).join(", ")}` : "send failed";
564
762
  console.warn(`[daemon] Quick reply not sent (${reason}), falling through to session dispatch`);
565
763
  }
@@ -589,7 +787,16 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
589
787
  // the session as a delivered one.
590
788
  let holdingDelivered = false;
591
789
  let obligationKeyForItem = null;
592
- const ackVerdict = shouldAcknowledge({ willSpawnSession: true, item, source: "inbox" });
790
+ // FAIL-SAFE, the storm's OTHER half: never post a holding "let me look into
791
+ // it" the seat cannot keep. If the Claude CLI is not even available, the
792
+ // reply session cannot run, so an ack here would be a promise broken the
793
+ // instant it is made — a non-elected-looking silence with a lie in front of
794
+ // it. When generation is unavailable we SUPPRESS the ack (ack:false) while
795
+ // still opening the durable debt, so the assurance sweep escalates instead of
796
+ // a human staring at "on it" that never resolves.
797
+ const ackVerdict = _claudeAvailable()
798
+ ? shouldAcknowledge({ willSpawnSession: true, item, source: "inbox" })
799
+ : { ack: false, reason: "claude-unavailable" };
593
800
  try {
594
801
  // The DEBT is opened whether or not we can speak. Those are two different
595
802
  // questions and conflating them is what made a whole class of ask