@cohortapp/agent-sdk 2.5.1 → 2.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (107) hide show
  1. package/bin/maestro.mjs +305 -89
  2. package/bin/maestro.test.mjs +357 -48
  3. package/docs/runbooks/backup-restore.md +65 -33
  4. package/framework-features.json +4 -4
  5. package/lib/backup/policy.mjs +710 -0
  6. package/lib/backup/policy.test.mjs +305 -0
  7. package/lib/budget-escalate.mjs +133 -0
  8. package/lib/budget-escalate.test.mjs +232 -0
  9. package/lib/budget-guard.envelope.test.mjs +476 -0
  10. package/lib/budget-guard.mjs +853 -75
  11. package/lib/budget-guard.test.mjs +91 -42
  12. package/lib/cadences.mjs +33 -0
  13. package/lib/channels/orgmail/adapter.mjs +88 -3
  14. package/lib/channels/orgmail/adapter.test.mjs +137 -0
  15. package/lib/channels/repeat-suppressor.mjs +198 -0
  16. package/lib/channels/repeat-suppressor.test.mjs +134 -0
  17. package/lib/comms/receipts.mjs +297 -0
  18. package/lib/cost/ledger-row.mjs +333 -0
  19. package/lib/cost/ledger-row.test.mjs +183 -0
  20. package/lib/execution/drive.mjs +28 -1
  21. package/lib/execution/effects.mjs +191 -12
  22. package/lib/execution/effects.test.mjs +50 -11
  23. package/lib/goals/admission.mjs +13 -1
  24. package/lib/goals/admission.test.mjs +26 -1
  25. package/lib/goals/loop.mjs +13 -0
  26. package/lib/kpi-sensors.test.mjs +3 -0
  27. package/lib/mandate/cache.mjs +13 -5
  28. package/lib/mandate/derive.mjs +146 -21
  29. package/lib/mandate/derive.test.mjs +50 -6
  30. package/lib/mandate/model.mjs +32 -4
  31. package/lib/mandate/refresh.test.mjs +16 -2
  32. package/lib/mcp/server.test.mjs +12 -3
  33. package/lib/model-router/economics.mjs +107 -76
  34. package/lib/model-router/economics.test.mjs +64 -46
  35. package/lib/model-router/integration-coverage.test.mjs +39 -37
  36. package/lib/model-router/ledger.mjs +75 -22
  37. package/lib/model-router/ledger.test.mjs +35 -2
  38. package/lib/org/client.mjs +14 -0
  39. package/lib/org/cost-sync.mjs +16 -2
  40. package/lib/org/doctor.mjs +62 -1
  41. package/lib/org/doctor.test.mjs +36 -3
  42. package/lib/org/email-remedy.mjs +49 -0
  43. package/lib/org/engagement-ledger.mjs +376 -0
  44. package/lib/org/engagement-ledger.test.mjs +112 -0
  45. package/lib/org/engagement.mjs +1056 -0
  46. package/lib/org/engagement.test.mjs +739 -0
  47. package/lib/org/messaging.mjs +230 -3
  48. package/lib/org/messaging.test.mjs +110 -1
  49. package/lib/org/param-contract.mjs +56 -2
  50. package/lib/org/param-contract.test.mjs +26 -0
  51. package/lib/org/protocol.checksum +1 -1
  52. package/lib/org/protocol.mjs +5 -0
  53. package/lib/org/protocol.test.mjs +7 -1
  54. package/lib/org/tool-surface.mjs +506 -10
  55. package/lib/org/tool-surface.test.mjs +191 -7
  56. package/lib/org/ui-parity.mjs +333 -6
  57. package/lib/org/ui-parity.test.mjs +96 -3
  58. package/lib/org/work-ledger.mjs +241 -0
  59. package/lib/org/work-ledger.test.mjs +237 -0
  60. package/lib/plan/adoption-e2e.test.mjs +366 -0
  61. package/lib/plan/budget-enforcement.test.mjs +400 -0
  62. package/lib/plan/budget-runtime.mjs +215 -0
  63. package/lib/plan/compile.mjs +201 -5
  64. package/lib/plan/compile.test.mjs +19 -5
  65. package/lib/plan/emit.mjs +8 -0
  66. package/lib/plan/emit.test.mjs +18 -0
  67. package/lib/resource-governor.mjs +58 -12
  68. package/lib/resource-governor.test.mjs +41 -1
  69. package/lib/security/audit-engine.mjs +45 -8
  70. package/lib/security/audit-engine.test.mjs +35 -0
  71. package/lib/setup/enroll-from-cohort.mjs +14 -1
  72. package/lib/setup/sections/mandate.mjs +48 -7
  73. package/lib/setup/sections/mandate.test.mjs +17 -2
  74. package/lib/setup/sections/orgmail.mjs +10 -2
  75. package/lib/setup/state.mjs +83 -2
  76. package/lib/telemetry/collect.mjs +360 -20
  77. package/lib/telemetry/collect.test.mjs +266 -0
  78. package/package.json +1 -1
  79. package/scripts/cost/track-claude-usage.mjs +207 -48
  80. package/scripts/cost/track-claude-usage.test.mjs +148 -0
  81. package/scripts/daemon/agent-daemon.mjs +315 -17
  82. package/scripts/daemon/assurance-e2e.test.mjs +421 -0
  83. package/scripts/daemon/assurance.mjs +944 -0
  84. package/scripts/daemon/assurance.test.mjs +668 -0
  85. package/scripts/daemon/cadence-consumer-governance.test.mjs +56 -0
  86. package/scripts/daemon/cadence-consumer.mjs +147 -9
  87. package/scripts/daemon/cadence-consumer.test.mjs +6 -0
  88. package/scripts/daemon/cadence-handlers.mjs +158 -0
  89. package/scripts/daemon/cadence-handlers.test.mjs +64 -0
  90. package/scripts/daemon/classifier.test.mjs +18 -9
  91. package/scripts/daemon/deliver.mjs +314 -0
  92. package/scripts/daemon/dispatcher-governance.test.mjs +10 -0
  93. package/scripts/daemon/dispatcher.mjs +64 -6
  94. package/scripts/daemon/responder-cost.test.mjs +68 -0
  95. package/scripts/daemon/responder.mjs +351 -298
  96. package/scripts/local-triggers/generate-plists.test.mjs +7 -4
  97. package/scripts/maintenance/backup-run.mjs +415 -0
  98. package/scripts/maintenance/backup-to-cloud.sh +16 -116
  99. package/scripts/org/send-orgmail.mjs +16 -0
  100. package/scripts/record-receipt.sh +63 -0
  101. package/scripts/restore-from-backup.sh +14 -3
  102. package/scripts/restore-from-backup.test.mjs +8 -5
  103. package/scripts/send-email-threaded.py +47 -0
  104. package/scripts/send-sms.sh +4 -0
  105. package/scripts/send-whatsapp.sh +4 -0
  106. package/scripts/setup/init-backup.mjs +93 -38
  107. package/scripts/slack-send.sh +12 -0
@@ -0,0 +1,333 @@
1
+ /**
2
+ * lib/cost/ledger-row.mjs — the ONE definition of "what did this ledger row
3
+ * cost, and do we actually know?"
4
+ *
5
+ * WHY THIS MODULE EXISTS
6
+ * ----------------------
7
+ * `state/cost-tracking/<date>.jsonl` has two writers with two different row
8
+ * shapes:
9
+ *
10
+ * v1 — scripts/cost/track-claude-usage.mjs `record` (dispatcher +
11
+ * cadence-consumer + responder: 100% of measured LLM spend today)
12
+ * v2 — lib/model-router/ledger.mjs `writeLedgerRow` (llm-task + messaging
13
+ * attribution)
14
+ *
15
+ * Every reader (budget-guard, doctor's cost tripwire, `maestro cost report`,
16
+ * fleet-digest) used to sum `estimated_usd` and count every line as a session.
17
+ * That was wrong three ways, and each way made the budget governor blinder:
18
+ *
19
+ * 1. `estimated_usd` on a v1 row was priced from a token table WITH NO CACHE
20
+ * TIER. A real dispatcher row on 2026-08-12 reads input_tokens:16,
21
+ * cache_read_input_tokens:553181, total_cost_usd:$0.533 — and priced out
22
+ * at $0.075. Cache reads are ~90% of every prompt this agent sends, so the
23
+ * day's spend read 2.6x-6.1x low ($90.79 vs $235.62 on 2026-08-11).
24
+ * 2. A row whose tokens were never measured (the tracker defaulted an absent
25
+ * --input-tokens flag to 0) is INDISTINGUISHABLE from a genuine zero-token
26
+ * row. A zero that means "unmeasured" is exactly what produces a governor
27
+ * that can never trip.
28
+ * 3. Zero-LLM attribution rows (a message send: no model, no tokens, $0 by
29
+ * construction) were counted as sessions, which armed doctor's "sessions
30
+ * ran but spend is $0" tripwire on days when nothing but sends happened.
31
+ *
32
+ * So: classification and billing live HERE, once, and every reader imports it.
33
+ * Zero dependencies beyond the language — budget-guard is on the spawn hot path
34
+ * and its contract is Node built-ins only, never throws.
35
+ *
36
+ * ROW CONTRACT (forward-looking fields; legacy rows are inferred, see below)
37
+ * measurement: "measured" | "unknown" | "n/a"
38
+ * measured — input_tokens/output_tokens are real counts lifted from a
39
+ * `claude --output-format json` envelope.
40
+ * unknown — a session ran, we could not read its usage. Tokens are null,
41
+ * `unmeasured_reason` says why. NEVER write 0 here.
42
+ * n/a — no model was invoked (message-send attribution). $0 is a fact,
43
+ * not an absence.
44
+ * total_cost_usd — the CLI's authoritative figure. Preferred over any estimate.
45
+ * estimated_usd — token-derived fallback. Kept on every row for compatibility.
46
+ *
47
+ * @module lib/cost/ledger-row
48
+ */
49
+
50
+ /** @typedef {"measured"|"unknown"|"n/a"|"invalid"} RowClass */
51
+
52
+ export const MEASURED = "measured";
53
+ export const UNKNOWN = "unknown";
54
+ export const NON_LLM = "n/a";
55
+ export const INVALID = "invalid";
56
+
57
+ /**
58
+ * Numeric coercion that treats ABSENCE as absence.
59
+ *
60
+ * `Number(null)` and `Number("")` are both 0, so a naive coercion would turn a
61
+ * deliberately-null `input_tokens` back into a measured zero — re-creating the
62
+ * exact "unmeasured looks free" bug this module exists to prevent. Guard first.
63
+ */
64
+ function num(v) {
65
+ if (v === null || v === undefined || v === "") return null;
66
+ const n = Number(v);
67
+ return Number.isFinite(n) ? n : null;
68
+ }
69
+
70
+ /**
71
+ * Classify a ledger row. Explicit `measurement` wins; otherwise we infer from
72
+ * the row's shape so the ~200 rows already on disk classify correctly without a
73
+ * migration.
74
+ *
75
+ * Legacy inference:
76
+ * - null/absent tokens on BOTH sides → unknown (the tracker only ever wrote
77
+ * literal 0s, so this shape can only come from the new writer; harmless).
78
+ * - no model AND zero tokens on both sides → n/a. This is precisely the
79
+ * v2 messaging attribution row (`model:null, input_tokens:0`) and cannot
80
+ * match a real session: every spawn path stamps a model.
81
+ * - anything else → measured.
82
+ *
83
+ * @param {object} row
84
+ * @returns {RowClass}
85
+ */
86
+ export function classifyRow(row) {
87
+ if (!row || typeof row !== "object") return INVALID;
88
+
89
+ const m = row.measurement;
90
+ if (m === MEASURED || m === UNKNOWN || m === NON_LLM) return m;
91
+
92
+ const inTok = num(row.input_tokens);
93
+ const outTok = num(row.output_tokens);
94
+ const hasModel = typeof row.model === "string" && row.model.length > 0;
95
+
96
+ // Checked BEFORE the cost test: a message-send row carries a real 0 cost, and
97
+ // we must not read that as "priced, therefore measured".
98
+ if (!hasModel && (inTok || 0) === 0 && (outTok || 0) === 0) return NON_LLM;
99
+
100
+ // A row that carries a NON-ZERO COST is measured, whatever its token fields
101
+ // look like. Cost is what readers bill; tokens are diagnostics. (This also
102
+ // keeps older cost-only rows billable instead of reclassifying history as
103
+ // blind.) A cost of exactly 0 on a row that NAMES A MODEL is not evidence of
104
+ // a free session — it is the writer's default standing in for "unpriceable",
105
+ // so it falls through to the token test below.
106
+ const cost = num(row.total_cost_usd) ?? num(row.estimated_usd);
107
+ if (cost !== null && cost > 0) return MEASURED;
108
+
109
+ // No usable cost. A model ran, so it generated tokens; zero or absent counts
110
+ // on BOTH sides therefore mean we failed to measure it, not that it was free.
111
+ // This is the same "a zero that means unmeasured" defect one layer down, and
112
+ // it is why an unpriced or catalog-missing model must never bill as $0.
113
+ if ((inTok || 0) === 0 && (outTok || 0) === 0) return UNKNOWN;
114
+
115
+ return MEASURED;
116
+ }
117
+
118
+ /**
119
+ * What did this row cost, and on what authority?
120
+ *
121
+ * Precedence — authoritative > estimated > unknown. `total_cost_usd` is the
122
+ * figure the Claude CLI reports for the session; it already accounts for cache
123
+ * read/write tiers, model class and any account-level pricing, none of which a
124
+ * local price table can track. We fall back to `estimated_usd` only when the
125
+ * authoritative number is absent, and we SAY SO in `basis` so the caller can
126
+ * log the degradation rather than silently under-count.
127
+ *
128
+ * @param {object} row
129
+ * @returns {{usd:number|null, basis:"authoritative"|"estimated"|"unmeasured"|"non_llm"|"invalid", cls:RowClass}}
130
+ */
131
+ export function billableUsd(row) {
132
+ const cls = classifyRow(row);
133
+ if (cls === INVALID) return { usd: null, basis: "invalid", cls };
134
+ // A message send costs $0 because no model ran. That is a measurement, not a
135
+ // gap — it must not arm any blindness alarm.
136
+ if (cls === NON_LLM) return { usd: 0, basis: "non_llm", cls };
137
+ // A session we could not measure has an UNKNOWN cost. Returning 0 here is the
138
+ // original bug; the caller has to decide what to do with null (budget-guard
139
+ // imputes and logs; doctor goes red).
140
+ if (cls === UNKNOWN) return { usd: null, basis: "unmeasured", cls };
141
+
142
+ // `> 0`, not `>= 0`: a zero price on a row that named a model is the writer's
143
+ // "I could not price this" default (an absent catalog row, an unreadable usage
144
+ // envelope), and billing it as a free session is exactly how a pricing outage
145
+ // silences the governor. Fall through to UNKNOWN and let the caller impute.
146
+ const authoritative = num(row.total_cost_usd);
147
+ if (authoritative !== null && authoritative > 0) {
148
+ return { usd: authoritative, basis: "authoritative", cls };
149
+ }
150
+ const estimated = num(row.estimated_usd);
151
+ if (estimated !== null && estimated > 0) {
152
+ return { usd: estimated, basis: "estimated", cls };
153
+ }
154
+ // The row HAS usage but nothing could price it (no authoritative figure, and
155
+ // the catalog had no row for the model). Distinct from "unmeasured": the
156
+ // telemetry worked and the PRICING did not, which is a different bug with a
157
+ // different fix, and doctor says so. Either way it is not free.
158
+ return { usd: null, basis: "unpriced", cls: UNKNOWN };
159
+ }
160
+
161
+ function round6(n) {
162
+ return Number(n.toFixed(6));
163
+ }
164
+
165
+ /**
166
+ * Fold a day's rows into the numbers every reader needs, including the ones
167
+ * that used to be invisible.
168
+ *
169
+ * `sessions` counts LLM sessions only (measured + unmeasured) — non-LLM
170
+ * attribution rows are reported separately as `nonLlmRows`. Anything that
171
+ * degraded the number is listed in `degradations[]`: fail-open is fine, silent
172
+ * is not.
173
+ *
174
+ * @param {Array<object>} rows
175
+ * @returns {{
176
+ * sessions:number, measured:number, unmeasured:number, nonLlmRows:number,
177
+ * invalidRows:number, measuredUsd:number, authoritativeUsd:number,
178
+ * estimatedFallbackUsd:number, estimatedFallbackRows:number,
179
+ * blind:boolean, degradations:string[]
180
+ * }}
181
+ */
182
+ export function summariseRows(rows) {
183
+ let measured = 0;
184
+ let unmeasured = 0;
185
+ let nonLlmRows = 0;
186
+ let invalidRows = 0;
187
+ let measuredUsd = 0;
188
+ let authoritativeUsd = 0;
189
+ let estimatedFallbackUsd = 0;
190
+ let estimatedFallbackRows = 0;
191
+ // `unmeasured` is the imputation base; these two say WHICH failure produced it.
192
+ let noTokenRows = 0; // a session ran and nobody read its usage
193
+ let unpricedRows = 0; // usage was read but nothing could price it
194
+
195
+ for (const row of rows || []) {
196
+ const { usd, basis } = billableUsd(row);
197
+ switch (basis) {
198
+ case "authoritative":
199
+ measured++;
200
+ measuredUsd += usd;
201
+ authoritativeUsd += usd;
202
+ break;
203
+ case "estimated":
204
+ measured++;
205
+ measuredUsd += usd;
206
+ estimatedFallbackUsd += usd;
207
+ estimatedFallbackRows++;
208
+ break;
209
+ case "unmeasured":
210
+ unmeasured++;
211
+ noTokenRows++;
212
+ break;
213
+ case "unpriced":
214
+ unmeasured++;
215
+ unpricedRows++;
216
+ break;
217
+ case "non_llm":
218
+ nonLlmRows++;
219
+ break;
220
+ default:
221
+ invalidRows++;
222
+ }
223
+ }
224
+
225
+ const degradations = [];
226
+ if (noTokenRows > 0) {
227
+ degradations.push(
228
+ `${noTokenRows} session row(s) recorded NO token counts (measurement:"unknown") — their spend is unknown, not zero`
229
+ );
230
+ }
231
+ if (unpricedRows > 0) {
232
+ degradations.push(
233
+ `${unpricedRows} session row(s) carry usage but NO price (no total_cost_usd, and the model is not in the catalog) — pricing is broken, and the rows are not free`
234
+ );
235
+ }
236
+ if (estimatedFallbackRows > 0) {
237
+ degradations.push(
238
+ `${estimatedFallbackRows} row(s) had no authoritative total_cost_usd; fell back to the local token estimate ($${round6(estimatedFallbackUsd)}), which under-prices cache reads`
239
+ );
240
+ }
241
+ if (invalidRows > 0) {
242
+ degradations.push(`${invalidRows} malformed ledger row(s) skipped`);
243
+ }
244
+
245
+ return {
246
+ sessions: measured + unmeasured,
247
+ measured,
248
+ unmeasured,
249
+ nonLlmRows,
250
+ invalidRows,
251
+ noTokenRows,
252
+ unpricedRows,
253
+ measuredUsd: round6(measuredUsd),
254
+ authoritativeUsd: round6(authoritativeUsd),
255
+ estimatedFallbackUsd: round6(estimatedFallbackUsd),
256
+ estimatedFallbackRows,
257
+ // Blind = sessions ran and we measured NONE of them. This is the precise
258
+ // condition doctor's tripwire is supposed to detect. An empty ledger is not
259
+ // blind — it is quiet.
260
+ blind: measured === 0 && unmeasured > 0,
261
+ degradations,
262
+ };
263
+ }
264
+
265
+ /**
266
+ * Impute a cost for the sessions we failed to measure, so a systematic parse
267
+ * regression cannot starve the budget governor into never tripping.
268
+ *
269
+ * The alternative — counting unmeasured sessions as $0 — is the exact failure
270
+ * this module exists to kill: a governor that goes quiet precisely when
271
+ * telemetry breaks.
272
+ *
273
+ * WHAT CHANGED, AND WHY THERE IS NO LONGER A MAGIC NUMBER
274
+ * ------------------------------------------------------
275
+ * This used to fall back to a hard-coded `floorUsd = 0.25` per session when the
276
+ * day had nothing measured. That number was sourced from nothing, and it was
277
+ * ~8x BELOW this fleet's observed mean ($1.95/session on 2026-08-12), so it
278
+ * defeated the property the docstring claimed: a total telemetry outage on a day
279
+ * that really cost $68.40 (274 % of a $25 cap → the refuse rung) imputed $8.75
280
+ * and reported band 0, mode normal. Under-imputing is not conservative; it is
281
+ * the silent-governor bug wearing a different hat.
282
+ *
283
+ * So the ladder is now evidence-only, and it ADMITS when it has none:
284
+ * 1. `day_mean` — this day's own mean measured session cost. Best evidence.
285
+ * 2. `period_mean`— the caller's `fallbackPerSessionUsd` (budget-guard passes
286
+ * the MONTH's mean measured cost, which survives a day that
287
+ * is blind end to end).
288
+ * 3. `unpriceable`— no evidence at ANY horizon. We return $0 and set
289
+ * `unpriceable:true` rather than invent a price. The caller
290
+ * must not read that 0 as "cheap": `summariseRows().blind`
291
+ * is the signal, and `lib/budget-guard.dailyStatus` floors
292
+ * the enforcement band at DEGRADE when it is set. "We cannot
293
+ * measure, therefore we degrade" needs no price table.
294
+ *
295
+ * Every imputation is reported by the caller; none of it is silent.
296
+ *
297
+ * @param {ReturnType<typeof summariseRows>} summary
298
+ * @param {{fallbackPerSessionUsd?:number}|number} [opts] month/period mean, when
299
+ * the day itself measured nothing. A bare number is accepted for
300
+ * backward compatibility with the old positional `floorUsd`.
301
+ * @returns {{imputedUsd:number, perSessionUsd:number,
302
+ * basis:"day_mean"|"period_mean"|"unpriceable"|"none",
303
+ * unpriceable:boolean}}
304
+ */
305
+ export function imputeUnmeasured(summary, opts = {}) {
306
+ if (!summary || summary.unmeasured <= 0) {
307
+ return { imputedUsd: 0, perSessionUsd: 0, basis: "none", unpriceable: false };
308
+ }
309
+ const fallback = typeof opts === "number"
310
+ ? opts
311
+ : num(opts && opts.fallbackPerSessionUsd);
312
+
313
+ const dayMean = summary.measured > 0 ? summary.measuredUsd / summary.measured : null;
314
+ if (dayMean !== null && dayMean > 0) {
315
+ return {
316
+ imputedUsd: round6(dayMean * summary.unmeasured),
317
+ perSessionUsd: round6(dayMean),
318
+ basis: "day_mean",
319
+ unpriceable: false,
320
+ };
321
+ }
322
+ if (fallback !== null && fallback > 0) {
323
+ return {
324
+ imputedUsd: round6(fallback * summary.unmeasured),
325
+ perSessionUsd: round6(fallback),
326
+ basis: "period_mean",
327
+ unpriceable: false,
328
+ };
329
+ }
330
+ return { imputedUsd: 0, perSessionUsd: 0, basis: "unpriceable", unpriceable: true };
331
+ }
332
+
333
+ export default { classifyRow, billableUsd, summariseRows, imputeUnmeasured, MEASURED, UNKNOWN, NON_LLM, INVALID };
@@ -0,0 +1,183 @@
1
+ /**
2
+ * lib/cost/ledger-row.test.mjs — the measured / unmeasured / non-LLM contract.
3
+ *
4
+ * The bug these lock down: an absent token count and a genuine zero used to
5
+ * produce identical rows, and every reader summed both as $0. That is what made
6
+ * the budget governor unable to trip.
7
+ */
8
+
9
+ import test from "node:test";
10
+ import assert from "node:assert/strict";
11
+
12
+ import {
13
+ classifyRow,
14
+ billableUsd,
15
+ summariseRows,
16
+ imputeUnmeasured,
17
+ MEASURED,
18
+ UNKNOWN,
19
+ NON_LLM,
20
+ INVALID,
21
+ } from "./ledger-row.mjs";
22
+
23
+ // A real dispatcher row copied from the live ledger (2026-08-12). Note the
24
+ // shape that broke the old estimator: 16 uncached input tokens against 553k
25
+ // cache reads, and an authoritative cost 7x the token-table estimate.
26
+ const REAL_ROW = {
27
+ ts: "2026-08-12T00:09:36.073Z", cadence: "inbox", source: "dispatcher",
28
+ model: "sonnet", duration_ms: 64728, input_tokens: 16, output_tokens: 5000,
29
+ exit_code: 0, cache_read_input_tokens: 553181,
30
+ estimated_usd: 0.075048, total_cost_usd: 0.533412,
31
+ };
32
+
33
+ test("classifyRow: explicit measurement wins", () => {
34
+ assert.equal(classifyRow({ measurement: "measured", input_tokens: 1 }), MEASURED);
35
+ assert.equal(classifyRow({ measurement: "unknown", input_tokens: null }), UNKNOWN);
36
+ assert.equal(classifyRow({ measurement: "n/a" }), NON_LLM);
37
+ assert.equal(classifyRow(null), INVALID);
38
+ assert.equal(classifyRow("nope"), INVALID);
39
+ });
40
+
41
+ test("classifyRow: legacy rows are inferred without a migration", () => {
42
+ // Real dispatcher row, no `measurement` field.
43
+ assert.equal(classifyRow(REAL_ROW), MEASURED);
44
+ // Real messaging attribution row: no model, zero tokens => not a session.
45
+ assert.equal(
46
+ classifyRow({ source: "messaging", model: null, input_tokens: 0, output_tokens: 0, estimated_usd: 0 }),
47
+ NON_LLM
48
+ );
49
+ // Null tokens on both sides => unmeasured.
50
+ assert.equal(classifyRow({ model: "sonnet", input_tokens: null, output_tokens: null }), UNKNOWN);
51
+ });
52
+
53
+ test("classifyRow: a real session that used zero output tokens is still MEASURED", () => {
54
+ // The distinction that matters: a model ran and produced nothing is a
55
+ // measurement; it must not be confused with "we never looked".
56
+ assert.equal(classifyRow({ model: "sonnet", input_tokens: 120, output_tokens: 0 }), MEASURED);
57
+ });
58
+
59
+ test("billableUsd: authoritative CLI cost beats the local estimate", () => {
60
+ const { usd, basis } = billableUsd(REAL_ROW);
61
+ assert.equal(basis, "authoritative");
62
+ assert.equal(usd, 0.533412);
63
+ // The whole point: the old path would have billed the estimate.
64
+ assert.notEqual(usd, REAL_ROW.estimated_usd);
65
+ });
66
+
67
+ test("billableUsd: falls back to the estimate and says so", () => {
68
+ const { usd, basis } = billableUsd({ ...REAL_ROW, total_cost_usd: undefined });
69
+ assert.equal(basis, "estimated");
70
+ assert.equal(usd, 0.075048);
71
+ });
72
+
73
+ test("billableUsd: an unmeasured session costs NULL, not zero", () => {
74
+ const { usd, basis } = billableUsd({ measurement: "unknown", input_tokens: null, output_tokens: null });
75
+ assert.equal(usd, null);
76
+ assert.equal(basis, "unmeasured");
77
+ });
78
+
79
+ test("billableUsd: usage with no price is UNPRICED — a distinct bug from unmeasured, and still not free", () => {
80
+ const { usd, basis, cls } = billableUsd({ model: "sonnet", input_tokens: 5, output_tokens: 5 });
81
+ assert.equal(usd, null, "never 0 — a pricing outage must not read as a free session");
82
+ assert.equal(basis, "unpriced", "the telemetry worked; the PRICING did not");
83
+ assert.equal(cls, UNKNOWN);
84
+ });
85
+
86
+ test("billableUsd: a $0 price on a row that named a model is unpriceable, not free", () => {
87
+ // lib/model-router/ledger.mjs used to default `estimated_usd` to 0 whenever
88
+ // estimateCost returned null (a model missing from the catalog), and that 0
89
+ // billed as a measured free session. Both zeroes now fall through.
90
+ const zeroed = billableUsd({ model: "opus", input_tokens: 900, output_tokens: 40, total_cost_usd: 0, estimated_usd: 0 });
91
+ assert.equal(zeroed.usd, null);
92
+ assert.equal(zeroed.basis, "unpriced");
93
+ });
94
+
95
+ test("billableUsd: a zero-model attribution row is a real $0, not a gap", () => {
96
+ const { usd, basis } = billableUsd({ source: "messaging", model: null, input_tokens: 0, output_tokens: 0 });
97
+ assert.equal(usd, 0);
98
+ assert.equal(basis, "non_llm");
99
+ });
100
+
101
+ test("summariseRows: separates sessions from attribution rows", () => {
102
+ const s = summariseRows([
103
+ REAL_ROW,
104
+ { ...REAL_ROW, total_cost_usd: 1 },
105
+ { source: "messaging", model: null, input_tokens: 0, output_tokens: 0 },
106
+ { measurement: "unknown", input_tokens: null, output_tokens: null },
107
+ ]);
108
+ assert.equal(s.sessions, 3, "2 measured + 1 unmeasured; the send is not a session");
109
+ assert.equal(s.measured, 2);
110
+ assert.equal(s.unmeasured, 1);
111
+ assert.equal(s.nonLlmRows, 1);
112
+ assert.equal(s.measuredUsd, 1.533412);
113
+ assert.equal(s.blind, false);
114
+ });
115
+
116
+ test("summariseRows: blind only when sessions ran and NONE were measured", () => {
117
+ assert.equal(summariseRows([]).blind, false, "an empty ledger is quiet, not blind");
118
+ assert.equal(
119
+ summariseRows([{ source: "messaging", model: null, input_tokens: 0, output_tokens: 0 }]).blind,
120
+ false,
121
+ "attribution rows alone must not arm the alarm"
122
+ );
123
+ assert.equal(
124
+ summariseRows([{ measurement: "unknown", input_tokens: null, output_tokens: null }]).blind,
125
+ true
126
+ );
127
+ });
128
+
129
+ test("summariseRows: every degradation is reported, never swallowed", () => {
130
+ const s = summariseRows([
131
+ { measurement: "unknown", input_tokens: null, output_tokens: null },
132
+ { ...REAL_ROW, total_cost_usd: undefined },
133
+ "malformed",
134
+ ]);
135
+ assert.equal(s.degradations.length, 3);
136
+ assert.match(s.degradations.join(" | "), /recorded NO token counts/);
137
+ assert.match(s.degradations.join(" | "), /no authoritative total_cost_usd/);
138
+ assert.match(s.degradations.join(" | "), /malformed/);
139
+ });
140
+
141
+ test("imputeUnmeasured: unmeasured sessions cost the day's mean, not zero", () => {
142
+ const s = summariseRows([
143
+ { ...REAL_ROW, total_cost_usd: 2 },
144
+ { measurement: "unknown", input_tokens: null, output_tokens: null },
145
+ { measurement: "unknown", input_tokens: null, output_tokens: null },
146
+ ]);
147
+ const imp = imputeUnmeasured(s);
148
+ assert.equal(imp.basis, "day_mean");
149
+ assert.equal(imp.perSessionUsd, 2);
150
+ assert.equal(imp.imputedUsd, 4);
151
+ });
152
+
153
+ test("imputeUnmeasured: a blind day imputes from the PERIOD mean when the day has nothing", () => {
154
+ // The anti-starvation property, without a magic number. The old fallback was a
155
+ // hard-coded $0.25/session — ~8x below this fleet's observed $1.95 mean — so a
156
+ // total outage on a day that really cost $68.40 (274% of a $25 cap) imputed
157
+ // $8.75 and reported band 0. Under-imputing is the silent-governor bug.
158
+ const s = summariseRows([{ measurement: "unknown", input_tokens: null, output_tokens: null }]);
159
+ const imp = imputeUnmeasured(s, { fallbackPerSessionUsd: 1.95 });
160
+ assert.equal(imp.basis, "period_mean");
161
+ assert.equal(imp.perSessionUsd, 1.95);
162
+ assert.equal(imp.imputedUsd, 1.95);
163
+ assert.equal(imp.unpriceable, false);
164
+ });
165
+
166
+ test("imputeUnmeasured: with NO evidence at any horizon it refuses to invent a price", () => {
167
+ // And says so. The caller (lib/budget-guard.dailyStatus) floors the band at
168
+ // DEGRADE on `unpriceable`, which needs no price table: we cannot measure,
169
+ // therefore we degrade. Returning a made-up dollar figure here is what let a
170
+ // $68 day read as band 0.
171
+ const s = summariseRows([{ measurement: "unknown", input_tokens: null, output_tokens: null }]);
172
+ const imp = imputeUnmeasured(s);
173
+ assert.equal(imp.basis, "unpriceable");
174
+ assert.equal(imp.imputedUsd, 0);
175
+ assert.equal(imp.unpriceable, true);
176
+ assert.equal(s.blind, true, "the signal the band floor keys off");
177
+ });
178
+
179
+ test("imputeUnmeasured: nothing unmeasured means nothing imputed", () => {
180
+ const imp = imputeUnmeasured(summariseRows([REAL_ROW]));
181
+ assert.equal(imp.imputedUsd, 0);
182
+ assert.equal(imp.basis, "none");
183
+ });
@@ -48,6 +48,24 @@ export const EFFECT_FOR = Object.freeze({
48
48
  /** Dispositions after which the agent has SPOKEN (deepens a reply chain). */
49
49
  const SPEAKING = Object.freeze(["react_now"]);
50
50
 
51
+ /**
52
+ * Did this effect actually put words in front of a person?
53
+ *
54
+ * The disposition alone cannot answer that any more. `escalate` and `delegate`
55
+ * now reach a human through the engagement judgement when one is warranted, and
56
+ * do not when it isn't — so whether the agent spoke is a property of what the
57
+ * effect DID, not of which lane it took. An effect that reached someone reports
58
+ * `ref.spoke === true`; anything else is taken at its disposition's word.
59
+ *
60
+ * This matters beyond bookkeeping: `spoke` is what a self-healing sweep reads to
61
+ * decide whether a human is still owed an answer. A lane that claims to have
62
+ * spoken when it only wrote a row makes that sweep conclude everything is fine.
63
+ */
64
+ function didSpeak(decision, r) {
65
+ if (r && r.ref && typeof r.ref === "object" && r.ref.spoke === true) return true;
66
+ return r.ok && SPEAKING.includes(decision.disposition);
67
+ }
68
+
51
69
  /**
52
70
  * Run one effect with uniform error capture. Returns a normalised result rather
53
71
  * than throwing, so the caller's sequencing stays linear.
@@ -331,6 +349,15 @@ export async function drive(decision, o = {}) {
331
349
  });
332
350
  }
333
351
  const r = await runEffect(effects[name], name, { decision, candidate: o.candidate, item: o.item });
352
+ // An effect that half-succeeded says so on its ref. Surfacing it here is what
353
+ // keeps "the person was reached but the audit row was not written" from
354
+ // reading as an unqualified success.
355
+ if (r.ref && typeof r.ref === "object" && Array.isArray(r.ref.degraded)) {
356
+ for (const d of r.ref.degraded) {
357
+ degraded.push(String(d));
358
+ log("warn", `${name} effect degraded: ${d}`, { key: decision.key });
359
+ }
360
+ }
334
361
  if (r.missing) {
335
362
  degraded.push(`${name}_effect_missing`);
336
363
  log("error", `no \`${name}\` effect wired — the event was decided but not acted on`, { key: decision.key });
@@ -341,7 +368,7 @@ export async function drive(decision, o = {}) {
341
368
  return finish({
342
369
  effect: name,
343
370
  ok: r.ok,
344
- spoke: r.ok && SPEAKING.includes(decision.disposition),
371
+ spoke: didSpeak(decision, r),
345
372
  ref: r.ref,
346
373
  error: r.error,
347
374
  degraded,