@cohortapp/agent-sdk 2.5.1 → 2.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (107) hide show
  1. package/bin/maestro.mjs +305 -89
  2. package/bin/maestro.test.mjs +357 -48
  3. package/docs/runbooks/backup-restore.md +65 -33
  4. package/framework-features.json +4 -4
  5. package/lib/backup/policy.mjs +710 -0
  6. package/lib/backup/policy.test.mjs +305 -0
  7. package/lib/budget-escalate.mjs +133 -0
  8. package/lib/budget-escalate.test.mjs +232 -0
  9. package/lib/budget-guard.envelope.test.mjs +476 -0
  10. package/lib/budget-guard.mjs +853 -75
  11. package/lib/budget-guard.test.mjs +91 -42
  12. package/lib/cadences.mjs +33 -0
  13. package/lib/channels/orgmail/adapter.mjs +88 -3
  14. package/lib/channels/orgmail/adapter.test.mjs +137 -0
  15. package/lib/channels/repeat-suppressor.mjs +198 -0
  16. package/lib/channels/repeat-suppressor.test.mjs +134 -0
  17. package/lib/comms/receipts.mjs +297 -0
  18. package/lib/cost/ledger-row.mjs +333 -0
  19. package/lib/cost/ledger-row.test.mjs +183 -0
  20. package/lib/execution/drive.mjs +28 -1
  21. package/lib/execution/effects.mjs +191 -12
  22. package/lib/execution/effects.test.mjs +50 -11
  23. package/lib/goals/admission.mjs +13 -1
  24. package/lib/goals/admission.test.mjs +26 -1
  25. package/lib/goals/loop.mjs +13 -0
  26. package/lib/kpi-sensors.test.mjs +3 -0
  27. package/lib/mandate/cache.mjs +13 -5
  28. package/lib/mandate/derive.mjs +146 -21
  29. package/lib/mandate/derive.test.mjs +50 -6
  30. package/lib/mandate/model.mjs +32 -4
  31. package/lib/mandate/refresh.test.mjs +16 -2
  32. package/lib/mcp/server.test.mjs +12 -3
  33. package/lib/model-router/economics.mjs +107 -76
  34. package/lib/model-router/economics.test.mjs +64 -46
  35. package/lib/model-router/integration-coverage.test.mjs +39 -37
  36. package/lib/model-router/ledger.mjs +75 -22
  37. package/lib/model-router/ledger.test.mjs +35 -2
  38. package/lib/org/client.mjs +14 -0
  39. package/lib/org/cost-sync.mjs +16 -2
  40. package/lib/org/doctor.mjs +62 -1
  41. package/lib/org/doctor.test.mjs +36 -3
  42. package/lib/org/email-remedy.mjs +49 -0
  43. package/lib/org/engagement-ledger.mjs +376 -0
  44. package/lib/org/engagement-ledger.test.mjs +112 -0
  45. package/lib/org/engagement.mjs +1056 -0
  46. package/lib/org/engagement.test.mjs +739 -0
  47. package/lib/org/messaging.mjs +230 -3
  48. package/lib/org/messaging.test.mjs +110 -1
  49. package/lib/org/param-contract.mjs +56 -2
  50. package/lib/org/param-contract.test.mjs +26 -0
  51. package/lib/org/protocol.checksum +1 -1
  52. package/lib/org/protocol.mjs +5 -0
  53. package/lib/org/protocol.test.mjs +7 -1
  54. package/lib/org/tool-surface.mjs +506 -10
  55. package/lib/org/tool-surface.test.mjs +191 -7
  56. package/lib/org/ui-parity.mjs +333 -6
  57. package/lib/org/ui-parity.test.mjs +96 -3
  58. package/lib/org/work-ledger.mjs +241 -0
  59. package/lib/org/work-ledger.test.mjs +237 -0
  60. package/lib/plan/adoption-e2e.test.mjs +366 -0
  61. package/lib/plan/budget-enforcement.test.mjs +400 -0
  62. package/lib/plan/budget-runtime.mjs +215 -0
  63. package/lib/plan/compile.mjs +201 -5
  64. package/lib/plan/compile.test.mjs +19 -5
  65. package/lib/plan/emit.mjs +8 -0
  66. package/lib/plan/emit.test.mjs +18 -0
  67. package/lib/resource-governor.mjs +58 -12
  68. package/lib/resource-governor.test.mjs +41 -1
  69. package/lib/security/audit-engine.mjs +45 -8
  70. package/lib/security/audit-engine.test.mjs +35 -0
  71. package/lib/setup/enroll-from-cohort.mjs +14 -1
  72. package/lib/setup/sections/mandate.mjs +48 -7
  73. package/lib/setup/sections/mandate.test.mjs +17 -2
  74. package/lib/setup/sections/orgmail.mjs +10 -2
  75. package/lib/setup/state.mjs +83 -2
  76. package/lib/telemetry/collect.mjs +360 -20
  77. package/lib/telemetry/collect.test.mjs +266 -0
  78. package/package.json +1 -1
  79. package/scripts/cost/track-claude-usage.mjs +207 -48
  80. package/scripts/cost/track-claude-usage.test.mjs +148 -0
  81. package/scripts/daemon/agent-daemon.mjs +315 -17
  82. package/scripts/daemon/assurance-e2e.test.mjs +421 -0
  83. package/scripts/daemon/assurance.mjs +944 -0
  84. package/scripts/daemon/assurance.test.mjs +668 -0
  85. package/scripts/daemon/cadence-consumer-governance.test.mjs +56 -0
  86. package/scripts/daemon/cadence-consumer.mjs +147 -9
  87. package/scripts/daemon/cadence-consumer.test.mjs +6 -0
  88. package/scripts/daemon/cadence-handlers.mjs +158 -0
  89. package/scripts/daemon/cadence-handlers.test.mjs +64 -0
  90. package/scripts/daemon/classifier.test.mjs +18 -9
  91. package/scripts/daemon/deliver.mjs +314 -0
  92. package/scripts/daemon/dispatcher-governance.test.mjs +10 -0
  93. package/scripts/daemon/dispatcher.mjs +64 -6
  94. package/scripts/daemon/responder-cost.test.mjs +68 -0
  95. package/scripts/daemon/responder.mjs +351 -298
  96. package/scripts/local-triggers/generate-plists.test.mjs +7 -4
  97. package/scripts/maintenance/backup-run.mjs +415 -0
  98. package/scripts/maintenance/backup-to-cloud.sh +16 -116
  99. package/scripts/org/send-orgmail.mjs +16 -0
  100. package/scripts/record-receipt.sh +63 -0
  101. package/scripts/restore-from-backup.sh +14 -3
  102. package/scripts/restore-from-backup.test.mjs +8 -5
  103. package/scripts/send-email-threaded.py +47 -0
  104. package/scripts/send-sms.sh +4 -0
  105. package/scripts/send-whatsapp.sh +4 -0
  106. package/scripts/setup/init-backup.mjs +93 -38
  107. package/scripts/slack-send.sh +12 -0
package/bin/maestro.mjs CHANGED
@@ -21,6 +21,7 @@ import {
21
21
  readdirSync,
22
22
  statSync,
23
23
  lstatSync,
24
+ unlinkSync,
24
25
  openSync,
25
26
  readSync,
26
27
  closeSync,
@@ -30,6 +31,7 @@ import { execFileSync, spawnSync } from "node:child_process";
30
31
  import { createHash } from "node:crypto";
31
32
  import { homedir } from "node:os";
32
33
  import { checkOwnershipSet } from "../lib/fs-ownership.mjs";
34
+ import { summariseRows as summariseCostRows } from "../lib/cost/ledger-row.mjs";
33
35
  import { resolveArchetype, migrateLegacyArchetype } from "../lib/archetype.mjs";
34
36
  import { runAudit, applyFixes, buildAttestation } from "../lib/security/audit-engine.mjs";
35
37
  import { selectProvider } from "../lib/secrets/providers.mjs";
@@ -766,6 +768,23 @@ const UPGRADE_PATHS = [
766
768
  { path: "archetypes", mode: "smart" },
767
769
  ];
768
770
 
771
+ // Directories that sit under a "smart" upgrade root but hold MACHINE-GENERATED
772
+ // artifacts rather than framework files — launchd plists rendered per-machine
773
+ // from config/agent.json. Upstream ships no copy of them by design, so to the
774
+ // prune pass they look exactly like orphans: tracked, clean, and absent from
775
+ // the SDK. Git cannot tell the difference, because there is no difference to
776
+ // see — the intent lives in the directory, not the file state. Deleting them
777
+ // would destroy this machine's schedule (and would race the cadence-bus
778
+ // migration, which backs these same plists up and regenerates them).
779
+ const PRUNE_EXCLUDE_PREFIXES = [
780
+ "scripts/local-triggers/plists/",
781
+ "scripts/poller-launchd/",
782
+ ];
783
+
784
+ function isPruneExcluded(repoRel) {
785
+ return PRUNE_EXCLUDE_PREFIXES.some((p) => repoRel.startsWith(p));
786
+ }
787
+
769
788
  function sha256File(p) {
770
789
  return createHash("sha256").update(readFileSync(p)).digest("hex");
771
790
  }
@@ -1073,11 +1092,13 @@ function parseUpgradeFlags(args) {
1073
1092
  forceOverwrite: false,
1074
1093
  noIncoming: false,
1075
1094
  verbose: false,
1095
+ noPrune: false,
1076
1096
  };
1077
1097
  for (const a of args) {
1078
1098
  if (a === "--dry-run" || a === "-n") flags.dryRun = true;
1079
1099
  else if (a === "--force-overwrite" || a === "--force") flags.forceOverwrite = true;
1080
1100
  else if (a === "--no-incoming") flags.noIncoming = true;
1101
+ else if (a === "--no-prune") flags.noPrune = true;
1081
1102
  else if (a === "--verbose" || a === "-v") flags.verbose = true;
1082
1103
  else if (a === "--help" || a === "-h") return null;
1083
1104
  else { fail(`Unknown flag: ${a}`); process.exit(1); }
@@ -1095,6 +1116,7 @@ Flags:
1095
1116
  --dry-run, -n Preview changes without writing
1096
1117
  --force-overwrite Overwrite even locally-modified files (backs them up)
1097
1118
  --no-incoming Don't write .maestro/incoming/ shadows for preserved files
1119
+ --no-prune Keep framework files that upstream has deleted
1098
1120
  --verbose, -v Print classification for every file
1099
1121
  --help, -h Show this help
1100
1122
 
@@ -1107,6 +1129,9 @@ Per-file behaviour:
1107
1129
  .maestro/incoming/<path> for manual diff
1108
1130
  mergeKept — under agents/ → never overwrite (custom agents preserved)
1109
1131
  forced — overwritten by --force-overwrite; backup at .maestro/backup/<path>
1132
+ pruned — deleted upstream and pristine here → removed; backup at
1133
+ .maestro/backup/<path>. Only ever applies to files git reports as
1134
+ tracked-and-clean, so your own files are never at risk.
1110
1135
 
1111
1136
  .maestroignore format (gitignore-style, top-down, last match wins):
1112
1137
  scripts/slack-send.sh exact file
@@ -1144,9 +1169,11 @@ Per-file behaviour:
1144
1169
  const banner = flags.dryRun ? "DRY RUN — " : "";
1145
1170
  log(`${banner}Upgrading framework files from @cohortapp/agent-sdk...`);
1146
1171
 
1147
- const counts = { added: 0, updated: 0, same: 0, ignored: 0, preserved: 0, mergeKept: 0, forced: 0 };
1172
+ const counts = { added: 0, updated: 0, same: 0, ignored: 0, preserved: 0, mergeKept: 0, forced: 0, pruned: 0, pruneKept: 0 };
1148
1173
  const preservedFiles = [];
1149
1174
  const ignoredFiles = [];
1175
+ const prunedFiles = [];
1176
+ const pruneKeptFiles = [];
1150
1177
 
1151
1178
  for (const { path: relRoot, mode } of UPGRADE_PATHS) {
1152
1179
  const srcRoot = join(MAESTRO_ROOT, relRoot);
@@ -1234,6 +1261,86 @@ Per-file behaviour:
1234
1261
  }
1235
1262
  }
1236
1263
 
1264
+ // ── Prune: framework files this agent still holds but upstream deleted ─────
1265
+ //
1266
+ // The copy loop above is one-directional: it adds and overwrites, but never
1267
+ // removes. So every file the SDK has ever shipped stays resident on an agent
1268
+ // machine forever. That is not merely untidy — an orphan is a *stale module*
1269
+ // that keeps importing symbols its collaborators no longer export, so it
1270
+ // fails permanently and drags the agent's own test suite red. (Found in the
1271
+ // field: scripts/daemon/responder.test.mjs survived the transport refactor
1272
+ // and kept importing a deleted `loadConversationHistory`.)
1273
+ //
1274
+ // Prune asks the converse of the copy loop's question — "does the agent hold
1275
+ // a file upstream deleted?" — and reuses the identical safety rules:
1276
+ //
1277
+ // * .maestroignore still wins; an ignored path is never touched.
1278
+ // * "merge" roots (agents/) are skipped wholesale — files there are
1279
+ // *supposed* to exist only locally; that is the mode's entire purpose.
1280
+ // * Only files git reports as tracked-and-clean are removed. dirtyPathSet()
1281
+ // is built with --untracked-files=all, so a path absent from it is
1282
+ // provably a pristine framework file the operator never touched.
1283
+ // Anything else — edited, untracked, operator-authored — is kept and
1284
+ // reported, never deleted.
1285
+ // * Outside a git repo we cannot prove any of the above, so prune does
1286
+ // nothing at all rather than guess.
1287
+ //
1288
+ // Removals are backed up under .maestro/backup/ exactly like a forced
1289
+ // overwrite, so a prune is always reversible.
1290
+ if (flags.noPrune) {
1291
+ log("Prune skipped (--no-prune): upstream-deleted files left in place.");
1292
+ } else if (!inGit) {
1293
+ warn("Prune skipped: not a git repo, so pristine framework files can't be told from yours.");
1294
+ } else {
1295
+ for (const { path: relRoot, mode } of UPGRADE_PATHS) {
1296
+ // merge-mode roots hold deliberately-local files — nothing to reconcile.
1297
+ if (mode === "merge") continue;
1298
+ const srcRoot = join(MAESTRO_ROOT, relRoot);
1299
+ const dstRoot = join(cwd, relRoot);
1300
+ // If upstream dropped the whole root, treat it as out of scope rather
1301
+ // than deleting an entire directory tree on the agent's machine.
1302
+ if (!existsSync(srcRoot) || !existsSync(dstRoot)) continue;
1303
+
1304
+ for (const dstFile of walkFiles(dstRoot)) {
1305
+ const relFromRoot = relative(dstRoot, dstFile);
1306
+ if (existsSync(join(srcRoot, relFromRoot))) continue; // still shipped
1307
+
1308
+ const repoRel = relative(cwd, dstFile).split(sep).join("/");
1309
+
1310
+ // Machine-generated, not framework-shipped — see PRUNE_EXCLUDE_PREFIXES.
1311
+ if (isPruneExcluded(repoRel)) {
1312
+ if (flags.verbose) console.log(` · ${repoRel} (machine-generated, never pruned)`);
1313
+ continue;
1314
+ }
1315
+
1316
+ if (matchesIgnore(repoRel, ignorePatterns)) {
1317
+ counts.ignored++;
1318
+ ignoredFiles.push(repoRel);
1319
+ if (flags.verbose) console.log(` · ${repoRel} (orphan, ignored via .maestroignore)`);
1320
+ continue;
1321
+ }
1322
+
1323
+ // Dirty covers edited AND untracked — either way it isn't ours to delete.
1324
+ if (dirty.has(repoRel)) {
1325
+ counts.pruneKept++;
1326
+ pruneKeptFiles.push(repoRel);
1327
+ if (flags.verbose) console.log(` ~ ${repoRel} (deleted upstream, but yours — kept)`);
1328
+ continue;
1329
+ }
1330
+
1331
+ if (!flags.dryRun) {
1332
+ const backup = join(cwd, ".maestro", "backup", repoRel);
1333
+ mkdirSync(dirname(backup), { recursive: true });
1334
+ copyFileSync(dstFile, backup);
1335
+ unlinkSync(dstFile);
1336
+ }
1337
+ counts.pruned++;
1338
+ prunedFiles.push(repoRel);
1339
+ if (flags.verbose) warn(`- ${repoRel} (deleted upstream; backup written)`);
1340
+ }
1341
+ }
1342
+ }
1343
+
1237
1344
  // Ensure standard runtime directories exist.
1238
1345
  const ensureDirs = [
1239
1346
  "state/handoffs", "state/huddle", "state/indexes", "state/rag",
@@ -1382,8 +1489,21 @@ Per-file behaviour:
1382
1489
  if (counts.mergeKept) console.log(` ~ ${counts.mergeKept} merge-mode kept (agents/ custom files preserved)`);
1383
1490
  if (counts.preserved) warn(`${counts.preserved} preserved (local edits — kept your version)`);
1384
1491
  if (counts.forced) warn(`${counts.forced} force-overwritten (backups in .maestro/backup/)`);
1492
+ if (counts.pruned) ok(`${counts.pruned} pruned (deleted upstream; backups in .maestro/backup/)`);
1493
+ if (counts.pruneKept) warn(`${counts.pruneKept} deleted upstream but kept (yours — edited or untracked)`);
1385
1494
  if (newDirs) ok(`${newDirs} new directories created`);
1386
1495
 
1496
+ if (prunedFiles.length && flags.verbose === false) {
1497
+ for (const p of prunedFiles.slice(0, 5)) console.log(` - ${p}`);
1498
+ if (prunedFiles.length > 5) console.log(` … and ${prunedFiles.length - 5} more`);
1499
+ }
1500
+ if (pruneKeptFiles.length) {
1501
+ console.log();
1502
+ log("These are gone from the framework but still present here — delete if you no longer want them:");
1503
+ for (const p of pruneKeptFiles.slice(0, 5)) console.log(` ${p}`);
1504
+ if (pruneKeptFiles.length > 5) console.log(` … and ${pruneKeptFiles.length - 5} more`);
1505
+ }
1506
+
1387
1507
  if (preservedFiles.length && !flags.noIncoming && !flags.dryRun) {
1388
1508
  console.log();
1389
1509
  log("Upstream versions of your locally-modified files saved to .maestro/incoming/");
@@ -1448,70 +1568,132 @@ Per-file behaviour:
1448
1568
  // in isolation from disk + the doctor's side-effecting ok/warn/fail output.
1449
1569
 
1450
1570
  /**
1451
- * Sum estimated_usd and count rows from already-parsed ledger rows.
1452
- * Tolerant of missing/NaN estimated_usd (treated as 0) and ignores non-objects.
1571
+ * Fold already-parsed ledger rows into the billable total plus the measurement
1572
+ * provenance the tripwire needs.
1573
+ *
1574
+ * Bills against lib/cost/ledger-row.mjs — the CLI's authoritative
1575
+ * `total_cost_usd` where present, the cache-aware token estimate as a logged
1576
+ * fallback. It used to sum `estimated_usd` blindly, which both under-counted
1577
+ * (no cache tier) and treated an unmeasured session as a free one.
1578
+ *
1453
1579
  * @param {Array<object>} rows
1454
- * @returns {{ rowCount: number, totalUsd: number }}
1580
+ * @returns {{ rowCount:number, totalUsd:number, measured:number, unmeasured:number, nonLlmRows:number, degradations:string[] }}
1455
1581
  */
1456
1582
  function summariseLedgerRows(rows) {
1457
- let rowCount = 0;
1458
- let totalUsd = 0;
1459
- for (const row of rows || []) {
1460
- if (!row || typeof row !== "object") continue;
1461
- rowCount++;
1462
- const usd = Number(row.estimated_usd);
1463
- if (Number.isFinite(usd)) totalUsd += usd;
1464
- }
1465
- return { rowCount, totalUsd: +totalUsd.toFixed(6) };
1583
+ const s = summariseCostRows(rows || []);
1584
+ return {
1585
+ // rowCount stays "LLM session rows" — attribution rows for zero-model work
1586
+ // are not sessions and must not arm a blindness alarm on their own.
1587
+ rowCount: s.sessions,
1588
+ totalUsd: s.measuredUsd,
1589
+ measured: s.measured,
1590
+ unmeasured: s.unmeasured,
1591
+ // WHICH failure produced `unmeasured`: usage unread, or usage read and
1592
+ // unpriceable. Different bugs, different fixes, different sentences.
1593
+ noTokenRows: s.noTokenRows,
1594
+ unpricedRows: s.unpricedRows,
1595
+ nonLlmRows: s.nonLlmRows,
1596
+ degradations: s.degradations,
1597
+ };
1466
1598
  }
1467
1599
 
1468
1600
  /**
1469
- * Decide the cost-telemetry posture from today's ledger rows and a count of
1470
- * sessions known to have run today (rows themselves OR a daemon heartbeat that
1471
- * proves activity). The contract the spec pins down:
1601
+ * Decide the cost-telemetry posture from today's ledger rows and the number of
1602
+ * sessions the daemon logged today.
1603
+ *
1604
+ * The contract, restated so it detects the real defect instead of the clock:
1605
+ *
1606
+ * RED — sessions ran and we cannot price them. Three distinct ways:
1607
+ * (a) unmeasured rows exist — a caller recorded measurement:"unknown"
1608
+ * (b) sessions ran with NO ledger row at all — spawned but never recorded
1609
+ * (c) measured rows exist but bill to $0 — pricing itself is broken
1610
+ * GREEN — no sessions ran (nothing to measure), or every session that ran is
1611
+ * measured and bills above $0.
1472
1612
  *
1473
- * RED — sessions ran today (rows exist OR heartbeat active) but summed
1474
- * estimated_usd === 0 → "spend telemetry is blind".
1475
- * GREEN — no sessions ran (nothing to measure) OR spend > 0.
1613
+ * `sessionCount` MUST be evidence that a session actually ran (a logged session
1614
+ * start/completion), NOT a liveness signal. The previous version accepted a
1615
+ * fresh daemon heartbeat, so it went RED every UTC day between midnight and the
1616
+ * first session — an empty ledger plus a live daemon — and printed a fix that
1617
+ * had already shipped. A loud wrong alarm trains the operator to ignore the one
1618
+ * channel meant to be trustworthy, so it is as much a defect as a quiet bug.
1476
1619
  *
1477
1620
  * @param {{ rows?: Array<object>, sessionCount?: number }} input
1478
- * rows — parsed ledger rows for today (may be empty)
1479
- * sessionCount — sessions evidenced today from outside the ledger (e.g. a
1480
- * fresh daemon heartbeat). Activity = (rows.length>0) || (sessionCount>0).
1481
- * @returns {{ red: boolean, status: "red"|"green", reason: string, rowCount: number, totalUsd: number, sessionCount: number }}
1621
+ * @returns {{ red:boolean, status:"red"|"green", reason:string, fix:string|null, rowCount:number, totalUsd:number, sessionCount:number, unmeasured:number, unrecorded:number }}
1482
1622
  */
1483
- function evaluateCostTripwire({ rows = [], sessionCount = 0 } = {}) {
1484
- const { rowCount, totalUsd } = summariseLedgerRows(rows);
1623
+ function evaluateCostTripwire({ rows = [], sessionCount = 0, sessionSource = "dispatcher" } = {}) {
1624
+ const { rowCount, totalUsd, measured, unmeasured, noTokenRows, unpricedRows, nonLlmRows } = summariseLedgerRows(rows);
1485
1625
  const sessions = Math.max(0, Number(sessionCount) || 0);
1486
- const hadActivity = rowCount > 0 || sessions > 0;
1487
1626
 
1488
- if (!hadActivity) {
1627
+ // LIKE FOR LIKE. `sessionCount` comes from logs/daemon/<date>-sessions.jsonl,
1628
+ // which ONLY the dispatcher writes (`responder` and the cadence consumer have
1629
+ // no logSession). Comparing it against rows from EVERY writer meant the
1630
+ // responder rows this changeset added would mask a dispatcher that had stopped
1631
+ // recording entirely: 35 dispatcher sessions unpriced + 40 responder rows read
1632
+ // as `unrecorded: 0`, status green. Count only the rows from the writer the
1633
+ // session log belongs to.
1634
+ const sourceRows = rows.filter((r) => r && (r.source ?? null) === sessionSource);
1635
+ const sourceSessions = summariseLedgerRows(sourceRows).rowCount;
1636
+ const unrecorded = Math.max(0, sessions - sourceSessions);
1637
+ const base = { rowCount, totalUsd, sessionCount: sessions, unmeasured, unrecorded, sourceRowCount: sourceSessions };
1638
+
1639
+ if (rowCount === 0 && sessions === 0) {
1489
1640
  return {
1641
+ ...base,
1490
1642
  red: false,
1491
1643
  status: "green",
1492
- reason: "no sessions ran today — nothing to measure",
1493
- rowCount,
1494
- totalUsd,
1495
- sessionCount: sessions,
1644
+ fix: null,
1645
+ reason: nonLlmRows > 0
1646
+ ? `no LLM sessions ran today — nothing to measure (${nonLlmRows} zero-model attribution row(s))`
1647
+ : "no sessions ran today — nothing to measure",
1496
1648
  };
1497
1649
  }
1498
- if (totalUsd > 0) {
1650
+
1651
+ if (noTokenRows > 0) {
1499
1652
  return {
1500
- red: false,
1501
- status: "green",
1502
- reason: `spend telemetry live ($${totalUsd} across ${rowCount} ledger row(s))`,
1503
- rowCount,
1504
- totalUsd,
1505
- sessionCount: sessions,
1653
+ ...base,
1654
+ red: true,
1655
+ status: "red",
1656
+ reason: `spend telemetry is partially blind — ${noTokenRows}/${rowCount} session row(s) recorded NO token counts (measurement:"unknown"). Their cost is unknown, not zero.`,
1657
+ fix: "check logs/daemon for cost_usage_parse_failed — a caller ran a session but could not read usage from the `claude --output-format json` envelope. Each row's unmeasured_reason names the parse failure.",
1506
1658
  };
1507
1659
  }
1660
+
1661
+ if (unpricedRows > 0) {
1662
+ return {
1663
+ ...base,
1664
+ red: true,
1665
+ status: "red",
1666
+ reason: `spend telemetry is blind — ${unpricedRows}/${rowCount} session row(s) bill to $0 despite carrying usage. Pricing is broken, not the day.`,
1667
+ fix: "callers must pass real --input-tokens/--output-tokens (and --cache-read-tokens/--cache-creation-tokens/--total-cost-usd) to scripts/cost/track-claude-usage.mjs, lifted from the `claude --output-format json` envelope. A model missing from the router catalog also lands here.",
1668
+ };
1669
+ }
1670
+
1671
+ if (unrecorded > 0) {
1672
+ return {
1673
+ ...base,
1674
+ red: true,
1675
+ status: "red",
1676
+ reason: `spend telemetry is incomplete — the ${sessionSource} logged ${sessions} session(s) today but only ${sourceSessions} ${sessionSource} row(s) reached the cost ledger; ${unrecorded} ran unpriced.`,
1677
+ fix: "a spawn path is not calling scripts/cost/track-claude-usage.mjs at all. Every path that spawns `claude` must record a row — measured, or explicitly --tokens-unknown.",
1678
+ };
1679
+ }
1680
+
1681
+ if (measured > 0 && totalUsd <= 0) {
1682
+ return {
1683
+ ...base,
1684
+ red: true,
1685
+ status: "red",
1686
+ reason: `spend telemetry is blind — ${measured} measured session(s) bill to $0. Pricing is broken, not the day.`,
1687
+ fix: "callers must pass real --input-tokens/--output-tokens (and --cache-read-tokens/--cache-creation-tokens/--total-cost-usd) to scripts/cost/track-claude-usage.mjs, lifted from the `claude --output-format json` envelope.",
1688
+ };
1689
+ }
1690
+
1508
1691
  return {
1509
- red: true,
1510
- status: "red",
1511
- reason: `spend telemetry is blind — ${hadActivity ? "sessions ran today" : ""} but ledger sums $0 (${rowCount} row(s)${sessions ? `, heartbeat-active` : ""}). Callers are recording 0 tokens; budget governor can never trip on spend.`,
1512
- rowCount,
1513
- totalUsd,
1514
- sessionCount: sessions,
1692
+ ...base,
1693
+ red: false,
1694
+ status: "green",
1695
+ fix: null,
1696
+ reason: `spend telemetry live ($${totalUsd} across ${measured} measured session(s) today${unmeasured ? `, ${unmeasured} unmeasured` : ""})`,
1515
1697
  };
1516
1698
  }
1517
1699
 
@@ -1942,13 +2124,34 @@ async function doctor() {
1942
2124
  }
1943
2125
  } catch { /* breaker read best-effort */ }
1944
2126
 
1945
- // Today's budget band.
2127
+ // Today's budget band + the enforcement posture it implies.
2128
+ //
2129
+ // The band is measured against the seat's hq-FUNDED envelope
2130
+ // (Employee.payBasis.meteredBudget → mandate body seatBudgetCents), with the
2131
+ // local cap as a safety net. An unfunded seat is reported as exactly that —
2132
+ // "no ceiling is being enforced" is a finding, not a clean bill of health.
1946
2133
  try {
1947
2134
  const { dailyStatus } = await import(join(MAESTRO_ROOT, "lib", "budget-guard.mjs"));
1948
2135
  const b = dailyStatus({ agentRoot: cwd });
1949
- if (b.essentialOnly) { warn(`Daily budget cap reached ($${b.spentUSD}/$${b.capUSD}) — essential-only (inbox replies continue; backlog+cadence deferred)`); issues++; }
1950
- else if (b.band >= 75) { warn(`Daily budget at ${b.band}% ($${b.spentUSD}/$${b.capUSD})`); }
1951
- else ok(`Daily budget band ${b.band}% ($${b.spentUSD}/$${b.capUSD})`);
2136
+ const where = `$${b.spentUSD.toFixed(2)}/$${b.capUSD.toFixed(2)} today (${b.pct}%, ${b.capSource})`;
2137
+ const monthly = b.envelope && b.envelope.monthUSD != null
2138
+ ? `; envelope $${b.envelope.monthUSD.toFixed(2)}/mo, MTD $${b.month.spentUSD.toFixed(2)} (${b.monthPct}%)`
2139
+ : "";
2140
+ if (b.mode === "refused") { fail(`Budget REFUSED — ${where}${monthly}: spawns other than direct human replies are deferred`); issues++; }
2141
+ else if (b.mode === "suspended") { warn(`Budget SUSPENDED — ${where}${monthly}: OUTCOME obligations suspended; inbox + offline-safe cadences continue`); issues++; }
2142
+ else if (b.mode === "degraded") { warn(`Budget DEGRADED — ${where}${monthly}: cheapest model class, fan-out 1, self-directed work stopped`); issues++; }
2143
+ else if (b.band >= 80) { warn(`Budget at ${b.band}% — ${where}${monthly} (supervisor notified; nothing degraded yet)`); }
2144
+ else ok(`Budget band ${b.band}% — ${where}${monthly}`);
2145
+ if (!b.funded) {
2146
+ warn(
2147
+ "Seat is UNFUNDED: hq published no seatBudgetCents, so the org's ceiling is not being enforced — " +
2148
+ "the number above is this seat's own config/recovery.yaml. Set Employee.payBasis.meteredBudget for this member in Cohort."
2149
+ );
2150
+ issues++;
2151
+ }
2152
+ if (!b.supervisorMemberId) {
2153
+ warn("No supervisor edge on the mandate — budget escalations have nowhere to go, and are NOT sent to the owner by design");
2154
+ }
1952
2155
  } catch { /* ledger read best-effort */ }
1953
2156
 
1954
2157
  // Cost-telemetry tripwire (observability F2). Read today's cost ledger and
@@ -1967,25 +2170,36 @@ async function doctor() {
1967
2170
  try { rows.push(JSON.parse(line)); } catch { /* skip malformed row */ }
1968
2171
  }
1969
2172
  }
1970
- // Heartbeat as out-of-ledger session evidence: a fresh consumer heartbeat
1971
- // (<5m) means the daemon is live and spawning, so $0 spend is suspicious
1972
- // even before the first ledger row lands.
1973
- let heartbeatActive = 0;
2173
+ // Out-of-ledger session evidence: sessions the daemon LOGGED as completed
2174
+ // today. This deliberately replaced a liveness heartbeat. A fresh
2175
+ // heartbeat only proves the daemon is up, so "heartbeat + empty ledger"
2176
+ // fired RED every UTC day between midnight and the first session and
2177
+ // printed a fix that had already shipped — an alarm that cries wolf daily
2178
+ // is worse than no alarm, because the operator learns to skip it.
2179
+ // A logged completion is proof a session actually ran and therefore that
2180
+ // a ledger row is genuinely owed.
2181
+ let loggedSessions = 0;
1974
2182
  try {
1975
- const health = JSON.parse(readFileSync(join(cwd, "state/cadence-bus/health.json"), "utf-8"));
1976
- const ageMs = Date.now() - new Date(health.ts).getTime();
1977
- if (Number.isFinite(ageMs) && ageMs >= 0 && ageMs < 5 * 60_000) heartbeatActive = 1;
1978
- } catch { /* no heartbeat — rely on ledger rows alone */ }
2183
+ const sessionLog = join(cwd, "logs/daemon", `${ledgerDate}-sessions.jsonl`);
2184
+ if (existsSync(sessionLog)) {
2185
+ for (const line of readFileSync(sessionLog, "utf-8").split("\n")) {
2186
+ if (!line.trim()) continue;
2187
+ try {
2188
+ if (JSON.parse(line).event === "completed") loggedSessions++;
2189
+ } catch { /* skip malformed line */ }
2190
+ }
2191
+ }
2192
+ } catch { /* no session log — rely on ledger rows alone */ }
1979
2193
 
1980
- const tripwire = evaluateCostTripwire({ rows, sessionCount: heartbeatActive });
2194
+ // `sessionSource` names the writer the session log belongs to, so the two
2195
+ // populations being compared are the same population.
2196
+ const tripwire = evaluateCostTripwire({ rows, sessionCount: loggedSessions, sessionSource: "dispatcher" });
1981
2197
  if (tripwire.red) {
1982
2198
  fail(`Cost telemetry BLIND — ${tripwire.reason}`);
1983
- fail(` Fix: callers must pass real --input-tokens/--output-tokens to scripts/cost/track-claude-usage.mjs (lift usage from the claude --output-format json envelope).`);
2199
+ if (tripwire.fix) fail(` Fix: ${tripwire.fix}`);
1984
2200
  issues++;
1985
- } else if (tripwire.rowCount === 0 && heartbeatActive === 0) {
1986
- ok("Cost telemetry: no sessions recorded today (nothing to measure)");
1987
2201
  } else {
1988
- ok(`Cost telemetry live ($${tripwire.totalUsd} across ${tripwire.rowCount} ledger row(s) today)`);
2202
+ ok(`Cost telemetry: ${tripwire.reason}`);
1989
2203
  }
1990
2204
  } catch { /* cost ledger read best-effort */ }
1991
2205
 
@@ -2325,35 +2539,37 @@ async function doctor() {
2325
2539
  }
2326
2540
  }
2327
2541
 
2328
- // ── Backup configured + fresh (DR-config, gaps-enterprise-ops G16) ──────
2329
- // RED when backup is unconfigured: at fleet scale a machine with no DR target
2330
- // is a silent data-loss risk. Previously doctor exited 0 on a missing backup
2331
- // target; now it FAILs so the gap surfaces. GREEN only when a target is
2332
- // configured AND a recent (<25h) successful run is recorded.
2333
- const backupCfg = join(cwd, ".maestro/backup-config.yaml");
2334
- if (existsSync(backupCfg)) {
2335
- const cfg = readFileSync(backupCfg, "utf-8");
2336
- if (/enabled:\s*true/.test(cfg)) {
2337
- const lastBackup = join(cwd, ".maestro/last-backup.json");
2338
- if (existsSync(lastBackup)) {
2339
- try {
2340
- const lb = JSON.parse(readFileSync(lastBackup, "utf-8"));
2341
- const age = Date.now() - new Date(lb.completed_at).getTime();
2342
- if (age < 25 * 60 * 60 * 1000) ok(`Backup configured + fresh: last run ${Math.round(age / 3600000)}h ago (${lb.provider}://${lb.bucket})`);
2343
- else { warn(`Last backup ${Math.round(age / 3600000)}h ago — investigate scripts/maintenance/backup-to-cloud.sh`); issues++; }
2344
- } catch { warn("last-backup.json is malformed"); }
2345
- } else {
2346
- warn("Backup enabled but no successful run yet — schedule scripts/maintenance/backup-to-cloud.sh");
2347
- issues++;
2348
- }
2349
- } else {
2350
- fail("Backup target present but disabled (.maestro/backup-config.yaml: enabled is not true).");
2351
- console.log(" fix: set enabled: true and schedule scripts/maintenance/backup-to-cloud.sh");
2352
- issues++;
2542
+ // ── Backup posture (DR, gaps-enterprise-ops G16) ────────────────────────
2543
+ // The verdict is NOT computed here — lib/backup/policy.mjs owns it, shared
2544
+ // with the nightly-backup cadence and the runner, because three call sites
2545
+ // each deriving their own idea of "configured" is how they drifted apart.
2546
+ //
2547
+ // The ladder it applies (see backupVerdict):
2548
+ // fail — no config / unparseable / disabled / nothing would be archived
2549
+ // fail — enabled but no run has EVER succeeded (there is no restore point)
2550
+ // warn — restore points are stale, OR they are local-only (a dead machine
2551
+ // still loses everything — named in words, not left implied)
2552
+ // ok — fresh AND off-machine
2553
+ //
2554
+ // "local-only" is a WARN and not a FAIL on purpose: a check that is red on
2555
+ // every machine in the fleet is a check nobody reads, and "no restore point
2556
+ // at all" is a genuinely worse state than "restore points that would not
2557
+ // survive a house fire". Doctor now tells you which one you have.
2558
+ try {
2559
+ const { resolveBackupPlan, backupVerdict } = await import("../lib/backup/policy.mjs");
2560
+ const { backupFreshness } = await import("../lib/diagnostics/backup-freshness.mjs");
2561
+ const plan = resolveBackupPlan({ agentRoot: cwd });
2562
+ const verdict = backupVerdict(plan, backupFreshness({ agentRoot: cwd }));
2563
+ if (verdict.level === "ok") ok(verdict.msg);
2564
+ else if (verdict.level === "warn") { warn(verdict.msg); issues++; }
2565
+ else { fail(verdict.msg); issues++; }
2566
+ if (verdict.fix) console.log(` fix: ${verdict.fix}`);
2567
+ for (const v of plan.violations || []) {
2568
+ warn(` backup include path "${v.path}" is DROPPED by the credential deny-list (${v.pattern}) — it will never be archived.`);
2353
2569
  }
2354
- } else {
2355
- fail("No backup target configured — fleet machines need a DR backup (.maestro/backup-config.yaml missing).");
2356
- console.log(" fix: create .maestro/backup-config.yaml (enabled: true, provider, bucket) and schedule scripts/maintenance/backup-to-cloud.sh");
2570
+ } catch (err) {
2571
+ // Fail-open, never silent: a doctor probe must not take doctor down.
2572
+ warn(`Backup posture could not be evaluated (${err && err.message}) — treat as unbacked until this resolves.`);
2357
2573
  issues++;
2358
2574
  }
2359
2575