@cohortapp/agent-sdk 2.5.0 → 2.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. package/bin/maestro.mjs +185 -88
  2. package/bin/maestro.test.mjs +175 -48
  3. package/docs/runbooks/backup-restore.md +65 -33
  4. package/framework-features.json +4 -4
  5. package/lib/backup/policy.mjs +710 -0
  6. package/lib/backup/policy.test.mjs +305 -0
  7. package/lib/budget-escalate.mjs +133 -0
  8. package/lib/budget-escalate.test.mjs +232 -0
  9. package/lib/budget-guard.envelope.test.mjs +476 -0
  10. package/lib/budget-guard.mjs +853 -75
  11. package/lib/budget-guard.test.mjs +91 -42
  12. package/lib/cadences.mjs +33 -0
  13. package/lib/channels/orgmail/adapter.mjs +88 -3
  14. package/lib/channels/orgmail/adapter.test.mjs +137 -0
  15. package/lib/channels/repeat-suppressor.mjs +198 -0
  16. package/lib/channels/repeat-suppressor.test.mjs +134 -0
  17. package/lib/comms/receipts.mjs +297 -0
  18. package/lib/cost/ledger-row.mjs +333 -0
  19. package/lib/cost/ledger-row.test.mjs +183 -0
  20. package/lib/execution/drive.mjs +28 -1
  21. package/lib/execution/effects.mjs +191 -12
  22. package/lib/execution/effects.test.mjs +50 -11
  23. package/lib/goals/admission.mjs +13 -1
  24. package/lib/goals/admission.test.mjs +26 -1
  25. package/lib/goals/loop.mjs +13 -0
  26. package/lib/kpi-sensors.test.mjs +3 -0
  27. package/lib/mandate/cache.mjs +13 -5
  28. package/lib/mandate/derive.mjs +146 -21
  29. package/lib/mandate/derive.test.mjs +50 -6
  30. package/lib/mandate/model.mjs +32 -4
  31. package/lib/mandate/refresh.test.mjs +16 -2
  32. package/lib/mcp/server.test.mjs +12 -3
  33. package/lib/model-router/economics.mjs +107 -76
  34. package/lib/model-router/economics.test.mjs +64 -46
  35. package/lib/model-router/integration-coverage.test.mjs +39 -37
  36. package/lib/model-router/ledger.mjs +75 -22
  37. package/lib/model-router/ledger.test.mjs +35 -2
  38. package/lib/org/client.mjs +14 -0
  39. package/lib/org/cost-sync.mjs +16 -2
  40. package/lib/org/doctor.mjs +62 -1
  41. package/lib/org/doctor.test.mjs +36 -3
  42. package/lib/org/email-remedy.mjs +49 -0
  43. package/lib/org/engagement-ledger.mjs +376 -0
  44. package/lib/org/engagement-ledger.test.mjs +112 -0
  45. package/lib/org/engagement.mjs +1056 -0
  46. package/lib/org/engagement.test.mjs +739 -0
  47. package/lib/org/inbound/hydrate.mjs +107 -15
  48. package/lib/org/inbound/hydrate.test.mjs +127 -0
  49. package/lib/org/messaging.mjs +230 -3
  50. package/lib/org/messaging.test.mjs +110 -1
  51. package/lib/org/param-contract.mjs +56 -2
  52. package/lib/org/param-contract.test.mjs +26 -0
  53. package/lib/org/protocol.checksum +1 -1
  54. package/lib/org/protocol.mjs +5 -0
  55. package/lib/org/protocol.test.mjs +7 -1
  56. package/lib/org/tool-surface.mjs +506 -10
  57. package/lib/org/tool-surface.test.mjs +191 -7
  58. package/lib/org/ui-parity.mjs +333 -6
  59. package/lib/org/ui-parity.test.mjs +96 -3
  60. package/lib/org/work-ledger.mjs +241 -0
  61. package/lib/org/work-ledger.test.mjs +237 -0
  62. package/lib/plan/adoption-e2e.test.mjs +366 -0
  63. package/lib/plan/budget-enforcement.test.mjs +400 -0
  64. package/lib/plan/budget-runtime.mjs +215 -0
  65. package/lib/plan/compile.mjs +201 -5
  66. package/lib/plan/compile.test.mjs +19 -5
  67. package/lib/plan/emit.mjs +8 -0
  68. package/lib/plan/emit.test.mjs +18 -0
  69. package/lib/resource-governor.mjs +58 -12
  70. package/lib/resource-governor.test.mjs +41 -1
  71. package/lib/security/audit-engine.mjs +45 -8
  72. package/lib/security/audit-engine.test.mjs +35 -0
  73. package/lib/setup/enroll-from-cohort.mjs +14 -1
  74. package/lib/setup/sections/mandate.mjs +48 -7
  75. package/lib/setup/sections/mandate.test.mjs +17 -2
  76. package/lib/setup/sections/orgmail.mjs +10 -2
  77. package/lib/setup/state.mjs +83 -2
  78. package/lib/telemetry/collect.mjs +360 -20
  79. package/lib/telemetry/collect.test.mjs +266 -0
  80. package/package.json +1 -1
  81. package/scripts/cost/track-claude-usage.mjs +207 -48
  82. package/scripts/cost/track-claude-usage.test.mjs +148 -0
  83. package/scripts/daemon/agent-daemon.mjs +315 -17
  84. package/scripts/daemon/assurance-e2e.test.mjs +421 -0
  85. package/scripts/daemon/assurance.mjs +944 -0
  86. package/scripts/daemon/assurance.test.mjs +668 -0
  87. package/scripts/daemon/cadence-consumer-governance.test.mjs +56 -0
  88. package/scripts/daemon/cadence-consumer.mjs +147 -9
  89. package/scripts/daemon/cadence-consumer.test.mjs +6 -0
  90. package/scripts/daemon/cadence-handlers.mjs +158 -0
  91. package/scripts/daemon/cadence-handlers.test.mjs +64 -0
  92. package/scripts/daemon/deliver.mjs +314 -0
  93. package/scripts/daemon/dispatcher-governance.test.mjs +10 -0
  94. package/scripts/daemon/dispatcher.mjs +64 -6
  95. package/scripts/daemon/responder-cost.test.mjs +68 -0
  96. package/scripts/daemon/responder.mjs +351 -298
  97. package/scripts/local-triggers/generate-plists.test.mjs +7 -4
  98. package/scripts/maintenance/backup-run.mjs +415 -0
  99. package/scripts/maintenance/backup-to-cloud.sh +16 -116
  100. package/scripts/org/send-orgmail.mjs +16 -0
  101. package/scripts/record-receipt.sh +63 -0
  102. package/scripts/restore-from-backup.sh +14 -3
  103. package/scripts/restore-from-backup.test.mjs +8 -5
  104. package/scripts/send-email-threaded.py +47 -0
  105. package/scripts/send-sms.sh +4 -0
  106. package/scripts/send-whatsapp.sh +4 -0
  107. package/scripts/setup/init-backup.mjs +93 -38
  108. package/scripts/slack-send.sh +12 -0
@@ -24,6 +24,11 @@ import {
24
24
  parsePowermetrics,
25
25
  parseDf,
26
26
  collectMachine,
27
+ collectPowermetrics,
28
+ powermetricsSamplers,
29
+ parseSystemProfilerHardware,
30
+ ramLabel,
31
+ collectSpend24h,
27
32
  _internals,
28
33
  } from "./collect.mjs";
29
34
 
@@ -200,3 +205,264 @@ test("execFileSafe resolves '' on a throwing exec impl (no reject)", async () =>
200
205
  const out = await _internals.execFileSafe(() => { throw new Error("nope"); }, "x", [], 50);
201
206
  assert.equal(out, "");
202
207
  });
208
+
209
+ // ---------------------------------------------------------------------------
210
+ // The Fleet telemetry vocabulary — the fields the hq Fleet row reads by name.
211
+ //
212
+ // These four (device / ram / uptimeDays / spend24h) were READ by
213
+ // hq/src/components/fleet/machine.ts#buildFleetMachine and emitted by NOTHING,
214
+ // while four others (loadavg / memUsedPct / cpuPct / diskUsedPct) were emitted
215
+ // here and read by nothing. The intersection in production was EMPTY, so the
216
+ // Fleet view em-dashed every hardware column on the one machine that was
217
+ // genuinely alive. These tests pin the producer half of that contract.
218
+ // ---------------------------------------------------------------------------
219
+
220
+ const SP_HARDWARE = [
221
+ "Hardware:",
222
+ " Hardware Overview:",
223
+ " Model Name: Mac mini",
224
+ " Model Identifier: Mac16,11",
225
+ " Chip: Apple M4 Pro",
226
+ " Total Number of Cores: 12 (8 performance and 4 efficiency)",
227
+ " Memory: 24 GB",
228
+ "",
229
+ ].join("\n");
230
+
231
+ test("collectMachine emits the INVENTORY the Fleet row reads: device, ram, cores", async () => {
232
+ const m = await collectMachine({
233
+ os: fakeOs({ totalmem: 24 * 1024 ** 3, freemem: 2e9, cpus: 12 }),
234
+ execFile: fakeExecFile({ "/usr/sbin/system_profiler": SP_HARDWARE }),
235
+ powermetrics: false,
236
+ disk: false,
237
+ spend: false,
238
+ });
239
+ assert.equal(m.device, "Mac mini M4 Pro");
240
+ assert.equal(m.ram, "24 GB");
241
+ assert.equal(m.cores, 12);
242
+ });
243
+
244
+ test("collectMachine falls back to node:os inventory when system_profiler yields nothing", async () => {
245
+ const m = await collectMachine({
246
+ os: {
247
+ loadavg: () => [1, 1, 1],
248
+ totalmem: () => 24 * 1024 ** 3,
249
+ freemem: () => 12 * 1024 ** 3,
250
+ cpus: () => new Array(12).fill({ model: "Apple M4 Pro" }),
251
+ platform: () => "darwin",
252
+ uptime: () => 0,
253
+ },
254
+ execFile: fakeExecFile({}), // system_profiler returns ""
255
+ powermetrics: false,
256
+ disk: false,
257
+ spend: false,
258
+ });
259
+ // Never nothing: node:os alone still proves the chip, the RAM and the cores.
260
+ assert.equal(m.device, "Apple M4 Pro");
261
+ assert.equal(m.ram, "24 GB");
262
+ assert.equal(m.cores, 12);
263
+ });
264
+
265
+ test("collectMachine emits uptimeDays", async () => {
266
+ const osImpl = fakeOs({});
267
+ osImpl.uptime = () => 5756923; // ~66.6 days
268
+ const m = await collectMachine({
269
+ os: osImpl, powermetrics: false, disk: false, inventory: false, spend: false,
270
+ });
271
+ assert.equal(m.uptimeDays, 66.6);
272
+ });
273
+
274
+ test("ramLabel renders binary GB the way the spec sheet does", () => {
275
+ assert.equal(ramLabel(24 * 1024 ** 3), "24 GB");
276
+ assert.equal(ramLabel(192 * 1024 ** 3), "192 GB");
277
+ assert.equal(ramLabel(8 * 1024 ** 3), "8 GB");
278
+ assert.equal(ramLabel(0), "");
279
+ assert.equal(ramLabel("nonsense"), "");
280
+ });
281
+
282
+ test("parseSystemProfilerHardware is a tolerant pure parser", () => {
283
+ assert.deepEqual(parseSystemProfilerHardware(SP_HARDWARE), {
284
+ device: "Mac mini M4 Pro",
285
+ ram: "24 GB",
286
+ cores: 12,
287
+ });
288
+ // Partial output degrades field-by-field rather than throwing.
289
+ assert.deepEqual(parseSystemProfilerHardware(" Memory: 16 GB\n"), { ram: "16 GB" });
290
+ assert.deepEqual(parseSystemProfilerHardware(""), {});
291
+ assert.deepEqual(parseSystemProfilerHardware(null), {});
292
+ });
293
+
294
+ // ---------------------------------------------------------------------------
295
+ // tempC — the sampler bug, and the rule that a failure is never silent
296
+ // ---------------------------------------------------------------------------
297
+
298
+ test("powermetricsSamplers uses the arm64 spelling — 'smc' is Intel-only", () => {
299
+ // The shipped constant was "smc,cpu_power" unconditionally. On Apple Silicon
300
+ // powermetrics rejects that argument list outright and exits non-zero BEFORE
301
+ // the root check, so the probe was unreachable on every machine in the fleet.
302
+ assert.equal(powermetricsSamplers("arm64"), "thermal,cpu_power");
303
+ assert.equal(powermetricsSamplers("x64"), "smc,cpu_power");
304
+ });
305
+
306
+ test("collectPowermetrics passes the arch-correct sampler to the binary", async () => {
307
+ const seen = [];
308
+ const spy = (cmd, args, optsOrCb, maybeCb) => {
309
+ const cb = typeof optsOrCb === "function" ? optsOrCb : maybeCb;
310
+ seen.push({ cmd, args });
311
+ queueMicrotask(() => cb(null, "CPU die temperature: 54.2 C\n", ""));
312
+ return { on() {} };
313
+ };
314
+ const pm = await collectPowermetrics({ execFile: spy, platform: "darwin", arch: "arm64" });
315
+ assert.equal(pm.tempC, 54.2);
316
+ assert.equal(pm.source, "powermetrics");
317
+ const samplerArg = seen[0].args[seen[0].args.indexOf("--samplers") + 1];
318
+ assert.equal(samplerArg, "thermal,cpu_power");
319
+ assert.ok(!samplerArg.includes("smc"), "must not send the Intel-only sampler to arm64");
320
+ });
321
+
322
+ test("collectPowermetrics ALWAYS says why it has no temperature (fail-open, never silent)", async () => {
323
+ // The unprivileged reality on every seat: the probe yields nothing.
324
+ const pm = await collectPowermetrics({
325
+ execFile: fakeExecFile({}), platform: "darwin", arch: "arm64", sudoPowermetrics: false,
326
+ });
327
+ assert.equal(pm.tempC, undefined);
328
+ assert.match(pm.detail, /requires root/);
329
+ assert.match(pm.detail, /TELEMETRY_SUDO_POWERMETRICS/);
330
+
331
+ // A non-darwin host is a DIFFERENT absence and says so.
332
+ const linux = await collectPowermetrics({ platform: "linux" });
333
+ assert.match(linux.detail, /platform linux/);
334
+ });
335
+
336
+ test("collectMachine surfaces the temperature failure as machine.tempDetail", async () => {
337
+ const m = await collectMachine({
338
+ os: fakeOs({}),
339
+ execFile: fakeExecFile({}),
340
+ platform: "darwin",
341
+ arch: "arm64",
342
+ sudoPowermetrics: false,
343
+ disk: false,
344
+ inventory: false,
345
+ spend: false,
346
+ });
347
+ assert.equal(m.tempC, undefined);
348
+ assert.equal(m.tempSource, undefined);
349
+ assert.match(m.tempDetail, /requires root/);
350
+ // And the cpuPct it DOES have is labelled as the estimate it is.
351
+ assert.equal(m.cpuSource, "loadavg");
352
+ });
353
+
354
+ test("collectPowermetrics tries the opt-in sudo path only when enabled", async () => {
355
+ const seen = [];
356
+ const spy = (cmd, args, optsOrCb, maybeCb) => {
357
+ const cb = typeof optsOrCb === "function" ? optsOrCb : maybeCb;
358
+ seen.push(cmd);
359
+ // Only the sudo invocation yields.
360
+ const out = cmd === "/usr/bin/sudo" ? "CPU die temperature: 61.0 C\n" : "";
361
+ queueMicrotask(() => cb(null, out, ""));
362
+ return { on() {} };
363
+ };
364
+ const off = await collectPowermetrics({
365
+ execFile: spy, platform: "darwin", arch: "arm64", sudoPowermetrics: false,
366
+ });
367
+ assert.equal(off.tempC, undefined, "must not shell out to sudo unless asked");
368
+ assert.ok(!seen.includes("/usr/bin/sudo"));
369
+
370
+ seen.length = 0;
371
+ const on = await collectPowermetrics({
372
+ execFile: spy, platform: "darwin", arch: "arm64", sudoPowermetrics: true,
373
+ });
374
+ assert.equal(on.tempC, 61);
375
+ assert.equal(on.source, "sudo-powermetrics");
376
+ assert.ok(seen.includes("/usr/bin/sudo"));
377
+ // `-n` so a machine with no NOPASSWD rule fails instantly instead of hanging
378
+ // the beat on a password prompt nobody will ever see.
379
+ const sudoCall = seen.indexOf("/usr/bin/sudo");
380
+ assert.ok(sudoCall >= 0);
381
+ });
382
+
383
+ // ---------------------------------------------------------------------------
384
+ // spend24h
385
+ // ---------------------------------------------------------------------------
386
+
387
+ test("collectSpend24h sums BILLABLE rows in a true rolling 24h window", () => {
388
+ const root = mkdtempSync(join(tmpdir(), "spend-"));
389
+ try {
390
+ const dir = join(root, "state", "cost-tracking");
391
+ mkdirSync(dir, { recursive: true });
392
+ const now = Date.parse("2026-08-13T06:00:00Z");
393
+ const row = (ts, cost) => JSON.stringify({
394
+ ts, model: "claude-opus-4-5", input_tokens: 10, output_tokens: 5, total_cost_usd: cost,
395
+ });
396
+ // Yesterday's file: one row inside the window, one outside it.
397
+ writeFileSync(join(dir, "2026-08-12.jsonl"), [
398
+ row("2026-08-12T05:00:00Z", 1.5), // 25h ago — OUTSIDE
399
+ row("2026-08-12T07:00:00Z", 2.25), // 23h ago — inside
400
+ ].join("\n") + "\n");
401
+ writeFileSync(join(dir, "2026-08-13.jsonl"), [
402
+ row("2026-08-13T01:00:00Z", 0.75),
403
+ // A non-LLM attribution row: $0 is a FACT (no model ran), not an absence.
404
+ JSON.stringify({ ts: "2026-08-13T02:00:00Z", model: null, task_class: "messaging.send", total_cost_usd: null, estimated_usd: 0 }),
405
+ ].join("\n") + "\n");
406
+
407
+ assert.equal(collectSpend24h({ agentRoot: root, now }), 3);
408
+ } finally {
409
+ rmSync(root, { recursive: true, force: true });
410
+ }
411
+ });
412
+
413
+ test("collectSpend24h: no ledger is UNDEFINED, not a confident zero", () => {
414
+ assert.equal(collectSpend24h({ agentRoot: "/nonexistent-root-xyz" }), undefined);
415
+ assert.equal(collectSpend24h({}), undefined);
416
+ });
417
+
418
+ test("collectSpend24h does not count an UNMEASURED row as free", () => {
419
+ const root = mkdtempSync(join(tmpdir(), "spend2-"));
420
+ try {
421
+ const dir = join(root, "state", "cost-tracking");
422
+ mkdirSync(dir, { recursive: true });
423
+ const now = Date.parse("2026-08-13T06:00:00Z");
424
+ writeFileSync(join(dir, "2026-08-13.jsonl"), [
425
+ // measurement:"unknown" → billableUsd().usd === null. Skipped, not zeroed.
426
+ JSON.stringify({ ts: "2026-08-13T01:00:00Z", model: "claude-opus-4-5", measurement: "unknown", input_tokens: null, output_tokens: null, total_cost_usd: null }),
427
+ JSON.stringify({ ts: "2026-08-13T02:00:00Z", model: "claude-opus-4-5", input_tokens: 1, output_tokens: 1, total_cost_usd: 4.5 }),
428
+ "{ not json",
429
+ ].join("\n") + "\n");
430
+ assert.equal(collectSpend24h({ agentRoot: root, now }), 4.5);
431
+ } finally {
432
+ rmSync(root, { recursive: true, force: true });
433
+ }
434
+ });
435
+
436
+ test("collectStatus ships the FULL machine vocabulary the Fleet row reads", async () => {
437
+ const root = mkdtempSync(join(tmpdir(), "vocab-"));
438
+ try {
439
+ const osImpl = fakeOs({ totalmem: 24 * 1024 ** 3, freemem: 1e9, loadavg: [1.26, 1.2, 1.08], cpus: 12 });
440
+ osImpl.uptime = () => 5756923;
441
+ const status = await collectStatus({
442
+ agentRoot: root,
443
+ now: Date.parse("2026-08-13T06:00:00Z"),
444
+ os: osImpl,
445
+ arch: "arm64",
446
+ sudoPowermetrics: false,
447
+ execFile: fakeExecFile({
448
+ "/usr/sbin/system_profiler": SP_HARDWARE,
449
+ "/bin/df": "Filesystem 1024-blocks Used Available Capacity Mounted\n/dev/disk1 100 47 53 47% /\n",
450
+ }),
451
+ subAgentsRunning: 0,
452
+ });
453
+ const m = status.machine;
454
+ // The five keys hq reads that this collector used to emit NONE of.
455
+ assert.equal(m.device, "Mac mini M4 Pro");
456
+ assert.equal(m.ram, "24 GB");
457
+ assert.equal(m.uptimeDays, 66.6);
458
+ assert.equal(typeof m.tempDetail, "string"); // tempC needs root; the REASON ships
459
+ // The four it emitted that hq used to read NONE of.
460
+ assert.deepEqual(m.loadavg, [1.26, 1.2, 1.08]);
461
+ assert.ok(m.memUsedPct > 90);
462
+ assert.equal(typeof m.cpuPct, "number");
463
+ assert.equal(m.diskUsedPct, 47);
464
+ assert.equal(m.cpuSource, "loadavg");
465
+ } finally {
466
+ rmSync(root, { recursive: true, force: true });
467
+ }
468
+ });
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@cohortapp/agent-sdk",
3
- "version": "2.5.0",
3
+ "version": "2.6.0",
4
4
  "description": "Cohort Agent SDK \u2014 autonomous AI colleague runtime. Deploy senior AI colleagues on dedicated Mac minis, wired to the Cohort operating surface.",
5
5
  "type": "module",
6
6
  "bin": {
@@ -10,38 +10,103 @@
10
10
  *
11
11
  * node scripts/cost/track-claude-usage.mjs record \
12
12
  * --cadence inbox-processor --model sonnet --duration-ms 12345 \
13
- * --input-tokens 1500 --output-tokens 320 --exit 0
13
+ * --input-tokens 1500 --output-tokens 320 \
14
+ * --cache-read-tokens 553181 --cache-creation-tokens 7800 \
15
+ * --total-cost-usd 0.533412 --exit 0
16
+ *
17
+ * A caller that ran a session but could NOT read its usage must say so:
18
+ *
19
+ * node scripts/cost/track-claude-usage.mjs record \
20
+ * --cadence inbox --source dispatcher --tokens-unknown no-usage-field
21
+ *
22
+ * That writes measurement:"unknown" with NULL tokens. It must never write
23
+ * zeros — see "MEASURED, UNKNOWN, N/A" below.
14
24
  *
15
25
  * 2. `summarise` — Rollup the last N days into state/dashboards/cost-summary.yaml.
16
26
  * Called daily (cadence: nightly-cost-rollup) and on demand.
17
27
  *
18
28
  * node scripts/cost/track-claude-usage.mjs summarise --days 7
19
29
  *
20
- * Token-cost estimation uses a static price table (./pricing.json if present,
21
- * otherwise sensible defaults). Numbers are estimates — the source of truth
22
- * for billing is the Anthropic console. Goal here is local observability,
23
- * not exact accounting.
30
+ * MEASURED, UNKNOWN, N/A
31
+ * ----------------------
32
+ * Every row carries `measurement`. The old behaviour — `Number(flags[x] || 0)`
33
+ * — turned an ABSENT token flag into a literal 0, so "we didn't measure this
34
+ * session" and "this session used no tokens" wrote identical rows. The budget
35
+ * governor summed both as $0 and could never trip. Absent flags now produce
36
+ * null tokens + measurement:"unknown" + an `unmeasured_reason`, and every
37
+ * reader (lib/cost/ledger-row.mjs) treats that as UNKNOWN, not free.
38
+ *
39
+ * PRICING
40
+ * -------
41
+ * `total_cost_usd` from the `claude --output-format json` envelope is the
42
+ * authoritative figure and is what readers bill against. The local table is a
43
+ * FALLBACK for when the envelope lacks it. It is cache-tier aware, because on
44
+ * this workload cache reads are ~99% of prompt tokens: a real row reads
45
+ * input_tokens:16 / cache_read:553181, and a cache-blind table priced that
46
+ * session at $0.075 against an actual $0.533. Rates are USD per MILLION tokens
47
+ * (Anthropic's own unit — the old table was per-1K and had drifted stale);
48
+ * cache reads bill at 0.1x input and 5-minute cache writes at 1.25x input.
24
49
  */
25
50
 
26
51
  import { existsSync, mkdirSync, appendFileSync, readFileSync, writeFileSync, readdirSync } from "node:fs";
27
52
  import { join, resolve, dirname } from "node:path";
28
53
  import { fileURLToPath } from "node:url";
29
54
 
55
+ import { billableUsd, summariseRows } from "../../lib/cost/ledger-row.mjs";
56
+
30
57
  const __dirname = dirname(fileURLToPath(import.meta.url));
31
58
  const AGENT_DIR = process.env.AGENT_ROOT || process.env.AGENT_DIR || resolve(__dirname, "..", "..");
32
59
  const LEDGER_DIR = join(AGENT_DIR, "state/cost-tracking");
33
60
 
34
- // Default pricing (USD per 1K tokens). Update when Anthropic changes prices.
61
+ // Default pricing (USD per MILLION tokens). Update when Anthropic changes
62
+ // prices. cache_read = 0.1x input, cache_write (5m TTL, what the CLI uses) =
63
+ // 1.25x input — derived rather than hand-listed so the ratios can't drift apart.
64
+ const MTOK = 1e6;
65
+ const CACHE_READ_MULTIPLIER = 0.1;
66
+ const CACHE_WRITE_MULTIPLIER = 1.25;
35
67
  const DEFAULT_PRICES = {
36
- opus: { input: 0.015, output: 0.075 },
37
- sonnet: { input: 0.003, output: 0.015 },
38
- haiku: { input: 0.0008, output: 0.004 },
68
+ opus: { input: 5, output: 25 },
69
+ sonnet: { input: 3, output: 15 },
70
+ haiku: { input: 1, output: 5 },
39
71
  };
40
72
  const PRICING_FILE = join(__dirname, "pricing.json");
41
73
  function loadPrices() {
42
74
  if (!existsSync(PRICING_FILE)) return DEFAULT_PRICES;
43
- try { return { ...DEFAULT_PRICES, ...JSON.parse(readFileSync(PRICING_FILE, "utf-8")) }; }
44
- catch { return DEFAULT_PRICES; }
75
+ try {
76
+ const raw = JSON.parse(readFileSync(PRICING_FILE, "utf-8"));
77
+ const merged = { ...DEFAULT_PRICES };
78
+ for (const [model, p] of Object.entries(raw || {})) {
79
+ if (!p || typeof p !== "object") continue;
80
+ const input = Number(p.input);
81
+ const output = Number(p.output);
82
+ if (!Number.isFinite(input) || !Number.isFinite(output)) continue;
83
+ // Unit guard: this table was per-1K until the cache-tier fix and an
84
+ // operator's pricing.json may still be in the old unit. No Claude model
85
+ // costs under $0.10/MTok, so a sub-0.10 input rate is per-1K — rescale
86
+ // rather than silently under-price the day by 1000x.
87
+ const perThousand = input < 0.1;
88
+ merged[model] = perThousand
89
+ ? { input: input * 1000, output: output * 1000, unit_rescaled: true }
90
+ : { input, output };
91
+ }
92
+ return merged;
93
+ } catch { return DEFAULT_PRICES; }
94
+ }
95
+
96
+ /**
97
+ * Price a session from token counts. FALLBACK ONLY — readers prefer the CLI's
98
+ * authoritative total_cost_usd (see lib/cost/ledger-row.mjs). Token semantics
99
+ * follow the Anthropic usage envelope: `input_tokens` is the UNCACHED remainder,
100
+ * with cache reads and cache writes counted separately (not a subset of it).
101
+ */
102
+ function estimateUsd(model, { inputTokens, outputTokens, cacheReadTokens, cacheWriteTokens }) {
103
+ const prices = loadPrices()[model] || DEFAULT_PRICES.sonnet;
104
+ const usd =
105
+ (inputTokens / MTOK) * prices.input +
106
+ (outputTokens / MTOK) * prices.output +
107
+ (cacheReadTokens / MTOK) * prices.input * CACHE_READ_MULTIPLIER +
108
+ (cacheWriteTokens / MTOK) * prices.input * CACHE_WRITE_MULTIPLIER;
109
+ return +usd.toFixed(6);
45
110
  }
46
111
 
47
112
  function todayUtc() { return new Date().toISOString().slice(0, 10); }
@@ -59,38 +124,94 @@ function parseFlags(argv) {
59
124
  return flags;
60
125
  }
61
126
 
127
+ /** Read a numeric flag. Returns null when ABSENT — never coerces absence to 0. */
128
+ function numFlag(flags, name) {
129
+ const raw = flags[name];
130
+ if (raw == null || raw === "true") return null;
131
+ const n = Number(raw);
132
+ return Number.isFinite(n) ? n : null;
133
+ }
134
+
62
135
  function recordCmd(flags) {
63
136
  mkdirSync(LEDGER_DIR, { recursive: true });
137
+
138
+ // --tokens-unknown <reason> is the explicit "a session ran, we could not read
139
+ // its usage" marker. A caller that has no counts MUST pass it; the absence of
140
+ // token flags alone is also treated as unknown (below) so an un-migrated
141
+ // caller degrades loudly instead of writing a false $0.
142
+ const unknownFlag = flags["tokens-unknown"];
143
+ const inputTokens = numFlag(flags, "input-tokens");
144
+ const outputTokens = numFlag(flags, "output-tokens");
145
+ // --non-llm marks an attribution row for work that invoked no model at all
146
+ // (a message send). $0 is a fact there, not a measurement gap.
147
+ const nonLlm = flags["non-llm"] != null;
148
+
149
+ let measurement = "measured";
150
+ let unmeasuredReason = null;
151
+ if (nonLlm) {
152
+ measurement = "n/a";
153
+ } else if (unknownFlag != null || (inputTokens === null && outputTokens === null)) {
154
+ measurement = "unknown";
155
+ unmeasuredReason =
156
+ typeof unknownFlag === "string" && unknownFlag !== "true"
157
+ ? unknownFlag
158
+ : "caller-passed-no-token-counts";
159
+ }
160
+
64
161
  const row = {
65
162
  ts: new Date().toISOString(),
66
163
  cadence: flags.cadence || "unknown",
67
164
  source: flags.source || "cadence-consumer",
68
- model: (flags.model || "sonnet").toLowerCase(),
69
- duration_ms: Number(flags["duration-ms"] || 0),
70
- input_tokens: Number(flags["input-tokens"] || 0),
71
- output_tokens: Number(flags["output-tokens"] || 0),
165
+ model: measurement === "n/a" ? null : (flags.model || "sonnet").toLowerCase(),
166
+ duration_ms: numFlag(flags, "duration-ms") ?? 0,
167
+ // NULL, not 0, when unmeasured. This single distinction is what lets the
168
+ // budget governor tell "cheap day" from "blind day".
169
+ input_tokens: measurement === "measured" ? (inputTokens ?? 0) : (measurement === "n/a" ? 0 : null),
170
+ output_tokens: measurement === "measured" ? (outputTokens ?? 0) : (measurement === "n/a" ? 0 : null),
72
171
  exit_code: Number(flags.exit ?? 0),
172
+ measurement,
73
173
  };
74
- // Cache-read tokens (when the CLI reports them) are billed at a discount, but
75
- // we record them for observability without folding them into estimated_usd —
76
- // the static price table has no cache tier, and budget-guard sums
77
- // estimated_usd, so keep that field token-derived and stable.
78
- if (flags["cache-read-tokens"] != null) {
79
- row.cache_read_input_tokens = Number(flags["cache-read-tokens"] || 0);
80
- }
81
- const prices = loadPrices()[row.model] || DEFAULT_PRICES.sonnet;
82
- row.estimated_usd = +((row.input_tokens / 1000) * prices.input + (row.output_tokens / 1000) * prices.output).toFixed(6);
83
- // When the caller has the CLI-reported authoritative cost (total_cost_usd from
84
- // `claude --print --output-format json`), store it alongside the estimate for
85
- // reconciliation. We deliberately DO NOT use it as estimated_usd: budget-guard
86
- // sums estimated_usd, and the shared contract keeps that token-derived.
87
- if (flags["total-cost-usd"] != null) {
88
- const tc = Number(flags["total-cost-usd"]);
89
- if (Number.isFinite(tc) && tc >= 0) row.total_cost_usd = +tc.toFixed(6);
174
+ if (unmeasuredReason) row.unmeasured_reason = unmeasuredReason;
175
+ if (flags["decision-id"]) row.decision_id = String(flags["decision-id"]);
176
+
177
+ if (measurement === "measured") {
178
+ // Cache tokens are billed at their own rates (read 0.1x, write 1.25x) and
179
+ // dominate this workload — record them and PRICE them.
180
+ const cacheRead = numFlag(flags, "cache-read-tokens") ?? 0;
181
+ const cacheWrite = numFlag(flags, "cache-creation-tokens") ?? 0;
182
+ row.cache_read_input_tokens = cacheRead;
183
+ row.cache_creation_input_tokens = cacheWrite;
184
+ row.estimated_usd = estimateUsd(row.model, {
185
+ inputTokens: row.input_tokens,
186
+ outputTokens: row.output_tokens,
187
+ cacheReadTokens: cacheRead,
188
+ cacheWriteTokens: cacheWrite,
189
+ });
190
+ // The CLI's own figure. Readers bill against THIS when present: it accounts
191
+ // for cache tiers, resolved model and account pricing that no local table
192
+ // tracks. estimated_usd stays alongside it as the reconciliation fallback.
193
+ const tc = numFlag(flags, "total-cost-usd");
194
+ if (tc !== null && tc >= 0) row.total_cost_usd = +tc.toFixed(6);
195
+ } else if (measurement === "n/a") {
196
+ row.estimated_usd = 0;
197
+ row.total_cost_usd = 0;
198
+ } else {
199
+ // Unmeasured: no cost claim at all. Readers must not read these as free.
200
+ row.estimated_usd = null;
201
+ row.total_cost_usd = null;
90
202
  }
203
+
91
204
  const file = join(LEDGER_DIR, `${todayUtc()}.jsonl`);
92
205
  appendFileSync(file, JSON.stringify(row) + "\n");
93
- process.stdout.write(JSON.stringify({ ok: true, file, estimated_usd: row.estimated_usd }) + "\n");
206
+ const { usd, basis } = billableUsd(row);
207
+ if (measurement === "unknown") {
208
+ // Loud on stderr: a caller that cannot measure its own session is a defect,
209
+ // and this is the only place that observes it at write time.
210
+ process.stderr.write(
211
+ `[track-claude-usage] UNMEASURED session recorded (source=${row.source} cadence=${row.cadence} reason=${unmeasuredReason}) — spend for this session is unknown, not zero\n`
212
+ );
213
+ }
214
+ process.stdout.write(JSON.stringify({ ok: true, file, measurement, billable_usd: usd, basis }) + "\n");
94
215
  }
95
216
 
96
217
  function summariseCmd(flags) {
@@ -103,6 +224,7 @@ function summariseCmd(flags) {
103
224
  const totals = empty();
104
225
  const byModel = {};
105
226
  const byCadence = {};
227
+ const windowRows = [];
106
228
  for (const name of readdirSync(LEDGER_DIR)) {
107
229
  if (!name.endsWith(".jsonl")) continue;
108
230
  let body;
@@ -112,26 +234,63 @@ function summariseCmd(flags) {
112
234
  let row;
113
235
  try { row = JSON.parse(line); } catch { continue; }
114
236
  if (new Date(row.ts).getTime() < cutoff) continue;
115
- totals.sessions += 1;
116
- totals.input_tokens += row.input_tokens || 0;
117
- totals.output_tokens += row.output_tokens || 0;
118
- totals.estimated_usd = +(totals.estimated_usd + (row.estimated_usd || 0)).toFixed(6);
119
- byModel[row.model] = byModel[row.model] || empty();
120
- byModel[row.model].sessions += 1;
121
- byModel[row.model].input_tokens += row.input_tokens || 0;
122
- byModel[row.model].output_tokens += row.output_tokens || 0;
123
- byModel[row.model].estimated_usd = +(byModel[row.model].estimated_usd + (row.estimated_usd || 0)).toFixed(6);
124
- byCadence[row.cadence] = byCadence[row.cadence] || empty();
125
- byCadence[row.cadence].sessions += 1;
126
- byCadence[row.cadence].input_tokens += row.input_tokens || 0;
127
- byCadence[row.cadence].output_tokens += row.output_tokens || 0;
128
- byCadence[row.cadence].estimated_usd = +(byCadence[row.cadence].estimated_usd + (row.estimated_usd || 0)).toFixed(6);
237
+ windowRows.push(row);
238
+ const { usd, basis } = billableUsd(row);
239
+ // Non-LLM attribution rows are not sessions; unmeasured ones are sessions
240
+ // whose cost we do not know. Neither may be folded into a spend total as
241
+ // though it were a measured $0.
242
+ if (basis === "non_llm" || basis === "invalid") continue;
243
+ const key = row.model || "unknown";
244
+ const cad = row.cadence || "unknown";
245
+ byModel[key] = byModel[key] || empty();
246
+ byCadence[cad] = byCadence[cad] || empty();
247
+ for (const bucket of [totals, byModel[key], byCadence[cad]]) {
248
+ bucket.sessions += 1;
249
+ if (basis === "unmeasured") { bucket.unmeasured_sessions += 1; continue; }
250
+ bucket.input_tokens += row.input_tokens || 0;
251
+ bucket.output_tokens += row.output_tokens || 0;
252
+ bucket.cache_read_tokens += row.cache_read_input_tokens || row.cache_read_tokens || 0;
253
+ bucket.billable_usd = +(bucket.billable_usd + usd).toFixed(6);
254
+ if (basis === "estimated") bucket.estimated_fallback_sessions += 1;
255
+ }
129
256
  }
130
257
  }
131
- writeSummary({ window_days: days, totals, by_model: byModel, by_cadence: byCadence });
258
+ // Keep the legacy field name populated so existing dashboards/readers don't
259
+ // silently read undefined — but it now carries the BILLABLE figure (CLI
260
+ // authoritative where available), not the cache-blind token estimate.
261
+ for (const bucket of [totals, ...Object.values(byModel), ...Object.values(byCadence)]) {
262
+ bucket.estimated_usd = bucket.billable_usd;
263
+ }
264
+ const audit = summariseRows(windowRows);
265
+ writeSummary({
266
+ window_days: days,
267
+ totals,
268
+ by_model: byModel,
269
+ by_cadence: byCadence,
270
+ measurement: {
271
+ measured_sessions: audit.measured,
272
+ unmeasured_sessions: audit.unmeasured,
273
+ non_llm_rows: audit.nonLlmRows,
274
+ authoritative_usd: audit.authoritativeUsd,
275
+ estimated_fallback_usd: audit.estimatedFallbackUsd,
276
+ blind: audit.blind,
277
+ degradations: audit.degradations,
278
+ },
279
+ });
132
280
  }
133
281
 
134
- function empty() { return { sessions: 0, input_tokens: 0, output_tokens: 0, estimated_usd: 0 }; }
282
+ function empty() {
283
+ return {
284
+ sessions: 0,
285
+ unmeasured_sessions: 0,
286
+ estimated_fallback_sessions: 0,
287
+ input_tokens: 0,
288
+ output_tokens: 0,
289
+ cache_read_tokens: 0,
290
+ billable_usd: 0,
291
+ estimated_usd: 0,
292
+ };
293
+ }
135
294
 
136
295
  function writeSummary(payload) {
137
296
  const out = {