@cohortapp/agent-sdk 2.5.1 → 2.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/maestro.mjs +305 -89
- package/bin/maestro.test.mjs +357 -48
- package/docs/runbooks/backup-restore.md +65 -33
- package/framework-features.json +4 -4
- package/lib/backup/policy.mjs +710 -0
- package/lib/backup/policy.test.mjs +305 -0
- package/lib/budget-escalate.mjs +133 -0
- package/lib/budget-escalate.test.mjs +232 -0
- package/lib/budget-guard.envelope.test.mjs +476 -0
- package/lib/budget-guard.mjs +853 -75
- package/lib/budget-guard.test.mjs +91 -42
- package/lib/cadences.mjs +33 -0
- package/lib/channels/orgmail/adapter.mjs +88 -3
- package/lib/channels/orgmail/adapter.test.mjs +137 -0
- package/lib/channels/repeat-suppressor.mjs +198 -0
- package/lib/channels/repeat-suppressor.test.mjs +134 -0
- package/lib/comms/receipts.mjs +297 -0
- package/lib/cost/ledger-row.mjs +333 -0
- package/lib/cost/ledger-row.test.mjs +183 -0
- package/lib/execution/drive.mjs +28 -1
- package/lib/execution/effects.mjs +191 -12
- package/lib/execution/effects.test.mjs +50 -11
- package/lib/goals/admission.mjs +13 -1
- package/lib/goals/admission.test.mjs +26 -1
- package/lib/goals/loop.mjs +13 -0
- package/lib/kpi-sensors.test.mjs +3 -0
- package/lib/mandate/cache.mjs +13 -5
- package/lib/mandate/derive.mjs +146 -21
- package/lib/mandate/derive.test.mjs +50 -6
- package/lib/mandate/model.mjs +32 -4
- package/lib/mandate/refresh.test.mjs +16 -2
- package/lib/mcp/server.test.mjs +12 -3
- package/lib/model-router/economics.mjs +107 -76
- package/lib/model-router/economics.test.mjs +64 -46
- package/lib/model-router/integration-coverage.test.mjs +39 -37
- package/lib/model-router/ledger.mjs +75 -22
- package/lib/model-router/ledger.test.mjs +35 -2
- package/lib/org/client.mjs +14 -0
- package/lib/org/cost-sync.mjs +16 -2
- package/lib/org/doctor.mjs +62 -1
- package/lib/org/doctor.test.mjs +36 -3
- package/lib/org/email-remedy.mjs +49 -0
- package/lib/org/engagement-ledger.mjs +376 -0
- package/lib/org/engagement-ledger.test.mjs +112 -0
- package/lib/org/engagement.mjs +1056 -0
- package/lib/org/engagement.test.mjs +739 -0
- package/lib/org/messaging.mjs +230 -3
- package/lib/org/messaging.test.mjs +110 -1
- package/lib/org/param-contract.mjs +56 -2
- package/lib/org/param-contract.test.mjs +26 -0
- package/lib/org/protocol.checksum +1 -1
- package/lib/org/protocol.mjs +5 -0
- package/lib/org/protocol.test.mjs +7 -1
- package/lib/org/tool-surface.mjs +506 -10
- package/lib/org/tool-surface.test.mjs +191 -7
- package/lib/org/ui-parity.mjs +333 -6
- package/lib/org/ui-parity.test.mjs +96 -3
- package/lib/org/work-ledger.mjs +241 -0
- package/lib/org/work-ledger.test.mjs +237 -0
- package/lib/plan/adoption-e2e.test.mjs +366 -0
- package/lib/plan/budget-enforcement.test.mjs +400 -0
- package/lib/plan/budget-runtime.mjs +215 -0
- package/lib/plan/compile.mjs +201 -5
- package/lib/plan/compile.test.mjs +19 -5
- package/lib/plan/emit.mjs +8 -0
- package/lib/plan/emit.test.mjs +18 -0
- package/lib/resource-governor.mjs +58 -12
- package/lib/resource-governor.test.mjs +41 -1
- package/lib/security/audit-engine.mjs +45 -8
- package/lib/security/audit-engine.test.mjs +35 -0
- package/lib/setup/enroll-from-cohort.mjs +14 -1
- package/lib/setup/sections/mandate.mjs +48 -7
- package/lib/setup/sections/mandate.test.mjs +17 -2
- package/lib/setup/sections/orgmail.mjs +10 -2
- package/lib/setup/state.mjs +83 -2
- package/lib/telemetry/collect.mjs +360 -20
- package/lib/telemetry/collect.test.mjs +266 -0
- package/package.json +1 -1
- package/scripts/cost/track-claude-usage.mjs +207 -48
- package/scripts/cost/track-claude-usage.test.mjs +148 -0
- package/scripts/daemon/agent-daemon.mjs +315 -17
- package/scripts/daemon/assurance-e2e.test.mjs +421 -0
- package/scripts/daemon/assurance.mjs +944 -0
- package/scripts/daemon/assurance.test.mjs +668 -0
- package/scripts/daemon/cadence-consumer-governance.test.mjs +56 -0
- package/scripts/daemon/cadence-consumer.mjs +147 -9
- package/scripts/daemon/cadence-consumer.test.mjs +6 -0
- package/scripts/daemon/cadence-handlers.mjs +158 -0
- package/scripts/daemon/cadence-handlers.test.mjs +64 -0
- package/scripts/daemon/classifier.test.mjs +18 -9
- package/scripts/daemon/deliver.mjs +314 -0
- package/scripts/daemon/dispatcher-governance.test.mjs +10 -0
- package/scripts/daemon/dispatcher.mjs +64 -6
- package/scripts/daemon/responder-cost.test.mjs +68 -0
- package/scripts/daemon/responder.mjs +351 -298
- package/scripts/local-triggers/generate-plists.test.mjs +7 -4
- package/scripts/maintenance/backup-run.mjs +415 -0
- package/scripts/maintenance/backup-to-cloud.sh +16 -116
- package/scripts/org/send-orgmail.mjs +16 -0
- package/scripts/record-receipt.sh +63 -0
- package/scripts/restore-from-backup.sh +14 -3
- package/scripts/restore-from-backup.test.mjs +8 -5
- package/scripts/send-email-threaded.py +47 -0
- package/scripts/send-sms.sh +4 -0
- package/scripts/send-whatsapp.sh +4 -0
- package/scripts/setup/init-backup.mjs +93 -38
- package/scripts/slack-send.sh +12 -0
|
@@ -0,0 +1,333 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* lib/cost/ledger-row.mjs — the ONE definition of "what did this ledger row
|
|
3
|
+
* cost, and do we actually know?"
|
|
4
|
+
*
|
|
5
|
+
* WHY THIS MODULE EXISTS
|
|
6
|
+
* ----------------------
|
|
7
|
+
* `state/cost-tracking/<date>.jsonl` has two writers with two different row
|
|
8
|
+
* shapes:
|
|
9
|
+
*
|
|
10
|
+
* v1 — scripts/cost/track-claude-usage.mjs `record` (dispatcher +
|
|
11
|
+
* cadence-consumer + responder: 100% of measured LLM spend today)
|
|
12
|
+
* v2 — lib/model-router/ledger.mjs `writeLedgerRow` (llm-task + messaging
|
|
13
|
+
* attribution)
|
|
14
|
+
*
|
|
15
|
+
* Every reader (budget-guard, doctor's cost tripwire, `maestro cost report`,
|
|
16
|
+
* fleet-digest) used to sum `estimated_usd` and count every line as a session.
|
|
17
|
+
* That was wrong three ways, and each way made the budget governor blinder:
|
|
18
|
+
*
|
|
19
|
+
* 1. `estimated_usd` on a v1 row was priced from a token table WITH NO CACHE
|
|
20
|
+
* TIER. A real dispatcher row on 2026-08-12 reads input_tokens:16,
|
|
21
|
+
* cache_read_input_tokens:553181, total_cost_usd:$0.533 — and priced out
|
|
22
|
+
* at $0.075. Cache reads are ~90% of every prompt this agent sends, so the
|
|
23
|
+
* day's spend read 2.6x-6.1x low ($90.79 vs $235.62 on 2026-08-11).
|
|
24
|
+
* 2. A row whose tokens were never measured (the tracker defaulted an absent
|
|
25
|
+
* --input-tokens flag to 0) is INDISTINGUISHABLE from a genuine zero-token
|
|
26
|
+
* row. A zero that means "unmeasured" is exactly what produces a governor
|
|
27
|
+
* that can never trip.
|
|
28
|
+
* 3. Zero-LLM attribution rows (a message send: no model, no tokens, $0 by
|
|
29
|
+
* construction) were counted as sessions, which armed doctor's "sessions
|
|
30
|
+
* ran but spend is $0" tripwire on days when nothing but sends happened.
|
|
31
|
+
*
|
|
32
|
+
* So: classification and billing live HERE, once, and every reader imports it.
|
|
33
|
+
* Zero dependencies beyond the language — budget-guard is on the spawn hot path
|
|
34
|
+
* and its contract is Node built-ins only, never throws.
|
|
35
|
+
*
|
|
36
|
+
* ROW CONTRACT (forward-looking fields; legacy rows are inferred, see below)
|
|
37
|
+
* measurement: "measured" | "unknown" | "n/a"
|
|
38
|
+
* measured — input_tokens/output_tokens are real counts lifted from a
|
|
39
|
+
* `claude --output-format json` envelope.
|
|
40
|
+
* unknown — a session ran, we could not read its usage. Tokens are null,
|
|
41
|
+
* `unmeasured_reason` says why. NEVER write 0 here.
|
|
42
|
+
* n/a — no model was invoked (message-send attribution). $0 is a fact,
|
|
43
|
+
* not an absence.
|
|
44
|
+
* total_cost_usd — the CLI's authoritative figure. Preferred over any estimate.
|
|
45
|
+
* estimated_usd — token-derived fallback. Kept on every row for compatibility.
|
|
46
|
+
*
|
|
47
|
+
* @module lib/cost/ledger-row
|
|
48
|
+
*/
|
|
49
|
+
|
|
50
|
+
/** @typedef {"measured"|"unknown"|"n/a"|"invalid"} RowClass */
|
|
51
|
+
|
|
52
|
+
export const MEASURED = "measured";
|
|
53
|
+
export const UNKNOWN = "unknown";
|
|
54
|
+
export const NON_LLM = "n/a";
|
|
55
|
+
export const INVALID = "invalid";
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Numeric coercion that treats ABSENCE as absence.
|
|
59
|
+
*
|
|
60
|
+
* `Number(null)` and `Number("")` are both 0, so a naive coercion would turn a
|
|
61
|
+
* deliberately-null `input_tokens` back into a measured zero — re-creating the
|
|
62
|
+
* exact "unmeasured looks free" bug this module exists to prevent. Guard first.
|
|
63
|
+
*/
|
|
64
|
+
function num(v) {
|
|
65
|
+
if (v === null || v === undefined || v === "") return null;
|
|
66
|
+
const n = Number(v);
|
|
67
|
+
return Number.isFinite(n) ? n : null;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* Classify a ledger row. Explicit `measurement` wins; otherwise we infer from
|
|
72
|
+
* the row's shape so the ~200 rows already on disk classify correctly without a
|
|
73
|
+
* migration.
|
|
74
|
+
*
|
|
75
|
+
* Legacy inference:
|
|
76
|
+
* - null/absent tokens on BOTH sides → unknown (the tracker only ever wrote
|
|
77
|
+
* literal 0s, so this shape can only come from the new writer; harmless).
|
|
78
|
+
* - no model AND zero tokens on both sides → n/a. This is precisely the
|
|
79
|
+
* v2 messaging attribution row (`model:null, input_tokens:0`) and cannot
|
|
80
|
+
* match a real session: every spawn path stamps a model.
|
|
81
|
+
* - anything else → measured.
|
|
82
|
+
*
|
|
83
|
+
* @param {object} row
|
|
84
|
+
* @returns {RowClass}
|
|
85
|
+
*/
|
|
86
|
+
export function classifyRow(row) {
|
|
87
|
+
if (!row || typeof row !== "object") return INVALID;
|
|
88
|
+
|
|
89
|
+
const m = row.measurement;
|
|
90
|
+
if (m === MEASURED || m === UNKNOWN || m === NON_LLM) return m;
|
|
91
|
+
|
|
92
|
+
const inTok = num(row.input_tokens);
|
|
93
|
+
const outTok = num(row.output_tokens);
|
|
94
|
+
const hasModel = typeof row.model === "string" && row.model.length > 0;
|
|
95
|
+
|
|
96
|
+
// Checked BEFORE the cost test: a message-send row carries a real 0 cost, and
|
|
97
|
+
// we must not read that as "priced, therefore measured".
|
|
98
|
+
if (!hasModel && (inTok || 0) === 0 && (outTok || 0) === 0) return NON_LLM;
|
|
99
|
+
|
|
100
|
+
// A row that carries a NON-ZERO COST is measured, whatever its token fields
|
|
101
|
+
// look like. Cost is what readers bill; tokens are diagnostics. (This also
|
|
102
|
+
// keeps older cost-only rows billable instead of reclassifying history as
|
|
103
|
+
// blind.) A cost of exactly 0 on a row that NAMES A MODEL is not evidence of
|
|
104
|
+
// a free session — it is the writer's default standing in for "unpriceable",
|
|
105
|
+
// so it falls through to the token test below.
|
|
106
|
+
const cost = num(row.total_cost_usd) ?? num(row.estimated_usd);
|
|
107
|
+
if (cost !== null && cost > 0) return MEASURED;
|
|
108
|
+
|
|
109
|
+
// No usable cost. A model ran, so it generated tokens; zero or absent counts
|
|
110
|
+
// on BOTH sides therefore mean we failed to measure it, not that it was free.
|
|
111
|
+
// This is the same "a zero that means unmeasured" defect one layer down, and
|
|
112
|
+
// it is why an unpriced or catalog-missing model must never bill as $0.
|
|
113
|
+
if ((inTok || 0) === 0 && (outTok || 0) === 0) return UNKNOWN;
|
|
114
|
+
|
|
115
|
+
return MEASURED;
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/**
|
|
119
|
+
* What did this row cost, and on what authority?
|
|
120
|
+
*
|
|
121
|
+
* Precedence — authoritative > estimated > unknown. `total_cost_usd` is the
|
|
122
|
+
* figure the Claude CLI reports for the session; it already accounts for cache
|
|
123
|
+
* read/write tiers, model class and any account-level pricing, none of which a
|
|
124
|
+
* local price table can track. We fall back to `estimated_usd` only when the
|
|
125
|
+
* authoritative number is absent, and we SAY SO in `basis` so the caller can
|
|
126
|
+
* log the degradation rather than silently under-count.
|
|
127
|
+
*
|
|
128
|
+
* @param {object} row
|
|
129
|
+
* @returns {{usd:number|null, basis:"authoritative"|"estimated"|"unmeasured"|"non_llm"|"invalid", cls:RowClass}}
|
|
130
|
+
*/
|
|
131
|
+
export function billableUsd(row) {
|
|
132
|
+
const cls = classifyRow(row);
|
|
133
|
+
if (cls === INVALID) return { usd: null, basis: "invalid", cls };
|
|
134
|
+
// A message send costs $0 because no model ran. That is a measurement, not a
|
|
135
|
+
// gap — it must not arm any blindness alarm.
|
|
136
|
+
if (cls === NON_LLM) return { usd: 0, basis: "non_llm", cls };
|
|
137
|
+
// A session we could not measure has an UNKNOWN cost. Returning 0 here is the
|
|
138
|
+
// original bug; the caller has to decide what to do with null (budget-guard
|
|
139
|
+
// imputes and logs; doctor goes red).
|
|
140
|
+
if (cls === UNKNOWN) return { usd: null, basis: "unmeasured", cls };
|
|
141
|
+
|
|
142
|
+
// `> 0`, not `>= 0`: a zero price on a row that named a model is the writer's
|
|
143
|
+
// "I could not price this" default (an absent catalog row, an unreadable usage
|
|
144
|
+
// envelope), and billing it as a free session is exactly how a pricing outage
|
|
145
|
+
// silences the governor. Fall through to UNKNOWN and let the caller impute.
|
|
146
|
+
const authoritative = num(row.total_cost_usd);
|
|
147
|
+
if (authoritative !== null && authoritative > 0) {
|
|
148
|
+
return { usd: authoritative, basis: "authoritative", cls };
|
|
149
|
+
}
|
|
150
|
+
const estimated = num(row.estimated_usd);
|
|
151
|
+
if (estimated !== null && estimated > 0) {
|
|
152
|
+
return { usd: estimated, basis: "estimated", cls };
|
|
153
|
+
}
|
|
154
|
+
// The row HAS usage but nothing could price it (no authoritative figure, and
|
|
155
|
+
// the catalog had no row for the model). Distinct from "unmeasured": the
|
|
156
|
+
// telemetry worked and the PRICING did not, which is a different bug with a
|
|
157
|
+
// different fix, and doctor says so. Either way it is not free.
|
|
158
|
+
return { usd: null, basis: "unpriced", cls: UNKNOWN };
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
function round6(n) {
|
|
162
|
+
return Number(n.toFixed(6));
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
/**
|
|
166
|
+
* Fold a day's rows into the numbers every reader needs, including the ones
|
|
167
|
+
* that used to be invisible.
|
|
168
|
+
*
|
|
169
|
+
* `sessions` counts LLM sessions only (measured + unmeasured) — non-LLM
|
|
170
|
+
* attribution rows are reported separately as `nonLlmRows`. Anything that
|
|
171
|
+
* degraded the number is listed in `degradations[]`: fail-open is fine, silent
|
|
172
|
+
* is not.
|
|
173
|
+
*
|
|
174
|
+
* @param {Array<object>} rows
|
|
175
|
+
* @returns {{
|
|
176
|
+
* sessions:number, measured:number, unmeasured:number, nonLlmRows:number,
|
|
177
|
+
* invalidRows:number, measuredUsd:number, authoritativeUsd:number,
|
|
178
|
+
* estimatedFallbackUsd:number, estimatedFallbackRows:number,
|
|
179
|
+
* blind:boolean, degradations:string[]
|
|
180
|
+
* }}
|
|
181
|
+
*/
|
|
182
|
+
export function summariseRows(rows) {
|
|
183
|
+
let measured = 0;
|
|
184
|
+
let unmeasured = 0;
|
|
185
|
+
let nonLlmRows = 0;
|
|
186
|
+
let invalidRows = 0;
|
|
187
|
+
let measuredUsd = 0;
|
|
188
|
+
let authoritativeUsd = 0;
|
|
189
|
+
let estimatedFallbackUsd = 0;
|
|
190
|
+
let estimatedFallbackRows = 0;
|
|
191
|
+
// `unmeasured` is the imputation base; these two say WHICH failure produced it.
|
|
192
|
+
let noTokenRows = 0; // a session ran and nobody read its usage
|
|
193
|
+
let unpricedRows = 0; // usage was read but nothing could price it
|
|
194
|
+
|
|
195
|
+
for (const row of rows || []) {
|
|
196
|
+
const { usd, basis } = billableUsd(row);
|
|
197
|
+
switch (basis) {
|
|
198
|
+
case "authoritative":
|
|
199
|
+
measured++;
|
|
200
|
+
measuredUsd += usd;
|
|
201
|
+
authoritativeUsd += usd;
|
|
202
|
+
break;
|
|
203
|
+
case "estimated":
|
|
204
|
+
measured++;
|
|
205
|
+
measuredUsd += usd;
|
|
206
|
+
estimatedFallbackUsd += usd;
|
|
207
|
+
estimatedFallbackRows++;
|
|
208
|
+
break;
|
|
209
|
+
case "unmeasured":
|
|
210
|
+
unmeasured++;
|
|
211
|
+
noTokenRows++;
|
|
212
|
+
break;
|
|
213
|
+
case "unpriced":
|
|
214
|
+
unmeasured++;
|
|
215
|
+
unpricedRows++;
|
|
216
|
+
break;
|
|
217
|
+
case "non_llm":
|
|
218
|
+
nonLlmRows++;
|
|
219
|
+
break;
|
|
220
|
+
default:
|
|
221
|
+
invalidRows++;
|
|
222
|
+
}
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
const degradations = [];
|
|
226
|
+
if (noTokenRows > 0) {
|
|
227
|
+
degradations.push(
|
|
228
|
+
`${noTokenRows} session row(s) recorded NO token counts (measurement:"unknown") — their spend is unknown, not zero`
|
|
229
|
+
);
|
|
230
|
+
}
|
|
231
|
+
if (unpricedRows > 0) {
|
|
232
|
+
degradations.push(
|
|
233
|
+
`${unpricedRows} session row(s) carry usage but NO price (no total_cost_usd, and the model is not in the catalog) — pricing is broken, and the rows are not free`
|
|
234
|
+
);
|
|
235
|
+
}
|
|
236
|
+
if (estimatedFallbackRows > 0) {
|
|
237
|
+
degradations.push(
|
|
238
|
+
`${estimatedFallbackRows} row(s) had no authoritative total_cost_usd; fell back to the local token estimate ($${round6(estimatedFallbackUsd)}), which under-prices cache reads`
|
|
239
|
+
);
|
|
240
|
+
}
|
|
241
|
+
if (invalidRows > 0) {
|
|
242
|
+
degradations.push(`${invalidRows} malformed ledger row(s) skipped`);
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
return {
|
|
246
|
+
sessions: measured + unmeasured,
|
|
247
|
+
measured,
|
|
248
|
+
unmeasured,
|
|
249
|
+
nonLlmRows,
|
|
250
|
+
invalidRows,
|
|
251
|
+
noTokenRows,
|
|
252
|
+
unpricedRows,
|
|
253
|
+
measuredUsd: round6(measuredUsd),
|
|
254
|
+
authoritativeUsd: round6(authoritativeUsd),
|
|
255
|
+
estimatedFallbackUsd: round6(estimatedFallbackUsd),
|
|
256
|
+
estimatedFallbackRows,
|
|
257
|
+
// Blind = sessions ran and we measured NONE of them. This is the precise
|
|
258
|
+
// condition doctor's tripwire is supposed to detect. An empty ledger is not
|
|
259
|
+
// blind — it is quiet.
|
|
260
|
+
blind: measured === 0 && unmeasured > 0,
|
|
261
|
+
degradations,
|
|
262
|
+
};
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
/**
|
|
266
|
+
* Impute a cost for the sessions we failed to measure, so a systematic parse
|
|
267
|
+
* regression cannot starve the budget governor into never tripping.
|
|
268
|
+
*
|
|
269
|
+
* The alternative — counting unmeasured sessions as $0 — is the exact failure
|
|
270
|
+
* this module exists to kill: a governor that goes quiet precisely when
|
|
271
|
+
* telemetry breaks.
|
|
272
|
+
*
|
|
273
|
+
* WHAT CHANGED, AND WHY THERE IS NO LONGER A MAGIC NUMBER
|
|
274
|
+
* ------------------------------------------------------
|
|
275
|
+
* This used to fall back to a hard-coded `floorUsd = 0.25` per session when the
|
|
276
|
+
* day had nothing measured. That number was sourced from nothing, and it was
|
|
277
|
+
* ~8x BELOW this fleet's observed mean ($1.95/session on 2026-08-12), so it
|
|
278
|
+
* defeated the property the docstring claimed: a total telemetry outage on a day
|
|
279
|
+
* that really cost $68.40 (274 % of a $25 cap → the refuse rung) imputed $8.75
|
|
280
|
+
* and reported band 0, mode normal. Under-imputing is not conservative; it is
|
|
281
|
+
* the silent-governor bug wearing a different hat.
|
|
282
|
+
*
|
|
283
|
+
* So the ladder is now evidence-only, and it ADMITS when it has none:
|
|
284
|
+
* 1. `day_mean` — this day's own mean measured session cost. Best evidence.
|
|
285
|
+
* 2. `period_mean`— the caller's `fallbackPerSessionUsd` (budget-guard passes
|
|
286
|
+
* the MONTH's mean measured cost, which survives a day that
|
|
287
|
+
* is blind end to end).
|
|
288
|
+
* 3. `unpriceable`— no evidence at ANY horizon. We return $0 and set
|
|
289
|
+
* `unpriceable:true` rather than invent a price. The caller
|
|
290
|
+
* must not read that 0 as "cheap": `summariseRows().blind`
|
|
291
|
+
* is the signal, and `lib/budget-guard.dailyStatus` floors
|
|
292
|
+
* the enforcement band at DEGRADE when it is set. "We cannot
|
|
293
|
+
* measure, therefore we degrade" needs no price table.
|
|
294
|
+
*
|
|
295
|
+
* Every imputation is reported by the caller; none of it is silent.
|
|
296
|
+
*
|
|
297
|
+
* @param {ReturnType<typeof summariseRows>} summary
|
|
298
|
+
* @param {{fallbackPerSessionUsd?:number}|number} [opts] month/period mean, when
|
|
299
|
+
* the day itself measured nothing. A bare number is accepted for
|
|
300
|
+
* backward compatibility with the old positional `floorUsd`.
|
|
301
|
+
* @returns {{imputedUsd:number, perSessionUsd:number,
|
|
302
|
+
* basis:"day_mean"|"period_mean"|"unpriceable"|"none",
|
|
303
|
+
* unpriceable:boolean}}
|
|
304
|
+
*/
|
|
305
|
+
export function imputeUnmeasured(summary, opts = {}) {
|
|
306
|
+
if (!summary || summary.unmeasured <= 0) {
|
|
307
|
+
return { imputedUsd: 0, perSessionUsd: 0, basis: "none", unpriceable: false };
|
|
308
|
+
}
|
|
309
|
+
const fallback = typeof opts === "number"
|
|
310
|
+
? opts
|
|
311
|
+
: num(opts && opts.fallbackPerSessionUsd);
|
|
312
|
+
|
|
313
|
+
const dayMean = summary.measured > 0 ? summary.measuredUsd / summary.measured : null;
|
|
314
|
+
if (dayMean !== null && dayMean > 0) {
|
|
315
|
+
return {
|
|
316
|
+
imputedUsd: round6(dayMean * summary.unmeasured),
|
|
317
|
+
perSessionUsd: round6(dayMean),
|
|
318
|
+
basis: "day_mean",
|
|
319
|
+
unpriceable: false,
|
|
320
|
+
};
|
|
321
|
+
}
|
|
322
|
+
if (fallback !== null && fallback > 0) {
|
|
323
|
+
return {
|
|
324
|
+
imputedUsd: round6(fallback * summary.unmeasured),
|
|
325
|
+
perSessionUsd: round6(fallback),
|
|
326
|
+
basis: "period_mean",
|
|
327
|
+
unpriceable: false,
|
|
328
|
+
};
|
|
329
|
+
}
|
|
330
|
+
return { imputedUsd: 0, perSessionUsd: 0, basis: "unpriceable", unpriceable: true };
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
export default { classifyRow, billableUsd, summariseRows, imputeUnmeasured, MEASURED, UNKNOWN, NON_LLM, INVALID };
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* lib/cost/ledger-row.test.mjs — the measured / unmeasured / non-LLM contract.
|
|
3
|
+
*
|
|
4
|
+
* The bug these lock down: an absent token count and a genuine zero used to
|
|
5
|
+
* produce identical rows, and every reader summed both as $0. That is what made
|
|
6
|
+
* the budget governor unable to trip.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import test from "node:test";
|
|
10
|
+
import assert from "node:assert/strict";
|
|
11
|
+
|
|
12
|
+
import {
|
|
13
|
+
classifyRow,
|
|
14
|
+
billableUsd,
|
|
15
|
+
summariseRows,
|
|
16
|
+
imputeUnmeasured,
|
|
17
|
+
MEASURED,
|
|
18
|
+
UNKNOWN,
|
|
19
|
+
NON_LLM,
|
|
20
|
+
INVALID,
|
|
21
|
+
} from "./ledger-row.mjs";
|
|
22
|
+
|
|
23
|
+
// A real dispatcher row copied from the live ledger (2026-08-12). Note the
|
|
24
|
+
// shape that broke the old estimator: 16 uncached input tokens against 553k
|
|
25
|
+
// cache reads, and an authoritative cost 7x the token-table estimate.
|
|
26
|
+
const REAL_ROW = {
|
|
27
|
+
ts: "2026-08-12T00:09:36.073Z", cadence: "inbox", source: "dispatcher",
|
|
28
|
+
model: "sonnet", duration_ms: 64728, input_tokens: 16, output_tokens: 5000,
|
|
29
|
+
exit_code: 0, cache_read_input_tokens: 553181,
|
|
30
|
+
estimated_usd: 0.075048, total_cost_usd: 0.533412,
|
|
31
|
+
};
|
|
32
|
+
|
|
33
|
+
test("classifyRow: explicit measurement wins", () => {
|
|
34
|
+
assert.equal(classifyRow({ measurement: "measured", input_tokens: 1 }), MEASURED);
|
|
35
|
+
assert.equal(classifyRow({ measurement: "unknown", input_tokens: null }), UNKNOWN);
|
|
36
|
+
assert.equal(classifyRow({ measurement: "n/a" }), NON_LLM);
|
|
37
|
+
assert.equal(classifyRow(null), INVALID);
|
|
38
|
+
assert.equal(classifyRow("nope"), INVALID);
|
|
39
|
+
});
|
|
40
|
+
|
|
41
|
+
test("classifyRow: legacy rows are inferred without a migration", () => {
|
|
42
|
+
// Real dispatcher row, no `measurement` field.
|
|
43
|
+
assert.equal(classifyRow(REAL_ROW), MEASURED);
|
|
44
|
+
// Real messaging attribution row: no model, zero tokens => not a session.
|
|
45
|
+
assert.equal(
|
|
46
|
+
classifyRow({ source: "messaging", model: null, input_tokens: 0, output_tokens: 0, estimated_usd: 0 }),
|
|
47
|
+
NON_LLM
|
|
48
|
+
);
|
|
49
|
+
// Null tokens on both sides => unmeasured.
|
|
50
|
+
assert.equal(classifyRow({ model: "sonnet", input_tokens: null, output_tokens: null }), UNKNOWN);
|
|
51
|
+
});
|
|
52
|
+
|
|
53
|
+
test("classifyRow: a real session that used zero output tokens is still MEASURED", () => {
|
|
54
|
+
// The distinction that matters: a model ran and produced nothing is a
|
|
55
|
+
// measurement; it must not be confused with "we never looked".
|
|
56
|
+
assert.equal(classifyRow({ model: "sonnet", input_tokens: 120, output_tokens: 0 }), MEASURED);
|
|
57
|
+
});
|
|
58
|
+
|
|
59
|
+
test("billableUsd: authoritative CLI cost beats the local estimate", () => {
|
|
60
|
+
const { usd, basis } = billableUsd(REAL_ROW);
|
|
61
|
+
assert.equal(basis, "authoritative");
|
|
62
|
+
assert.equal(usd, 0.533412);
|
|
63
|
+
// The whole point: the old path would have billed the estimate.
|
|
64
|
+
assert.notEqual(usd, REAL_ROW.estimated_usd);
|
|
65
|
+
});
|
|
66
|
+
|
|
67
|
+
test("billableUsd: falls back to the estimate and says so", () => {
|
|
68
|
+
const { usd, basis } = billableUsd({ ...REAL_ROW, total_cost_usd: undefined });
|
|
69
|
+
assert.equal(basis, "estimated");
|
|
70
|
+
assert.equal(usd, 0.075048);
|
|
71
|
+
});
|
|
72
|
+
|
|
73
|
+
test("billableUsd: an unmeasured session costs NULL, not zero", () => {
|
|
74
|
+
const { usd, basis } = billableUsd({ measurement: "unknown", input_tokens: null, output_tokens: null });
|
|
75
|
+
assert.equal(usd, null);
|
|
76
|
+
assert.equal(basis, "unmeasured");
|
|
77
|
+
});
|
|
78
|
+
|
|
79
|
+
test("billableUsd: usage with no price is UNPRICED — a distinct bug from unmeasured, and still not free", () => {
|
|
80
|
+
const { usd, basis, cls } = billableUsd({ model: "sonnet", input_tokens: 5, output_tokens: 5 });
|
|
81
|
+
assert.equal(usd, null, "never 0 — a pricing outage must not read as a free session");
|
|
82
|
+
assert.equal(basis, "unpriced", "the telemetry worked; the PRICING did not");
|
|
83
|
+
assert.equal(cls, UNKNOWN);
|
|
84
|
+
});
|
|
85
|
+
|
|
86
|
+
test("billableUsd: a $0 price on a row that named a model is unpriceable, not free", () => {
|
|
87
|
+
// lib/model-router/ledger.mjs used to default `estimated_usd` to 0 whenever
|
|
88
|
+
// estimateCost returned null (a model missing from the catalog), and that 0
|
|
89
|
+
// billed as a measured free session. Both zeroes now fall through.
|
|
90
|
+
const zeroed = billableUsd({ model: "opus", input_tokens: 900, output_tokens: 40, total_cost_usd: 0, estimated_usd: 0 });
|
|
91
|
+
assert.equal(zeroed.usd, null);
|
|
92
|
+
assert.equal(zeroed.basis, "unpriced");
|
|
93
|
+
});
|
|
94
|
+
|
|
95
|
+
test("billableUsd: a zero-model attribution row is a real $0, not a gap", () => {
|
|
96
|
+
const { usd, basis } = billableUsd({ source: "messaging", model: null, input_tokens: 0, output_tokens: 0 });
|
|
97
|
+
assert.equal(usd, 0);
|
|
98
|
+
assert.equal(basis, "non_llm");
|
|
99
|
+
});
|
|
100
|
+
|
|
101
|
+
test("summariseRows: separates sessions from attribution rows", () => {
|
|
102
|
+
const s = summariseRows([
|
|
103
|
+
REAL_ROW,
|
|
104
|
+
{ ...REAL_ROW, total_cost_usd: 1 },
|
|
105
|
+
{ source: "messaging", model: null, input_tokens: 0, output_tokens: 0 },
|
|
106
|
+
{ measurement: "unknown", input_tokens: null, output_tokens: null },
|
|
107
|
+
]);
|
|
108
|
+
assert.equal(s.sessions, 3, "2 measured + 1 unmeasured; the send is not a session");
|
|
109
|
+
assert.equal(s.measured, 2);
|
|
110
|
+
assert.equal(s.unmeasured, 1);
|
|
111
|
+
assert.equal(s.nonLlmRows, 1);
|
|
112
|
+
assert.equal(s.measuredUsd, 1.533412);
|
|
113
|
+
assert.equal(s.blind, false);
|
|
114
|
+
});
|
|
115
|
+
|
|
116
|
+
test("summariseRows: blind only when sessions ran and NONE were measured", () => {
|
|
117
|
+
assert.equal(summariseRows([]).blind, false, "an empty ledger is quiet, not blind");
|
|
118
|
+
assert.equal(
|
|
119
|
+
summariseRows([{ source: "messaging", model: null, input_tokens: 0, output_tokens: 0 }]).blind,
|
|
120
|
+
false,
|
|
121
|
+
"attribution rows alone must not arm the alarm"
|
|
122
|
+
);
|
|
123
|
+
assert.equal(
|
|
124
|
+
summariseRows([{ measurement: "unknown", input_tokens: null, output_tokens: null }]).blind,
|
|
125
|
+
true
|
|
126
|
+
);
|
|
127
|
+
});
|
|
128
|
+
|
|
129
|
+
test("summariseRows: every degradation is reported, never swallowed", () => {
|
|
130
|
+
const s = summariseRows([
|
|
131
|
+
{ measurement: "unknown", input_tokens: null, output_tokens: null },
|
|
132
|
+
{ ...REAL_ROW, total_cost_usd: undefined },
|
|
133
|
+
"malformed",
|
|
134
|
+
]);
|
|
135
|
+
assert.equal(s.degradations.length, 3);
|
|
136
|
+
assert.match(s.degradations.join(" | "), /recorded NO token counts/);
|
|
137
|
+
assert.match(s.degradations.join(" | "), /no authoritative total_cost_usd/);
|
|
138
|
+
assert.match(s.degradations.join(" | "), /malformed/);
|
|
139
|
+
});
|
|
140
|
+
|
|
141
|
+
test("imputeUnmeasured: unmeasured sessions cost the day's mean, not zero", () => {
|
|
142
|
+
const s = summariseRows([
|
|
143
|
+
{ ...REAL_ROW, total_cost_usd: 2 },
|
|
144
|
+
{ measurement: "unknown", input_tokens: null, output_tokens: null },
|
|
145
|
+
{ measurement: "unknown", input_tokens: null, output_tokens: null },
|
|
146
|
+
]);
|
|
147
|
+
const imp = imputeUnmeasured(s);
|
|
148
|
+
assert.equal(imp.basis, "day_mean");
|
|
149
|
+
assert.equal(imp.perSessionUsd, 2);
|
|
150
|
+
assert.equal(imp.imputedUsd, 4);
|
|
151
|
+
});
|
|
152
|
+
|
|
153
|
+
test("imputeUnmeasured: a blind day imputes from the PERIOD mean when the day has nothing", () => {
|
|
154
|
+
// The anti-starvation property, without a magic number. The old fallback was a
|
|
155
|
+
// hard-coded $0.25/session — ~8x below this fleet's observed $1.95 mean — so a
|
|
156
|
+
// total outage on a day that really cost $68.40 (274% of a $25 cap) imputed
|
|
157
|
+
// $8.75 and reported band 0. Under-imputing is the silent-governor bug.
|
|
158
|
+
const s = summariseRows([{ measurement: "unknown", input_tokens: null, output_tokens: null }]);
|
|
159
|
+
const imp = imputeUnmeasured(s, { fallbackPerSessionUsd: 1.95 });
|
|
160
|
+
assert.equal(imp.basis, "period_mean");
|
|
161
|
+
assert.equal(imp.perSessionUsd, 1.95);
|
|
162
|
+
assert.equal(imp.imputedUsd, 1.95);
|
|
163
|
+
assert.equal(imp.unpriceable, false);
|
|
164
|
+
});
|
|
165
|
+
|
|
166
|
+
test("imputeUnmeasured: with NO evidence at any horizon it refuses to invent a price", () => {
|
|
167
|
+
// And says so. The caller (lib/budget-guard.dailyStatus) floors the band at
|
|
168
|
+
// DEGRADE on `unpriceable`, which needs no price table: we cannot measure,
|
|
169
|
+
// therefore we degrade. Returning a made-up dollar figure here is what let a
|
|
170
|
+
// $68 day read as band 0.
|
|
171
|
+
const s = summariseRows([{ measurement: "unknown", input_tokens: null, output_tokens: null }]);
|
|
172
|
+
const imp = imputeUnmeasured(s);
|
|
173
|
+
assert.equal(imp.basis, "unpriceable");
|
|
174
|
+
assert.equal(imp.imputedUsd, 0);
|
|
175
|
+
assert.equal(imp.unpriceable, true);
|
|
176
|
+
assert.equal(s.blind, true, "the signal the band floor keys off");
|
|
177
|
+
});
|
|
178
|
+
|
|
179
|
+
test("imputeUnmeasured: nothing unmeasured means nothing imputed", () => {
|
|
180
|
+
const imp = imputeUnmeasured(summariseRows([REAL_ROW]));
|
|
181
|
+
assert.equal(imp.imputedUsd, 0);
|
|
182
|
+
assert.equal(imp.basis, "none");
|
|
183
|
+
});
|
package/lib/execution/drive.mjs
CHANGED
|
@@ -48,6 +48,24 @@ export const EFFECT_FOR = Object.freeze({
|
|
|
48
48
|
/** Dispositions after which the agent has SPOKEN (deepens a reply chain). */
|
|
49
49
|
const SPEAKING = Object.freeze(["react_now"]);
|
|
50
50
|
|
|
51
|
+
/**
|
|
52
|
+
* Did this effect actually put words in front of a person?
|
|
53
|
+
*
|
|
54
|
+
* The disposition alone cannot answer that any more. `escalate` and `delegate`
|
|
55
|
+
* now reach a human through the engagement judgement when one is warranted, and
|
|
56
|
+
* do not when it isn't — so whether the agent spoke is a property of what the
|
|
57
|
+
* effect DID, not of which lane it took. An effect that reached someone reports
|
|
58
|
+
* `ref.spoke === true`; anything else is taken at its disposition's word.
|
|
59
|
+
*
|
|
60
|
+
* This matters beyond bookkeeping: `spoke` is what a self-healing sweep reads to
|
|
61
|
+
* decide whether a human is still owed an answer. A lane that claims to have
|
|
62
|
+
* spoken when it only wrote a row makes that sweep conclude everything is fine.
|
|
63
|
+
*/
|
|
64
|
+
function didSpeak(decision, r) {
|
|
65
|
+
if (r && r.ref && typeof r.ref === "object" && r.ref.spoke === true) return true;
|
|
66
|
+
return r.ok && SPEAKING.includes(decision.disposition);
|
|
67
|
+
}
|
|
68
|
+
|
|
51
69
|
/**
|
|
52
70
|
* Run one effect with uniform error capture. Returns a normalised result rather
|
|
53
71
|
* than throwing, so the caller's sequencing stays linear.
|
|
@@ -331,6 +349,15 @@ export async function drive(decision, o = {}) {
|
|
|
331
349
|
});
|
|
332
350
|
}
|
|
333
351
|
const r = await runEffect(effects[name], name, { decision, candidate: o.candidate, item: o.item });
|
|
352
|
+
// An effect that half-succeeded says so on its ref. Surfacing it here is what
|
|
353
|
+
// keeps "the person was reached but the audit row was not written" from
|
|
354
|
+
// reading as an unqualified success.
|
|
355
|
+
if (r.ref && typeof r.ref === "object" && Array.isArray(r.ref.degraded)) {
|
|
356
|
+
for (const d of r.ref.degraded) {
|
|
357
|
+
degraded.push(String(d));
|
|
358
|
+
log("warn", `${name} effect degraded: ${d}`, { key: decision.key });
|
|
359
|
+
}
|
|
360
|
+
}
|
|
334
361
|
if (r.missing) {
|
|
335
362
|
degraded.push(`${name}_effect_missing`);
|
|
336
363
|
log("error", `no \`${name}\` effect wired — the event was decided but not acted on`, { key: decision.key });
|
|
@@ -341,7 +368,7 @@ export async function drive(decision, o = {}) {
|
|
|
341
368
|
return finish({
|
|
342
369
|
effect: name,
|
|
343
370
|
ok: r.ok,
|
|
344
|
-
spoke:
|
|
371
|
+
spoke: didSpeak(decision, r),
|
|
345
372
|
ref: r.ref,
|
|
346
373
|
error: r.error,
|
|
347
374
|
degraded,
|