@cohortapp/agent-sdk 2.3.2 → 2.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (148) hide show
  1. package/framework-features.json +30 -0
  2. package/lib/backlog.mjs +136 -0
  3. package/lib/cadences.mjs +63 -2
  4. package/lib/cadences.test.mjs +105 -0
  5. package/lib/capability/inventory.mjs +542 -0
  6. package/lib/capability/inventory.test.mjs +232 -0
  7. package/lib/capability/probe.mjs +255 -0
  8. package/lib/channels/contract.mjs +37 -1
  9. package/lib/channels/contract.test.mjs +25 -1
  10. package/lib/claude-bin.mjs +37 -3
  11. package/lib/claude-bin.test.mjs +42 -8
  12. package/lib/execution/disposition.mjs +501 -0
  13. package/lib/execution/disposition.test.mjs +482 -0
  14. package/lib/execution/drive.mjs +352 -0
  15. package/lib/execution/drive.test.mjs +270 -0
  16. package/lib/execution/effects.mjs +340 -0
  17. package/lib/execution/effects.test.mjs +193 -0
  18. package/lib/execution/index.mjs +152 -0
  19. package/lib/execution/intake.mjs +581 -0
  20. package/lib/execution/intake.test.mjs +343 -0
  21. package/lib/execution/journal.mjs +374 -0
  22. package/lib/execution/journal.test.mjs +261 -0
  23. package/lib/execution/match.mjs +331 -0
  24. package/lib/execution/match.test.mjs +235 -0
  25. package/lib/execution/pipeline.mjs +341 -0
  26. package/lib/execution/pipeline.test.mjs +389 -0
  27. package/lib/execution/route.mjs +332 -0
  28. package/lib/execution/route.test.mjs +186 -0
  29. package/lib/execution/surface-policy.mjs +446 -0
  30. package/lib/execution/surface-policy.test.mjs +162 -0
  31. package/lib/goals/admission.mjs +209 -0
  32. package/lib/goals/admission.test.mjs +139 -0
  33. package/lib/goals/classify.mjs +206 -0
  34. package/lib/goals/classify.test.mjs +109 -0
  35. package/lib/goals/collaborate.mjs +415 -0
  36. package/lib/goals/collaborate.test.mjs +324 -0
  37. package/lib/goals/gaps.mjs +111 -0
  38. package/lib/goals/gaps.test.mjs +284 -0
  39. package/lib/goals/loop.mjs +537 -0
  40. package/lib/goals/loop.test.mjs +719 -0
  41. package/lib/identity/persona.mjs +247 -0
  42. package/lib/identity/persona.test.mjs +117 -0
  43. package/lib/kpi.mjs +469 -0
  44. package/lib/kpi.test.mjs +244 -0
  45. package/lib/mandate/audit.mjs +168 -0
  46. package/lib/mandate/audit.test.mjs +195 -0
  47. package/lib/mandate/cache.mjs +162 -0
  48. package/lib/mandate/derive.mjs +317 -0
  49. package/lib/mandate/derive.test.mjs +224 -0
  50. package/lib/mandate/model.mjs +352 -0
  51. package/lib/mandate/model.test.mjs +145 -0
  52. package/lib/mandate/refresh.mjs +187 -0
  53. package/lib/mandate/refresh.test.mjs +293 -0
  54. package/lib/mcp/server.test.mjs +4 -4
  55. package/lib/org/approvals.mjs +14 -2
  56. package/lib/org/client.mjs +58 -22
  57. package/lib/org/client.test.mjs +3 -1
  58. package/lib/org/inbound/directedness.mjs +720 -0
  59. package/lib/org/inbound/directedness.test.mjs +543 -0
  60. package/lib/org/inbound/facts.mjs +501 -0
  61. package/lib/org/inbound/facts.test.mjs +375 -0
  62. package/lib/org/inbound/hydrate.mjs +535 -0
  63. package/lib/org/inbound/hydrate.test.mjs +326 -0
  64. package/lib/org/inbound/index.mjs +233 -0
  65. package/lib/org/inbound/index.test.mjs +324 -0
  66. package/lib/org/inbound/io.mjs +141 -0
  67. package/lib/org/inbound/project.mjs +201 -0
  68. package/lib/org/inbound/project.test.mjs +287 -0
  69. package/lib/org/inbound/surfaces.mjs +257 -0
  70. package/lib/org/knowledge.mjs +10 -1
  71. package/lib/org/knowledge.test.mjs +8 -1
  72. package/lib/org/leases.mjs +5 -0
  73. package/lib/org/mesh.mjs +17 -2
  74. package/lib/org/messaging.mjs +40 -4
  75. package/lib/org/messaging.test.mjs +40 -0
  76. package/lib/org/param-contract.mjs +694 -0
  77. package/lib/org/param-contract.test.mjs +451 -0
  78. package/lib/org/protocol.checksum +1 -1
  79. package/lib/org/protocol.mjs +8 -0
  80. package/lib/org/protocol.test.mjs +5 -1
  81. package/lib/org/push.mjs +1025 -0
  82. package/lib/org/push.test.mjs +690 -0
  83. package/lib/org/tool-surface.mjs +138 -38
  84. package/lib/org/tool-surface.test.mjs +13 -8
  85. package/lib/org/typing.mjs +341 -0
  86. package/lib/org/typing.test.mjs +291 -0
  87. package/lib/plan/compile.mjs +510 -0
  88. package/lib/plan/compile.test.mjs +286 -0
  89. package/lib/plan/emit.mjs +256 -0
  90. package/lib/plan/emit.test.mjs +246 -0
  91. package/lib/plan/explain.mjs +226 -0
  92. package/lib/plan/explain.test.mjs +188 -0
  93. package/lib/plan/schema.mjs +140 -0
  94. package/lib/resource-governor.mjs +47 -1
  95. package/lib/resource-governor.test.mjs +21 -1
  96. package/lib/setup/enroll-from-cohort.mjs +105 -17
  97. package/lib/setup/enroll-from-cohort.test.mjs +68 -1
  98. package/lib/setup/sections/identity.mjs +15 -4
  99. package/lib/setup/sections/identity.test.mjs +94 -0
  100. package/lib/setup/sections/inventory.mjs +178 -0
  101. package/lib/setup/sections/inventory.test.mjs +198 -0
  102. package/lib/setup/sections/mandate.mjs +392 -0
  103. package/lib/setup/sections/mandate.test.mjs +373 -0
  104. package/lib/setup/sections/subagents.mjs +427 -0
  105. package/lib/setup/sections/subagents.test.mjs +429 -0
  106. package/lib/setup/sections/verify.mjs +121 -0
  107. package/lib/setup/sections/verify.test.mjs +175 -0
  108. package/lib/setup/sot.mjs +2 -0
  109. package/lib/subagents/cli.mjs +463 -0
  110. package/lib/subagents/cli.test.mjs +389 -0
  111. package/lib/subagents/client.mjs +373 -0
  112. package/lib/subagents/client.test.mjs +309 -0
  113. package/lib/subagents/gap.mjs +268 -0
  114. package/lib/subagents/gap.test.mjs +234 -0
  115. package/lib/subagents/lock.mjs +296 -0
  116. package/lib/subagents/lock.test.mjs +248 -0
  117. package/lib/subagents/manifest.mjs +224 -0
  118. package/lib/subagents/manifest.test.mjs +175 -0
  119. package/lib/subagents/refs.mjs +274 -0
  120. package/lib/subagents/refs.test.mjs +204 -0
  121. package/lib/subagents/resolve.mjs +455 -0
  122. package/lib/subagents/resolve.test.mjs +422 -0
  123. package/lib/subagents/schema.mjs +467 -0
  124. package/lib/subagents/schema.test.mjs +306 -0
  125. package/package.json +8 -3
  126. package/plugins/maestro-skills/.claude-plugin/marketplace.json +16 -0
  127. package/policies/ai-disclosure.yaml +42 -2
  128. package/scaffold/CLAUDE.md +16 -2
  129. package/schedules/triggers/goal-steward.md +79 -0
  130. package/scripts/ci/conformance-org-api.mjs +792 -0
  131. package/scripts/ci/conformance-org-api.test.mjs +417 -0
  132. package/scripts/daemon/agent-daemon.mjs +36 -4
  133. package/scripts/daemon/cadence-handlers.mjs +145 -1
  134. package/scripts/daemon/goal-steward-cadence.test.mjs +243 -0
  135. package/scripts/daemon/inbox-deferral.mjs +45 -2
  136. package/scripts/daemon/inbox-deferral.test.mjs +56 -0
  137. package/scripts/daemon/inbox-wake.mjs +282 -0
  138. package/scripts/daemon/inbox-wake.test.mjs +199 -0
  139. package/scripts/daemon/prompt-builder.mjs +41 -1
  140. package/scripts/daemon/typing-registry.mjs +55 -2
  141. package/scripts/daemon/typing-registry.test.mjs +25 -0
  142. package/scripts/local-triggers/generate-plists.test.mjs +5 -5
  143. package/scripts/setup/gen-subagent-manifest.mjs +95 -0
  144. package/scripts/setup/gen-subagent-manifest.test.mjs +124 -0
  145. package/scripts/setup/generate-plan.mjs +108 -0
  146. package/scripts/setup/init-capability-manifest.mjs +70 -0
  147. package/scripts/setup/init-skill-marketplace.mjs +155 -0
  148. package/scripts/setup/init-skill-marketplace.test.mjs +193 -0
@@ -0,0 +1,244 @@
1
+ /**
2
+ * kpi.test.mjs — measurement, the series, and the gap.
3
+ * Run: `node --test lib/kpi.test.mjs`
4
+ */
5
+
6
+ "use strict";
7
+
8
+ import { test } from "node:test";
9
+ import assert from "node:assert/strict";
10
+ import { mkdtempSync, rmSync } from "node:fs";
11
+ import { tmpdir } from "node:os";
12
+ import { join } from "node:path";
13
+
14
+ import {
15
+ windowFor,
16
+ isoWeek,
17
+ isDue,
18
+ readSeries,
19
+ recordMeasurement,
20
+ recordIntervention,
21
+ readInterventions,
22
+ measureKpi,
23
+ gapFor,
24
+ periodSeconds,
25
+ } from "./kpi.mjs";
26
+
27
+ function root() {
28
+ return mkdtempSync(join(tmpdir(), "kpi-test-"));
29
+ }
30
+ function cleanup(p) {
31
+ try { rmSync(p, { recursive: true, force: true }); } catch { /* */ }
32
+ }
33
+
34
+ const T = Date.parse("2026-08-11T09:00:00Z"); // a Tuesday, ISO week 33
35
+ const now = () => T;
36
+
37
+ const OBJ = {
38
+ key: "pipeline",
39
+ metric: "pipeline_coverage_x",
40
+ target: 3,
41
+ tolerance: 0.2,
42
+ direction: "up",
43
+ cadence: "weekly",
44
+ sensor: { capability: "crm_list_deals", params: { stage: "open" }, source: "method" },
45
+ };
46
+
47
+ // ---------------------------------------------------------------------------
48
+ // windows
49
+ // ---------------------------------------------------------------------------
50
+
51
+ test("windowFor produces the right idempotency key per cadence", () => {
52
+ assert.equal(windowFor("weekly", T), "2026-W33");
53
+ assert.equal(windowFor("monthly", T), "2026-08");
54
+ assert.equal(windowFor("quarterly", T), "2026-Q3");
55
+ assert.equal(windowFor("daily", T), "2026-08-11");
56
+ assert.equal(isoWeek(new Date(T)).week, 33);
57
+ });
58
+
59
+ test("periodSeconds covers every cadence and defaults to a day", () => {
60
+ assert.equal(periodSeconds("weekly"), 604800);
61
+ assert.equal(periodSeconds("monthly"), 2592000);
62
+ assert.equal(periodSeconds("nonsense"), 86400);
63
+ });
64
+
65
+ // ---------------------------------------------------------------------------
66
+ // the ledger
67
+ // ---------------------------------------------------------------------------
68
+
69
+ test("recordMeasurement is idempotent on (objective, window) and refuses a weaker duplicate", () => {
70
+ const r = root();
71
+ try {
72
+ const a = recordMeasurement(r, { objectiveKey: "pipeline", value: 2.1, window: "2026-W33", source: "method" }, { now });
73
+ assert.equal(a.recorded, true);
74
+
75
+ const dup = recordMeasurement(r, { objectiveKey: "pipeline", value: 9.9, window: "2026-W33", source: "llm" }, { now });
76
+ assert.equal(dup.recorded, false);
77
+ assert.equal(dup.reason, "duplicate-window");
78
+ assert.equal(readSeries(r, "pipeline")[0].value, 2.1, "the llm guess must not overwrite the measured number");
79
+
80
+ // A STRONGER source may supersede a weaker one in the same window.
81
+ const r2 = root();
82
+ try {
83
+ recordMeasurement(r2, { objectiveKey: "p", value: 1, window: "w", source: "llm" }, { now });
84
+ const better = recordMeasurement(r2, { objectiveKey: "p", value: 5, window: "w", source: "method" }, { now });
85
+ assert.equal(better.recorded, true);
86
+ assert.equal(readSeries(r2, "p")[0].value, 5);
87
+ assert.equal(readSeries(r2, "p").length, 1, "the window collapses to the strongest source");
88
+ } finally { cleanup(r2); }
89
+ } finally { cleanup(r); }
90
+ });
91
+
92
+ test("recordMeasurement refuses a non-numeric value and logs it", () => {
93
+ const r = root();
94
+ const logs = [];
95
+ try {
96
+ const res = recordMeasurement(r, { objectiveKey: "p", value: "banana", window: "w", source: "method" }, { now, log: (l, m) => logs.push(`${l}:${m}`) });
97
+ assert.equal(res.recorded, false);
98
+ assert.equal(res.reason, "non-numeric value");
99
+ assert.equal(logs.length, 1, "a refusal is never silent");
100
+ } finally { cleanup(r); }
101
+ });
102
+
103
+ test("interventions are the learning signal and read back newest-last", () => {
104
+ const r = root();
105
+ try {
106
+ for (const d of [-1, 0.5, 2]) {
107
+ recordIntervention(r, { objectiveKey: "pipeline", taskId: `t${d}`, expectedDelta: 1, realizedDelta: d }, { now });
108
+ }
109
+ const last2 = readInterventions(r, "pipeline", 2);
110
+ assert.deepEqual(last2.map((i) => i.realizedDelta), [0.5, 2]);
111
+ } finally { cleanup(r); }
112
+ });
113
+
114
+ test("isDue opens once per window and closes after a sample; no cadence is never due", () => {
115
+ const r = root();
116
+ try {
117
+ assert.equal(isDue(OBJ, [], { now }).due, true);
118
+ recordMeasurement(r, { objectiveKey: "pipeline", value: 2, window: "2026-W33", source: "method" }, { now });
119
+ assert.equal(isDue(OBJ, readSeries(r, "pipeline"), { now }).due, false);
120
+ const none = isDue({ ...OBJ, cadence: null }, [], { now });
121
+ assert.equal(none.due, false);
122
+ assert.match(none.reason, /no measure cadence/);
123
+ } finally { cleanup(r); }
124
+ });
125
+
126
+ // ---------------------------------------------------------------------------
127
+ // measurement — the negative cases matter most
128
+ // ---------------------------------------------------------------------------
129
+
130
+ test("measureKpi executes a registered sensor and stamps method-sourced evidence", async () => {
131
+ const seen = [];
132
+ const m = await measureKpi(OBJ, {
133
+ sensors: { crm_list_deals: async ({ params }) => { seen.push(params); return { value: 2.4, rowCount: 17 }; } },
134
+ reachable: new Set(["crm_list_deals"]),
135
+ });
136
+ assert.equal(m.ok, true);
137
+ assert.equal(m.value, 2.4);
138
+ assert.equal(m.source, "method");
139
+ assert.equal(m.evidence.method, "crm_list_deals");
140
+ assert.equal(m.evidence.rowCount, 17);
141
+ assert.deepEqual(seen, [{ stage: "open" }]);
142
+ });
143
+
144
+ test("NEGATIVE — an unreachable capability refuses to measure rather than recording a zero", async () => {
145
+ const logs = [];
146
+ const m = await measureKpi(OBJ, {
147
+ sensors: { crm_list_deals: async () => 2.4 },
148
+ reachable: new Set(["something_else"]),
149
+ log: (l, msg) => logs.push(msg),
150
+ });
151
+ assert.equal(m.ok, false);
152
+ assert.equal(m.reason, "unreachable-capability");
153
+ assert.equal(m.value, null);
154
+ assert.match(logs.join(" "), /NOT reachable/);
155
+ });
156
+
157
+ test("NEGATIVE — a missing sensor implementation, a throwing sensor and a non-numeric result all fail loudly", async () => {
158
+ const logs = [];
159
+ const log = (l, msg) => logs.push(msg);
160
+
161
+ const missing = await measureKpi(OBJ, { sensors: {}, reachable: new Set(["crm_list_deals"]), log });
162
+ assert.equal(missing.reason, "no-implementation");
163
+
164
+ const threw = await measureKpi(OBJ, {
165
+ sensors: { crm_list_deals: async () => { throw new Error("server unreachable"); } },
166
+ reachable: new Set(["crm_list_deals"]),
167
+ log,
168
+ });
169
+ assert.equal(threw.reason, "sensor-threw");
170
+
171
+ const junk = await measureKpi(OBJ, {
172
+ sensors: { crm_list_deals: async () => ({ value: "n/a" }) },
173
+ reachable: new Set(["crm_list_deals"]),
174
+ log,
175
+ });
176
+ assert.equal(junk.reason, "non-numeric");
177
+ assert.equal(logs.length, 3, "every degradation is logged");
178
+ });
179
+
180
+ test("an objective with a human sensor is not an error — it is a human obligation", async () => {
181
+ const m = await measureKpi({ ...OBJ, sensor: { capability: null, source: "human" } }, { sensors: {} });
182
+ assert.equal(m.ok, false);
183
+ assert.equal(m.reason, "no-sensor");
184
+ assert.equal(m.source, "human");
185
+ });
186
+
187
+ // ---------------------------------------------------------------------------
188
+ // the gap
189
+ // ---------------------------------------------------------------------------
190
+
191
+ function series(values, cadence = "weekly", at = "2026-08-11T09:00:00Z") {
192
+ return values.map((v, i) => ({ objectiveKey: "pipeline", value: v, window: `w${i}`, at, source: "method" }));
193
+ }
194
+
195
+ test("gapFor honours direction: up is a shortfall, down is an excess", () => {
196
+ const up = gapFor(OBJ, series([2.0]), { now });
197
+ assert.equal(up.gap, 1);
198
+ assert.equal(up.withinTolerance, false);
199
+ assert.equal(up.normalizedGap, +(1 / 3).toFixed(6));
200
+
201
+ const down = gapFor({ ...OBJ, direction: "down", target: 5 }, series([7]), { now });
202
+ assert.equal(down.gap, 2, "for a down metric, above target is the gap");
203
+
204
+ const under = gapFor({ ...OBJ, direction: "down", target: 5 }, series([3]), { now });
205
+ assert.equal(under.gap, -2);
206
+ assert.equal(under.withinTolerance, true, "beating a down-target is not a gap");
207
+ });
208
+
209
+ test("gapFor honours tolerance and band direction", () => {
210
+ const inTol = gapFor(OBJ, series([2.85]), { now });
211
+ assert.ok(inTol.gap <= 0.2 && inTol.withinTolerance, "0.15 short of 3 is inside a 0.2 tolerance");
212
+
213
+ const band = gapFor({ ...OBJ, direction: "band", target: 10, tolerance: 2 }, series([13]), { now });
214
+ assert.equal(band.gap, 1, "a band gap is the distance outside the band");
215
+ const inBand = gapFor({ ...OBJ, direction: "band", target: 10, tolerance: 2 }, series([11]), { now });
216
+ assert.equal(inBand.gap, 0);
217
+ });
218
+
219
+ test("gapFor computes a real trend over the gap, not the raw value", () => {
220
+ assert.equal(gapFor(OBJ, series([1.0, 1.8, 2.5]), { now }).trend, "improving");
221
+ assert.equal(gapFor(OBJ, series([2.5, 1.8, 1.0]), { now }).trend, "worsening");
222
+ assert.equal(gapFor(OBJ, series([2.0, 2.0, 2.0]), { now }).trend, "flat");
223
+ assert.equal(gapFor(OBJ, series([2.0]), { now }).trend, "unknown");
224
+ });
225
+
226
+ test("NEGATIVE — no samples, and no target, both report rather than compute", () => {
227
+ const none = gapFor(OBJ, [], { now });
228
+ assert.equal(none.gap, null);
229
+ assert.equal(none.reason, "no samples");
230
+
231
+ const noTarget = gapFor({ ...OBJ, target: null }, series([2]), { now });
232
+ assert.equal(noTarget.gap, null);
233
+ assert.match(noTarget.reason, /no numeric target/);
234
+ assert.equal(noTarget.target, null, "a null target must never coerce to 0");
235
+ });
236
+
237
+ test("a sensor silent for 2× its cadence is itself a gap", () => {
238
+ const stale = gapFor(OBJ, series([2], "weekly", "2026-07-01T09:00:00Z"), { now });
239
+ assert.equal(stale.stale, true);
240
+ assert.ok(stale.stalenessPeriods > 2);
241
+
242
+ const fresh = gapFor(OBJ, series([2], "weekly", "2026-08-10T09:00:00Z"), { now });
243
+ assert.equal(fresh.stale, false);
244
+ });
@@ -0,0 +1,168 @@
1
+ /**
2
+ * lib/mandate/audit.mjs — one audit line for every autonomous decision.
3
+ *
4
+ * The question this file exists to answer, offline, from local state alone:
5
+ * **"why did the agent do that?"** Every decision the goal loop makes — a
6
+ * measurement taken, a gap computed, a candidate admitted or REJECTED, a
7
+ * disposition chosen, an execution rung picked, a budget refusal, an approval
8
+ * gate — appends exactly one line here, carrying the provenance chain back to
9
+ * the pillar and the human who adopted the objective.
10
+ *
11
+ * Rejections are recorded with the same weight as admissions. A candidate that
12
+ * is silently dropped is indistinguishable from a bug — the recurring failure
13
+ * mode this codebase keeps rediscovering.
14
+ *
15
+ * Append-only JSONL, atomic append, never throws. When an org mirror is
16
+ * injected it is called best-effort AFTER the local write, so an unreachable
17
+ * server never loses the local audit line.
18
+ *
19
+ * Not to be confused with `lib/execution/journal.mjs`, which is the REACTIVE
20
+ * lane's ledger: it is keyed on inbound dedupe/thread keys and is scan-capped
21
+ * for flood detection. This ledger is keyed on `objectiveKey`/`obligationKey`
22
+ * and is never truncated, because it is the provenance record a human reads
23
+ * when asking why a goal produced a task. Different key space, different
24
+ * retention, deliberately separate files.
25
+ *
26
+ * @module lib/mandate/audit
27
+ */
28
+
29
+ "use strict";
30
+
31
+ import { existsSync, readFileSync, mkdirSync } from "node:fs";
32
+ import { join, dirname } from "node:path";
33
+
34
+ import { resolveAgentRoot } from "../agent-root.mjs";
35
+ import { appendJsonl } from "../fs-atomic.mjs";
36
+ import { provenanceOf } from "./model.mjs";
37
+
38
+ export const AUDIT_REL = join("state", "mandate", "audit.jsonl");
39
+
40
+ /** Decision kinds that may appear on an audit line. */
41
+ export const DECISIONS = Object.freeze([
42
+ "measured",
43
+ "measure_failed",
44
+ "gap",
45
+ "candidate_admitted",
46
+ "candidate_rejected",
47
+ "disposition",
48
+ "routed",
49
+ "route_failed",
50
+ "execution_rung",
51
+ "budget_refused",
52
+ "approval_gate",
53
+ "drift",
54
+ "observe_only",
55
+ "mandate_refreshed",
56
+ ]);
57
+
58
+ /** Absolute path of the audit ledger. */
59
+ export function auditPath(agentRoot) {
60
+ return join(resolveAgentRoot(agentRoot), AUDIT_REL);
61
+ }
62
+
63
+ function nowIso(deps) {
64
+ return new Date(deps && typeof deps.now === "function" ? deps.now() : Date.now()).toISOString();
65
+ }
66
+
67
+ /**
68
+ * Build the provenance block for an objective key: the chain of keys up to the
69
+ * pillar, the charter clause, and who adopted it. Delegates to
70
+ * `model.provenanceOf` — the walk has exactly one implementation. Pure; offline.
71
+ *
72
+ * @param {object|object[]} mandate cache record, body, or objective array
73
+ * @param {string} objectiveKey
74
+ * @returns {{chain:string[], pillar:string|null, clause:string|null, adoptedBy:string|null, origin:string|null}}
75
+ */
76
+ export function provenanceFor(mandate, objectiveKey) {
77
+ return provenanceOf(mandate, objectiveKey);
78
+ }
79
+
80
+ /**
81
+ * Append one audit line. Never throws.
82
+ *
83
+ * @param {string} agentRoot
84
+ * @param {object} record - { decision, objectiveKey?, obligationKey?, detail?, … }
85
+ * @param {object} [deps] { now, mandate?, mirror?, log? }
86
+ * @returns {object} the written record (with `at` + `provenance` stamped)
87
+ */
88
+ export function recordDecision(agentRoot, record, deps = {}) {
89
+ const rec = {
90
+ at: nowIso(deps),
91
+ decision: String((record && record.decision) || "unknown"),
92
+ objectiveKey: (record && record.objectiveKey) || null,
93
+ obligationKey: (record && record.obligationKey) || null,
94
+ ...record,
95
+ };
96
+ if (!DECISIONS.includes(rec.decision) && typeof deps.log === "function") {
97
+ deps.log("warn", `[mandate-audit] unrecognised decision kind "${rec.decision}" — recorded anyway`);
98
+ }
99
+ if (deps.mandate && rec.objectiveKey) {
100
+ rec.provenance = provenanceFor(deps.mandate, rec.objectiveKey);
101
+ }
102
+ try {
103
+ const p = auditPath(agentRoot);
104
+ mkdirSync(dirname(p), { recursive: true });
105
+ appendJsonl(p, rec);
106
+ } catch (err) {
107
+ if (typeof deps.log === "function") {
108
+ deps.log("error", `[mandate-audit] could not append audit line: ${err && err.message ? err.message : err}`);
109
+ }
110
+ }
111
+ if (typeof deps.mirror === "function") {
112
+ try {
113
+ const r = deps.mirror(rec);
114
+ if (r && typeof r.catch === "function") {
115
+ r.catch((err) => {
116
+ if (typeof deps.log === "function") {
117
+ deps.log("warn", `[mandate-audit] org mirror failed (local line is intact): ${err && err.message ? err.message : err}`);
118
+ }
119
+ });
120
+ }
121
+ } catch (err) {
122
+ if (typeof deps.log === "function") {
123
+ deps.log("warn", `[mandate-audit] org mirror threw (local line is intact): ${err && err.message ? err.message : err}`);
124
+ }
125
+ }
126
+ }
127
+ return rec;
128
+ }
129
+
130
+ /**
131
+ * Read audit lines back (newest last). Never throws; malformed lines skipped.
132
+ * @param {string} agentRoot @param {{limit?:number, decision?:string, objectiveKey?:string}} [q]
133
+ * @returns {object[]}
134
+ */
135
+ export function readAudit(agentRoot, q = {}) {
136
+ const p = auditPath(agentRoot);
137
+ if (!existsSync(p)) return [];
138
+ let body;
139
+ try { body = readFileSync(p, "utf-8"); } catch { return []; }
140
+ const out = [];
141
+ for (const line of body.split("\n")) {
142
+ if (!line.trim()) continue;
143
+ let rec;
144
+ try { rec = JSON.parse(line); } catch { continue; }
145
+ if (q.decision && rec.decision !== q.decision) continue;
146
+ if (q.objectiveKey && rec.objectiveKey !== q.objectiveKey) continue;
147
+ out.push(rec);
148
+ }
149
+ return Number.isFinite(q.limit) && q.limit > 0 ? out.slice(-q.limit) : out;
150
+ }
151
+
152
+ /**
153
+ * The offline "why did the agent do that?" answer for one objective: its
154
+ * provenance chain plus every decision taken against it. Needs no network —
155
+ * this is the client-side half of the audit story (the server-side half is a
156
+ * chain query joining Task.obligationKey → AgentPlanDigest → Objective).
157
+ * @param {string} agentRoot @param {object|object[]} mandate @param {string} objectiveKey
158
+ * @returns {{objectiveKey:string, provenance:object, decisions:object[]}}
159
+ */
160
+ export function explain(agentRoot, mandate, objectiveKey) {
161
+ return {
162
+ objectiveKey,
163
+ provenance: provenanceFor(mandate, objectiveKey),
164
+ decisions: readAudit(agentRoot, { objectiveKey }),
165
+ };
166
+ }
167
+
168
+ export default { AUDIT_REL, DECISIONS, auditPath, provenanceFor, recordDecision, readAudit, explain };
@@ -0,0 +1,195 @@
1
+ /**
2
+ * lib/mandate/audit.test.mjs — an audit line for every autonomous decision.
3
+ *
4
+ * The governance question this answers is "why did the agent do that?", and it
5
+ * has to be answerable OFFLINE, from the ledger plus the cached mandate alone.
6
+ * Every line therefore carries its own provenance walk: objective → pillar →
7
+ * charter clause → who adopted it.
8
+ *
9
+ * The failure mode being guarded against is the one this codebase keeps hitting:
10
+ * a decision that happens but leaves no trace. So the ledger never throws, never
11
+ * drops a line for an unrecognised decision kind, and logs when the org mirror
12
+ * fails rather than pretending it succeeded.
13
+ */
14
+
15
+ "use strict";
16
+
17
+ import { test } from "node:test";
18
+ import assert from "node:assert/strict";
19
+ import { mkdtempSync, rmSync, readFileSync, writeFileSync, existsSync } from "node:fs";
20
+ import { tmpdir } from "node:os";
21
+ import { join } from "node:path";
22
+
23
+ import { recordDecision, readAudit, explain, provenanceFor, auditPath, DECISIONS, AUDIT_REL } from "./audit.mjs";
24
+
25
+ const NOW = Date.parse("2026-08-11T09:00:00.000Z");
26
+
27
+ const MANDATE = {
28
+ body: {
29
+ memberId: "mem_self",
30
+ objectives: [
31
+ { key: "revenue", kind: "PILLAR", state: "active", charterSectionId: "cs_12", adoptedById: "mem_boss" },
32
+ {
33
+ key: "pipeline", kind: "OBJECTIVE", parentKey: "revenue", state: "active",
34
+ metric: "pipeline_coverage_x", target: 3, adoptedById: "mem_boss",
35
+ },
36
+ ],
37
+ },
38
+ };
39
+
40
+ function makeRoot() { return mkdtempSync(join(tmpdir(), "mandate-audit-")); }
41
+ function cleanup(root) { try { rmSync(root, { recursive: true, force: true }); } catch { /* best effort */ } }
42
+ function lines(root) {
43
+ return readFileSync(auditPath(root), "utf-8").trim().split("\n").map((l) => JSON.parse(l));
44
+ }
45
+
46
+ test("every decision kind the loop emits is in the declared vocabulary", () => {
47
+ // If the loop grows a decision the ledger has never heard of, that is a
48
+ // review signal, not a silent write.
49
+ for (const d of [
50
+ "measured", "measure_failed", "gap", "candidate_admitted", "candidate_rejected",
51
+ "disposition", "routed", "route_failed", "execution_rung", "budget_refused",
52
+ "approval_gate", "drift", "observe_only",
53
+ ]) {
54
+ assert.ok(DECISIONS.includes(d), `"${d}" must be a declared decision kind`);
55
+ }
56
+ });
57
+
58
+ test("a decision line stamps time, kind, and the full provenance chain", () => {
59
+ const root = makeRoot();
60
+ try {
61
+ const rec = recordDecision(
62
+ root,
63
+ { decision: "candidate_admitted", objectiveKey: "pipeline", obligationKey: "outcome.pipeline", detail: { title: "Book calls" } },
64
+ { now: () => NOW, mandate: MANDATE }
65
+ );
66
+ assert.equal(rec.at, new Date(NOW).toISOString());
67
+ assert.equal(rec.decision, "candidate_admitted");
68
+ assert.equal(rec.obligationKey, "outcome.pipeline");
69
+ // The offline "why" walk.
70
+ assert.deepEqual(rec.provenance.chain, ["pipeline", "revenue"]);
71
+ assert.equal(rec.provenance.pillar, "revenue");
72
+ assert.equal(rec.provenance.clause, "cs_12");
73
+ assert.equal(rec.provenance.adoptedBy, "mem_boss");
74
+
75
+ assert.equal(lines(root).length, 1);
76
+ assert.equal(lines(root)[0].provenance.clause, "cs_12");
77
+ } finally { cleanup(root); }
78
+ });
79
+
80
+ test("the ledger is append-only and read back oldest-first", () => {
81
+ const root = makeRoot();
82
+ try {
83
+ for (const d of ["measured", "gap", "candidate_admitted"]) {
84
+ recordDecision(root, { decision: d, objectiveKey: "pipeline" }, { now: () => NOW, mandate: MANDATE });
85
+ }
86
+ const all = readAudit(root);
87
+ assert.deepEqual(all.map((a) => a.decision), ["measured", "gap", "candidate_admitted"]);
88
+ assert.equal(readAudit(root, { decision: "gap" }).length, 1);
89
+ assert.equal(readAudit(root, { objectiveKey: "pipeline" }).length, 3);
90
+ assert.equal(readAudit(root, { objectiveKey: "nope" }).length, 0);
91
+ assert.equal(readAudit(root, { limit: 2 }).length, 2);
92
+ } finally { cleanup(root); }
93
+ });
94
+
95
+ test("an UNRECOGNISED decision kind is recorded anyway, with a warning", () => {
96
+ const root = makeRoot();
97
+ try {
98
+ const logs = [];
99
+ recordDecision(root, { decision: "invented_kind" }, { now: () => NOW, log: (l, m) => logs.push(`${l}:${m}`) });
100
+ // Recorded — dropping it would be the silent failure.
101
+ assert.equal(lines(root).length, 1);
102
+ assert.equal(lines(root)[0].decision, "invented_kind");
103
+ // …but loudly.
104
+ assert.ok(logs.some((l) => l.startsWith("warn:") && /unrecognised decision kind/.test(l)));
105
+ } finally { cleanup(root); }
106
+ });
107
+
108
+ test("a decision with no objective still records — provenance is optional, the line is not", () => {
109
+ const root = makeRoot();
110
+ try {
111
+ recordDecision(root, { decision: "drift", detail: { kind: "mandate_stale" } }, { now: () => NOW });
112
+ const [rec] = lines(root);
113
+ assert.equal(rec.objectiveKey, null);
114
+ assert.equal(rec.provenance, undefined);
115
+ assert.equal(rec.detail.kind, "mandate_stale");
116
+ } finally { cleanup(root); }
117
+ });
118
+
119
+ test("readAudit tolerates a malformed line rather than losing the whole ledger", () => {
120
+ const root = makeRoot();
121
+ try {
122
+ recordDecision(root, { decision: "gap", objectiveKey: "pipeline" }, { now: () => NOW, mandate: MANDATE });
123
+ // Simulate a torn write.
124
+ const p = auditPath(root);
125
+ const body = readFileSync(p, "utf-8");
126
+ writeFileSync(p, body + "{ this is not json\n");
127
+ recordDecision(root, { decision: "measured", objectiveKey: "pipeline" }, { now: () => NOW, mandate: MANDATE });
128
+ const all = readAudit(root);
129
+ assert.equal(all.length, 2, "the readable lines must survive one corrupt line");
130
+ } finally { cleanup(root); }
131
+ });
132
+
133
+ test("an unwritable ledger logs and returns — it never throws into the agent loop", () => {
134
+ const logs = [];
135
+ // A path that cannot be created.
136
+ const rec = recordDecision("/proc/nonexistent-root-xyz", { decision: "gap" }, {
137
+ now: () => NOW,
138
+ log: (l, m) => logs.push(`${l}:${m}`),
139
+ });
140
+ assert.equal(rec.decision, "gap", "the caller still gets its record back");
141
+ assert.ok(logs.some((l) => l.startsWith("error:")), "an unwritable ledger must be reported");
142
+ });
143
+
144
+ test("a FAILING org mirror never costs the local line, and is logged", async () => {
145
+ const root = makeRoot();
146
+ try {
147
+ const logs = [];
148
+ recordDecision(root, { decision: "gap", objectiveKey: "pipeline" }, {
149
+ now: () => NOW, mandate: MANDATE,
150
+ mirror: () => { throw new Error("ECONNREFUSED"); },
151
+ log: (l, m) => logs.push(`${l}:${m}`),
152
+ });
153
+ assert.equal(lines(root).length, 1, "the local line is intact even when hq is unreachable");
154
+ assert.ok(logs.some((l) => /mirror threw/.test(l)));
155
+
156
+ // …and the async rejection path too.
157
+ recordDecision(root, { decision: "gap", objectiveKey: "pipeline" }, {
158
+ now: () => NOW, mandate: MANDATE,
159
+ mirror: async () => { throw new Error("502"); },
160
+ log: (l, m) => logs.push(`${l}:${m}`),
161
+ });
162
+ await new Promise((r) => setTimeout(r, 10));
163
+ assert.equal(lines(root).length, 2);
164
+ assert.ok(logs.some((l) => /mirror failed/.test(l)));
165
+ } finally { cleanup(root); }
166
+ });
167
+
168
+ test("provenanceFor and explain answer 'why did the agent do that' offline", () => {
169
+ const root = makeRoot();
170
+ try {
171
+ const prov = provenanceFor(MANDATE, "pipeline");
172
+ assert.equal(prov.clause, "cs_12");
173
+ assert.equal(prov.adoptedBy, "mem_boss");
174
+
175
+ recordDecision(root, { decision: "candidate_admitted", objectiveKey: "pipeline" }, { now: () => NOW, mandate: MANDATE });
176
+ recordDecision(root, { decision: "routed", objectiveKey: "pipeline" }, { now: () => NOW, mandate: MANDATE });
177
+ recordDecision(root, { decision: "gap", objectiveKey: "other" }, { now: () => NOW, mandate: MANDATE });
178
+
179
+ const e = explain(root, MANDATE, "pipeline");
180
+ assert.equal(e.provenance.clause, "cs_12");
181
+ assert.equal(e.decisions.length, 2, "explain must scope to the objective asked about");
182
+ } finally { cleanup(root); }
183
+ });
184
+
185
+ test("an unknown objective key degrades to a null chain, not a throw", () => {
186
+ const prov = provenanceFor(MANDATE, "does-not-exist");
187
+ assert.deepEqual(prov.chain, []);
188
+ assert.equal(prov.clause, null);
189
+ });
190
+
191
+ test("the ledger path is the documented one", () => {
192
+ assert.equal(AUDIT_REL, join("state", "mandate", "audit.jsonl"));
193
+ assert.ok(auditPath("/tmp/x").endsWith(AUDIT_REL));
194
+ assert.equal(existsSync(join("/tmp/definitely-not-here", AUDIT_REL)), false);
195
+ });