@cohortapp/agent-sdk 2.3.2 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (148) hide show
  1. package/framework-features.json +30 -0
  2. package/lib/backlog.mjs +136 -0
  3. package/lib/cadences.mjs +63 -2
  4. package/lib/cadences.test.mjs +105 -0
  5. package/lib/capability/inventory.mjs +542 -0
  6. package/lib/capability/inventory.test.mjs +232 -0
  7. package/lib/capability/probe.mjs +255 -0
  8. package/lib/channels/contract.mjs +37 -1
  9. package/lib/channels/contract.test.mjs +25 -1
  10. package/lib/claude-bin.mjs +37 -3
  11. package/lib/claude-bin.test.mjs +42 -8
  12. package/lib/execution/disposition.mjs +501 -0
  13. package/lib/execution/disposition.test.mjs +482 -0
  14. package/lib/execution/drive.mjs +352 -0
  15. package/lib/execution/drive.test.mjs +270 -0
  16. package/lib/execution/effects.mjs +340 -0
  17. package/lib/execution/effects.test.mjs +193 -0
  18. package/lib/execution/index.mjs +152 -0
  19. package/lib/execution/intake.mjs +581 -0
  20. package/lib/execution/intake.test.mjs +343 -0
  21. package/lib/execution/journal.mjs +374 -0
  22. package/lib/execution/journal.test.mjs +261 -0
  23. package/lib/execution/match.mjs +331 -0
  24. package/lib/execution/match.test.mjs +235 -0
  25. package/lib/execution/pipeline.mjs +341 -0
  26. package/lib/execution/pipeline.test.mjs +389 -0
  27. package/lib/execution/route.mjs +332 -0
  28. package/lib/execution/route.test.mjs +186 -0
  29. package/lib/execution/surface-policy.mjs +446 -0
  30. package/lib/execution/surface-policy.test.mjs +162 -0
  31. package/lib/goals/admission.mjs +209 -0
  32. package/lib/goals/admission.test.mjs +139 -0
  33. package/lib/goals/classify.mjs +206 -0
  34. package/lib/goals/classify.test.mjs +109 -0
  35. package/lib/goals/collaborate.mjs +415 -0
  36. package/lib/goals/collaborate.test.mjs +324 -0
  37. package/lib/goals/gaps.mjs +111 -0
  38. package/lib/goals/gaps.test.mjs +284 -0
  39. package/lib/goals/loop.mjs +537 -0
  40. package/lib/goals/loop.test.mjs +719 -0
  41. package/lib/identity/persona.mjs +247 -0
  42. package/lib/identity/persona.test.mjs +117 -0
  43. package/lib/kpi.mjs +469 -0
  44. package/lib/kpi.test.mjs +244 -0
  45. package/lib/mandate/audit.mjs +168 -0
  46. package/lib/mandate/audit.test.mjs +195 -0
  47. package/lib/mandate/cache.mjs +162 -0
  48. package/lib/mandate/derive.mjs +317 -0
  49. package/lib/mandate/derive.test.mjs +224 -0
  50. package/lib/mandate/model.mjs +352 -0
  51. package/lib/mandate/model.test.mjs +145 -0
  52. package/lib/mandate/refresh.mjs +187 -0
  53. package/lib/mandate/refresh.test.mjs +293 -0
  54. package/lib/mcp/server.test.mjs +4 -4
  55. package/lib/org/approvals.mjs +14 -2
  56. package/lib/org/client.mjs +58 -22
  57. package/lib/org/client.test.mjs +3 -1
  58. package/lib/org/inbound/directedness.mjs +720 -0
  59. package/lib/org/inbound/directedness.test.mjs +543 -0
  60. package/lib/org/inbound/facts.mjs +501 -0
  61. package/lib/org/inbound/facts.test.mjs +375 -0
  62. package/lib/org/inbound/hydrate.mjs +535 -0
  63. package/lib/org/inbound/hydrate.test.mjs +326 -0
  64. package/lib/org/inbound/index.mjs +233 -0
  65. package/lib/org/inbound/index.test.mjs +324 -0
  66. package/lib/org/inbound/io.mjs +141 -0
  67. package/lib/org/inbound/project.mjs +201 -0
  68. package/lib/org/inbound/project.test.mjs +287 -0
  69. package/lib/org/inbound/surfaces.mjs +257 -0
  70. package/lib/org/knowledge.mjs +10 -1
  71. package/lib/org/knowledge.test.mjs +8 -1
  72. package/lib/org/leases.mjs +5 -0
  73. package/lib/org/mesh.mjs +17 -2
  74. package/lib/org/messaging.mjs +40 -4
  75. package/lib/org/messaging.test.mjs +40 -0
  76. package/lib/org/param-contract.mjs +694 -0
  77. package/lib/org/param-contract.test.mjs +451 -0
  78. package/lib/org/protocol.checksum +1 -1
  79. package/lib/org/protocol.mjs +8 -0
  80. package/lib/org/protocol.test.mjs +5 -1
  81. package/lib/org/push.mjs +1025 -0
  82. package/lib/org/push.test.mjs +690 -0
  83. package/lib/org/tool-surface.mjs +138 -38
  84. package/lib/org/tool-surface.test.mjs +13 -8
  85. package/lib/org/typing.mjs +341 -0
  86. package/lib/org/typing.test.mjs +291 -0
  87. package/lib/plan/compile.mjs +510 -0
  88. package/lib/plan/compile.test.mjs +286 -0
  89. package/lib/plan/emit.mjs +256 -0
  90. package/lib/plan/emit.test.mjs +246 -0
  91. package/lib/plan/explain.mjs +226 -0
  92. package/lib/plan/explain.test.mjs +188 -0
  93. package/lib/plan/schema.mjs +140 -0
  94. package/lib/resource-governor.mjs +47 -1
  95. package/lib/resource-governor.test.mjs +21 -1
  96. package/lib/setup/enroll-from-cohort.mjs +84 -16
  97. package/lib/setup/enroll-from-cohort.test.mjs +43 -1
  98. package/lib/setup/sections/identity.mjs +15 -4
  99. package/lib/setup/sections/identity.test.mjs +94 -0
  100. package/lib/setup/sections/inventory.mjs +178 -0
  101. package/lib/setup/sections/inventory.test.mjs +198 -0
  102. package/lib/setup/sections/mandate.mjs +392 -0
  103. package/lib/setup/sections/mandate.test.mjs +373 -0
  104. package/lib/setup/sections/subagents.mjs +427 -0
  105. package/lib/setup/sections/subagents.test.mjs +429 -0
  106. package/lib/setup/sections/verify.mjs +121 -0
  107. package/lib/setup/sections/verify.test.mjs +175 -0
  108. package/lib/setup/sot.mjs +2 -0
  109. package/lib/subagents/cli.mjs +463 -0
  110. package/lib/subagents/cli.test.mjs +389 -0
  111. package/lib/subagents/client.mjs +373 -0
  112. package/lib/subagents/client.test.mjs +309 -0
  113. package/lib/subagents/gap.mjs +268 -0
  114. package/lib/subagents/gap.test.mjs +234 -0
  115. package/lib/subagents/lock.mjs +296 -0
  116. package/lib/subagents/lock.test.mjs +248 -0
  117. package/lib/subagents/manifest.mjs +224 -0
  118. package/lib/subagents/manifest.test.mjs +175 -0
  119. package/lib/subagents/refs.mjs +274 -0
  120. package/lib/subagents/refs.test.mjs +204 -0
  121. package/lib/subagents/resolve.mjs +455 -0
  122. package/lib/subagents/resolve.test.mjs +422 -0
  123. package/lib/subagents/schema.mjs +467 -0
  124. package/lib/subagents/schema.test.mjs +306 -0
  125. package/package.json +8 -3
  126. package/plugins/maestro-skills/.claude-plugin/marketplace.json +16 -0
  127. package/policies/ai-disclosure.yaml +42 -2
  128. package/scaffold/CLAUDE.md +16 -2
  129. package/schedules/triggers/goal-steward.md +79 -0
  130. package/scripts/ci/conformance-org-api.mjs +792 -0
  131. package/scripts/ci/conformance-org-api.test.mjs +417 -0
  132. package/scripts/daemon/agent-daemon.mjs +36 -4
  133. package/scripts/daemon/cadence-handlers.mjs +145 -1
  134. package/scripts/daemon/goal-steward-cadence.test.mjs +243 -0
  135. package/scripts/daemon/inbox-deferral.mjs +45 -2
  136. package/scripts/daemon/inbox-deferral.test.mjs +56 -0
  137. package/scripts/daemon/inbox-wake.mjs +282 -0
  138. package/scripts/daemon/inbox-wake.test.mjs +199 -0
  139. package/scripts/daemon/prompt-builder.mjs +41 -1
  140. package/scripts/daemon/typing-registry.mjs +55 -2
  141. package/scripts/daemon/typing-registry.test.mjs +25 -0
  142. package/scripts/local-triggers/generate-plists.test.mjs +5 -5
  143. package/scripts/setup/gen-subagent-manifest.mjs +95 -0
  144. package/scripts/setup/gen-subagent-manifest.test.mjs +124 -0
  145. package/scripts/setup/generate-plan.mjs +108 -0
  146. package/scripts/setup/init-capability-manifest.mjs +70 -0
  147. package/scripts/setup/init-skill-marketplace.mjs +155 -0
  148. package/scripts/setup/init-skill-marketplace.test.mjs +193 -0
@@ -0,0 +1,332 @@
1
+ /**
2
+ * lib/execution/route.mjs — the execution RUNG decision rule (SPEC §7).
3
+ *
4
+ * Once `disposition.mjs` has decided the agent should act, this module decides
5
+ * *how much machinery* the action deserves. The doctrine, stated once:
6
+ *
7
+ * Pick the LOWEST rung whose predicate holds.
8
+ * Escalate exactly ONE rung on the second failure of the same obligation.
9
+ * Never skip a rung.
10
+ * Never escalate past rung 3 without an adopted objective behind the work.
11
+ *
12
+ * The reason this is a module rather than a comment in the dispatcher is that
13
+ * today the choice is implicit: every inbound item that survives the gate gets a
14
+ * full Claude Code session (rung 3), whether it needed one protocol call or a
15
+ * week of research. That is both the dominant cost line and the dominant latency
16
+ * line, and it is why "answer a DM" and "rebuild the pipeline model" currently
17
+ * cost the same.
18
+ *
19
+ * Two gates cut ACROSS every rung and are applied after the rung is chosen:
20
+ * - blast radius: an `external | irreversible | financial` action class forces
21
+ * an approval BEFORE any rung executes, rung 0 included;
22
+ * - budget: an estimate over the remaining envelope forces an enqueue, never a
23
+ * silent downgrade to a cheaper rung (a cheaper rung would do the work
24
+ * WORSE, not cheaper — downgrading on cost is how agents produce garbage).
25
+ *
26
+ * PURE. Every input arrives on the argument; nothing is read from disk or the
27
+ * network. That is what lets `explain.mjs`-style tooling replay a routing
28
+ * decision offline from the journal alone.
29
+ *
30
+ * @module lib/execution/route
31
+ */
32
+
33
+ "use strict";
34
+
35
+ import { GATED_CLASSES } from "../org/approvals.mjs";
36
+
37
+ /**
38
+ * The six rungs. `id` is the numeric rung, `mechanism` names the machinery that
39
+ * actually runs it, `offline` says whether the rung survives with no network to
40
+ * hq (rungs that spawn sessions do; rungs that call protocol methods do not).
41
+ */
42
+ export const RUNGS = Object.freeze([
43
+ Object.freeze({
44
+ id: 0,
45
+ key: "function_call",
46
+ mechanism: "tool-surface",
47
+ label: "single protocol method",
48
+ offline: false,
49
+ }),
50
+ Object.freeze({
51
+ id: 1,
52
+ key: "skill",
53
+ mechanism: "claude-skill",
54
+ label: "bounded skill session",
55
+ offline: true,
56
+ }),
57
+ Object.freeze({
58
+ id: 2,
59
+ key: "plugin",
60
+ mechanism: "integration",
61
+ label: "external SaaS via integration plugin",
62
+ offline: false,
63
+ }),
64
+ Object.freeze({
65
+ id: 3,
66
+ key: "session",
67
+ mechanism: "dispatcher-spawn",
68
+ label: "Claude Code session",
69
+ offline: true,
70
+ }),
71
+ Object.freeze({
72
+ id: 4,
73
+ key: "workflow",
74
+ mechanism: "workflow-runner",
75
+ label: "multi-stage durable workflow",
76
+ offline: true,
77
+ }),
78
+ Object.freeze({
79
+ id: 5,
80
+ key: "team",
81
+ mechanism: "subagent-fanout",
82
+ label: "sub-agent team",
83
+ offline: true,
84
+ }),
85
+ ]);
86
+
87
+ /** Rung lookup by numeric id. */
88
+ export function rungById(id) {
89
+ return RUNGS.find((r) => r.id === id) || null;
90
+ }
91
+
92
+ /**
93
+ * `MAX_STEPS` in hq's `agent-runner.ts` — the natural in-chat ceiling, and the
94
+ * reason 6 rather than some rounder number is the rung-3 step threshold. Kept as
95
+ * a named constant so the coupling is visible instead of being a magic 6.
96
+ */
97
+ export const IN_CHAT_STEP_CEILING = 6;
98
+
99
+ /** The number of failures of the SAME obligation that buys exactly one rung. */
100
+ export const FAILURES_PER_RUNG = 2;
101
+
102
+ /**
103
+ * @typedef {Object} Need
104
+ * @property {string[]} [methods] protocol methods the action would call
105
+ * @property {boolean} [paramsKnown] are all params already resolved (no lookup)?
106
+ * @property {number} [steps] estimated tool calls
107
+ * @property {boolean} [filesystem] does it touch a repo / the filesystem?
108
+ * @property {boolean} [openEnded] is it research with no known stopping point?
109
+ * @property {boolean} [externalSaas] does it need a capability outside the protocol?
110
+ * @property {string} [skillId] a skill that covers this procedure
111
+ * @property {number} [stages] distinct stages with gates between them
112
+ * @property {boolean} [gated] are there approval gates BETWEEN the stages?
113
+ * @property {boolean} [durable] must the run survive a restart?
114
+ * @property {number} [subProblems] independent sub-problems
115
+ * @property {boolean} [sharedState] do the sub-problems share mutable state?
116
+ * @property {string[]} [actionClasses] blast-radius classes of the action
117
+ * @property {string} [obligationKey]
118
+ * @property {string} [objectiveId]
119
+ */
120
+
121
+ /**
122
+ * @typedef {Object} RouteContext
123
+ * @property {Array<{id:string, reachable:boolean}>} [manifest] capability manifest
124
+ * @property {number} [failures] prior failures of THIS obligation
125
+ * @property {number} [estCents] estimated cost of the action
126
+ * @property {number} [remainingCents] remaining obligation/seat envelope
127
+ * @property {boolean} [hasAdoptedObjective] is an adopted objective behind this?
128
+ * @property {number} [maxRung] hard ceiling (e.g. staleness ladder)
129
+ */
130
+
131
+ /** Is `id` present in the manifest AND reachable? Absent manifest ⇒ unknown, not false. */
132
+ function reachable(manifest, id) {
133
+ if (!Array.isArray(manifest) || !id) return null;
134
+ const hit = manifest.find((c) => c && String(c.id) === String(id));
135
+ if (!hit) return false;
136
+ return hit.reachable !== false;
137
+ }
138
+
139
+ /**
140
+ * The blast-radius classes of an action, normalised against the ONE gated-class
141
+ * vocabulary (`lib/org/approvals.mjs GATED_CLASSES`) so this module cannot drift
142
+ * into a second, parallel list of what "risky" means.
143
+ *
144
+ * @param {string[]|undefined} classes
145
+ * @returns {string[]} the gated subset, sorted
146
+ */
147
+ export function gatedClassesOf(classes) {
148
+ const out = new Set();
149
+ for (const c of [].concat(classes || [])) {
150
+ const k = String(c || "").toLowerCase();
151
+ if (GATED_CLASSES.includes(k)) out.add(k);
152
+ }
153
+ return [...out].sort();
154
+ }
155
+
156
+ /**
157
+ * Choose the lowest rung whose predicate holds, before any cross-cutting gate.
158
+ * Split out from {@link routeRung} so a test can assert the ladder itself
159
+ * without the gates muddying it.
160
+ *
161
+ * @param {Need} need
162
+ * @param {RouteContext} ctx
163
+ * @returns {{rung:number, reason:string, why:string[]}}
164
+ */
165
+ export function baseRung(need = {}, ctx = {}) {
166
+ const why = [];
167
+ // A default parameter only fires on `undefined`; an explicit `null` from a
168
+ // caller that had nothing to say must not throw here — this is the routing
169
+ // path for live inbound events.
170
+ if (!need || typeof need !== "object") need = {};
171
+ if (!ctx || typeof ctx !== "object") ctx = {};
172
+ const methods = [].concat(need.methods || []);
173
+ const steps = Number.isFinite(need.steps) ? need.steps : methods.length || 1;
174
+
175
+ // Rung 5 and 4 are structural: they describe the SHAPE of the work, and no
176
+ // amount of "it's only a small task" makes a 3-stage gated process fit in a
177
+ // single session. Tested first so a genuinely large job is not walked up from
178
+ // rung 0 one failure at a time.
179
+ const subProblems = Number.isFinite(need.subProblems) ? need.subProblems : 0;
180
+ if (subProblems >= 3 && need.sharedState !== true) {
181
+ why.push(`subProblems=${subProblems} independent → team`);
182
+ return { rung: 5, reason: "independent_sub_problems", why };
183
+ }
184
+ const stages = Number.isFinite(need.stages) ? need.stages : 0;
185
+ if (stages >= 3 && (need.durable === true || need.gated === true)) {
186
+ why.push(`stages=${stages} with gates and durability → workflow`);
187
+ return { rung: 4, reason: "multi_stage_durable", why };
188
+ }
189
+
190
+ // Rung 0 — one method, params already in hand, trivially short, no filesystem.
191
+ if (
192
+ methods.length === 1 &&
193
+ need.paramsKnown === true &&
194
+ steps <= 3 &&
195
+ need.filesystem !== true &&
196
+ need.openEnded !== true &&
197
+ need.externalSaas !== true
198
+ ) {
199
+ why.push(`single method ${methods[0]}, params known, ${steps} step(s) → function call`);
200
+ return { rung: 0, reason: "single_method", why };
201
+ }
202
+ why.push(
203
+ `rung 0 declined (methods=${methods.length}, paramsKnown=${need.paramsKnown === true}, steps=${steps}, filesystem=${need.filesystem === true})`,
204
+ );
205
+
206
+ // Rung 1 — a repeatable procedure with a REACHABLE skill behind it. The
207
+ // reachability check is what stops rung 1 being a lie: citing a skill file
208
+ // that the marketplace never registered means the session starts with no skill
209
+ // loaded and silently behaves like a bare rung-3 session.
210
+ if (
211
+ need.skillId &&
212
+ steps >= 2 &&
213
+ steps <= 10 &&
214
+ need.filesystem !== true &&
215
+ need.openEnded !== true &&
216
+ need.externalSaas !== true
217
+ ) {
218
+ const ok = reachable(ctx.manifest, need.skillId);
219
+ if (ok === false) {
220
+ why.push(`rung 1 declined: skill ${need.skillId} is not reachable in the manifest`);
221
+ } else {
222
+ why.push(`skill ${need.skillId} covers ${steps} step(s)${ok === null ? " (manifest absent — assumed reachable)" : ""}`);
223
+ return { rung: 1, reason: "skill_covers", why };
224
+ }
225
+ }
226
+
227
+ // Rung 2 — the capability lives outside the protocol entirely.
228
+ if (need.externalSaas === true && need.filesystem !== true) {
229
+ why.push("capability is external SaaS outside the protocol → integration plugin");
230
+ return { rung: 2, reason: "external_capability", why };
231
+ }
232
+
233
+ // Rung 3 — filesystem, or too long for chat, or open-ended.
234
+ if (need.filesystem === true) {
235
+ why.push("touches the filesystem → session");
236
+ return { rung: 3, reason: "filesystem", why };
237
+ }
238
+ if (steps > IN_CHAT_STEP_CEILING) {
239
+ why.push(`steps=${steps} > in-chat ceiling ${IN_CHAT_STEP_CEILING} → session`);
240
+ return { rung: 3, reason: "too_many_steps", why };
241
+ }
242
+ if (need.openEnded === true) {
243
+ why.push("open-ended research → session");
244
+ return { rung: 3, reason: "open_ended", why };
245
+ }
246
+
247
+ // Nothing matched cleanly: a handful of protocol calls with judgment in
248
+ // between. Rung 1 without a skill is not available, so the honest floor is a
249
+ // bounded session.
250
+ why.push("no lower predicate held → bounded session");
251
+ return { rung: 3, reason: "default_session", why };
252
+ }
253
+
254
+ /**
255
+ * Route a need to a rung, applying the failure walk-up and both cross-cutting
256
+ * gates. Never throws.
257
+ *
258
+ * @param {Need} need
259
+ * @param {RouteContext} ctx
260
+ * @returns {{
261
+ * rung:number, mechanism:string, key:string, reason:string, why:string[],
262
+ * blocked:boolean, gate:null|{kind:string, detail:object},
263
+ * approval:null|{required:true, classes:string[]},
264
+ * escalatedFrom:number|null
265
+ * }}
266
+ */
267
+ export function routeRung(need = {}, ctx = {}) {
268
+ if (!need || typeof need !== "object") need = {};
269
+ if (!ctx || typeof ctx !== "object") ctx = {};
270
+ const base = baseRung(need, ctx);
271
+ const why = base.why.slice();
272
+ let rung = base.rung;
273
+ let escalatedFrom = null;
274
+
275
+ // ── the failure walk-up: exactly one rung per FAILURES_PER_RUNG failures,
276
+ // and never a skip. Two failures at rung 0 buys rung 1, not rung 3.
277
+ const failures = Number.isFinite(ctx.failures) ? Math.max(0, ctx.failures) : 0;
278
+ if (failures >= FAILURES_PER_RUNG) {
279
+ const steps = Math.floor(failures / FAILURES_PER_RUNG);
280
+ const target = Math.min(rung + steps, 5);
281
+ if (target !== rung) {
282
+ why.push(`${failures} prior failure(s) of this obligation → walk up ${rung}→${target}`);
283
+ escalatedFrom = rung;
284
+ rung = target;
285
+ }
286
+ }
287
+
288
+ // ── "never escalate past rung 3 without an adopted objective behind the work"
289
+ if (rung > 3 && ctx.hasAdoptedObjective !== true) {
290
+ why.push(`rung ${rung} needs an adopted objective behind it; none → capped at 3`);
291
+ rung = 3;
292
+ }
293
+
294
+ // ── an externally imposed ceiling (the staleness ladder, a beat directive)
295
+ if (Number.isFinite(ctx.maxRung) && rung > ctx.maxRung) {
296
+ why.push(`ceiling maxRung=${ctx.maxRung} → capped from ${rung}`);
297
+ rung = Math.max(0, ctx.maxRung);
298
+ }
299
+
300
+ const def = rungById(rung) || RUNGS[3];
301
+
302
+ // ── gate 1: blast radius. Applied AFTER the rung is known so the approval
303
+ // card can name the mechanism, but BEFORE anything executes.
304
+ const classes = gatedClassesOf(need.actionClasses);
305
+ const approval = classes.length ? { required: true, classes } : null;
306
+ if (approval) why.push(`blast radius ${classes.join("+")} → approval required before rung ${rung} runs`);
307
+
308
+ // ── gate 2: budget. An over-envelope estimate ENQUEUES; it never quietly
309
+ // picks a cheaper rung, because a cheaper rung is a worse answer, not a
310
+ // cheaper one.
311
+ let gate = null;
312
+ const est = Number.isFinite(ctx.estCents) ? ctx.estCents : null;
313
+ const remaining = Number.isFinite(ctx.remainingCents) ? ctx.remainingCents : null;
314
+ if (est !== null && remaining !== null && est > remaining) {
315
+ gate = { kind: "budget", detail: { estCents: est, remainingCents: remaining } };
316
+ why.push(`estimate ${est}c exceeds remaining envelope ${remaining}c → enqueue, not downgrade`);
317
+ }
318
+
319
+ return {
320
+ rung,
321
+ mechanism: def.mechanism,
322
+ key: def.key,
323
+ reason: base.reason,
324
+ why,
325
+ blocked: gate !== null,
326
+ gate,
327
+ approval,
328
+ escalatedFrom,
329
+ };
330
+ }
331
+
332
+ export default { RUNGS, rungById, baseRung, routeRung, gatedClassesOf, IN_CHAT_STEP_CEILING, FAILURES_PER_RUNG };
@@ -0,0 +1,186 @@
1
+ /**
2
+ * route.test.mjs — the execution rung ladder (SPEC §7).
3
+ * Run: node --test lib/execution/route.test.mjs
4
+ */
5
+ "use strict";
6
+
7
+ import { test } from "node:test";
8
+ import assert from "node:assert/strict";
9
+
10
+ import { GATED_CLASSES } from "../org/approvals.mjs";
11
+ import {
12
+ RUNGS,
13
+ rungById,
14
+ baseRung,
15
+ routeRung,
16
+ gatedClassesOf,
17
+ IN_CHAT_STEP_CEILING,
18
+ FAILURES_PER_RUNG,
19
+ } from "./route.mjs";
20
+
21
+ const ONE_CALL = { methods: ["messaging.send"], paramsKnown: true, steps: 1 };
22
+
23
+ test("the ladder is six rungs, contiguous, each with a mechanism", () => {
24
+ assert.equal(RUNGS.length, 6);
25
+ RUNGS.forEach((r, i) => {
26
+ assert.equal(r.id, i);
27
+ assert.ok(r.mechanism && r.key && r.label);
28
+ });
29
+ assert.equal(rungById(3).key, "session");
30
+ assert.equal(rungById(99), null);
31
+ });
32
+
33
+ test("rung 0: one method, params known, short, no filesystem", () => {
34
+ const r = baseRung(ONE_CALL);
35
+ assert.equal(r.rung, 0);
36
+ assert.equal(r.reason, "single_method");
37
+ });
38
+
39
+ test("rung 0 is declined when any of its four predicates fails", () => {
40
+ assert.notEqual(baseRung({ ...ONE_CALL, paramsKnown: false }).rung, 0);
41
+ assert.notEqual(baseRung({ ...ONE_CALL, methods: ["a", "b"] }).rung, 0);
42
+ assert.notEqual(baseRung({ ...ONE_CALL, steps: 4 }).rung, 0);
43
+ assert.notEqual(baseRung({ ...ONE_CALL, filesystem: true }).rung, 0);
44
+ // and a bare need with nothing known does not sneak into rung 0
45
+ assert.notEqual(baseRung({}).rung, 0);
46
+ });
47
+
48
+ test("rung 1: a reachable skill covers a 2–10 step procedure", () => {
49
+ const need = { methods: ["a", "b", "c"], steps: 4, skillId: "triage-inbox" };
50
+ const ctx = { manifest: [{ id: "triage-inbox", reachable: true }] };
51
+ const r = baseRung(need, ctx);
52
+ assert.equal(r.rung, 1);
53
+ assert.equal(r.reason, "skill_covers");
54
+ });
55
+
56
+ test("rung 1 is DECLINED when the skill is not reachable — the marketplace lie", () => {
57
+ // Citing a skill the marketplace never registered means the session starts
58
+ // with no skill loaded. That must fall through, not pretend to be rung 1.
59
+ const need = { methods: ["a", "b", "c"], steps: 4, skillId: "triage-inbox" };
60
+ const r = baseRung(need, { manifest: [{ id: "triage-inbox", reachable: false }] });
61
+ assert.notEqual(r.rung, 1);
62
+ assert.ok(r.why.some((w) => /not reachable/.test(w)));
63
+ });
64
+
65
+ test("rung 1 with an ABSENT manifest is allowed, and says so", () => {
66
+ const need = { methods: ["a", "b"], steps: 3, skillId: "triage" };
67
+ const r = baseRung(need, {});
68
+ assert.equal(r.rung, 1);
69
+ assert.ok(r.why.some((w) => /manifest absent/.test(w)));
70
+ });
71
+
72
+ test("rung 1 declines a skill outside the 2–10 step band", () => {
73
+ assert.notEqual(baseRung({ steps: 1, skillId: "s", methods: ["a"], paramsKnown: true }).rung, 1);
74
+ assert.equal(baseRung({ steps: 40, skillId: "s" }).rung, 3);
75
+ });
76
+
77
+ test("rung 2: an external SaaS capability outside the protocol", () => {
78
+ const r = baseRung({ externalSaas: true, steps: 2 });
79
+ assert.equal(r.rung, 2);
80
+ assert.equal(r.reason, "external_capability");
81
+ });
82
+
83
+ test("rung 3: filesystem, too many steps, or open-ended", () => {
84
+ assert.equal(baseRung({ filesystem: true }).reason, "filesystem");
85
+ assert.equal(baseRung({ steps: IN_CHAT_STEP_CEILING + 1 }).reason, "too_many_steps");
86
+ assert.equal(baseRung({ openEnded: true, steps: 2 }).reason, "open_ended");
87
+ assert.equal(baseRung({ filesystem: true }).rung, 3);
88
+ });
89
+
90
+ test("rung 3 is the honest floor when nothing lower matches", () => {
91
+ const r = baseRung({ methods: ["a", "b"], steps: 3 });
92
+ assert.equal(r.rung, 3);
93
+ assert.equal(r.reason, "default_session");
94
+ });
95
+
96
+ test("rungs 4 and 5 are STRUCTURAL — tested before the cheap rungs", () => {
97
+ // A genuinely large job must not be walked up from rung 0 one failure at a time.
98
+ const workflow = baseRung({ stages: 4, gated: true, durable: true, methods: ["a"], paramsKnown: true, steps: 1 });
99
+ assert.equal(workflow.rung, 4);
100
+ const team = baseRung({ subProblems: 4, methods: ["a"], paramsKnown: true, steps: 1 });
101
+ assert.equal(team.rung, 5);
102
+ });
103
+
104
+ test("rung 5 declines when the sub-problems share mutable state", () => {
105
+ const r = baseRung({ subProblems: 5, sharedState: true, filesystem: true });
106
+ assert.equal(r.rung, 3);
107
+ });
108
+
109
+ test("rung 4 declines when the stages have neither gates nor durability", () => {
110
+ assert.notEqual(baseRung({ stages: 5, steps: 3, methods: ["a", "b"] }).rung, 4);
111
+ });
112
+
113
+ test("failure walk-up: one rung per two failures, and never a skip", () => {
114
+ const need = { ...ONE_CALL };
115
+ assert.equal(routeRung(need, { failures: 0 }).rung, 0);
116
+ assert.equal(routeRung(need, { failures: 1 }).rung, 0, "one failure buys nothing");
117
+ assert.equal(routeRung(need, { failures: 2 }).rung, 1);
118
+ assert.equal(routeRung(need, { failures: 3 }).rung, 1);
119
+ assert.equal(routeRung(need, { failures: 4 }).rung, 2);
120
+ assert.equal(routeRung(need, { failures: 2 }).escalatedFrom, 0);
121
+ assert.equal(FAILURES_PER_RUNG, 2);
122
+ });
123
+
124
+ test("the walk-up cannot pass rung 3 without an adopted objective", () => {
125
+ const need = { ...ONE_CALL };
126
+ const capped = routeRung(need, { failures: 20, hasAdoptedObjective: false });
127
+ assert.equal(capped.rung, 3);
128
+ assert.ok(capped.why.some((w) => /adopted objective/.test(w)));
129
+
130
+ const allowed = routeRung(need, { failures: 20, hasAdoptedObjective: true });
131
+ assert.ok(allowed.rung > 3);
132
+ });
133
+
134
+ test("an external ceiling (staleness ladder / beat directive) caps the rung", () => {
135
+ const r = routeRung({ filesystem: true }, { maxRung: 1 });
136
+ assert.equal(r.rung, 1);
137
+ assert.ok(r.why.some((w) => /ceiling maxRung=1/.test(w)));
138
+ // maxRung 0 is a real value, not a falsy no-op
139
+ assert.equal(routeRung({ filesystem: true }, { maxRung: 0 }).rung, 0);
140
+ });
141
+
142
+ test("blast radius forces an approval BEFORE any rung — rung 0 included", () => {
143
+ const r = routeRung({ ...ONE_CALL, actionClasses: ["external"] });
144
+ assert.equal(r.rung, 0, "the gate does not change the rung");
145
+ assert.deepEqual(r.approval, { required: true, classes: ["external"] });
146
+ assert.ok(r.why.some((w) => /approval required before rung 0/.test(w)));
147
+ });
148
+
149
+ test("no gated class ⇒ no approval gate", () => {
150
+ assert.equal(routeRung(ONE_CALL).approval, null);
151
+ assert.equal(routeRung({ ...ONE_CALL, actionClasses: [] }).approval, null);
152
+ assert.equal(routeRung({ ...ONE_CALL, actionClasses: ["chatty"] }).approval, null);
153
+ });
154
+
155
+ test("gatedClassesOf normalises against the ONE gated-class vocabulary", () => {
156
+ assert.deepEqual(gatedClassesOf(["EXTERNAL", "financial", "nonsense"]), ["external", "financial"]);
157
+ assert.deepEqual(gatedClassesOf(undefined), []);
158
+ assert.deepEqual(gatedClassesOf("irreversible"), ["irreversible"]);
159
+ assert.deepEqual(gatedClassesOf(["external", "external"]), ["external"], "deduped");
160
+ // the vocabulary is not re-declared locally
161
+ for (const c of GATED_CLASSES) assert.deepEqual(gatedClassesOf([c]), [c]);
162
+ });
163
+
164
+ test("over-budget BLOCKS with an enqueue signal — it never downgrades the rung", () => {
165
+ const r = routeRung({ filesystem: true }, { estCents: 900, remainingCents: 100 });
166
+ assert.equal(r.blocked, true);
167
+ assert.equal(r.gate.kind, "budget");
168
+ assert.equal(r.rung, 3, "a cheaper rung would be a worse answer, not a cheaper one");
169
+ });
170
+
171
+ test("budget gate is inert when either side of the comparison is unknown", () => {
172
+ assert.equal(routeRung(ONE_CALL, { estCents: 900 }).blocked, false);
173
+ assert.equal(routeRung(ONE_CALL, { remainingCents: 10 }).blocked, false);
174
+ assert.equal(routeRung(ONE_CALL, {}).blocked, false);
175
+ // exactly-at-budget is allowed; only strictly-over blocks
176
+ assert.equal(routeRung(ONE_CALL, { estCents: 100, remainingCents: 100 }).blocked, false);
177
+ });
178
+
179
+ test("routeRung never throws and always names a mechanism", () => {
180
+ for (const need of [undefined, null, {}, { methods: null }, { steps: NaN }, { subProblems: "x" }]) {
181
+ const r = routeRung(need, {});
182
+ assert.ok(Number.isInteger(r.rung) && r.rung >= 0 && r.rung <= 5);
183
+ assert.ok(r.mechanism, "every routing result names its mechanism");
184
+ assert.ok(Array.isArray(r.why) && r.why.length > 0, "and explains itself");
185
+ }
186
+ });