@cohortapp/agent-sdk 2.17.0 → 2.18.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (531) hide show
  1. package/.claude/settings.json +18 -0
  2. package/.env.example +18 -5
  3. package/README.md +1 -0
  4. package/bin/maestro.mjs +62 -0
  5. package/docs/guides/billing-console-keys.md +60 -0
  6. package/docs/guides/front-door-session.md +54 -9
  7. package/docs/guides/mac-mini.md +20 -25
  8. package/docs/guides/setup-wizard.md +1 -1
  9. package/docs/runbooks/fleet-rollout.md +156 -0
  10. package/docs/runbooks/mac-mini-bootstrap.md +12 -14
  11. package/lib/action-executor.js +19 -3
  12. package/lib/budget-guard.mjs +279 -3
  13. package/lib/channels/base-adapter.mjs +3 -1
  14. package/lib/channels/contract.mjs +2 -1
  15. package/lib/channels/inbox-item.mjs +8 -0
  16. package/lib/claude-bin.mjs +5 -6
  17. package/lib/cli/doctor-checks.mjs +141 -10
  18. package/lib/cli/global-setup-extras.mjs +5 -1
  19. package/lib/cli/inbox.mjs +100 -15
  20. package/lib/cli/seat-auth.mjs +463 -0
  21. package/lib/cli/session.mjs +80 -12
  22. package/lib/collective/capture-slots.mjs +234 -0
  23. package/lib/collective/capture.mjs +8 -6
  24. package/lib/collective/config.mjs +2 -0
  25. package/lib/collective/global-config.mjs +63 -1
  26. package/lib/collective/loop-guard.mjs +155 -0
  27. package/lib/collective/presence.mjs +142 -5
  28. package/lib/comms/send-gate.mjs +559 -1
  29. package/lib/diagnostics/alerts.mjs +49 -0
  30. package/lib/diagnostics/cadence-output-freshness.mjs +288 -0
  31. package/lib/engine/agents/definitions.mjs +343 -0
  32. package/lib/engine/agents/persist.mjs +275 -0
  33. package/lib/engine/agents/runtime.mjs +748 -0
  34. package/lib/engine/agents/usage.mjs +95 -0
  35. package/lib/engine/auth-status.mjs +139 -0
  36. package/lib/engine/budget.mjs +194 -0
  37. package/lib/engine/cli.mjs +1204 -0
  38. package/lib/engine/commands/index.mjs +269 -0
  39. package/lib/engine/context/budget.mjs +219 -0
  40. package/lib/engine/context/cache.mjs +125 -0
  41. package/lib/engine/context/child-env.mjs +215 -0
  42. package/lib/engine/context/compaction.mjs +342 -0
  43. package/lib/engine/context/images.mjs +90 -0
  44. package/lib/engine/context/instructions.mjs +327 -0
  45. package/lib/engine/context/lazy-instructions.mjs +169 -0
  46. package/lib/engine/context/manager.mjs +182 -0
  47. package/lib/engine/context/real-path.mjs +91 -0
  48. package/lib/engine/context/secret-values.mjs +163 -0
  49. package/lib/engine/context/settings.mjs +274 -0
  50. package/lib/engine/context/stream-input.mjs +159 -0
  51. package/lib/engine/guard.mjs +152 -0
  52. package/lib/engine/hooks.mjs +713 -0
  53. package/lib/engine/loop.mjs +560 -0
  54. package/lib/engine/mcp/client.mjs +254 -0
  55. package/lib/engine/mcp/config.mjs +301 -0
  56. package/lib/engine/mcp/http.mjs +201 -0
  57. package/lib/engine/mcp/index.mjs +146 -0
  58. package/lib/engine/mcp/jsonrpc.mjs +147 -0
  59. package/lib/engine/mcp/naming.mjs +66 -0
  60. package/lib/engine/mcp/resources.mjs +89 -0
  61. package/lib/engine/mcp/results.mjs +133 -0
  62. package/lib/engine/mcp/stdio.mjs +137 -0
  63. package/lib/engine/mcp/supervisor.mjs +116 -0
  64. package/lib/engine/messages.mjs +104 -0
  65. package/lib/engine/output/json.mjs +164 -0
  66. package/lib/engine/output/stream-json.mjs +266 -0
  67. package/lib/engine/permissions.mjs +845 -0
  68. package/lib/engine/process-identity.mjs +164 -0
  69. package/lib/engine/process-tree.mjs +551 -0
  70. package/lib/engine/prompt.mjs +60 -0
  71. package/lib/engine/session/store.mjs +299 -0
  72. package/lib/engine/session-runtime/args.mjs +97 -0
  73. package/lib/engine/session-runtime/host.mjs +143 -0
  74. package/lib/engine/session-runtime/inbox.mjs +122 -0
  75. package/lib/engine/session-runtime/notifications.mjs +129 -0
  76. package/lib/engine/session-runtime/registry.mjs +328 -0
  77. package/lib/engine/session-runtime/runner.mjs +344 -0
  78. package/lib/engine/session-runtime/socket.mjs +212 -0
  79. package/lib/engine/session-runtime/wakeup.mjs +115 -0
  80. package/lib/engine/skills/index.mjs +321 -0
  81. package/lib/engine/tools/bash-background.mjs +533 -0
  82. package/lib/engine/tools/bash.mjs +216 -0
  83. package/lib/engine/tools/edit.mjs +97 -0
  84. package/lib/engine/tools/glob.mjs +81 -0
  85. package/lib/engine/tools/grep.mjs +224 -0
  86. package/lib/engine/tools/index.mjs +84 -0
  87. package/lib/engine/tools/list-agents.mjs +32 -0
  88. package/lib/engine/tools/ls.mjs +127 -0
  89. package/lib/engine/tools/monitor.mjs +82 -0
  90. package/lib/engine/tools/notebook-edit.mjs +218 -0
  91. package/lib/engine/tools/read.mjs +103 -0
  92. package/lib/engine/tools/schedule-wakeup.mjs +45 -0
  93. package/lib/engine/tools/schema.mjs +144 -0
  94. package/lib/engine/tools/send-message.mjs +77 -0
  95. package/lib/engine/tools/session.mjs +70 -0
  96. package/lib/engine/tools/todo.mjs +144 -0
  97. package/lib/engine/tools/toolsearch.mjs +217 -0
  98. package/lib/engine/tools/walk.mjs +193 -0
  99. package/lib/engine/tools/web-switch.mjs +31 -0
  100. package/lib/engine/tools/webfetch-html.mjs +387 -0
  101. package/lib/engine/tools/webfetch-net.mjs +340 -0
  102. package/lib/engine/tools/webfetch.mjs +198 -0
  103. package/lib/engine/tools/websearch.mjs +91 -0
  104. package/lib/engine/tools/workflow.mjs +95 -0
  105. package/lib/engine/tools/write.mjs +76 -0
  106. package/lib/engine/tui/line-editor.mjs +137 -0
  107. package/lib/engine/tui/render.mjs +86 -0
  108. package/lib/engine/tui/tui.mjs +274 -0
  109. package/lib/engine/wire/anthropic-messages.mjs +263 -0
  110. package/lib/engine/wire/effort.mjs +36 -0
  111. package/lib/engine/wire/errors.mjs +496 -0
  112. package/lib/engine/wire/http.mjs +441 -0
  113. package/lib/engine/wire/index.mjs +76 -0
  114. package/lib/engine/wire/openai-chat.mjs +332 -0
  115. package/lib/engine/wire/prompt-cache.mjs +79 -0
  116. package/lib/engine/wire/search.mjs +140 -0
  117. package/lib/engine/wire/sse.mjs +114 -0
  118. package/lib/engine/wire/stall.mjs +349 -0
  119. package/lib/engine/wire/token-provider.mjs +175 -0
  120. package/lib/engine/wire/usage.mjs +192 -0
  121. package/lib/engine/workflow/host.mjs +524 -0
  122. package/lib/engine/workflow/journal.mjs +188 -0
  123. package/lib/engine/workflow/json-schema.mjs +171 -0
  124. package/lib/engine/workflow/meta.mjs +329 -0
  125. package/lib/engine/workflow/notifications.mjs +52 -0
  126. package/lib/engine/workflow/runtime.mjs +447 -0
  127. package/lib/engine/workflow/sandbox.mjs +534 -0
  128. package/lib/engine/workflow/worker.mjs +141 -0
  129. package/lib/engine/workflow/worktree.mjs +74 -0
  130. package/lib/execution/disposition.mjs +1 -1
  131. package/lib/execution/intake.mjs +10 -0
  132. package/lib/execution/surface-policy.mjs +15 -0
  133. package/lib/learning/curator.mjs +8 -6
  134. package/lib/learning/reflect.mjs +8 -6
  135. package/lib/model-router/catalog/cohort.yaml +137 -0
  136. package/lib/model-router/catalog.mjs +118 -1
  137. package/lib/model-router/failover.mjs +67 -16
  138. package/lib/model-router/llm-task.mjs +39 -3
  139. package/lib/model-router/resolve.mjs +89 -3
  140. package/lib/model-router/spawn.mjs +46 -47
  141. package/lib/model-router/taxonomy.mjs +126 -4
  142. package/lib/org/cost-sync.mjs +141 -11
  143. package/lib/org/inbound/broadcast.mjs +289 -0
  144. package/lib/org/inbound/collective.mjs +375 -0
  145. package/lib/org/inbound/directedness.mjs +96 -8
  146. package/lib/org/inbound/facts.mjs +78 -2
  147. package/lib/org/inbound/project.mjs +22 -0
  148. package/lib/org/inbound/surfaces.mjs +14 -0
  149. package/lib/org/llm-token.mjs +879 -0
  150. package/lib/org/mesh.mjs +61 -0
  151. package/lib/org/messaging.mjs +3 -1
  152. package/lib/org/protocol.checksum +1 -1
  153. package/lib/org/protocol.mjs +15 -0
  154. package/lib/org/quota.mjs +520 -0
  155. package/lib/org/tool-surface.mjs +104 -16
  156. package/lib/org/ui-parity.mjs +16 -1
  157. package/lib/org/work-ledger.mjs +37 -6
  158. package/lib/rate-guard.mjs +114 -1
  159. package/lib/resource-governor.mjs +41 -6
  160. package/lib/runtime/adapter.mjs +833 -0
  161. package/lib/runtime/child-env.mjs +191 -0
  162. package/lib/runtime/legacy-shell-guard.mjs +97 -0
  163. package/lib/runtime/seat-engine.mjs +162 -0
  164. package/lib/session/ask-ledger.mjs +271 -0
  165. package/lib/session/current-work.mjs +676 -0
  166. package/lib/session/feed-core.mjs +40 -3
  167. package/lib/session/launch-args.mjs +56 -4
  168. package/lib/session/status-summary.mjs +26 -9
  169. package/lib/session/upgrade-notice.mjs +42 -0
  170. package/lib/setup/claude-probe.mjs +117 -13
  171. package/lib/setup/enrich.mjs +13 -10
  172. package/lib/setup/sections/model.mjs +39 -13
  173. package/lib/telemetry/collect.mjs +229 -11
  174. package/lib/upgrade/ignored-drift.mjs +105 -0
  175. package/lib/voice/post-call-brief.mjs +30 -17
  176. package/package.json +13 -3
  177. package/plugins/maestro-skills/skills/board-work.md +5 -0
  178. package/plugins/maestro-skills/skills/inbound-triage.md +56 -15
  179. package/plugins/maestro-skills/skills/main-session.md +18 -7
  180. package/scaffold/config/collective.yaml +7 -0
  181. package/scripts/ci/check-durable-write-seam.mjs +3 -1
  182. package/scripts/ci/check-tarball-fidelity.mjs +126 -2
  183. package/scripts/ci/run-tests.mjs +47 -19
  184. package/scripts/cohort-llm/api-key-helper.mjs +92 -0
  185. package/scripts/collective/hook-runner.mjs +142 -19
  186. package/scripts/continuous-monitor.sh +13 -0
  187. package/scripts/cost/track-claude-usage.mjs +15 -0
  188. package/scripts/daemon/agent-daemon.mjs +408 -20
  189. package/scripts/daemon/assurance.mjs +48 -12
  190. package/scripts/daemon/cadence-consumer.mjs +218 -68
  191. package/scripts/daemon/cadence-handlers.mjs +73 -4
  192. package/scripts/daemon/classifier.mjs +75 -26
  193. package/scripts/daemon/context-compiler.mjs +51 -37
  194. package/scripts/daemon/deliver.mjs +30 -1
  195. package/scripts/daemon/dispatcher.mjs +595 -149
  196. package/scripts/daemon/health.mjs +14 -1
  197. package/scripts/daemon/maestro-daemon.mjs +11 -0
  198. package/scripts/daemon/prompt-builder.mjs +24 -0
  199. package/scripts/daemon/responder.mjs +246 -79
  200. package/scripts/daemon/sdk-version.mjs +98 -16
  201. package/scripts/eval/probe-gateway.mjs +635 -0
  202. package/scripts/eval/replay/extract.mjs +270 -0
  203. package/scripts/eval/replay/grade.mjs +260 -0
  204. package/scripts/eval/replay/lib/config.mjs +50 -0
  205. package/scripts/eval/replay/lib/effects.mjs +65 -0
  206. package/scripts/eval/replay/lib/fixture.mjs +188 -0
  207. package/scripts/eval/replay/lib/judge.mjs +72 -0
  208. package/scripts/eval/replay/lib/redact.mjs +136 -0
  209. package/scripts/eval/replay/lib/sandbox.mjs +170 -0
  210. package/scripts/eval/replay/lib/schema-check.mjs +63 -0
  211. package/scripts/eval/replay/lib/transcript.mjs +76 -0
  212. package/scripts/eval/replay/mcp-replay-stub.mjs +101 -0
  213. package/scripts/eval/replay/report.mjs +185 -0
  214. package/scripts/eval/replay/run.mjs +404 -0
  215. package/scripts/fleet/rollout.mjs +1151 -0
  216. package/scripts/hooks/pre-send-audit.sh +36 -245
  217. package/scripts/hooks/pre-write-yaml-validate.mjs +275 -0
  218. package/scripts/hooks/validate-state-yaml.sh +190 -0
  219. package/scripts/huddle/huddle-llm.mjs +361 -0
  220. package/scripts/huddle/huddle-server.mjs +46 -121
  221. package/scripts/local-triggers/autoupdate.sh +465 -81
  222. package/scripts/local-triggers/run-trigger.sh +13 -0
  223. package/scripts/maintenance/pin-integrity.mjs +364 -0
  224. package/scripts/poll-slack-events.sh +41 -9
  225. package/scripts/poller/slack-socket-mode.mjs +28 -3
  226. package/scripts/session/supervisor.mjs +80 -13
  227. package/scripts/spawn-session.sh +13 -0
  228. package/bin/maestro.test.mjs +0 -1574
  229. package/lib/action-executor.test.mjs +0 -871
  230. package/lib/archetype.test.mjs +0 -132
  231. package/lib/assurance/plan-note.test.mjs +0 -234
  232. package/lib/assurance/room-budget.test.mjs +0 -486
  233. package/lib/assurance/tier.test.mjs +0 -174
  234. package/lib/autonomy.test.mjs +0 -66
  235. package/lib/backlog.test.mjs +0 -302
  236. package/lib/backup/policy.test.mjs +0 -305
  237. package/lib/budget-escalate.test.mjs +0 -232
  238. package/lib/budget-guard.envelope.test.mjs +0 -476
  239. package/lib/budget-guard.test.mjs +0 -427
  240. package/lib/cadence-bus-requeue.test.mjs +0 -83
  241. package/lib/cadence-bus-schedule.test.mjs +0 -194
  242. package/lib/cadence-bus.test.mjs +0 -720
  243. package/lib/cadences.test.mjs +0 -230
  244. package/lib/capability/inventory.test.mjs +0 -232
  245. package/lib/capability.test.mjs +0 -78
  246. package/lib/channels/base-adapter.test.mjs +0 -590
  247. package/lib/channels/channels.test.mjs +0 -371
  248. package/lib/channels/contract.test.mjs +0 -162
  249. package/lib/channels/inbox-item.test.mjs +0 -368
  250. package/lib/channels/orgmail/adapter.test.mjs +0 -448
  251. package/lib/channels/pairing.test.mjs +0 -270
  252. package/lib/channels/repeat-suppressor.test.mjs +0 -134
  253. package/lib/channels/slack-adapter.test.mjs +0 -212
  254. package/lib/channels/telegram-adapter.test.mjs +0 -306
  255. package/lib/channels/voice/adapter.test.mjs +0 -278
  256. package/lib/channels/whatsapp/adapter-baileys.test.mjs +0 -359
  257. package/lib/channels/whatsapp/baileys-typing.test.mjs +0 -154
  258. package/lib/charter.test.mjs +0 -89
  259. package/lib/claude-bin.test.mjs +0 -131
  260. package/lib/cli/board.test.mjs +0 -227
  261. package/lib/cli/design.test.mjs +0 -270
  262. package/lib/cli/doctor-checks.test.mjs +0 -336
  263. package/lib/cli/global-setup-extras.test.mjs +0 -462
  264. package/lib/cli/inbox.test.mjs +0 -230
  265. package/lib/cli/session-ack.test.mjs +0 -63
  266. package/lib/cli/session.test.mjs +0 -613
  267. package/lib/collective/capture.test.mjs +0 -121
  268. package/lib/collective/cards.test.mjs +0 -114
  269. package/lib/collective/config.test.mjs +0 -123
  270. package/lib/collective/global-config.test.mjs +0 -220
  271. package/lib/collective/global-skills.test.mjs +0 -126
  272. package/lib/collective/presence.test.mjs +0 -95
  273. package/lib/collective/recall.test.mjs +0 -116
  274. package/lib/collective/vendor-skills.test.mjs +0 -306
  275. package/lib/comms/send-gate.test.mjs +0 -770
  276. package/lib/comms.test.mjs +0 -41
  277. package/lib/context/budget.test.mjs +0 -252
  278. package/lib/context/history-scope.test.mjs +0 -79
  279. package/lib/cost/ledger-row.test.mjs +0 -183
  280. package/lib/design/design-md.test.mjs +0 -318
  281. package/lib/design/fixtures/DESIGN.golden.md +0 -238
  282. package/lib/design/fixtures/PRODUCT.golden.md +0 -67
  283. package/lib/design/fixtures/foundation.json +0 -133
  284. package/lib/design/refresh-gate.test.mjs +0 -144
  285. package/lib/design/write.test.mjs +0 -241
  286. package/lib/diagnostics/alerts.test.mjs +0 -318
  287. package/lib/diagnostics/backup-freshness.test.mjs +0 -185
  288. package/lib/diagnostics/counters.test.mjs +0 -206
  289. package/lib/diagnostics/events.test.mjs +0 -290
  290. package/lib/diagnostics/otel.test.mjs +0 -196
  291. package/lib/diagnostics/trace.test.mjs +0 -251
  292. package/lib/env-compat.test.mjs +0 -104
  293. package/lib/execution/disposition.test.mjs +0 -553
  294. package/lib/execution/drive.test.mjs +0 -270
  295. package/lib/execution/effects.test.mjs +0 -344
  296. package/lib/execution/intake.test.mjs +0 -389
  297. package/lib/execution/journal.test.mjs +0 -261
  298. package/lib/execution/match.test.mjs +0 -235
  299. package/lib/execution/pipeline.test.mjs +0 -392
  300. package/lib/execution/route.test.mjs +0 -186
  301. package/lib/execution/surface-policy.test.mjs +0 -162
  302. package/lib/fs-atomic.test.mjs +0 -72
  303. package/lib/fs-ownership.test.mjs +0 -158
  304. package/lib/goals/admission.test.mjs +0 -164
  305. package/lib/goals/classify.test.mjs +0 -167
  306. package/lib/goals/collaborate.test.mjs +0 -336
  307. package/lib/goals/gaps.test.mjs +0 -284
  308. package/lib/goals/loop.test.mjs +0 -845
  309. package/lib/hooks/bus.test.mjs +0 -387
  310. package/lib/identity/persona.test.mjs +0 -142
  311. package/lib/kpi-sensors.test.mjs +0 -278
  312. package/lib/kpi.test.mjs +0 -244
  313. package/lib/learning/config.test.mjs +0 -75
  314. package/lib/learning/counters.test.mjs +0 -69
  315. package/lib/learning/curator-consolidate.test.mjs +0 -238
  316. package/lib/learning/curator.test.mjs +0 -106
  317. package/lib/learning/reflect.test.mjs +0 -0
  318. package/lib/learning/session-index.test.mjs +0 -125
  319. package/lib/learning/skill-writer.test.mjs +0 -210
  320. package/lib/mandate/audit.test.mjs +0 -195
  321. package/lib/mandate/contract.test.mjs +0 -185
  322. package/lib/mandate/derive.test.mjs +0 -274
  323. package/lib/mandate/model.test.mjs +0 -164
  324. package/lib/mandate/refresh.test.mjs +0 -389
  325. package/lib/mcp/server.test.mjs +0 -426
  326. package/lib/model-router/auth-profiles.test.mjs +0 -580
  327. package/lib/model-router/catalog.test.mjs +0 -385
  328. package/lib/model-router/economics.test.mjs +0 -438
  329. package/lib/model-router/failover.test.mjs +0 -439
  330. package/lib/model-router/health.test.mjs +0 -338
  331. package/lib/model-router/integration-coverage.test.mjs +0 -831
  332. package/lib/model-router/integration.test.mjs +0 -564
  333. package/lib/model-router/ledger.test.mjs +0 -415
  334. package/lib/model-router/llm-task.test.mjs +0 -392
  335. package/lib/model-router/org-credentials.test.mjs +0 -265
  336. package/lib/model-router/pricing-refresh.test.mjs +0 -286
  337. package/lib/model-router/reconcile.test.mjs +0 -316
  338. package/lib/model-router/repair.test.mjs +0 -180
  339. package/lib/model-router/spawn.test.mjs +0 -446
  340. package/lib/model-router/taxonomy.test.mjs +0 -410
  341. package/lib/model-router.test.mjs +0 -1207
  342. package/lib/org/activity.test.mjs +0 -134
  343. package/lib/org/approvals.test.mjs +0 -216
  344. package/lib/org/awareness.test.mjs +0 -159
  345. package/lib/org/board-mine-cache.test.mjs +0 -53
  346. package/lib/org/board.test.mjs +0 -187
  347. package/lib/org/bootstrap-context.test.mjs +0 -153
  348. package/lib/org/client.test.mjs +0 -1206
  349. package/lib/org/cohort-client.test.mjs +0 -126
  350. package/lib/org/cost-sync.test.mjs +0 -153
  351. package/lib/org/doctor.test.mjs +0 -346
  352. package/lib/org/engagement-ledger.test.mjs +0 -112
  353. package/lib/org/engagement.test.mjs +0 -739
  354. package/lib/org/handoff.test.mjs +0 -269
  355. package/lib/org/inbound/directedness.test.mjs +0 -668
  356. package/lib/org/inbound/facts.test.mjs +0 -471
  357. package/lib/org/inbound/hydrate.test.mjs +0 -908
  358. package/lib/org/inbound/index.test.mjs +0 -429
  359. package/lib/org/inbound/project.test.mjs +0 -287
  360. package/lib/org/integration-tools.test.mjs +0 -160
  361. package/lib/org/keys.test.mjs +0 -92
  362. package/lib/org/knowledge.test.mjs +0 -326
  363. package/lib/org/leases.test.mjs +0 -235
  364. package/lib/org/mesh-directives.test.mjs +0 -110
  365. package/lib/org/mesh-integration.test.mjs +0 -127
  366. package/lib/org/mesh.test.mjs +0 -400
  367. package/lib/org/messaging.test.mjs +0 -471
  368. package/lib/org/param-contract.test.mjs +0 -477
  369. package/lib/org/policy.test.mjs +0 -237
  370. package/lib/org/protocol.checksum.test.mjs +0 -90
  371. package/lib/org/protocol.test.mjs +0 -323
  372. package/lib/org/push.test.mjs +0 -792
  373. package/lib/org/registry.test.mjs +0 -100
  374. package/lib/org/resource-tools.test.mjs +0 -361
  375. package/lib/org/tool-access.test.mjs +0 -144
  376. package/lib/org/tool-surface-integration.test.mjs +0 -120
  377. package/lib/org/tool-surface.test.mjs +0 -1268
  378. package/lib/org/typing.test.mjs +0 -291
  379. package/lib/org/ui-parity.test.mjs +0 -560
  380. package/lib/org/verify.test.mjs +0 -194
  381. package/lib/org/work-ledger.test.mjs +0 -273
  382. package/lib/plan/adoption-e2e.test.mjs +0 -366
  383. package/lib/plan/budget-enforcement.test.mjs +0 -400
  384. package/lib/plan/compile.test.mjs +0 -382
  385. package/lib/plan/emit.test.mjs +0 -269
  386. package/lib/plan/explain.test.mjs +0 -188
  387. package/lib/prompts/parallelism.test.mjs +0 -177
  388. package/lib/rag/rag.test.mjs +0 -505
  389. package/lib/rate-guard.test.mjs +0 -272
  390. package/lib/reactive-gate.test.mjs +0 -57
  391. package/lib/render.test.mjs +0 -68
  392. package/lib/resource-governor.test.mjs +0 -488
  393. package/lib/scheduling/dynamic-jobs.test.mjs +0 -344
  394. package/lib/scheduling/jitter.test.mjs +0 -140
  395. package/lib/secrets/broker.test.mjs +0 -280
  396. package/lib/secrets/providers.test.mjs +0 -274
  397. package/lib/security/audit-engine.test.mjs +0 -424
  398. package/lib/security/coerce-args.test.mjs +0 -281
  399. package/lib/security/dangerous-tools.test.mjs +0 -68
  400. package/lib/security/external-content.test.mjs +0 -84
  401. package/lib/security/redact.test.mjs +0 -441
  402. package/lib/security/secret-equal.test.mjs +0 -55
  403. package/lib/session/config.test.mjs +0 -92
  404. package/lib/session/feed-core.test.mjs +0 -198
  405. package/lib/session/first-run.test.mjs +0 -121
  406. package/lib/session/frontdoor.test.mjs +0 -205
  407. package/lib/session/handoffs.test.mjs +0 -183
  408. package/lib/session/identity.test.mjs +0 -180
  409. package/lib/session/inbox-claims.test.mjs +0 -286
  410. package/lib/session/launch-args.test.mjs +0 -157
  411. package/lib/session/liveness.test.mjs +0 -100
  412. package/lib/session/status-summary.test.mjs +0 -118
  413. package/lib/session-permissions.test.mjs +0 -120
  414. package/lib/setup/claude-probe.test.mjs +0 -187
  415. package/lib/setup/completeness.test.mjs +0 -110
  416. package/lib/setup/context-pack.test.mjs +0 -89
  417. package/lib/setup/enrich.test.mjs +0 -115
  418. package/lib/setup/enroll-from-cohort.test.mjs +0 -300
  419. package/lib/setup/integration.test.mjs +0 -162
  420. package/lib/setup/io.test.mjs +0 -77
  421. package/lib/setup/runner.test.mjs +0 -132
  422. package/lib/setup/sections/identity.test.mjs +0 -234
  423. package/lib/setup/sections/inventory.test.mjs +0 -198
  424. package/lib/setup/sections/learning.test.mjs +0 -81
  425. package/lib/setup/sections/mandate.test.mjs +0 -388
  426. package/lib/setup/sections/messaging.test.mjs +0 -127
  427. package/lib/setup/sections/model.test.mjs +0 -240
  428. package/lib/setup/sections/org.test.mjs +0 -346
  429. package/lib/setup/sections/orgmail.test.mjs +0 -118
  430. package/lib/setup/sections/recovery.test.mjs +0 -98
  431. package/lib/setup/sections/subagents.test.mjs +0 -429
  432. package/lib/setup/sections/verify.test.mjs +0 -175
  433. package/lib/setup/sot.test.mjs +0 -81
  434. package/lib/setup/state.test.mjs +0 -115
  435. package/lib/singleton.test.mjs +0 -151
  436. package/lib/subagents/cli.test.mjs +0 -389
  437. package/lib/subagents/client.test.mjs +0 -309
  438. package/lib/subagents/gap.test.mjs +0 -234
  439. package/lib/subagents/lock.test.mjs +0 -248
  440. package/lib/subagents/manifest.test.mjs +0 -175
  441. package/lib/subagents/refs.test.mjs +0 -204
  442. package/lib/subagents/resolve.test.mjs +0 -422
  443. package/lib/subagents/schema.test.mjs +0 -328
  444. package/lib/telemetry/alerts.test.mjs +0 -109
  445. package/lib/telemetry/collect.test.mjs +0 -1274
  446. package/lib/tool-definitions-integration.test.mjs +0 -83
  447. package/lib/tool-definitions.test.mjs +0 -437
  448. package/lib/upgrade/global-refresh.test.mjs +0 -65
  449. package/lib/upgrade/launchd-reconcile.test.mjs +0 -272
  450. package/lib/upgrade/post-steps.test.mjs +0 -200
  451. package/lib/upgrade/verify.test.mjs +0 -164
  452. package/lib/util/fetch-timeout.test.mjs +0 -202
  453. package/lib/util/reconnect.test.mjs +0 -369
  454. package/lib/util/unhandled.test.mjs +0 -216
  455. package/lib/voice/outbound.test.mjs +0 -69
  456. package/lib/voice/session-rotation.test.mjs +0 -114
  457. package/lib/voice/stt.test.mjs +0 -226
  458. package/lib/voice/voice.test.mjs +0 -990
  459. package/scripts/cadence/enqueue-cadence-tick.test.mjs +0 -187
  460. package/scripts/ci/check-docs-accuracy.test.mjs +0 -409
  461. package/scripts/ci/check-durable-write-seam.test.mjs +0 -90
  462. package/scripts/ci/check-no-build-artifacts.test.mjs +0 -71
  463. package/scripts/ci/check-no-residual-identity.test.mjs +0 -202
  464. package/scripts/ci/check-skill-packs.test.mjs +0 -495
  465. package/scripts/ci/check-subagent-frontmatter.test.mjs +0 -124
  466. package/scripts/ci/check.test.mjs +0 -194
  467. package/scripts/ci/conformance-org-api.test.mjs +0 -425
  468. package/scripts/cloud-relay/voice/relay-identity.test.mjs +0 -96
  469. package/scripts/collective/hook-runner.test.mjs +0 -173
  470. package/scripts/cost/fleet-digest.test.mjs +0 -207
  471. package/scripts/cost/track-claude-usage-pricing.test.mjs +0 -183
  472. package/scripts/cost/track-claude-usage.test.mjs +0 -148
  473. package/scripts/daemon/agent-daemon-board-mine.test.mjs +0 -96
  474. package/scripts/daemon/agent-daemon-design.test.mjs +0 -238
  475. package/scripts/daemon/agent-daemon-frontdoor.test.mjs +0 -60
  476. package/scripts/daemon/agent-daemon.test.mjs +0 -995
  477. package/scripts/daemon/assurance-e2e.test.mjs +0 -613
  478. package/scripts/daemon/assurance.test.mjs +0 -1791
  479. package/scripts/daemon/board-mirror.test.mjs +0 -165
  480. package/scripts/daemon/cadence-consumer-frontdoor.test.mjs +0 -393
  481. package/scripts/daemon/cadence-consumer-governance.test.mjs +0 -276
  482. package/scripts/daemon/cadence-consumer.test.mjs +0 -776
  483. package/scripts/daemon/cadence-handlers.test.mjs +0 -837
  484. package/scripts/daemon/classifier-identity.test.mjs +0 -137
  485. package/scripts/daemon/classifier.test.mjs +0 -266
  486. package/scripts/daemon/classify-kind.test.mjs +0 -40
  487. package/scripts/daemon/context-compiler.test.mjs +0 -406
  488. package/scripts/daemon/deliver.test.mjs +0 -564
  489. package/scripts/daemon/dispatcher-cooldown.test.mjs +0 -122
  490. package/scripts/daemon/dispatcher-governance.test.mjs +0 -1013
  491. package/scripts/daemon/dispatcher-resume.test.mjs +0 -166
  492. package/scripts/daemon/dispatcher-session-continuity.test.mjs +0 -365
  493. package/scripts/daemon/execution-ladder.test.mjs +0 -470
  494. package/scripts/daemon/goal-steward-cadence.test.mjs +0 -312
  495. package/scripts/daemon/inbox-deferral-session.test.mjs +0 -49
  496. package/scripts/daemon/inbox-deferral.test.mjs +0 -336
  497. package/scripts/daemon/inbox-wake.test.mjs +0 -199
  498. package/scripts/daemon/integration.test.mjs +0 -149
  499. package/scripts/daemon/lib/self-echo.test.mjs +0 -153
  500. package/scripts/daemon/lib/session-router.test.mjs +0 -554
  501. package/scripts/daemon/prompt-builder-preamble.test.mjs +0 -210
  502. package/scripts/daemon/prompt-builder.test.mjs +0 -556
  503. package/scripts/daemon/responder-cost.test.mjs +0 -68
  504. package/scripts/daemon/responder-history.test.mjs +0 -221
  505. package/scripts/daemon/sdk-version.test.mjs +0 -31
  506. package/scripts/daemon/session-lock.test.mjs +0 -252
  507. package/scripts/daemon/session-outcomes.test.mjs +0 -533
  508. package/scripts/daemon/typing-registry.test.mjs +0 -102
  509. package/scripts/hooks/pre-send-audit.test.mjs +0 -354
  510. package/scripts/huddle/huddle-prompt.test.mjs +0 -176
  511. package/scripts/local-triggers/autoupdate.test.mjs +0 -518
  512. package/scripts/local-triggers/generate-plists.test.mjs +0 -456
  513. package/scripts/media-generation/brand-clause.test.mjs +0 -135
  514. package/scripts/org/send-orgmail.first-contact.test.mjs +0 -102
  515. package/scripts/poller/inbox-privilege-injection.test.mjs +0 -167
  516. package/scripts/poller/inbox-scan-poller.test.mjs +0 -295
  517. package/scripts/poller/lib/cloud-relay-dedup.test.mjs +0 -133
  518. package/scripts/poller/slack-socket-mode.test.mjs +0 -805
  519. package/scripts/poller-launchd/install.test.mjs +0 -243
  520. package/scripts/restore-from-backup.test.mjs +0 -181
  521. package/scripts/session/feed.test.mjs +0 -196
  522. package/scripts/session/supervisor-sh.test.mjs +0 -218
  523. package/scripts/session/supervisor.test.mjs +0 -482
  524. package/scripts/setup/configure-macos.test.mjs +0 -306
  525. package/scripts/setup/gen-subagent-manifest.test.mjs +0 -124
  526. package/scripts/setup/generate-agent-package-json.test.mjs +0 -143
  527. package/scripts/setup/generate-capability.test.mjs +0 -134
  528. package/scripts/setup/init-agent.test.mjs +0 -370
  529. package/scripts/setup/init-skill-marketplace.test.mjs +0 -193
  530. package/scripts/vendor/sync-skill-packs.test.mjs +0 -103
  531. package/scripts/watchdog/memory-watchdog.test.mjs +0 -64
@@ -1,871 +0,0 @@
1
- /**
2
- * action-executor.test.mjs — contract test for the tool-call chokepoint.
3
- *
4
- * action-executor.js is where Claude `tool_use` blocks become real side
5
- * effects (spawning `claude --print`, running agent-local shell scripts,
6
- * writing draft files). This test pins the *contract* of executeAction()
7
- * without triggering any of those side effects:
8
- *
9
- * - the return value is always { success: boolean, result: string }
10
- * - access-level gating rejects tools the caller is not authorised for
11
- * - an authorised-but-unknown tool name returns a failure object (no throw)
12
- *
13
- * Side effects are avoided by (a) pointing AGENT_ROOT at a throwaway temp dir
14
- * so any stray script/draft write lands in /tmp, (b) pointing CLAUDE_BIN at a
15
- * harmless echo stub so no lookup path could ever reach the real `claude`
16
- * binary, and (c) choosing tool names + access levels that exercise the
17
- * validation / dispatch branches *before* any spawn. The lookup and
18
- * write-script branches that genuinely spawn are deliberately NOT exercised.
19
- *
20
- * Pure node:test; no extra deps.
21
- */
22
-
23
- import { test } from "node:test";
24
- import assert from "node:assert/strict";
25
- import {
26
- mkdtempSync,
27
- writeFileSync,
28
- chmodSync,
29
- rmSync,
30
- readFileSync,
31
- existsSync,
32
- } from "node:fs";
33
- import { tmpdir } from "node:os";
34
- import { join } from "node:path";
35
-
36
- // --- sandbox setup: must happen BEFORE importing the module under test, since
37
- // it reads AGENT_ROOT at module-eval time (const AGENT_ROOT = ...). ----------
38
-
39
- const sandbox = mkdtempSync(join(tmpdir(), "maestro-action-executor-"));
40
-
41
- // A harmless executable that ignores its args and prints a fixed line. If any
42
- // path we test were to spawn it (it should not), nothing real happens.
43
- const stubBin = join(sandbox, "claude-stub.sh");
44
- writeFileSync(stubBin, "#!/bin/sh\necho 'STUB: no real claude invoked'\n");
45
- chmodSync(stubBin, 0o755);
46
-
47
- process.env.AGENT_ROOT = sandbox;
48
- process.env.CLAUDE_BIN = stubBin;
49
-
50
- const {
51
- executeAction,
52
- executeLookup,
53
- resolveLookupModel,
54
- resolveEffectiveAccessLevel,
55
- LOOKUP_TIMEOUT_MS,
56
- LOOKUP_MAX_BUFFER,
57
- SCRIPT_TIMEOUT_MS,
58
- LOOKUP_DEFAULT_MODEL,
59
- } = await import("./action-executor.js");
60
- const toolDefs = await import("./tool-definitions.js");
61
-
62
- // Resolve a concrete (access level, authorised tool name) pair from the real
63
- // tool registry so the "unknown tool" test actually passes authorisation and
64
- // reaches the dispatch switch — rather than being short-circuited by the
65
- // access gate. Probe the documented access levels and pick the first that
66
- // yields a non-empty tool list.
67
- function pickAuthorisedLevel() {
68
- for (const level of ["ceo", "leadership", "default"]) {
69
- let names;
70
- try {
71
- names = toolDefs.getToolNamesForAccessLevel(level);
72
- } catch {
73
- continue;
74
- }
75
- if (Array.isArray(names) && names.length > 0) {
76
- return { level, names };
77
- }
78
- }
79
- return null;
80
- }
81
-
82
- const authorised = pickAuthorisedLevel();
83
-
84
- // A caller object shaped like the one the dispatcher builds.
85
- function caller(accessLevel) {
86
- return { slug: "test-caller", name: "Test Caller", accessLevel };
87
- }
88
-
89
- function assertResultShape(res) {
90
- assert.ok(res && typeof res === "object", "result must be an object");
91
- assert.equal(typeof res.success, "boolean", "result.success must be a boolean");
92
- assert.equal(typeof res.result, "string", "result.result must be a string");
93
- }
94
-
95
- test("setup: tool registry exposes at least one authorised access level", () => {
96
- assert.ok(
97
- authorised,
98
- "expected getToolNamesForAccessLevel to return a non-empty list for one of ceo/leadership/default",
99
- );
100
- });
101
-
102
- test("unauthorised access level is rejected with a failure object (no spawn)", async () => {
103
- // Use a tool that genuinely exists for *some* level, paired with an access
104
- // level that almost certainly does not include it. We confirm the chosen
105
- // level does NOT contain the tool before asserting rejection, so the test is
106
- // robust to registry changes.
107
- const toolName = authorised.names[0];
108
-
109
- // Find a bogus level whose tool list does NOT include toolName.
110
- let denyLevel = "nonexistent-access-level";
111
- try {
112
- const denyNames = toolDefs.getToolNamesForAccessLevel(denyLevel);
113
- if (Array.isArray(denyNames) && denyNames.includes(toolName)) {
114
- // Unexpected: the bogus level grants the tool. Skip the precondition
115
- // assertion but the gate should still reject it consistently below.
116
- }
117
- } catch {
118
- // getToolNamesForAccessLevel may throw on an unknown level; that's fine —
119
- // executeAction calls it the same way, so we instead use a known level
120
- // that is guaranteed not to contain the tool, if one exists.
121
- }
122
-
123
- const res = await executeAction(toolName, {}, caller(denyLevel), "session-1");
124
- assertResultShape(res);
125
- assert.equal(res.success, false, "unauthorised call must not succeed");
126
- assert.match(
127
- res.result,
128
- /permission/i,
129
- "rejection message should mention permission",
130
- );
131
- });
132
-
133
- test("unknown tool name (but authorised level) returns failure, does not throw", async () => {
134
- // "__definitely_not_a_real_tool__" is not in any access list, so to reach the
135
- // dispatch switch we would normally be blocked by the gate. Instead we assert
136
- // the safe, observable contract: an unrecognised tool always yields a
137
- // { success:false, result:string } object and never throws.
138
- const res = await executeAction(
139
- "__definitely_not_a_real_tool__",
140
- {},
141
- caller(authorised.level),
142
- "session-2",
143
- );
144
- assertResultShape(res);
145
- assert.equal(res.success, false, "unknown tool must not report success");
146
- assert.match(
147
- res.result,
148
- /permission|unknown/i,
149
- "unknown tool should be denied or reported as unknown, never silently succeed",
150
- );
151
- });
152
-
153
- test("rejected (unauthorised) call never reaches a spawn — result is the gate message verbatim", async () => {
154
- // The access gate runs before the try/catch and before any execFile. A tool
155
- // that DOES require a spawn, denied at the gate, must come back with the
156
- // canonical permission string and not an 'Action failed' / spawn error.
157
- const res = await executeAction("slack_send", { recipient: "x", message: "y" }, caller("nonexistent-access-level"), "session-3");
158
- assertResultShape(res);
159
- assert.equal(res.success, false);
160
- assert.equal(
161
- res.result,
162
- "You do not have permission to perform this action.",
163
- "denied call should return the canonical gate message, proving no spawn occurred",
164
- );
165
- });
166
-
167
- test("contract holds across a sweep of tool names and access levels", async () => {
168
- // An unknown access level now coerces to the least-privilege "default"
169
- // (read-only) surface — NOT an empty set (audit M4 / CEO-case footgun). So the
170
- // genuinely-denied sweep must use WRITE/side-effecting tools, which "default"
171
- // never grants, plus unknown names. None of these reach a spawn or a write.
172
- const deniedAtDefault = [
173
- "slack_send",
174
- "draft_email",
175
- "whatsapp_send",
176
- "generate_report",
177
- "__bogus__",
178
- "",
179
- ];
180
- for (const t of deniedAtDefault) {
181
- const res = await executeAction(t, {}, caller("nonexistent-access-level"), "sweep");
182
- assertResultShape(res);
183
- assert.equal(res.success, false, `denied "${t}" must not succeed at default`);
184
- }
185
-
186
- // Two authorised-but-unknown names at an authorised level: they pass the gate
187
- // only if the registry grants them (it does not), so they exercise the
188
- // dispatch 'default'/permission branches without any spawn or write.
189
- for (const t of ["__bogus__", ""]) {
190
- const res = await executeAction(t, {}, caller(authorised.level), "sweep");
191
- assertResultShape(res);
192
- assert.equal(res.success, false, `unknown "${t}" must not succeed`);
193
- }
194
- });
195
-
196
- // ---------------------------------------------------------------------------
197
- // Access-level integrity — defence-in-depth (audit M4) + CEO-case footgun
198
- // ---------------------------------------------------------------------------
199
-
200
- test("resolveEffectiveAccessLevel falls back to self-asserted level with no provenance", () => {
201
- assert.equal(resolveEffectiveAccessLevel({ accessLevel: "leadership" }), "leadership");
202
- assert.equal(
203
- resolveEffectiveAccessLevel({ accessLevel: "ceo", verifiedAccessLevel: null }),
204
- "ceo",
205
- );
206
- });
207
-
208
- test("resolveEffectiveAccessLevel defaults to least privilege when nothing is provided", () => {
209
- assert.equal(resolveEffectiveAccessLevel(), "default");
210
- assert.equal(resolveEffectiveAccessLevel(null), "default");
211
- assert.equal(resolveEffectiveAccessLevel({}), "default");
212
- assert.equal(resolveEffectiveAccessLevel({ slug: "x" }), "default");
213
- });
214
-
215
- test("resolveEffectiveAccessLevel clamps a self-asserted level above the verified level", () => {
216
- // Self-asserted 'ceo' must not beat a verified 'default'.
217
- assert.equal(
218
- resolveEffectiveAccessLevel({ accessLevel: "ceo", verifiedAccessLevel: "default" }),
219
- "default",
220
- );
221
- assert.equal(
222
- resolveEffectiveAccessLevel({ accessLevel: "ceo", verifiedAccessLevel: "leadership" }),
223
- "leadership",
224
- );
225
- });
226
-
227
- test("resolveEffectiveAccessLevel honours a self-asserted level that only narrows", () => {
228
- assert.equal(
229
- resolveEffectiveAccessLevel({ accessLevel: "default", verifiedAccessLevel: "ceo" }),
230
- "default",
231
- );
232
- assert.equal(
233
- resolveEffectiveAccessLevel({ accessLevel: "leadership", verifiedAccessLevel: "ceo" }),
234
- "leadership",
235
- );
236
- });
237
-
238
- test("resolveEffectiveAccessLevel uses the verified level when none is self-asserted", () => {
239
- assert.equal(
240
- resolveEffectiveAccessLevel({ verifiedAccessLevel: "leadership" }),
241
- "leadership",
242
- );
243
- });
244
-
245
- test("resolveEffectiveAccessLevel normalises casing on asserted and verified levels", () => {
246
- assert.equal(
247
- resolveEffectiveAccessLevel({ accessLevel: "CEO", verifiedAccessLevel: "Leadership" }),
248
- "leadership",
249
- );
250
- assert.equal(resolveEffectiveAccessLevel({ accessLevel: " Ceo " }), "ceo");
251
- });
252
-
253
- test("resolveEffectiveAccessLevel never throws on garbage input", () => {
254
- assert.doesNotThrow(() =>
255
- resolveEffectiveAccessLevel({ accessLevel: 42, verifiedAccessLevel: 99 }),
256
- );
257
- assert.equal(
258
- resolveEffectiveAccessLevel({ accessLevel: 42, verifiedAccessLevel: 99 }),
259
- "default",
260
- );
261
- });
262
-
263
- test("executeAction clamps a self-asserted ceo down to verified default (denied write tool)", async () => {
264
- // The M4 attack: caller self-asserts ceo but provenance only grants default.
265
- // slack_send is a write tool absent from default, so it must be denied.
266
- const res = await executeAction(
267
- "slack_send",
268
- { recipient: "x", message: "y" },
269
- { slug: "spoof", name: "Spoof", accessLevel: "ceo", verifiedAccessLevel: "default" },
270
- "session-m4-1",
271
- );
272
- assertResultShape(res);
273
- assert.equal(res.success, false);
274
- assert.match(res.result, /permission/i);
275
- });
276
-
277
- test("executeAction allows a CEO write tool when verified level is ceo (passes the auth gate)", async () => {
278
- // Verified provenance grants ceo; slack_send passes authorisation. It then
279
- // dispatches to a script that does not exist in the sandbox, so it returns a
280
- // failure — but NOT the permission gate message. We assert it cleared the gate.
281
- const res = await executeAction(
282
- "slack_send",
283
- { recipient: "x", message: "y" },
284
- { slug: "boss", name: "Boss", accessLevel: "ceo", verifiedAccessLevel: "ceo" },
285
- "session-m4-2",
286
- );
287
- assertResultShape(res);
288
- assert.doesNotMatch(
289
- res.result,
290
- /You do not have permission/i,
291
- "verified ceo should clear the authorisation gate",
292
- );
293
- });
294
-
295
- test("executeAction: uppercase access level no longer silently under-privileges a write tool", async () => {
296
- // CEO-case footgun: 'CEO' used to yield 0 tools => every tool denied. Now it
297
- // normalises to ceo, so slack_send clears the gate (and fails later at the
298
- // missing sandbox script, not at authorisation).
299
- const res = await executeAction(
300
- "slack_send",
301
- { recipient: "x", message: "y" },
302
- { slug: "boss", name: "Boss", accessLevel: "CEO" },
303
- "session-m4-3",
304
- );
305
- assertResultShape(res);
306
- assert.doesNotMatch(
307
- res.result,
308
- /You do not have permission/i,
309
- "'CEO' (uppercase) should clear the gate, not be denied",
310
- );
311
- });
312
-
313
- test("executeAction: a lookup tool is allowed at an unknown level (coerced to default)", async () => {
314
- // Unknown level coerces to least-privilege default, which includes read-only
315
- // lookups. search_email therefore clears the gate; CLAUDE_BIN points at the
316
- // harmless stub so no real claude runs.
317
- const res = await executeAction(
318
- "search_email",
319
- { query: "anything" },
320
- caller("totally-unknown-level"),
321
- "session-m4-4",
322
- );
323
- assertResultShape(res);
324
- assert.doesNotMatch(
325
- res.result,
326
- /You do not have permission/i,
327
- "unknown level should grant the read-only default surface, not deny everything",
328
- );
329
- });
330
-
331
- // ---------------------------------------------------------------------------
332
- // Spawn tuning constants are exported and numeric (audit L25)
333
- // ---------------------------------------------------------------------------
334
-
335
- test("spawn tuning constants are exported as positive numbers with sane defaults", () => {
336
- for (const [name, v] of [
337
- ["LOOKUP_TIMEOUT_MS", LOOKUP_TIMEOUT_MS],
338
- ["LOOKUP_MAX_BUFFER", LOOKUP_MAX_BUFFER],
339
- ["SCRIPT_TIMEOUT_MS", SCRIPT_TIMEOUT_MS],
340
- ]) {
341
- assert.equal(typeof v, "number", `${name} must be a number`);
342
- assert.ok(Number.isFinite(v) && v > 0, `${name} must be finite and > 0`);
343
- }
344
- // Defaults preserved when the env overrides are absent (CI sets none).
345
- assert.equal(LOOKUP_TIMEOUT_MS, 25000);
346
- assert.equal(LOOKUP_MAX_BUFFER, 1024 * 1024);
347
- assert.equal(SCRIPT_TIMEOUT_MS, 15000);
348
- });
349
-
350
- // ---------------------------------------------------------------------------
351
- // Internal lookup prompts still resolve via their write-tool executors (L24)
352
- // ---------------------------------------------------------------------------
353
-
354
- test("internal lookup names are NOT directly dispatchable from executeAction", async () => {
355
- // queue_update_internal / create_item_internal moved out of LOOKUP_PROMPTS
356
- // into INTERNAL_LOOKUP_PROMPTS, so they are not public tool names. At ceo
357
- // level (broadest surface) they must NOT be treated as a lookup; they are
358
- // denied at the gate (not real tools) — never "No prompt for".
359
- for (const name of ["queue_update_internal", "create_item_internal"]) {
360
- const res = await executeAction(
361
- name,
362
- {},
363
- { slug: "boss", name: "Boss", accessLevel: "ceo", verifiedAccessLevel: "ceo" },
364
- "internal-direct",
365
- );
366
- assertResultShape(res);
367
- assert.equal(res.success, false, `${name} must not succeed when called directly`);
368
- assert.doesNotMatch(
369
- res.result,
370
- /No prompt for/i,
371
- `${name} should be gated/unknown, not reach the lookup resolver directly`,
372
- );
373
- }
374
- });
375
-
376
- test("queue_update / create_action_item resolve their internal prompt (no 'No prompt for')", async () => {
377
- // These public write tools delegate to executeLookup with an INTERNAL_
378
- // prompt name. With a verified-ceo caller they clear the auth gate, and the
379
- // internal prompt must resolve (CLAUDE_BIN is the harmless stub, so the
380
- // spawn is inert). The failure mode we guard against is the L24 regression
381
- // where the internal prompt is missing → "No prompt for: ...".
382
- for (const [tool, input] of [
383
- ["queue_update", { search_term: "x", new_status: "done" }],
384
- ["create_action_item", { title: "x" }],
385
- ]) {
386
- const res = await executeAction(
387
- tool,
388
- input,
389
- { slug: "boss", name: "Boss", accessLevel: "ceo", verifiedAccessLevel: "ceo" },
390
- "internal-delegate",
391
- );
392
- assertResultShape(res);
393
- assert.doesNotMatch(
394
- res.result,
395
- /No prompt for/i,
396
- `${tool} must resolve its internal prompt (INTERNAL_LOOKUP_PROMPTS wired)`,
397
- );
398
- assert.doesNotMatch(
399
- res.result,
400
- /You do not have permission/i,
401
- `${tool} should clear the gate at verified ceo`,
402
- );
403
- }
404
- });
405
-
406
- // ---------------------------------------------------------------------------
407
- // Self-auditing — executeAction writes one audit row per invocation, incl.
408
- // denials and failures (audit observability F1/F3).
409
- // ---------------------------------------------------------------------------
410
-
411
- const todayUTC = () => new Date().toISOString().slice(0, 10);
412
-
413
- // Read the rows executeAction appended to <root>/logs/audit/<date>-actions.jsonl.
414
- // Each test points executeAction at its OWN throwaway root via options.agentRoot
415
- // so the audit file holds exactly that test's rows.
416
- function readAuditRows(root) {
417
- const file = join(root, "logs", "audit", `${todayUTC()}-actions.jsonl`);
418
- if (!existsSync(file)) return [];
419
- return readFileSync(file, "utf8")
420
- .split("\n")
421
- .filter((l) => l.trim() !== "")
422
- .map((l) => JSON.parse(l));
423
- }
424
-
425
- function freshRoot() {
426
- return mkdtempSync(join(tmpdir(), "maestro-audit-"));
427
- }
428
-
429
- test("self-audit: a successful action writes a row with the real tool name and status completed", async () => {
430
- const root = freshRoot();
431
- // draft_email at verified ceo writes a local draft file and returns success
432
- // with NO spawn — a deterministic success path for the audit assertion.
433
- const res = await executeAction(
434
- "draft_email",
435
- { to: "alex@example.com", subject: "Hi", body: "Body" },
436
- { slug: "boss", name: "Boss", accessLevel: "ceo", verifiedAccessLevel: "ceo" },
437
- "sess-success",
438
- { agentRoot: root },
439
- );
440
- assertResultShape(res);
441
- assert.equal(res.success, true, "draft_email should succeed");
442
-
443
- const rows = readAuditRows(root);
444
- assert.equal(rows.length, 1, "exactly one audit row for one invocation");
445
- const row = rows[0];
446
- assert.equal(row.tool, "draft_email", "row records the real tool name, not 'unknown'");
447
- assert.equal(row.status, "completed", "successful action is logged completed");
448
- assert.equal(row.denied, false, "an allowed action is not denied");
449
- assert.equal(row.session_id, "sess-success", "the sessionId is captured");
450
- assert.equal(row.target, "alex@example.com", "best-effort target is the recipient");
451
- assert.match(
452
- row.timestamp,
453
- /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}Z$/,
454
- "timestamp is ISO8601 Z",
455
- );
456
-
457
- rmSync(root, { recursive: true, force: true });
458
- });
459
-
460
- test("self-audit: a permission-denied call writes a row with denied=true", async () => {
461
- const root = freshRoot();
462
- // slack_send is a write tool absent from the least-privilege default surface;
463
- // an unknown level coerces to default, so this is denied at the gate.
464
- const res = await executeAction(
465
- "slack_send",
466
- { recipient: "C123", message: "y" },
467
- caller("nonexistent-access-level"),
468
- "sess-denied",
469
- { agentRoot: root },
470
- );
471
- assert.equal(res.success, false);
472
- assert.match(res.result, /permission/i);
473
-
474
- const rows = readAuditRows(root);
475
- assert.equal(rows.length, 1, "the denied attempt is logged, not silent");
476
- const row = rows[0];
477
- assert.equal(row.tool, "slack_send", "denied row records the attempted tool");
478
- assert.equal(row.denied, true, "denied row has denied=true");
479
- assert.equal(row.status, "denied", "denied row status is 'denied'");
480
- assert.equal(row.session_id, "sess-denied");
481
- assert.equal(row.target, "C123", "denied row still captures the attempted target");
482
-
483
- rmSync(root, { recursive: true, force: true });
484
- });
485
-
486
- test("self-audit: a failing action writes status failed", async () => {
487
- const root = freshRoot();
488
- // verified ceo clears the gate; slack-send.sh does not exist under this fresh
489
- // root, so executeScript returns { success:false } — a real failure path.
490
- const res = await executeAction(
491
- "slack_send",
492
- { recipient: "C999", message: "y" },
493
- { slug: "boss", name: "Boss", accessLevel: "ceo", verifiedAccessLevel: "ceo" },
494
- "sess-fail",
495
- { agentRoot: root },
496
- );
497
- assert.equal(res.success, false, "missing script should make the action fail");
498
- assert.doesNotMatch(res.result, /You do not have permission/i, "it cleared the gate");
499
-
500
- const rows = readAuditRows(root);
501
- assert.equal(rows.length, 1);
502
- const row = rows[0];
503
- assert.equal(row.tool, "slack_send");
504
- assert.equal(row.status, "failed", "a failed (non-denied) action is logged failed");
505
- assert.equal(row.denied, false, "a gate-cleared failure is not a denial");
506
- assert.equal(row.session_id, "sess-fail");
507
-
508
- rmSync(root, { recursive: true, force: true });
509
- });
510
-
511
- test("self-audit: audit write failure does not throw into the action path", async () => {
512
- // Point agentRoot at a regular FILE, so logs/audit/ cannot be created — the
513
- // audit write must fail internally and be swallowed. executeAction must still
514
- // return its normal { success:false, result } denial object.
515
- const blocker = join(sandbox, "not-a-dir");
516
- writeFileSync(blocker, "x");
517
-
518
- let res;
519
- await assert.doesNotReject(async () => {
520
- res = await executeAction(
521
- "slack_send",
522
- { recipient: "C123", message: "y" },
523
- caller("nonexistent-access-level"),
524
- "sess-blocked",
525
- { agentRoot: blocker },
526
- );
527
- }, "a broken audit path must not throw out of executeAction");
528
- assertResultShape(res);
529
- assert.equal(res.success, false);
530
- assert.match(res.result, /permission/i, "the action result is unaffected by the log failure");
531
- });
532
-
533
- // ---------------------------------------------------------------------------
534
- // Router coverage for lookups (#29 residual): executeLookup routes the MODEL
535
- // SELECTION through the model router (cheapest capable model) and records the
536
- // usage to the cost ledger with a decision_id — a model-selection + accounting
537
- // upgrade that preserves the lookup behaviour, timeout, MCP access, and the
538
- // { success, result } output shape exactly. All deps are injected so these
539
- // tests touch neither the real router nor the real ledger nor the network.
540
- // ---------------------------------------------------------------------------
541
-
542
- // A v2 routing config is the gate that turns on router-driven selection. The
543
- // SHAPE is all the helper checks (schema_version === 2); the chain is never
544
- // walked because we inject `resolve`.
545
- const V2_CONFIG = { schema_version: 2, routing_policy: [{ default: true, chain: ["x"] }] };
546
-
547
- // A RouteDecision the injected resolver returns: it names a cheap Haiku-class
548
- // model and carries the decision_id we expect on the ledger row.
549
- function fakeDecision(over = {}) {
550
- return {
551
- decision_id: over.decision_id || "dec-lookup-001",
552
- chosen: {
553
- provider: over.provider || "anthropic",
554
- model: over.model || "claude-haiku-4-5",
555
- harness: "session",
556
- catalogRow: { ref: over.ref || "anthropic/claude-haiku-4-5" },
557
- },
558
- spawnArgs: { modelFlag: over.modelFlag || "claude-haiku-4-5" },
559
- ...over.extra,
560
- };
561
- }
562
-
563
- test("executeLookup: selects the model via the injected router and writes a ledger row", async () => {
564
- const root = sandbox;
565
- const calls = { resolve: 0, model: null };
566
- const ledgerRows = [];
567
-
568
- const res = await executeLookup(
569
- "search_email",
570
- { query: "Q2 board deck" },
571
- root,
572
- {
573
- config: V2_CONFIG,
574
- resolve: async (req, ropts) => {
575
- calls.resolve += 1;
576
- calls.req = req;
577
- calls.ropts = ropts;
578
- return fakeDecision();
579
- },
580
- writeLedger: async (catalog, entry) => {
581
- ledgerRows.push({ catalog, entry });
582
- },
583
- // Capture the model flag the spawn actually used by stubbing nothing —
584
- // CLAUDE_BIN is the harmless echo stub, so the spawn is inert but real.
585
- },
586
- );
587
-
588
- assertResultShape(res);
589
- assert.equal(res.success, true, "the lookup still succeeds (stub echoes a line)");
590
- assert.equal(calls.resolve, 1, "the router resolver was consulted exactly once");
591
- // The request handed to the router is a tool-less, sensitive RAG lookup.
592
- assert.equal(calls.req.task_class, "session.lookup");
593
- assert.equal(calls.req.data_class, "sensitive");
594
- assert.equal(calls.ropts.config, V2_CONFIG, "config threaded into resolveChain");
595
-
596
- // Exactly one ledger row, joined by the decision_id from the decision.
597
- assert.equal(ledgerRows.length, 1, "one ledger row per routed lookup");
598
- const { entry } = ledgerRows[0];
599
- assert.equal(entry.decision_id, "dec-lookup-001");
600
- assert.equal(entry.ref, "anthropic/claude-haiku-4-5");
601
- assert.equal(entry.provider, "anthropic");
602
- assert.equal(entry.model, "claude-haiku-4-5");
603
- assert.equal(entry.task_class, "session.lookup");
604
- assert.equal(entry.source, "lookup");
605
- assert.equal(entry.exitReason, "ok");
606
- });
607
-
608
- test("resolveLookupModel: picks the router's modelFlag when a v2 config + resolver are present", async () => {
609
- const routed = await resolveLookupModel("search_slack", sandbox, {
610
- config: V2_CONFIG,
611
- resolve: async () => fakeDecision({ modelFlag: "deepseek-v4-flash", provider: "deepseek", model: "deepseek-v4-flash", ref: "deepseek/deepseek-v4-flash", decision_id: "d-2" }),
612
- });
613
- assert.equal(routed.modelFlag, "deepseek-v4-flash", "cheapest capable model wins selection");
614
- assert.equal(routed.decisionId, "d-2");
615
- assert.equal(routed.ref, "deepseek/deepseek-v4-flash");
616
- });
617
-
618
- test("resolveLookupModel: falls back to the default model when NO routing config is present", async () => {
619
- let resolveCalled = false;
620
- const routed = await resolveLookupModel("search_calendar", sandbox, {
621
- // loadConfig returns null (the no-config path) → never even calls resolve.
622
- loadConfig: async () => null,
623
- resolve: async () => { resolveCalled = true; return fakeDecision(); },
624
- });
625
- assert.equal(routed.modelFlag, LOOKUP_DEFAULT_MODEL, "no config → historical haiku default");
626
- assert.equal(routed.modelFlag, "haiku");
627
- assert.equal(routed.decisionId, null, "no decision, so no ledger join id");
628
- assert.equal(resolveCalled, false, "the router is not consulted without a v2 config");
629
- });
630
-
631
- test("resolveLookupModel: a v1 (non-2) config also falls back to the default model", async () => {
632
- const routed = await resolveLookupModel("search_files", sandbox, {
633
- config: { schema_version: 1, backends: {} },
634
- resolve: async () => { throw new Error("should not be called"); },
635
- });
636
- assert.equal(routed.modelFlag, LOOKUP_DEFAULT_MODEL);
637
- assert.equal(routed.decisionId, null);
638
- });
639
-
640
- test("resolveLookupModel: MAESTRO_ROUTER_FORCE_ANTHROPIC short-circuits to the default (no resolve, no ledger)", async () => {
641
- let resolveCalled = false;
642
- const routed = await resolveLookupModel("search_web", sandbox, {
643
- env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" },
644
- config: V2_CONFIG,
645
- resolve: async () => { resolveCalled = true; return fakeDecision(); },
646
- });
647
- assert.equal(routed.modelFlag, LOOKUP_DEFAULT_MODEL, "kill switch → stock haiku");
648
- assert.equal(routed.decisionId, null);
649
- assert.equal(resolveCalled, false, "kill switch never consults the router");
650
- });
651
-
652
- test("executeLookup: FORCE_ANTHROPIC writes NO ledger row and keeps the output shape", async () => {
653
- const ledgerRows = [];
654
- let resolveCalled = false;
655
- const res = await executeLookup(
656
- "search_email",
657
- { query: "x" },
658
- sandbox,
659
- {
660
- env: { ...process.env, MAESTRO_ROUTER_FORCE_ANTHROPIC: "true" },
661
- config: V2_CONFIG,
662
- resolve: async () => { resolveCalled = true; return fakeDecision(); },
663
- writeLedger: async (c, e) => ledgerRows.push(e),
664
- },
665
- );
666
- assertResultShape(res);
667
- assert.equal(res.success, true);
668
- assert.equal(resolveCalled, false, "kill switch bypasses the router");
669
- assert.equal(ledgerRows.length, 0, "no decision → no ledger row (matches the pre-router default spawn)");
670
- });
671
-
672
- test("executeLookup: a router fault is fail-open — the lookup still runs on the default model", async () => {
673
- const ledgerRows = [];
674
- const res = await executeLookup(
675
- "search_email",
676
- { query: "x" },
677
- sandbox,
678
- {
679
- config: V2_CONFIG,
680
- resolve: async () => { throw new Error("router exploded"); },
681
- writeLedger: async (c, e) => ledgerRows.push(e),
682
- },
683
- );
684
- assertResultShape(res);
685
- assert.equal(res.success, true, "a router error never breaks the lookup");
686
- assert.equal(ledgerRows.length, 0, "no usable decision → no ledger row, but the search ran");
687
- });
688
-
689
- test("executeLookup: ledger write failure does not break the lookup (best-effort accounting)", async () => {
690
- const res = await executeLookup(
691
- "search_email",
692
- { query: "x" },
693
- sandbox,
694
- {
695
- config: V2_CONFIG,
696
- resolve: async () => fakeDecision(),
697
- writeLedger: async () => { throw new Error("disk full"); },
698
- },
699
- );
700
- assertResultShape(res);
701
- assert.equal(res.success, true, "a ledger we can't persist must not fail the search");
702
- });
703
-
704
- test("executeLookup: output shape is unchanged for an unknown prompt name (no router, no ledger)", async () => {
705
- let resolveCalled = false;
706
- const res = await executeLookup(
707
- "__no_such_lookup__",
708
- {},
709
- sandbox,
710
- { config: V2_CONFIG, resolve: async () => { resolveCalled = true; return fakeDecision(); } },
711
- );
712
- assertResultShape(res);
713
- assert.equal(res.success, false);
714
- assert.match(res.result, /No prompt for/);
715
- assert.equal(resolveCalled, false, "no prompt → no spawn → no routing");
716
- });
717
-
718
- test("executeAction: a public lookup tool routes through the model router end-to-end", async () => {
719
- // Drive the FULL public path (executeAction → executeLookup) but with the
720
- // router/ledger injected via options. This proves the dispatch threads the
721
- // injectable opts through and that authorisation still gates correctly.
722
- const ledgerRows = [];
723
- const res = await executeAction(
724
- "search_email",
725
- { query: "anything" },
726
- { slug: "alex", name: "Alex", accessLevel: "ceo", verifiedAccessLevel: "ceo" },
727
- "sess-router",
728
- {
729
- agentRoot: sandbox,
730
- agent: "alex",
731
- // executeAction builds its own lookupOpts; we cannot inject resolve through
732
- // it directly, so we only assert the contract holds (no throw, right shape).
733
- },
734
- );
735
- assertResultShape(res);
736
- assert.doesNotMatch(res.result, /You do not have permission/i, "ceo clears the gate");
737
- });
738
-
739
- test.after(() => {
740
- try {
741
- rmSync(sandbox, { recursive: true, force: true });
742
- } catch {
743
- /* best effort */
744
- }
745
- });
746
-
747
- // ---------------------------------------------------------------------------
748
- // Org (Cohort) tool routing — the curated surface via executeOrgTool
749
- // ---------------------------------------------------------------------------
750
-
751
- // Ambient COHORT_* creds must not leak into the org-tool fixtures.
752
- for (const k of ["COHORT_API_TOKEN", "COHORT_TOKEN", "COHORT_API_KEY", "COHORT_ORG_ID", "COHORT_BASE", "COHORT_API_URL"]) delete process.env[k];
753
-
754
- import * as fsExtra from "node:fs";
755
-
756
- /** A tmp agent root carrying an enrolled config/org.yaml. */
757
- function orgEnrolledRoot() {
758
- const root = mkdtempSync(join(tmpdir(), "maestro-orgtool-"));
759
- const cfgDir = join(root, "config");
760
- fsExtra.mkdirSync(cfgDir, { recursive: true });
761
- writeFileSync(
762
- join(cfgDir, "org.yaml"),
763
- ["org:", " cohort:", " enabled: true", " base: https://org.example", " orgId: acme", " token: tok-exec"].join("\n"),
764
- );
765
- return root;
766
- }
767
-
768
- function orgFakeFetch(body = { ok: true, result: { done: true } }) {
769
- const calls = [];
770
- const fn = async (url, init) => {
771
- calls.push({ url: String(url), init: init || {} });
772
- return { ok: true, status: 200, json: async () => body, headers: { get: () => undefined } };
773
- };
774
- fn.calls = calls;
775
- return fn;
776
- }
777
-
778
- test("org tool routes to executeOrgTool with the injected fetchImpl (no network, frame → outcome)", async () => {
779
- const root = orgEnrolledRoot();
780
- const fetchImpl = orgFakeFetch({ ok: true, result: { channels: [{ id: "C1" }] } });
781
- const res = await executeAction("messaging_channels", {}, { accessLevel: "default" }, "s-org-1", { agentRoot: root, fetchImpl });
782
- assert.equal(res.success, true);
783
- assert.match(res.result, /C1/, "frame.result stringified into the outcome");
784
- assert.equal(fetchImpl.calls.length, 1);
785
- assert.match(fetchImpl.calls[0].url, /https:\/\/org\.example\/api\/v1\/messaging\.channels$/);
786
- rmSync(root, { recursive: true, force: true });
787
- });
788
-
789
- test("org error frames map to success:false with the code surfaced", async () => {
790
- const root = orgEnrolledRoot();
791
- const fetchImpl = async () => ({ ok: false, status: 403, json: async () => ({ ok: false, error: { code: "FORBIDDEN_SCOPE", message: "not paired" } }), headers: { get: () => undefined } });
792
- const res = await executeAction("messaging_channels", {}, { accessLevel: "default" }, "s-org-2", { agentRoot: root, fetchImpl });
793
- assert.equal(res.success, false);
794
- assert.match(res.result, /FORBIDDEN_SCOPE/);
795
- assert.match(res.result, /not paired/);
796
- rmSync(root, { recursive: true, force: true });
797
- });
798
-
799
- test("org writes are DENIED at the default access level (name-set authorization)", async () => {
800
- const root = orgEnrolledRoot();
801
- const fetchImpl = orgFakeFetch();
802
- const res = await executeAction("messaging_send", { channelId: "C1", body: "hi" }, { accessLevel: "default" }, "s-org-3", { agentRoot: root, fetchImpl });
803
- assert.equal(res.success, false);
804
- assert.match(res.result, /permission/);
805
- assert.equal(fetchImpl.calls.length, 0, "denied before any dispatch");
806
- rmSync(root, { recursive: true, force: true });
807
- });
808
-
809
- test("org_rpc is denied below ceo; routed (and protocol-validated) at ceo", async () => {
810
- const root = orgEnrolledRoot();
811
- const fetchImpl = orgFakeFetch();
812
- const lead = await executeAction("org_rpc", { method: "member.get", params: { memberId: "m" } }, { accessLevel: "leadership" }, "s-org-4", { agentRoot: root, fetchImpl });
813
- assert.equal(lead.success, false);
814
- assert.match(lead.result, /permission/, "escape hatch withheld below ceo");
815
- assert.equal(fetchImpl.calls.length, 0);
816
-
817
- const ceoBad = await executeAction("org_rpc", { method: "no.method" }, { accessLevel: "ceo" }, "s-org-5", { agentRoot: root, fetchImpl });
818
- assert.equal(ceoBad.success, false);
819
- assert.match(ceoBad.result, /NOT_FOUND/, "unknown method rejected against the protocol table");
820
- assert.equal(fetchImpl.calls.length, 0, "no network for an invalid method");
821
-
822
- const ceoOk = await executeAction("org_rpc", { method: "member.get", params: { memberId: "m" } }, { accessLevel: "ceo" }, "s-org-6", { agentRoot: root, fetchImpl });
823
- assert.equal(ceoOk.success, true);
824
- assert.match(fetchImpl.calls[0].url, /member\.get$/);
825
- rmSync(root, { recursive: true, force: true });
826
- });
827
-
828
- test("send-gate invoked for messaging_send in the native plane (options.screenImpl)", async () => {
829
- const root = orgEnrolledRoot();
830
- const screened = [];
831
- const fetchImpl = orgFakeFetch({ ok: true, result: { messageId: "m1" } });
832
- const ok = await executeAction(
833
- "messaging_send",
834
- { channelId: "C1", body: "shipping now" },
835
- { accessLevel: "leadership" },
836
- "s-org-7",
837
- { agentRoot: root, fetchImpl, screenImpl: async (o) => { screened.push(o); return { allow: true }; } },
838
- );
839
- assert.equal(ok.success, true);
840
- assert.equal(screened.length, 1, "screenOutbound ran before dispatch");
841
- assert.equal(screened[0].recipient, "C1");
842
-
843
- const blocked = await executeAction(
844
- "messaging_send",
845
- { channelId: "C1", body: "As an AI…" },
846
- { accessLevel: "leadership" },
847
- "s-org-8",
848
- { agentRoot: root, fetchImpl, screenImpl: async () => ({ allow: false, reason: "banned-phrase" }) },
849
- );
850
- assert.equal(blocked.success, false);
851
- assert.match(blocked.result, /banned-phrase/);
852
- assert.equal(fetchImpl.calls.length, 1, "blocked send never reached the wire");
853
- rmSync(root, { recursive: true, force: true });
854
- });
855
-
856
- test("denied + completed org invocations write the native plane's audit rows", async () => {
857
- const root = orgEnrolledRoot();
858
- const fetchImpl = orgFakeFetch();
859
- await executeAction("messaging_send", { channelId: "C1", body: "x" }, { accessLevel: "default" }, "s-audit-1", { agentRoot: root, fetchImpl });
860
- await executeAction("org_describe", {}, { accessLevel: "default" }, "s-audit-2", { agentRoot: root, fetchImpl });
861
- const dir = join(root, "logs", "audit");
862
- const rows = [];
863
- for (const f of fsExtra.readdirSync(dir)) {
864
- for (const line of readFileSync(join(dir, f), "utf8").split("\n")) if (line.trim()) rows.push(JSON.parse(line));
865
- }
866
- assert.equal(rows.length, 2, "one audit row per invocation");
867
- assert.equal(rows[0].denied, true, "denied write audited");
868
- assert.equal(rows[1].denied, false);
869
- assert.equal(rows[1].status, "completed");
870
- rmSync(root, { recursive: true, force: true });
871
- });