@cohortapp/agent-sdk 2.17.0 → 2.18.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/settings.json +18 -0
- package/.env.example +18 -5
- package/README.md +1 -0
- package/bin/maestro.mjs +62 -0
- package/docs/guides/billing-console-keys.md +60 -0
- package/docs/guides/front-door-session.md +54 -9
- package/docs/guides/mac-mini.md +20 -25
- package/docs/guides/setup-wizard.md +1 -1
- package/docs/runbooks/fleet-rollout.md +156 -0
- package/docs/runbooks/mac-mini-bootstrap.md +12 -14
- package/lib/action-executor.js +19 -3
- package/lib/budget-guard.mjs +279 -3
- package/lib/channels/base-adapter.mjs +3 -1
- package/lib/channels/contract.mjs +2 -1
- package/lib/channels/inbox-item.mjs +8 -0
- package/lib/claude-bin.mjs +5 -6
- package/lib/cli/doctor-checks.mjs +141 -10
- package/lib/cli/global-setup-extras.mjs +5 -1
- package/lib/cli/inbox.mjs +100 -15
- package/lib/cli/seat-auth.mjs +463 -0
- package/lib/cli/session.mjs +80 -12
- package/lib/collective/capture-slots.mjs +234 -0
- package/lib/collective/capture.mjs +8 -6
- package/lib/collective/config.mjs +2 -0
- package/lib/collective/global-config.mjs +63 -1
- package/lib/collective/loop-guard.mjs +155 -0
- package/lib/collective/presence.mjs +142 -5
- package/lib/comms/send-gate.mjs +559 -1
- package/lib/diagnostics/alerts.mjs +49 -0
- package/lib/diagnostics/cadence-output-freshness.mjs +288 -0
- package/lib/engine/agents/definitions.mjs +343 -0
- package/lib/engine/agents/persist.mjs +275 -0
- package/lib/engine/agents/runtime.mjs +748 -0
- package/lib/engine/agents/usage.mjs +95 -0
- package/lib/engine/auth-status.mjs +139 -0
- package/lib/engine/budget.mjs +194 -0
- package/lib/engine/cli.mjs +1204 -0
- package/lib/engine/commands/index.mjs +269 -0
- package/lib/engine/context/budget.mjs +219 -0
- package/lib/engine/context/cache.mjs +125 -0
- package/lib/engine/context/child-env.mjs +215 -0
- package/lib/engine/context/compaction.mjs +342 -0
- package/lib/engine/context/images.mjs +90 -0
- package/lib/engine/context/instructions.mjs +327 -0
- package/lib/engine/context/lazy-instructions.mjs +169 -0
- package/lib/engine/context/manager.mjs +182 -0
- package/lib/engine/context/real-path.mjs +91 -0
- package/lib/engine/context/secret-values.mjs +163 -0
- package/lib/engine/context/settings.mjs +274 -0
- package/lib/engine/context/stream-input.mjs +159 -0
- package/lib/engine/guard.mjs +152 -0
- package/lib/engine/hooks.mjs +713 -0
- package/lib/engine/loop.mjs +560 -0
- package/lib/engine/mcp/client.mjs +254 -0
- package/lib/engine/mcp/config.mjs +301 -0
- package/lib/engine/mcp/http.mjs +201 -0
- package/lib/engine/mcp/index.mjs +146 -0
- package/lib/engine/mcp/jsonrpc.mjs +147 -0
- package/lib/engine/mcp/naming.mjs +66 -0
- package/lib/engine/mcp/resources.mjs +89 -0
- package/lib/engine/mcp/results.mjs +133 -0
- package/lib/engine/mcp/stdio.mjs +137 -0
- package/lib/engine/mcp/supervisor.mjs +116 -0
- package/lib/engine/messages.mjs +104 -0
- package/lib/engine/output/json.mjs +164 -0
- package/lib/engine/output/stream-json.mjs +266 -0
- package/lib/engine/permissions.mjs +845 -0
- package/lib/engine/process-identity.mjs +164 -0
- package/lib/engine/process-tree.mjs +551 -0
- package/lib/engine/prompt.mjs +60 -0
- package/lib/engine/session/store.mjs +299 -0
- package/lib/engine/session-runtime/args.mjs +97 -0
- package/lib/engine/session-runtime/host.mjs +143 -0
- package/lib/engine/session-runtime/inbox.mjs +122 -0
- package/lib/engine/session-runtime/notifications.mjs +129 -0
- package/lib/engine/session-runtime/registry.mjs +328 -0
- package/lib/engine/session-runtime/runner.mjs +344 -0
- package/lib/engine/session-runtime/socket.mjs +212 -0
- package/lib/engine/session-runtime/wakeup.mjs +115 -0
- package/lib/engine/skills/index.mjs +321 -0
- package/lib/engine/tools/bash-background.mjs +533 -0
- package/lib/engine/tools/bash.mjs +216 -0
- package/lib/engine/tools/edit.mjs +97 -0
- package/lib/engine/tools/glob.mjs +81 -0
- package/lib/engine/tools/grep.mjs +224 -0
- package/lib/engine/tools/index.mjs +84 -0
- package/lib/engine/tools/list-agents.mjs +32 -0
- package/lib/engine/tools/ls.mjs +127 -0
- package/lib/engine/tools/monitor.mjs +82 -0
- package/lib/engine/tools/notebook-edit.mjs +218 -0
- package/lib/engine/tools/read.mjs +103 -0
- package/lib/engine/tools/schedule-wakeup.mjs +45 -0
- package/lib/engine/tools/schema.mjs +144 -0
- package/lib/engine/tools/send-message.mjs +77 -0
- package/lib/engine/tools/session.mjs +70 -0
- package/lib/engine/tools/todo.mjs +144 -0
- package/lib/engine/tools/toolsearch.mjs +217 -0
- package/lib/engine/tools/walk.mjs +193 -0
- package/lib/engine/tools/web-switch.mjs +31 -0
- package/lib/engine/tools/webfetch-html.mjs +387 -0
- package/lib/engine/tools/webfetch-net.mjs +340 -0
- package/lib/engine/tools/webfetch.mjs +198 -0
- package/lib/engine/tools/websearch.mjs +91 -0
- package/lib/engine/tools/workflow.mjs +95 -0
- package/lib/engine/tools/write.mjs +76 -0
- package/lib/engine/tui/line-editor.mjs +137 -0
- package/lib/engine/tui/render.mjs +86 -0
- package/lib/engine/tui/tui.mjs +274 -0
- package/lib/engine/wire/anthropic-messages.mjs +263 -0
- package/lib/engine/wire/effort.mjs +36 -0
- package/lib/engine/wire/errors.mjs +496 -0
- package/lib/engine/wire/http.mjs +441 -0
- package/lib/engine/wire/index.mjs +76 -0
- package/lib/engine/wire/openai-chat.mjs +332 -0
- package/lib/engine/wire/prompt-cache.mjs +79 -0
- package/lib/engine/wire/search.mjs +140 -0
- package/lib/engine/wire/sse.mjs +114 -0
- package/lib/engine/wire/stall.mjs +349 -0
- package/lib/engine/wire/token-provider.mjs +175 -0
- package/lib/engine/wire/usage.mjs +192 -0
- package/lib/engine/workflow/host.mjs +524 -0
- package/lib/engine/workflow/journal.mjs +188 -0
- package/lib/engine/workflow/json-schema.mjs +171 -0
- package/lib/engine/workflow/meta.mjs +329 -0
- package/lib/engine/workflow/notifications.mjs +52 -0
- package/lib/engine/workflow/runtime.mjs +447 -0
- package/lib/engine/workflow/sandbox.mjs +534 -0
- package/lib/engine/workflow/worker.mjs +141 -0
- package/lib/engine/workflow/worktree.mjs +74 -0
- package/lib/execution/disposition.mjs +1 -1
- package/lib/execution/intake.mjs +10 -0
- package/lib/execution/surface-policy.mjs +15 -0
- package/lib/learning/curator.mjs +8 -6
- package/lib/learning/reflect.mjs +8 -6
- package/lib/model-router/catalog/cohort.yaml +137 -0
- package/lib/model-router/catalog.mjs +118 -1
- package/lib/model-router/failover.mjs +67 -16
- package/lib/model-router/llm-task.mjs +39 -3
- package/lib/model-router/resolve.mjs +89 -3
- package/lib/model-router/spawn.mjs +46 -47
- package/lib/model-router/taxonomy.mjs +126 -4
- package/lib/org/cost-sync.mjs +141 -11
- package/lib/org/inbound/broadcast.mjs +289 -0
- package/lib/org/inbound/collective.mjs +375 -0
- package/lib/org/inbound/directedness.mjs +96 -8
- package/lib/org/inbound/facts.mjs +78 -2
- package/lib/org/inbound/project.mjs +22 -0
- package/lib/org/inbound/surfaces.mjs +14 -0
- package/lib/org/llm-token.mjs +879 -0
- package/lib/org/mesh.mjs +61 -0
- package/lib/org/messaging.mjs +3 -1
- package/lib/org/protocol.checksum +1 -1
- package/lib/org/protocol.mjs +15 -0
- package/lib/org/quota.mjs +520 -0
- package/lib/org/tool-surface.mjs +104 -16
- package/lib/org/ui-parity.mjs +16 -1
- package/lib/org/work-ledger.mjs +37 -6
- package/lib/rate-guard.mjs +114 -1
- package/lib/resource-governor.mjs +41 -6
- package/lib/runtime/adapter.mjs +833 -0
- package/lib/runtime/child-env.mjs +191 -0
- package/lib/runtime/legacy-shell-guard.mjs +97 -0
- package/lib/runtime/seat-engine.mjs +162 -0
- package/lib/session/ask-ledger.mjs +271 -0
- package/lib/session/current-work.mjs +676 -0
- package/lib/session/feed-core.mjs +40 -3
- package/lib/session/launch-args.mjs +56 -4
- package/lib/session/status-summary.mjs +26 -9
- package/lib/session/upgrade-notice.mjs +42 -0
- package/lib/setup/claude-probe.mjs +117 -13
- package/lib/setup/enrich.mjs +13 -10
- package/lib/setup/sections/model.mjs +39 -13
- package/lib/telemetry/collect.mjs +229 -11
- package/lib/upgrade/ignored-drift.mjs +105 -0
- package/lib/voice/post-call-brief.mjs +30 -17
- package/package.json +13 -3
- package/plugins/maestro-skills/skills/board-work.md +5 -0
- package/plugins/maestro-skills/skills/inbound-triage.md +56 -15
- package/plugins/maestro-skills/skills/main-session.md +18 -7
- package/scaffold/config/collective.yaml +7 -0
- package/scripts/ci/check-durable-write-seam.mjs +3 -1
- package/scripts/ci/check-tarball-fidelity.mjs +126 -2
- package/scripts/ci/run-tests.mjs +47 -19
- package/scripts/cohort-llm/api-key-helper.mjs +92 -0
- package/scripts/collective/hook-runner.mjs +142 -19
- package/scripts/continuous-monitor.sh +13 -0
- package/scripts/cost/track-claude-usage.mjs +15 -0
- package/scripts/daemon/agent-daemon.mjs +408 -20
- package/scripts/daemon/assurance.mjs +48 -12
- package/scripts/daemon/cadence-consumer.mjs +218 -68
- package/scripts/daemon/cadence-handlers.mjs +73 -4
- package/scripts/daemon/classifier.mjs +75 -26
- package/scripts/daemon/context-compiler.mjs +51 -37
- package/scripts/daemon/deliver.mjs +30 -1
- package/scripts/daemon/dispatcher.mjs +595 -149
- package/scripts/daemon/health.mjs +14 -1
- package/scripts/daemon/maestro-daemon.mjs +11 -0
- package/scripts/daemon/prompt-builder.mjs +24 -0
- package/scripts/daemon/responder.mjs +246 -79
- package/scripts/daemon/sdk-version.mjs +98 -16
- package/scripts/eval/probe-gateway.mjs +635 -0
- package/scripts/eval/replay/extract.mjs +270 -0
- package/scripts/eval/replay/grade.mjs +260 -0
- package/scripts/eval/replay/lib/config.mjs +50 -0
- package/scripts/eval/replay/lib/effects.mjs +65 -0
- package/scripts/eval/replay/lib/fixture.mjs +188 -0
- package/scripts/eval/replay/lib/judge.mjs +72 -0
- package/scripts/eval/replay/lib/redact.mjs +136 -0
- package/scripts/eval/replay/lib/sandbox.mjs +170 -0
- package/scripts/eval/replay/lib/schema-check.mjs +63 -0
- package/scripts/eval/replay/lib/transcript.mjs +76 -0
- package/scripts/eval/replay/mcp-replay-stub.mjs +101 -0
- package/scripts/eval/replay/report.mjs +185 -0
- package/scripts/eval/replay/run.mjs +404 -0
- package/scripts/fleet/rollout.mjs +1151 -0
- package/scripts/hooks/pre-send-audit.sh +36 -245
- package/scripts/hooks/pre-write-yaml-validate.mjs +275 -0
- package/scripts/hooks/validate-state-yaml.sh +190 -0
- package/scripts/huddle/huddle-llm.mjs +361 -0
- package/scripts/huddle/huddle-server.mjs +46 -121
- package/scripts/local-triggers/autoupdate.sh +465 -81
- package/scripts/local-triggers/run-trigger.sh +13 -0
- package/scripts/maintenance/pin-integrity.mjs +364 -0
- package/scripts/poll-slack-events.sh +41 -9
- package/scripts/poller/slack-socket-mode.mjs +28 -3
- package/scripts/session/supervisor.mjs +80 -13
- package/scripts/spawn-session.sh +13 -0
- package/bin/maestro.test.mjs +0 -1574
- package/lib/action-executor.test.mjs +0 -871
- package/lib/archetype.test.mjs +0 -132
- package/lib/assurance/plan-note.test.mjs +0 -234
- package/lib/assurance/room-budget.test.mjs +0 -486
- package/lib/assurance/tier.test.mjs +0 -174
- package/lib/autonomy.test.mjs +0 -66
- package/lib/backlog.test.mjs +0 -302
- package/lib/backup/policy.test.mjs +0 -305
- package/lib/budget-escalate.test.mjs +0 -232
- package/lib/budget-guard.envelope.test.mjs +0 -476
- package/lib/budget-guard.test.mjs +0 -427
- package/lib/cadence-bus-requeue.test.mjs +0 -83
- package/lib/cadence-bus-schedule.test.mjs +0 -194
- package/lib/cadence-bus.test.mjs +0 -720
- package/lib/cadences.test.mjs +0 -230
- package/lib/capability/inventory.test.mjs +0 -232
- package/lib/capability.test.mjs +0 -78
- package/lib/channels/base-adapter.test.mjs +0 -590
- package/lib/channels/channels.test.mjs +0 -371
- package/lib/channels/contract.test.mjs +0 -162
- package/lib/channels/inbox-item.test.mjs +0 -368
- package/lib/channels/orgmail/adapter.test.mjs +0 -448
- package/lib/channels/pairing.test.mjs +0 -270
- package/lib/channels/repeat-suppressor.test.mjs +0 -134
- package/lib/channels/slack-adapter.test.mjs +0 -212
- package/lib/channels/telegram-adapter.test.mjs +0 -306
- package/lib/channels/voice/adapter.test.mjs +0 -278
- package/lib/channels/whatsapp/adapter-baileys.test.mjs +0 -359
- package/lib/channels/whatsapp/baileys-typing.test.mjs +0 -154
- package/lib/charter.test.mjs +0 -89
- package/lib/claude-bin.test.mjs +0 -131
- package/lib/cli/board.test.mjs +0 -227
- package/lib/cli/design.test.mjs +0 -270
- package/lib/cli/doctor-checks.test.mjs +0 -336
- package/lib/cli/global-setup-extras.test.mjs +0 -462
- package/lib/cli/inbox.test.mjs +0 -230
- package/lib/cli/session-ack.test.mjs +0 -63
- package/lib/cli/session.test.mjs +0 -613
- package/lib/collective/capture.test.mjs +0 -121
- package/lib/collective/cards.test.mjs +0 -114
- package/lib/collective/config.test.mjs +0 -123
- package/lib/collective/global-config.test.mjs +0 -220
- package/lib/collective/global-skills.test.mjs +0 -126
- package/lib/collective/presence.test.mjs +0 -95
- package/lib/collective/recall.test.mjs +0 -116
- package/lib/collective/vendor-skills.test.mjs +0 -306
- package/lib/comms/send-gate.test.mjs +0 -770
- package/lib/comms.test.mjs +0 -41
- package/lib/context/budget.test.mjs +0 -252
- package/lib/context/history-scope.test.mjs +0 -79
- package/lib/cost/ledger-row.test.mjs +0 -183
- package/lib/design/design-md.test.mjs +0 -318
- package/lib/design/fixtures/DESIGN.golden.md +0 -238
- package/lib/design/fixtures/PRODUCT.golden.md +0 -67
- package/lib/design/fixtures/foundation.json +0 -133
- package/lib/design/refresh-gate.test.mjs +0 -144
- package/lib/design/write.test.mjs +0 -241
- package/lib/diagnostics/alerts.test.mjs +0 -318
- package/lib/diagnostics/backup-freshness.test.mjs +0 -185
- package/lib/diagnostics/counters.test.mjs +0 -206
- package/lib/diagnostics/events.test.mjs +0 -290
- package/lib/diagnostics/otel.test.mjs +0 -196
- package/lib/diagnostics/trace.test.mjs +0 -251
- package/lib/env-compat.test.mjs +0 -104
- package/lib/execution/disposition.test.mjs +0 -553
- package/lib/execution/drive.test.mjs +0 -270
- package/lib/execution/effects.test.mjs +0 -344
- package/lib/execution/intake.test.mjs +0 -389
- package/lib/execution/journal.test.mjs +0 -261
- package/lib/execution/match.test.mjs +0 -235
- package/lib/execution/pipeline.test.mjs +0 -392
- package/lib/execution/route.test.mjs +0 -186
- package/lib/execution/surface-policy.test.mjs +0 -162
- package/lib/fs-atomic.test.mjs +0 -72
- package/lib/fs-ownership.test.mjs +0 -158
- package/lib/goals/admission.test.mjs +0 -164
- package/lib/goals/classify.test.mjs +0 -167
- package/lib/goals/collaborate.test.mjs +0 -336
- package/lib/goals/gaps.test.mjs +0 -284
- package/lib/goals/loop.test.mjs +0 -845
- package/lib/hooks/bus.test.mjs +0 -387
- package/lib/identity/persona.test.mjs +0 -142
- package/lib/kpi-sensors.test.mjs +0 -278
- package/lib/kpi.test.mjs +0 -244
- package/lib/learning/config.test.mjs +0 -75
- package/lib/learning/counters.test.mjs +0 -69
- package/lib/learning/curator-consolidate.test.mjs +0 -238
- package/lib/learning/curator.test.mjs +0 -106
- package/lib/learning/reflect.test.mjs +0 -0
- package/lib/learning/session-index.test.mjs +0 -125
- package/lib/learning/skill-writer.test.mjs +0 -210
- package/lib/mandate/audit.test.mjs +0 -195
- package/lib/mandate/contract.test.mjs +0 -185
- package/lib/mandate/derive.test.mjs +0 -274
- package/lib/mandate/model.test.mjs +0 -164
- package/lib/mandate/refresh.test.mjs +0 -389
- package/lib/mcp/server.test.mjs +0 -426
- package/lib/model-router/auth-profiles.test.mjs +0 -580
- package/lib/model-router/catalog.test.mjs +0 -385
- package/lib/model-router/economics.test.mjs +0 -438
- package/lib/model-router/failover.test.mjs +0 -439
- package/lib/model-router/health.test.mjs +0 -338
- package/lib/model-router/integration-coverage.test.mjs +0 -831
- package/lib/model-router/integration.test.mjs +0 -564
- package/lib/model-router/ledger.test.mjs +0 -415
- package/lib/model-router/llm-task.test.mjs +0 -392
- package/lib/model-router/org-credentials.test.mjs +0 -265
- package/lib/model-router/pricing-refresh.test.mjs +0 -286
- package/lib/model-router/reconcile.test.mjs +0 -316
- package/lib/model-router/repair.test.mjs +0 -180
- package/lib/model-router/spawn.test.mjs +0 -446
- package/lib/model-router/taxonomy.test.mjs +0 -410
- package/lib/model-router.test.mjs +0 -1207
- package/lib/org/activity.test.mjs +0 -134
- package/lib/org/approvals.test.mjs +0 -216
- package/lib/org/awareness.test.mjs +0 -159
- package/lib/org/board-mine-cache.test.mjs +0 -53
- package/lib/org/board.test.mjs +0 -187
- package/lib/org/bootstrap-context.test.mjs +0 -153
- package/lib/org/client.test.mjs +0 -1206
- package/lib/org/cohort-client.test.mjs +0 -126
- package/lib/org/cost-sync.test.mjs +0 -153
- package/lib/org/doctor.test.mjs +0 -346
- package/lib/org/engagement-ledger.test.mjs +0 -112
- package/lib/org/engagement.test.mjs +0 -739
- package/lib/org/handoff.test.mjs +0 -269
- package/lib/org/inbound/directedness.test.mjs +0 -668
- package/lib/org/inbound/facts.test.mjs +0 -471
- package/lib/org/inbound/hydrate.test.mjs +0 -908
- package/lib/org/inbound/index.test.mjs +0 -429
- package/lib/org/inbound/project.test.mjs +0 -287
- package/lib/org/integration-tools.test.mjs +0 -160
- package/lib/org/keys.test.mjs +0 -92
- package/lib/org/knowledge.test.mjs +0 -326
- package/lib/org/leases.test.mjs +0 -235
- package/lib/org/mesh-directives.test.mjs +0 -110
- package/lib/org/mesh-integration.test.mjs +0 -127
- package/lib/org/mesh.test.mjs +0 -400
- package/lib/org/messaging.test.mjs +0 -471
- package/lib/org/param-contract.test.mjs +0 -477
- package/lib/org/policy.test.mjs +0 -237
- package/lib/org/protocol.checksum.test.mjs +0 -90
- package/lib/org/protocol.test.mjs +0 -323
- package/lib/org/push.test.mjs +0 -792
- package/lib/org/registry.test.mjs +0 -100
- package/lib/org/resource-tools.test.mjs +0 -361
- package/lib/org/tool-access.test.mjs +0 -144
- package/lib/org/tool-surface-integration.test.mjs +0 -120
- package/lib/org/tool-surface.test.mjs +0 -1268
- package/lib/org/typing.test.mjs +0 -291
- package/lib/org/ui-parity.test.mjs +0 -560
- package/lib/org/verify.test.mjs +0 -194
- package/lib/org/work-ledger.test.mjs +0 -273
- package/lib/plan/adoption-e2e.test.mjs +0 -366
- package/lib/plan/budget-enforcement.test.mjs +0 -400
- package/lib/plan/compile.test.mjs +0 -382
- package/lib/plan/emit.test.mjs +0 -269
- package/lib/plan/explain.test.mjs +0 -188
- package/lib/prompts/parallelism.test.mjs +0 -177
- package/lib/rag/rag.test.mjs +0 -505
- package/lib/rate-guard.test.mjs +0 -272
- package/lib/reactive-gate.test.mjs +0 -57
- package/lib/render.test.mjs +0 -68
- package/lib/resource-governor.test.mjs +0 -488
- package/lib/scheduling/dynamic-jobs.test.mjs +0 -344
- package/lib/scheduling/jitter.test.mjs +0 -140
- package/lib/secrets/broker.test.mjs +0 -280
- package/lib/secrets/providers.test.mjs +0 -274
- package/lib/security/audit-engine.test.mjs +0 -424
- package/lib/security/coerce-args.test.mjs +0 -281
- package/lib/security/dangerous-tools.test.mjs +0 -68
- package/lib/security/external-content.test.mjs +0 -84
- package/lib/security/redact.test.mjs +0 -441
- package/lib/security/secret-equal.test.mjs +0 -55
- package/lib/session/config.test.mjs +0 -92
- package/lib/session/feed-core.test.mjs +0 -198
- package/lib/session/first-run.test.mjs +0 -121
- package/lib/session/frontdoor.test.mjs +0 -205
- package/lib/session/handoffs.test.mjs +0 -183
- package/lib/session/identity.test.mjs +0 -180
- package/lib/session/inbox-claims.test.mjs +0 -286
- package/lib/session/launch-args.test.mjs +0 -157
- package/lib/session/liveness.test.mjs +0 -100
- package/lib/session/status-summary.test.mjs +0 -118
- package/lib/session-permissions.test.mjs +0 -120
- package/lib/setup/claude-probe.test.mjs +0 -187
- package/lib/setup/completeness.test.mjs +0 -110
- package/lib/setup/context-pack.test.mjs +0 -89
- package/lib/setup/enrich.test.mjs +0 -115
- package/lib/setup/enroll-from-cohort.test.mjs +0 -300
- package/lib/setup/integration.test.mjs +0 -162
- package/lib/setup/io.test.mjs +0 -77
- package/lib/setup/runner.test.mjs +0 -132
- package/lib/setup/sections/identity.test.mjs +0 -234
- package/lib/setup/sections/inventory.test.mjs +0 -198
- package/lib/setup/sections/learning.test.mjs +0 -81
- package/lib/setup/sections/mandate.test.mjs +0 -388
- package/lib/setup/sections/messaging.test.mjs +0 -127
- package/lib/setup/sections/model.test.mjs +0 -240
- package/lib/setup/sections/org.test.mjs +0 -346
- package/lib/setup/sections/orgmail.test.mjs +0 -118
- package/lib/setup/sections/recovery.test.mjs +0 -98
- package/lib/setup/sections/subagents.test.mjs +0 -429
- package/lib/setup/sections/verify.test.mjs +0 -175
- package/lib/setup/sot.test.mjs +0 -81
- package/lib/setup/state.test.mjs +0 -115
- package/lib/singleton.test.mjs +0 -151
- package/lib/subagents/cli.test.mjs +0 -389
- package/lib/subagents/client.test.mjs +0 -309
- package/lib/subagents/gap.test.mjs +0 -234
- package/lib/subagents/lock.test.mjs +0 -248
- package/lib/subagents/manifest.test.mjs +0 -175
- package/lib/subagents/refs.test.mjs +0 -204
- package/lib/subagents/resolve.test.mjs +0 -422
- package/lib/subagents/schema.test.mjs +0 -328
- package/lib/telemetry/alerts.test.mjs +0 -109
- package/lib/telemetry/collect.test.mjs +0 -1274
- package/lib/tool-definitions-integration.test.mjs +0 -83
- package/lib/tool-definitions.test.mjs +0 -437
- package/lib/upgrade/global-refresh.test.mjs +0 -65
- package/lib/upgrade/launchd-reconcile.test.mjs +0 -272
- package/lib/upgrade/post-steps.test.mjs +0 -200
- package/lib/upgrade/verify.test.mjs +0 -164
- package/lib/util/fetch-timeout.test.mjs +0 -202
- package/lib/util/reconnect.test.mjs +0 -369
- package/lib/util/unhandled.test.mjs +0 -216
- package/lib/voice/outbound.test.mjs +0 -69
- package/lib/voice/session-rotation.test.mjs +0 -114
- package/lib/voice/stt.test.mjs +0 -226
- package/lib/voice/voice.test.mjs +0 -990
- package/scripts/cadence/enqueue-cadence-tick.test.mjs +0 -187
- package/scripts/ci/check-docs-accuracy.test.mjs +0 -409
- package/scripts/ci/check-durable-write-seam.test.mjs +0 -90
- package/scripts/ci/check-no-build-artifacts.test.mjs +0 -71
- package/scripts/ci/check-no-residual-identity.test.mjs +0 -202
- package/scripts/ci/check-skill-packs.test.mjs +0 -495
- package/scripts/ci/check-subagent-frontmatter.test.mjs +0 -124
- package/scripts/ci/check.test.mjs +0 -194
- package/scripts/ci/conformance-org-api.test.mjs +0 -425
- package/scripts/cloud-relay/voice/relay-identity.test.mjs +0 -96
- package/scripts/collective/hook-runner.test.mjs +0 -173
- package/scripts/cost/fleet-digest.test.mjs +0 -207
- package/scripts/cost/track-claude-usage-pricing.test.mjs +0 -183
- package/scripts/cost/track-claude-usage.test.mjs +0 -148
- package/scripts/daemon/agent-daemon-board-mine.test.mjs +0 -96
- package/scripts/daemon/agent-daemon-design.test.mjs +0 -238
- package/scripts/daemon/agent-daemon-frontdoor.test.mjs +0 -60
- package/scripts/daemon/agent-daemon.test.mjs +0 -995
- package/scripts/daemon/assurance-e2e.test.mjs +0 -613
- package/scripts/daemon/assurance.test.mjs +0 -1791
- package/scripts/daemon/board-mirror.test.mjs +0 -165
- package/scripts/daemon/cadence-consumer-frontdoor.test.mjs +0 -393
- package/scripts/daemon/cadence-consumer-governance.test.mjs +0 -276
- package/scripts/daemon/cadence-consumer.test.mjs +0 -776
- package/scripts/daemon/cadence-handlers.test.mjs +0 -837
- package/scripts/daemon/classifier-identity.test.mjs +0 -137
- package/scripts/daemon/classifier.test.mjs +0 -266
- package/scripts/daemon/classify-kind.test.mjs +0 -40
- package/scripts/daemon/context-compiler.test.mjs +0 -406
- package/scripts/daemon/deliver.test.mjs +0 -564
- package/scripts/daemon/dispatcher-cooldown.test.mjs +0 -122
- package/scripts/daemon/dispatcher-governance.test.mjs +0 -1013
- package/scripts/daemon/dispatcher-resume.test.mjs +0 -166
- package/scripts/daemon/dispatcher-session-continuity.test.mjs +0 -365
- package/scripts/daemon/execution-ladder.test.mjs +0 -470
- package/scripts/daemon/goal-steward-cadence.test.mjs +0 -312
- package/scripts/daemon/inbox-deferral-session.test.mjs +0 -49
- package/scripts/daemon/inbox-deferral.test.mjs +0 -336
- package/scripts/daemon/inbox-wake.test.mjs +0 -199
- package/scripts/daemon/integration.test.mjs +0 -149
- package/scripts/daemon/lib/self-echo.test.mjs +0 -153
- package/scripts/daemon/lib/session-router.test.mjs +0 -554
- package/scripts/daemon/prompt-builder-preamble.test.mjs +0 -210
- package/scripts/daemon/prompt-builder.test.mjs +0 -556
- package/scripts/daemon/responder-cost.test.mjs +0 -68
- package/scripts/daemon/responder-history.test.mjs +0 -221
- package/scripts/daemon/sdk-version.test.mjs +0 -31
- package/scripts/daemon/session-lock.test.mjs +0 -252
- package/scripts/daemon/session-outcomes.test.mjs +0 -533
- package/scripts/daemon/typing-registry.test.mjs +0 -102
- package/scripts/hooks/pre-send-audit.test.mjs +0 -354
- package/scripts/huddle/huddle-prompt.test.mjs +0 -176
- package/scripts/local-triggers/autoupdate.test.mjs +0 -518
- package/scripts/local-triggers/generate-plists.test.mjs +0 -456
- package/scripts/media-generation/brand-clause.test.mjs +0 -135
- package/scripts/org/send-orgmail.first-contact.test.mjs +0 -102
- package/scripts/poller/inbox-privilege-injection.test.mjs +0 -167
- package/scripts/poller/inbox-scan-poller.test.mjs +0 -295
- package/scripts/poller/lib/cloud-relay-dedup.test.mjs +0 -133
- package/scripts/poller/slack-socket-mode.test.mjs +0 -805
- package/scripts/poller-launchd/install.test.mjs +0 -243
- package/scripts/restore-from-backup.test.mjs +0 -181
- package/scripts/session/feed.test.mjs +0 -196
- package/scripts/session/supervisor-sh.test.mjs +0 -218
- package/scripts/session/supervisor.test.mjs +0 -482
- package/scripts/setup/configure-macos.test.mjs +0 -306
- package/scripts/setup/gen-subagent-manifest.test.mjs +0 -124
- package/scripts/setup/generate-agent-package-json.test.mjs +0 -143
- package/scripts/setup/generate-capability.test.mjs +0 -134
- package/scripts/setup/init-agent.test.mjs +0 -370
- package/scripts/setup/init-skill-marketplace.test.mjs +0 -193
- package/scripts/vendor/sync-skill-packs.test.mjs +0 -103
- package/scripts/watchdog/memory-watchdog.test.mjs +0 -64
|
@@ -1,871 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* action-executor.test.mjs — contract test for the tool-call chokepoint.
|
|
3
|
-
*
|
|
4
|
-
* action-executor.js is where Claude `tool_use` blocks become real side
|
|
5
|
-
* effects (spawning `claude --print`, running agent-local shell scripts,
|
|
6
|
-
* writing draft files). This test pins the *contract* of executeAction()
|
|
7
|
-
* without triggering any of those side effects:
|
|
8
|
-
*
|
|
9
|
-
* - the return value is always { success: boolean, result: string }
|
|
10
|
-
* - access-level gating rejects tools the caller is not authorised for
|
|
11
|
-
* - an authorised-but-unknown tool name returns a failure object (no throw)
|
|
12
|
-
*
|
|
13
|
-
* Side effects are avoided by (a) pointing AGENT_ROOT at a throwaway temp dir
|
|
14
|
-
* so any stray script/draft write lands in /tmp, (b) pointing CLAUDE_BIN at a
|
|
15
|
-
* harmless echo stub so no lookup path could ever reach the real `claude`
|
|
16
|
-
* binary, and (c) choosing tool names + access levels that exercise the
|
|
17
|
-
* validation / dispatch branches *before* any spawn. The lookup and
|
|
18
|
-
* write-script branches that genuinely spawn are deliberately NOT exercised.
|
|
19
|
-
*
|
|
20
|
-
* Pure node:test; no extra deps.
|
|
21
|
-
*/
|
|
22
|
-
|
|
23
|
-
import { test } from "node:test";
|
|
24
|
-
import assert from "node:assert/strict";
|
|
25
|
-
import {
|
|
26
|
-
mkdtempSync,
|
|
27
|
-
writeFileSync,
|
|
28
|
-
chmodSync,
|
|
29
|
-
rmSync,
|
|
30
|
-
readFileSync,
|
|
31
|
-
existsSync,
|
|
32
|
-
} from "node:fs";
|
|
33
|
-
import { tmpdir } from "node:os";
|
|
34
|
-
import { join } from "node:path";
|
|
35
|
-
|
|
36
|
-
// --- sandbox setup: must happen BEFORE importing the module under test, since
|
|
37
|
-
// it reads AGENT_ROOT at module-eval time (const AGENT_ROOT = ...). ----------
|
|
38
|
-
|
|
39
|
-
const sandbox = mkdtempSync(join(tmpdir(), "maestro-action-executor-"));
|
|
40
|
-
|
|
41
|
-
// A harmless executable that ignores its args and prints a fixed line. If any
|
|
42
|
-
// path we test were to spawn it (it should not), nothing real happens.
|
|
43
|
-
const stubBin = join(sandbox, "claude-stub.sh");
|
|
44
|
-
writeFileSync(stubBin, "#!/bin/sh\necho 'STUB: no real claude invoked'\n");
|
|
45
|
-
chmodSync(stubBin, 0o755);
|
|
46
|
-
|
|
47
|
-
process.env.AGENT_ROOT = sandbox;
|
|
48
|
-
process.env.CLAUDE_BIN = stubBin;
|
|
49
|
-
|
|
50
|
-
const {
|
|
51
|
-
executeAction,
|
|
52
|
-
executeLookup,
|
|
53
|
-
resolveLookupModel,
|
|
54
|
-
resolveEffectiveAccessLevel,
|
|
55
|
-
LOOKUP_TIMEOUT_MS,
|
|
56
|
-
LOOKUP_MAX_BUFFER,
|
|
57
|
-
SCRIPT_TIMEOUT_MS,
|
|
58
|
-
LOOKUP_DEFAULT_MODEL,
|
|
59
|
-
} = await import("./action-executor.js");
|
|
60
|
-
const toolDefs = await import("./tool-definitions.js");
|
|
61
|
-
|
|
62
|
-
// Resolve a concrete (access level, authorised tool name) pair from the real
|
|
63
|
-
// tool registry so the "unknown tool" test actually passes authorisation and
|
|
64
|
-
// reaches the dispatch switch — rather than being short-circuited by the
|
|
65
|
-
// access gate. Probe the documented access levels and pick the first that
|
|
66
|
-
// yields a non-empty tool list.
|
|
67
|
-
function pickAuthorisedLevel() {
|
|
68
|
-
for (const level of ["ceo", "leadership", "default"]) {
|
|
69
|
-
let names;
|
|
70
|
-
try {
|
|
71
|
-
names = toolDefs.getToolNamesForAccessLevel(level);
|
|
72
|
-
} catch {
|
|
73
|
-
continue;
|
|
74
|
-
}
|
|
75
|
-
if (Array.isArray(names) && names.length > 0) {
|
|
76
|
-
return { level, names };
|
|
77
|
-
}
|
|
78
|
-
}
|
|
79
|
-
return null;
|
|
80
|
-
}
|
|
81
|
-
|
|
82
|
-
const authorised = pickAuthorisedLevel();
|
|
83
|
-
|
|
84
|
-
// A caller object shaped like the one the dispatcher builds.
|
|
85
|
-
function caller(accessLevel) {
|
|
86
|
-
return { slug: "test-caller", name: "Test Caller", accessLevel };
|
|
87
|
-
}
|
|
88
|
-
|
|
89
|
-
function assertResultShape(res) {
|
|
90
|
-
assert.ok(res && typeof res === "object", "result must be an object");
|
|
91
|
-
assert.equal(typeof res.success, "boolean", "result.success must be a boolean");
|
|
92
|
-
assert.equal(typeof res.result, "string", "result.result must be a string");
|
|
93
|
-
}
|
|
94
|
-
|
|
95
|
-
test("setup: tool registry exposes at least one authorised access level", () => {
|
|
96
|
-
assert.ok(
|
|
97
|
-
authorised,
|
|
98
|
-
"expected getToolNamesForAccessLevel to return a non-empty list for one of ceo/leadership/default",
|
|
99
|
-
);
|
|
100
|
-
});
|
|
101
|
-
|
|
102
|
-
test("unauthorised access level is rejected with a failure object (no spawn)", async () => {
|
|
103
|
-
// Use a tool that genuinely exists for *some* level, paired with an access
|
|
104
|
-
// level that almost certainly does not include it. We confirm the chosen
|
|
105
|
-
// level does NOT contain the tool before asserting rejection, so the test is
|
|
106
|
-
// robust to registry changes.
|
|
107
|
-
const toolName = authorised.names[0];
|
|
108
|
-
|
|
109
|
-
// Find a bogus level whose tool list does NOT include toolName.
|
|
110
|
-
let denyLevel = "nonexistent-access-level";
|
|
111
|
-
try {
|
|
112
|
-
const denyNames = toolDefs.getToolNamesForAccessLevel(denyLevel);
|
|
113
|
-
if (Array.isArray(denyNames) && denyNames.includes(toolName)) {
|
|
114
|
-
// Unexpected: the bogus level grants the tool. Skip the precondition
|
|
115
|
-
// assertion but the gate should still reject it consistently below.
|
|
116
|
-
}
|
|
117
|
-
} catch {
|
|
118
|
-
// getToolNamesForAccessLevel may throw on an unknown level; that's fine —
|
|
119
|
-
// executeAction calls it the same way, so we instead use a known level
|
|
120
|
-
// that is guaranteed not to contain the tool, if one exists.
|
|
121
|
-
}
|
|
122
|
-
|
|
123
|
-
const res = await executeAction(toolName, {}, caller(denyLevel), "session-1");
|
|
124
|
-
assertResultShape(res);
|
|
125
|
-
assert.equal(res.success, false, "unauthorised call must not succeed");
|
|
126
|
-
assert.match(
|
|
127
|
-
res.result,
|
|
128
|
-
/permission/i,
|
|
129
|
-
"rejection message should mention permission",
|
|
130
|
-
);
|
|
131
|
-
});
|
|
132
|
-
|
|
133
|
-
test("unknown tool name (but authorised level) returns failure, does not throw", async () => {
|
|
134
|
-
// "__definitely_not_a_real_tool__" is not in any access list, so to reach the
|
|
135
|
-
// dispatch switch we would normally be blocked by the gate. Instead we assert
|
|
136
|
-
// the safe, observable contract: an unrecognised tool always yields a
|
|
137
|
-
// { success:false, result:string } object and never throws.
|
|
138
|
-
const res = await executeAction(
|
|
139
|
-
"__definitely_not_a_real_tool__",
|
|
140
|
-
{},
|
|
141
|
-
caller(authorised.level),
|
|
142
|
-
"session-2",
|
|
143
|
-
);
|
|
144
|
-
assertResultShape(res);
|
|
145
|
-
assert.equal(res.success, false, "unknown tool must not report success");
|
|
146
|
-
assert.match(
|
|
147
|
-
res.result,
|
|
148
|
-
/permission|unknown/i,
|
|
149
|
-
"unknown tool should be denied or reported as unknown, never silently succeed",
|
|
150
|
-
);
|
|
151
|
-
});
|
|
152
|
-
|
|
153
|
-
test("rejected (unauthorised) call never reaches a spawn — result is the gate message verbatim", async () => {
|
|
154
|
-
// The access gate runs before the try/catch and before any execFile. A tool
|
|
155
|
-
// that DOES require a spawn, denied at the gate, must come back with the
|
|
156
|
-
// canonical permission string and not an 'Action failed' / spawn error.
|
|
157
|
-
const res = await executeAction("slack_send", { recipient: "x", message: "y" }, caller("nonexistent-access-level"), "session-3");
|
|
158
|
-
assertResultShape(res);
|
|
159
|
-
assert.equal(res.success, false);
|
|
160
|
-
assert.equal(
|
|
161
|
-
res.result,
|
|
162
|
-
"You do not have permission to perform this action.",
|
|
163
|
-
"denied call should return the canonical gate message, proving no spawn occurred",
|
|
164
|
-
);
|
|
165
|
-
});
|
|
166
|
-
|
|
167
|
-
test("contract holds across a sweep of tool names and access levels", async () => {
|
|
168
|
-
// An unknown access level now coerces to the least-privilege "default"
|
|
169
|
-
// (read-only) surface — NOT an empty set (audit M4 / CEO-case footgun). So the
|
|
170
|
-
// genuinely-denied sweep must use WRITE/side-effecting tools, which "default"
|
|
171
|
-
// never grants, plus unknown names. None of these reach a spawn or a write.
|
|
172
|
-
const deniedAtDefault = [
|
|
173
|
-
"slack_send",
|
|
174
|
-
"draft_email",
|
|
175
|
-
"whatsapp_send",
|
|
176
|
-
"generate_report",
|
|
177
|
-
"__bogus__",
|
|
178
|
-
"",
|
|
179
|
-
];
|
|
180
|
-
for (const t of deniedAtDefault) {
|
|
181
|
-
const res = await executeAction(t, {}, caller("nonexistent-access-level"), "sweep");
|
|
182
|
-
assertResultShape(res);
|
|
183
|
-
assert.equal(res.success, false, `denied "${t}" must not succeed at default`);
|
|
184
|
-
}
|
|
185
|
-
|
|
186
|
-
// Two authorised-but-unknown names at an authorised level: they pass the gate
|
|
187
|
-
// only if the registry grants them (it does not), so they exercise the
|
|
188
|
-
// dispatch 'default'/permission branches without any spawn or write.
|
|
189
|
-
for (const t of ["__bogus__", ""]) {
|
|
190
|
-
const res = await executeAction(t, {}, caller(authorised.level), "sweep");
|
|
191
|
-
assertResultShape(res);
|
|
192
|
-
assert.equal(res.success, false, `unknown "${t}" must not succeed`);
|
|
193
|
-
}
|
|
194
|
-
});
|
|
195
|
-
|
|
196
|
-
// ---------------------------------------------------------------------------
|
|
197
|
-
// Access-level integrity — defence-in-depth (audit M4) + CEO-case footgun
|
|
198
|
-
// ---------------------------------------------------------------------------
|
|
199
|
-
|
|
200
|
-
test("resolveEffectiveAccessLevel falls back to self-asserted level with no provenance", () => {
|
|
201
|
-
assert.equal(resolveEffectiveAccessLevel({ accessLevel: "leadership" }), "leadership");
|
|
202
|
-
assert.equal(
|
|
203
|
-
resolveEffectiveAccessLevel({ accessLevel: "ceo", verifiedAccessLevel: null }),
|
|
204
|
-
"ceo",
|
|
205
|
-
);
|
|
206
|
-
});
|
|
207
|
-
|
|
208
|
-
test("resolveEffectiveAccessLevel defaults to least privilege when nothing is provided", () => {
|
|
209
|
-
assert.equal(resolveEffectiveAccessLevel(), "default");
|
|
210
|
-
assert.equal(resolveEffectiveAccessLevel(null), "default");
|
|
211
|
-
assert.equal(resolveEffectiveAccessLevel({}), "default");
|
|
212
|
-
assert.equal(resolveEffectiveAccessLevel({ slug: "x" }), "default");
|
|
213
|
-
});
|
|
214
|
-
|
|
215
|
-
test("resolveEffectiveAccessLevel clamps a self-asserted level above the verified level", () => {
|
|
216
|
-
// Self-asserted 'ceo' must not beat a verified 'default'.
|
|
217
|
-
assert.equal(
|
|
218
|
-
resolveEffectiveAccessLevel({ accessLevel: "ceo", verifiedAccessLevel: "default" }),
|
|
219
|
-
"default",
|
|
220
|
-
);
|
|
221
|
-
assert.equal(
|
|
222
|
-
resolveEffectiveAccessLevel({ accessLevel: "ceo", verifiedAccessLevel: "leadership" }),
|
|
223
|
-
"leadership",
|
|
224
|
-
);
|
|
225
|
-
});
|
|
226
|
-
|
|
227
|
-
test("resolveEffectiveAccessLevel honours a self-asserted level that only narrows", () => {
|
|
228
|
-
assert.equal(
|
|
229
|
-
resolveEffectiveAccessLevel({ accessLevel: "default", verifiedAccessLevel: "ceo" }),
|
|
230
|
-
"default",
|
|
231
|
-
);
|
|
232
|
-
assert.equal(
|
|
233
|
-
resolveEffectiveAccessLevel({ accessLevel: "leadership", verifiedAccessLevel: "ceo" }),
|
|
234
|
-
"leadership",
|
|
235
|
-
);
|
|
236
|
-
});
|
|
237
|
-
|
|
238
|
-
test("resolveEffectiveAccessLevel uses the verified level when none is self-asserted", () => {
|
|
239
|
-
assert.equal(
|
|
240
|
-
resolveEffectiveAccessLevel({ verifiedAccessLevel: "leadership" }),
|
|
241
|
-
"leadership",
|
|
242
|
-
);
|
|
243
|
-
});
|
|
244
|
-
|
|
245
|
-
test("resolveEffectiveAccessLevel normalises casing on asserted and verified levels", () => {
|
|
246
|
-
assert.equal(
|
|
247
|
-
resolveEffectiveAccessLevel({ accessLevel: "CEO", verifiedAccessLevel: "Leadership" }),
|
|
248
|
-
"leadership",
|
|
249
|
-
);
|
|
250
|
-
assert.equal(resolveEffectiveAccessLevel({ accessLevel: " Ceo " }), "ceo");
|
|
251
|
-
});
|
|
252
|
-
|
|
253
|
-
test("resolveEffectiveAccessLevel never throws on garbage input", () => {
|
|
254
|
-
assert.doesNotThrow(() =>
|
|
255
|
-
resolveEffectiveAccessLevel({ accessLevel: 42, verifiedAccessLevel: 99 }),
|
|
256
|
-
);
|
|
257
|
-
assert.equal(
|
|
258
|
-
resolveEffectiveAccessLevel({ accessLevel: 42, verifiedAccessLevel: 99 }),
|
|
259
|
-
"default",
|
|
260
|
-
);
|
|
261
|
-
});
|
|
262
|
-
|
|
263
|
-
test("executeAction clamps a self-asserted ceo down to verified default (denied write tool)", async () => {
|
|
264
|
-
// The M4 attack: caller self-asserts ceo but provenance only grants default.
|
|
265
|
-
// slack_send is a write tool absent from default, so it must be denied.
|
|
266
|
-
const res = await executeAction(
|
|
267
|
-
"slack_send",
|
|
268
|
-
{ recipient: "x", message: "y" },
|
|
269
|
-
{ slug: "spoof", name: "Spoof", accessLevel: "ceo", verifiedAccessLevel: "default" },
|
|
270
|
-
"session-m4-1",
|
|
271
|
-
);
|
|
272
|
-
assertResultShape(res);
|
|
273
|
-
assert.equal(res.success, false);
|
|
274
|
-
assert.match(res.result, /permission/i);
|
|
275
|
-
});
|
|
276
|
-
|
|
277
|
-
test("executeAction allows a CEO write tool when verified level is ceo (passes the auth gate)", async () => {
|
|
278
|
-
// Verified provenance grants ceo; slack_send passes authorisation. It then
|
|
279
|
-
// dispatches to a script that does not exist in the sandbox, so it returns a
|
|
280
|
-
// failure — but NOT the permission gate message. We assert it cleared the gate.
|
|
281
|
-
const res = await executeAction(
|
|
282
|
-
"slack_send",
|
|
283
|
-
{ recipient: "x", message: "y" },
|
|
284
|
-
{ slug: "boss", name: "Boss", accessLevel: "ceo", verifiedAccessLevel: "ceo" },
|
|
285
|
-
"session-m4-2",
|
|
286
|
-
);
|
|
287
|
-
assertResultShape(res);
|
|
288
|
-
assert.doesNotMatch(
|
|
289
|
-
res.result,
|
|
290
|
-
/You do not have permission/i,
|
|
291
|
-
"verified ceo should clear the authorisation gate",
|
|
292
|
-
);
|
|
293
|
-
});
|
|
294
|
-
|
|
295
|
-
test("executeAction: uppercase access level no longer silently under-privileges a write tool", async () => {
|
|
296
|
-
// CEO-case footgun: 'CEO' used to yield 0 tools => every tool denied. Now it
|
|
297
|
-
// normalises to ceo, so slack_send clears the gate (and fails later at the
|
|
298
|
-
// missing sandbox script, not at authorisation).
|
|
299
|
-
const res = await executeAction(
|
|
300
|
-
"slack_send",
|
|
301
|
-
{ recipient: "x", message: "y" },
|
|
302
|
-
{ slug: "boss", name: "Boss", accessLevel: "CEO" },
|
|
303
|
-
"session-m4-3",
|
|
304
|
-
);
|
|
305
|
-
assertResultShape(res);
|
|
306
|
-
assert.doesNotMatch(
|
|
307
|
-
res.result,
|
|
308
|
-
/You do not have permission/i,
|
|
309
|
-
"'CEO' (uppercase) should clear the gate, not be denied",
|
|
310
|
-
);
|
|
311
|
-
});
|
|
312
|
-
|
|
313
|
-
test("executeAction: a lookup tool is allowed at an unknown level (coerced to default)", async () => {
|
|
314
|
-
// Unknown level coerces to least-privilege default, which includes read-only
|
|
315
|
-
// lookups. search_email therefore clears the gate; CLAUDE_BIN points at the
|
|
316
|
-
// harmless stub so no real claude runs.
|
|
317
|
-
const res = await executeAction(
|
|
318
|
-
"search_email",
|
|
319
|
-
{ query: "anything" },
|
|
320
|
-
caller("totally-unknown-level"),
|
|
321
|
-
"session-m4-4",
|
|
322
|
-
);
|
|
323
|
-
assertResultShape(res);
|
|
324
|
-
assert.doesNotMatch(
|
|
325
|
-
res.result,
|
|
326
|
-
/You do not have permission/i,
|
|
327
|
-
"unknown level should grant the read-only default surface, not deny everything",
|
|
328
|
-
);
|
|
329
|
-
});
|
|
330
|
-
|
|
331
|
-
// ---------------------------------------------------------------------------
|
|
332
|
-
// Spawn tuning constants are exported and numeric (audit L25)
|
|
333
|
-
// ---------------------------------------------------------------------------
|
|
334
|
-
|
|
335
|
-
test("spawn tuning constants are exported as positive numbers with sane defaults", () => {
|
|
336
|
-
for (const [name, v] of [
|
|
337
|
-
["LOOKUP_TIMEOUT_MS", LOOKUP_TIMEOUT_MS],
|
|
338
|
-
["LOOKUP_MAX_BUFFER", LOOKUP_MAX_BUFFER],
|
|
339
|
-
["SCRIPT_TIMEOUT_MS", SCRIPT_TIMEOUT_MS],
|
|
340
|
-
]) {
|
|
341
|
-
assert.equal(typeof v, "number", `${name} must be a number`);
|
|
342
|
-
assert.ok(Number.isFinite(v) && v > 0, `${name} must be finite and > 0`);
|
|
343
|
-
}
|
|
344
|
-
// Defaults preserved when the env overrides are absent (CI sets none).
|
|
345
|
-
assert.equal(LOOKUP_TIMEOUT_MS, 25000);
|
|
346
|
-
assert.equal(LOOKUP_MAX_BUFFER, 1024 * 1024);
|
|
347
|
-
assert.equal(SCRIPT_TIMEOUT_MS, 15000);
|
|
348
|
-
});
|
|
349
|
-
|
|
350
|
-
// ---------------------------------------------------------------------------
|
|
351
|
-
// Internal lookup prompts still resolve via their write-tool executors (L24)
|
|
352
|
-
// ---------------------------------------------------------------------------
|
|
353
|
-
|
|
354
|
-
test("internal lookup names are NOT directly dispatchable from executeAction", async () => {
|
|
355
|
-
// queue_update_internal / create_item_internal moved out of LOOKUP_PROMPTS
|
|
356
|
-
// into INTERNAL_LOOKUP_PROMPTS, so they are not public tool names. At ceo
|
|
357
|
-
// level (broadest surface) they must NOT be treated as a lookup; they are
|
|
358
|
-
// denied at the gate (not real tools) — never "No prompt for".
|
|
359
|
-
for (const name of ["queue_update_internal", "create_item_internal"]) {
|
|
360
|
-
const res = await executeAction(
|
|
361
|
-
name,
|
|
362
|
-
{},
|
|
363
|
-
{ slug: "boss", name: "Boss", accessLevel: "ceo", verifiedAccessLevel: "ceo" },
|
|
364
|
-
"internal-direct",
|
|
365
|
-
);
|
|
366
|
-
assertResultShape(res);
|
|
367
|
-
assert.equal(res.success, false, `${name} must not succeed when called directly`);
|
|
368
|
-
assert.doesNotMatch(
|
|
369
|
-
res.result,
|
|
370
|
-
/No prompt for/i,
|
|
371
|
-
`${name} should be gated/unknown, not reach the lookup resolver directly`,
|
|
372
|
-
);
|
|
373
|
-
}
|
|
374
|
-
});
|
|
375
|
-
|
|
376
|
-
test("queue_update / create_action_item resolve their internal prompt (no 'No prompt for')", async () => {
|
|
377
|
-
// These public write tools delegate to executeLookup with an INTERNAL_
|
|
378
|
-
// prompt name. With a verified-ceo caller they clear the auth gate, and the
|
|
379
|
-
// internal prompt must resolve (CLAUDE_BIN is the harmless stub, so the
|
|
380
|
-
// spawn is inert). The failure mode we guard against is the L24 regression
|
|
381
|
-
// where the internal prompt is missing → "No prompt for: ...".
|
|
382
|
-
for (const [tool, input] of [
|
|
383
|
-
["queue_update", { search_term: "x", new_status: "done" }],
|
|
384
|
-
["create_action_item", { title: "x" }],
|
|
385
|
-
]) {
|
|
386
|
-
const res = await executeAction(
|
|
387
|
-
tool,
|
|
388
|
-
input,
|
|
389
|
-
{ slug: "boss", name: "Boss", accessLevel: "ceo", verifiedAccessLevel: "ceo" },
|
|
390
|
-
"internal-delegate",
|
|
391
|
-
);
|
|
392
|
-
assertResultShape(res);
|
|
393
|
-
assert.doesNotMatch(
|
|
394
|
-
res.result,
|
|
395
|
-
/No prompt for/i,
|
|
396
|
-
`${tool} must resolve its internal prompt (INTERNAL_LOOKUP_PROMPTS wired)`,
|
|
397
|
-
);
|
|
398
|
-
assert.doesNotMatch(
|
|
399
|
-
res.result,
|
|
400
|
-
/You do not have permission/i,
|
|
401
|
-
`${tool} should clear the gate at verified ceo`,
|
|
402
|
-
);
|
|
403
|
-
}
|
|
404
|
-
});
|
|
405
|
-
|
|
406
|
-
// ---------------------------------------------------------------------------
|
|
407
|
-
// Self-auditing — executeAction writes one audit row per invocation, incl.
|
|
408
|
-
// denials and failures (audit observability F1/F3).
|
|
409
|
-
// ---------------------------------------------------------------------------
|
|
410
|
-
|
|
411
|
-
const todayUTC = () => new Date().toISOString().slice(0, 10);
|
|
412
|
-
|
|
413
|
-
// Read the rows executeAction appended to <root>/logs/audit/<date>-actions.jsonl.
|
|
414
|
-
// Each test points executeAction at its OWN throwaway root via options.agentRoot
|
|
415
|
-
// so the audit file holds exactly that test's rows.
|
|
416
|
-
function readAuditRows(root) {
|
|
417
|
-
const file = join(root, "logs", "audit", `${todayUTC()}-actions.jsonl`);
|
|
418
|
-
if (!existsSync(file)) return [];
|
|
419
|
-
return readFileSync(file, "utf8")
|
|
420
|
-
.split("\n")
|
|
421
|
-
.filter((l) => l.trim() !== "")
|
|
422
|
-
.map((l) => JSON.parse(l));
|
|
423
|
-
}
|
|
424
|
-
|
|
425
|
-
function freshRoot() {
|
|
426
|
-
return mkdtempSync(join(tmpdir(), "maestro-audit-"));
|
|
427
|
-
}
|
|
428
|
-
|
|
429
|
-
test("self-audit: a successful action writes a row with the real tool name and status completed", async () => {
|
|
430
|
-
const root = freshRoot();
|
|
431
|
-
// draft_email at verified ceo writes a local draft file and returns success
|
|
432
|
-
// with NO spawn — a deterministic success path for the audit assertion.
|
|
433
|
-
const res = await executeAction(
|
|
434
|
-
"draft_email",
|
|
435
|
-
{ to: "alex@example.com", subject: "Hi", body: "Body" },
|
|
436
|
-
{ slug: "boss", name: "Boss", accessLevel: "ceo", verifiedAccessLevel: "ceo" },
|
|
437
|
-
"sess-success",
|
|
438
|
-
{ agentRoot: root },
|
|
439
|
-
);
|
|
440
|
-
assertResultShape(res);
|
|
441
|
-
assert.equal(res.success, true, "draft_email should succeed");
|
|
442
|
-
|
|
443
|
-
const rows = readAuditRows(root);
|
|
444
|
-
assert.equal(rows.length, 1, "exactly one audit row for one invocation");
|
|
445
|
-
const row = rows[0];
|
|
446
|
-
assert.equal(row.tool, "draft_email", "row records the real tool name, not 'unknown'");
|
|
447
|
-
assert.equal(row.status, "completed", "successful action is logged completed");
|
|
448
|
-
assert.equal(row.denied, false, "an allowed action is not denied");
|
|
449
|
-
assert.equal(row.session_id, "sess-success", "the sessionId is captured");
|
|
450
|
-
assert.equal(row.target, "alex@example.com", "best-effort target is the recipient");
|
|
451
|
-
assert.match(
|
|
452
|
-
row.timestamp,
|
|
453
|
-
/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}Z$/,
|
|
454
|
-
"timestamp is ISO8601 Z",
|
|
455
|
-
);
|
|
456
|
-
|
|
457
|
-
rmSync(root, { recursive: true, force: true });
|
|
458
|
-
});
|
|
459
|
-
|
|
460
|
-
test("self-audit: a permission-denied call writes a row with denied=true", async () => {
|
|
461
|
-
const root = freshRoot();
|
|
462
|
-
// slack_send is a write tool absent from the least-privilege default surface;
|
|
463
|
-
// an unknown level coerces to default, so this is denied at the gate.
|
|
464
|
-
const res = await executeAction(
|
|
465
|
-
"slack_send",
|
|
466
|
-
{ recipient: "C123", message: "y" },
|
|
467
|
-
caller("nonexistent-access-level"),
|
|
468
|
-
"sess-denied",
|
|
469
|
-
{ agentRoot: root },
|
|
470
|
-
);
|
|
471
|
-
assert.equal(res.success, false);
|
|
472
|
-
assert.match(res.result, /permission/i);
|
|
473
|
-
|
|
474
|
-
const rows = readAuditRows(root);
|
|
475
|
-
assert.equal(rows.length, 1, "the denied attempt is logged, not silent");
|
|
476
|
-
const row = rows[0];
|
|
477
|
-
assert.equal(row.tool, "slack_send", "denied row records the attempted tool");
|
|
478
|
-
assert.equal(row.denied, true, "denied row has denied=true");
|
|
479
|
-
assert.equal(row.status, "denied", "denied row status is 'denied'");
|
|
480
|
-
assert.equal(row.session_id, "sess-denied");
|
|
481
|
-
assert.equal(row.target, "C123", "denied row still captures the attempted target");
|
|
482
|
-
|
|
483
|
-
rmSync(root, { recursive: true, force: true });
|
|
484
|
-
});
|
|
485
|
-
|
|
486
|
-
test("self-audit: a failing action writes status failed", async () => {
|
|
487
|
-
const root = freshRoot();
|
|
488
|
-
// verified ceo clears the gate; slack-send.sh does not exist under this fresh
|
|
489
|
-
// root, so executeScript returns { success:false } — a real failure path.
|
|
490
|
-
const res = await executeAction(
|
|
491
|
-
"slack_send",
|
|
492
|
-
{ recipient: "C999", message: "y" },
|
|
493
|
-
{ slug: "boss", name: "Boss", accessLevel: "ceo", verifiedAccessLevel: "ceo" },
|
|
494
|
-
"sess-fail",
|
|
495
|
-
{ agentRoot: root },
|
|
496
|
-
);
|
|
497
|
-
assert.equal(res.success, false, "missing script should make the action fail");
|
|
498
|
-
assert.doesNotMatch(res.result, /You do not have permission/i, "it cleared the gate");
|
|
499
|
-
|
|
500
|
-
const rows = readAuditRows(root);
|
|
501
|
-
assert.equal(rows.length, 1);
|
|
502
|
-
const row = rows[0];
|
|
503
|
-
assert.equal(row.tool, "slack_send");
|
|
504
|
-
assert.equal(row.status, "failed", "a failed (non-denied) action is logged failed");
|
|
505
|
-
assert.equal(row.denied, false, "a gate-cleared failure is not a denial");
|
|
506
|
-
assert.equal(row.session_id, "sess-fail");
|
|
507
|
-
|
|
508
|
-
rmSync(root, { recursive: true, force: true });
|
|
509
|
-
});
|
|
510
|
-
|
|
511
|
-
test("self-audit: audit write failure does not throw into the action path", async () => {
|
|
512
|
-
// Point agentRoot at a regular FILE, so logs/audit/ cannot be created — the
|
|
513
|
-
// audit write must fail internally and be swallowed. executeAction must still
|
|
514
|
-
// return its normal { success:false, result } denial object.
|
|
515
|
-
const blocker = join(sandbox, "not-a-dir");
|
|
516
|
-
writeFileSync(blocker, "x");
|
|
517
|
-
|
|
518
|
-
let res;
|
|
519
|
-
await assert.doesNotReject(async () => {
|
|
520
|
-
res = await executeAction(
|
|
521
|
-
"slack_send",
|
|
522
|
-
{ recipient: "C123", message: "y" },
|
|
523
|
-
caller("nonexistent-access-level"),
|
|
524
|
-
"sess-blocked",
|
|
525
|
-
{ agentRoot: blocker },
|
|
526
|
-
);
|
|
527
|
-
}, "a broken audit path must not throw out of executeAction");
|
|
528
|
-
assertResultShape(res);
|
|
529
|
-
assert.equal(res.success, false);
|
|
530
|
-
assert.match(res.result, /permission/i, "the action result is unaffected by the log failure");
|
|
531
|
-
});
|
|
532
|
-
|
|
533
|
-
// ---------------------------------------------------------------------------
|
|
534
|
-
// Router coverage for lookups (#29 residual): executeLookup routes the MODEL
|
|
535
|
-
// SELECTION through the model router (cheapest capable model) and records the
|
|
536
|
-
// usage to the cost ledger with a decision_id — a model-selection + accounting
|
|
537
|
-
// upgrade that preserves the lookup behaviour, timeout, MCP access, and the
|
|
538
|
-
// { success, result } output shape exactly. All deps are injected so these
|
|
539
|
-
// tests touch neither the real router nor the real ledger nor the network.
|
|
540
|
-
// ---------------------------------------------------------------------------
|
|
541
|
-
|
|
542
|
-
// A v2 routing config is the gate that turns on router-driven selection. The
|
|
543
|
-
// SHAPE is all the helper checks (schema_version === 2); the chain is never
|
|
544
|
-
// walked because we inject `resolve`.
|
|
545
|
-
const V2_CONFIG = { schema_version: 2, routing_policy: [{ default: true, chain: ["x"] }] };
|
|
546
|
-
|
|
547
|
-
// A RouteDecision the injected resolver returns: it names a cheap Haiku-class
|
|
548
|
-
// model and carries the decision_id we expect on the ledger row.
|
|
549
|
-
function fakeDecision(over = {}) {
|
|
550
|
-
return {
|
|
551
|
-
decision_id: over.decision_id || "dec-lookup-001",
|
|
552
|
-
chosen: {
|
|
553
|
-
provider: over.provider || "anthropic",
|
|
554
|
-
model: over.model || "claude-haiku-4-5",
|
|
555
|
-
harness: "session",
|
|
556
|
-
catalogRow: { ref: over.ref || "anthropic/claude-haiku-4-5" },
|
|
557
|
-
},
|
|
558
|
-
spawnArgs: { modelFlag: over.modelFlag || "claude-haiku-4-5" },
|
|
559
|
-
...over.extra,
|
|
560
|
-
};
|
|
561
|
-
}
|
|
562
|
-
|
|
563
|
-
test("executeLookup: selects the model via the injected router and writes a ledger row", async () => {
|
|
564
|
-
const root = sandbox;
|
|
565
|
-
const calls = { resolve: 0, model: null };
|
|
566
|
-
const ledgerRows = [];
|
|
567
|
-
|
|
568
|
-
const res = await executeLookup(
|
|
569
|
-
"search_email",
|
|
570
|
-
{ query: "Q2 board deck" },
|
|
571
|
-
root,
|
|
572
|
-
{
|
|
573
|
-
config: V2_CONFIG,
|
|
574
|
-
resolve: async (req, ropts) => {
|
|
575
|
-
calls.resolve += 1;
|
|
576
|
-
calls.req = req;
|
|
577
|
-
calls.ropts = ropts;
|
|
578
|
-
return fakeDecision();
|
|
579
|
-
},
|
|
580
|
-
writeLedger: async (catalog, entry) => {
|
|
581
|
-
ledgerRows.push({ catalog, entry });
|
|
582
|
-
},
|
|
583
|
-
// Capture the model flag the spawn actually used by stubbing nothing —
|
|
584
|
-
// CLAUDE_BIN is the harmless echo stub, so the spawn is inert but real.
|
|
585
|
-
},
|
|
586
|
-
);
|
|
587
|
-
|
|
588
|
-
assertResultShape(res);
|
|
589
|
-
assert.equal(res.success, true, "the lookup still succeeds (stub echoes a line)");
|
|
590
|
-
assert.equal(calls.resolve, 1, "the router resolver was consulted exactly once");
|
|
591
|
-
// The request handed to the router is a tool-less, sensitive RAG lookup.
|
|
592
|
-
assert.equal(calls.req.task_class, "session.lookup");
|
|
593
|
-
assert.equal(calls.req.data_class, "sensitive");
|
|
594
|
-
assert.equal(calls.ropts.config, V2_CONFIG, "config threaded into resolveChain");
|
|
595
|
-
|
|
596
|
-
// Exactly one ledger row, joined by the decision_id from the decision.
|
|
597
|
-
assert.equal(ledgerRows.length, 1, "one ledger row per routed lookup");
|
|
598
|
-
const { entry } = ledgerRows[0];
|
|
599
|
-
assert.equal(entry.decision_id, "dec-lookup-001");
|
|
600
|
-
assert.equal(entry.ref, "anthropic/claude-haiku-4-5");
|
|
601
|
-
assert.equal(entry.provider, "anthropic");
|
|
602
|
-
assert.equal(entry.model, "claude-haiku-4-5");
|
|
603
|
-
assert.equal(entry.task_class, "session.lookup");
|
|
604
|
-
assert.equal(entry.source, "lookup");
|
|
605
|
-
assert.equal(entry.exitReason, "ok");
|
|
606
|
-
});
|
|
607
|
-
|
|
608
|
-
test("resolveLookupModel: picks the router's modelFlag when a v2 config + resolver are present", async () => {
|
|
609
|
-
const routed = await resolveLookupModel("search_slack", sandbox, {
|
|
610
|
-
config: V2_CONFIG,
|
|
611
|
-
resolve: async () => fakeDecision({ modelFlag: "deepseek-v4-flash", provider: "deepseek", model: "deepseek-v4-flash", ref: "deepseek/deepseek-v4-flash", decision_id: "d-2" }),
|
|
612
|
-
});
|
|
613
|
-
assert.equal(routed.modelFlag, "deepseek-v4-flash", "cheapest capable model wins selection");
|
|
614
|
-
assert.equal(routed.decisionId, "d-2");
|
|
615
|
-
assert.equal(routed.ref, "deepseek/deepseek-v4-flash");
|
|
616
|
-
});
|
|
617
|
-
|
|
618
|
-
test("resolveLookupModel: falls back to the default model when NO routing config is present", async () => {
|
|
619
|
-
let resolveCalled = false;
|
|
620
|
-
const routed = await resolveLookupModel("search_calendar", sandbox, {
|
|
621
|
-
// loadConfig returns null (the no-config path) → never even calls resolve.
|
|
622
|
-
loadConfig: async () => null,
|
|
623
|
-
resolve: async () => { resolveCalled = true; return fakeDecision(); },
|
|
624
|
-
});
|
|
625
|
-
assert.equal(routed.modelFlag, LOOKUP_DEFAULT_MODEL, "no config → historical haiku default");
|
|
626
|
-
assert.equal(routed.modelFlag, "haiku");
|
|
627
|
-
assert.equal(routed.decisionId, null, "no decision, so no ledger join id");
|
|
628
|
-
assert.equal(resolveCalled, false, "the router is not consulted without a v2 config");
|
|
629
|
-
});
|
|
630
|
-
|
|
631
|
-
test("resolveLookupModel: a v1 (non-2) config also falls back to the default model", async () => {
|
|
632
|
-
const routed = await resolveLookupModel("search_files", sandbox, {
|
|
633
|
-
config: { schema_version: 1, backends: {} },
|
|
634
|
-
resolve: async () => { throw new Error("should not be called"); },
|
|
635
|
-
});
|
|
636
|
-
assert.equal(routed.modelFlag, LOOKUP_DEFAULT_MODEL);
|
|
637
|
-
assert.equal(routed.decisionId, null);
|
|
638
|
-
});
|
|
639
|
-
|
|
640
|
-
test("resolveLookupModel: MAESTRO_ROUTER_FORCE_ANTHROPIC short-circuits to the default (no resolve, no ledger)", async () => {
|
|
641
|
-
let resolveCalled = false;
|
|
642
|
-
const routed = await resolveLookupModel("search_web", sandbox, {
|
|
643
|
-
env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" },
|
|
644
|
-
config: V2_CONFIG,
|
|
645
|
-
resolve: async () => { resolveCalled = true; return fakeDecision(); },
|
|
646
|
-
});
|
|
647
|
-
assert.equal(routed.modelFlag, LOOKUP_DEFAULT_MODEL, "kill switch → stock haiku");
|
|
648
|
-
assert.equal(routed.decisionId, null);
|
|
649
|
-
assert.equal(resolveCalled, false, "kill switch never consults the router");
|
|
650
|
-
});
|
|
651
|
-
|
|
652
|
-
test("executeLookup: FORCE_ANTHROPIC writes NO ledger row and keeps the output shape", async () => {
|
|
653
|
-
const ledgerRows = [];
|
|
654
|
-
let resolveCalled = false;
|
|
655
|
-
const res = await executeLookup(
|
|
656
|
-
"search_email",
|
|
657
|
-
{ query: "x" },
|
|
658
|
-
sandbox,
|
|
659
|
-
{
|
|
660
|
-
env: { ...process.env, MAESTRO_ROUTER_FORCE_ANTHROPIC: "true" },
|
|
661
|
-
config: V2_CONFIG,
|
|
662
|
-
resolve: async () => { resolveCalled = true; return fakeDecision(); },
|
|
663
|
-
writeLedger: async (c, e) => ledgerRows.push(e),
|
|
664
|
-
},
|
|
665
|
-
);
|
|
666
|
-
assertResultShape(res);
|
|
667
|
-
assert.equal(res.success, true);
|
|
668
|
-
assert.equal(resolveCalled, false, "kill switch bypasses the router");
|
|
669
|
-
assert.equal(ledgerRows.length, 0, "no decision → no ledger row (matches the pre-router default spawn)");
|
|
670
|
-
});
|
|
671
|
-
|
|
672
|
-
test("executeLookup: a router fault is fail-open — the lookup still runs on the default model", async () => {
|
|
673
|
-
const ledgerRows = [];
|
|
674
|
-
const res = await executeLookup(
|
|
675
|
-
"search_email",
|
|
676
|
-
{ query: "x" },
|
|
677
|
-
sandbox,
|
|
678
|
-
{
|
|
679
|
-
config: V2_CONFIG,
|
|
680
|
-
resolve: async () => { throw new Error("router exploded"); },
|
|
681
|
-
writeLedger: async (c, e) => ledgerRows.push(e),
|
|
682
|
-
},
|
|
683
|
-
);
|
|
684
|
-
assertResultShape(res);
|
|
685
|
-
assert.equal(res.success, true, "a router error never breaks the lookup");
|
|
686
|
-
assert.equal(ledgerRows.length, 0, "no usable decision → no ledger row, but the search ran");
|
|
687
|
-
});
|
|
688
|
-
|
|
689
|
-
test("executeLookup: ledger write failure does not break the lookup (best-effort accounting)", async () => {
|
|
690
|
-
const res = await executeLookup(
|
|
691
|
-
"search_email",
|
|
692
|
-
{ query: "x" },
|
|
693
|
-
sandbox,
|
|
694
|
-
{
|
|
695
|
-
config: V2_CONFIG,
|
|
696
|
-
resolve: async () => fakeDecision(),
|
|
697
|
-
writeLedger: async () => { throw new Error("disk full"); },
|
|
698
|
-
},
|
|
699
|
-
);
|
|
700
|
-
assertResultShape(res);
|
|
701
|
-
assert.equal(res.success, true, "a ledger we can't persist must not fail the search");
|
|
702
|
-
});
|
|
703
|
-
|
|
704
|
-
test("executeLookup: output shape is unchanged for an unknown prompt name (no router, no ledger)", async () => {
|
|
705
|
-
let resolveCalled = false;
|
|
706
|
-
const res = await executeLookup(
|
|
707
|
-
"__no_such_lookup__",
|
|
708
|
-
{},
|
|
709
|
-
sandbox,
|
|
710
|
-
{ config: V2_CONFIG, resolve: async () => { resolveCalled = true; return fakeDecision(); } },
|
|
711
|
-
);
|
|
712
|
-
assertResultShape(res);
|
|
713
|
-
assert.equal(res.success, false);
|
|
714
|
-
assert.match(res.result, /No prompt for/);
|
|
715
|
-
assert.equal(resolveCalled, false, "no prompt → no spawn → no routing");
|
|
716
|
-
});
|
|
717
|
-
|
|
718
|
-
test("executeAction: a public lookup tool routes through the model router end-to-end", async () => {
|
|
719
|
-
// Drive the FULL public path (executeAction → executeLookup) but with the
|
|
720
|
-
// router/ledger injected via options. This proves the dispatch threads the
|
|
721
|
-
// injectable opts through and that authorisation still gates correctly.
|
|
722
|
-
const ledgerRows = [];
|
|
723
|
-
const res = await executeAction(
|
|
724
|
-
"search_email",
|
|
725
|
-
{ query: "anything" },
|
|
726
|
-
{ slug: "alex", name: "Alex", accessLevel: "ceo", verifiedAccessLevel: "ceo" },
|
|
727
|
-
"sess-router",
|
|
728
|
-
{
|
|
729
|
-
agentRoot: sandbox,
|
|
730
|
-
agent: "alex",
|
|
731
|
-
// executeAction builds its own lookupOpts; we cannot inject resolve through
|
|
732
|
-
// it directly, so we only assert the contract holds (no throw, right shape).
|
|
733
|
-
},
|
|
734
|
-
);
|
|
735
|
-
assertResultShape(res);
|
|
736
|
-
assert.doesNotMatch(res.result, /You do not have permission/i, "ceo clears the gate");
|
|
737
|
-
});
|
|
738
|
-
|
|
739
|
-
test.after(() => {
|
|
740
|
-
try {
|
|
741
|
-
rmSync(sandbox, { recursive: true, force: true });
|
|
742
|
-
} catch {
|
|
743
|
-
/* best effort */
|
|
744
|
-
}
|
|
745
|
-
});
|
|
746
|
-
|
|
747
|
-
// ---------------------------------------------------------------------------
|
|
748
|
-
// Org (Cohort) tool routing — the curated surface via executeOrgTool
|
|
749
|
-
// ---------------------------------------------------------------------------
|
|
750
|
-
|
|
751
|
-
// Ambient COHORT_* creds must not leak into the org-tool fixtures.
|
|
752
|
-
for (const k of ["COHORT_API_TOKEN", "COHORT_TOKEN", "COHORT_API_KEY", "COHORT_ORG_ID", "COHORT_BASE", "COHORT_API_URL"]) delete process.env[k];
|
|
753
|
-
|
|
754
|
-
import * as fsExtra from "node:fs";
|
|
755
|
-
|
|
756
|
-
/** A tmp agent root carrying an enrolled config/org.yaml. */
|
|
757
|
-
function orgEnrolledRoot() {
|
|
758
|
-
const root = mkdtempSync(join(tmpdir(), "maestro-orgtool-"));
|
|
759
|
-
const cfgDir = join(root, "config");
|
|
760
|
-
fsExtra.mkdirSync(cfgDir, { recursive: true });
|
|
761
|
-
writeFileSync(
|
|
762
|
-
join(cfgDir, "org.yaml"),
|
|
763
|
-
["org:", " cohort:", " enabled: true", " base: https://org.example", " orgId: acme", " token: tok-exec"].join("\n"),
|
|
764
|
-
);
|
|
765
|
-
return root;
|
|
766
|
-
}
|
|
767
|
-
|
|
768
|
-
function orgFakeFetch(body = { ok: true, result: { done: true } }) {
|
|
769
|
-
const calls = [];
|
|
770
|
-
const fn = async (url, init) => {
|
|
771
|
-
calls.push({ url: String(url), init: init || {} });
|
|
772
|
-
return { ok: true, status: 200, json: async () => body, headers: { get: () => undefined } };
|
|
773
|
-
};
|
|
774
|
-
fn.calls = calls;
|
|
775
|
-
return fn;
|
|
776
|
-
}
|
|
777
|
-
|
|
778
|
-
test("org tool routes to executeOrgTool with the injected fetchImpl (no network, frame → outcome)", async () => {
|
|
779
|
-
const root = orgEnrolledRoot();
|
|
780
|
-
const fetchImpl = orgFakeFetch({ ok: true, result: { channels: [{ id: "C1" }] } });
|
|
781
|
-
const res = await executeAction("messaging_channels", {}, { accessLevel: "default" }, "s-org-1", { agentRoot: root, fetchImpl });
|
|
782
|
-
assert.equal(res.success, true);
|
|
783
|
-
assert.match(res.result, /C1/, "frame.result stringified into the outcome");
|
|
784
|
-
assert.equal(fetchImpl.calls.length, 1);
|
|
785
|
-
assert.match(fetchImpl.calls[0].url, /https:\/\/org\.example\/api\/v1\/messaging\.channels$/);
|
|
786
|
-
rmSync(root, { recursive: true, force: true });
|
|
787
|
-
});
|
|
788
|
-
|
|
789
|
-
test("org error frames map to success:false with the code surfaced", async () => {
|
|
790
|
-
const root = orgEnrolledRoot();
|
|
791
|
-
const fetchImpl = async () => ({ ok: false, status: 403, json: async () => ({ ok: false, error: { code: "FORBIDDEN_SCOPE", message: "not paired" } }), headers: { get: () => undefined } });
|
|
792
|
-
const res = await executeAction("messaging_channels", {}, { accessLevel: "default" }, "s-org-2", { agentRoot: root, fetchImpl });
|
|
793
|
-
assert.equal(res.success, false);
|
|
794
|
-
assert.match(res.result, /FORBIDDEN_SCOPE/);
|
|
795
|
-
assert.match(res.result, /not paired/);
|
|
796
|
-
rmSync(root, { recursive: true, force: true });
|
|
797
|
-
});
|
|
798
|
-
|
|
799
|
-
test("org writes are DENIED at the default access level (name-set authorization)", async () => {
|
|
800
|
-
const root = orgEnrolledRoot();
|
|
801
|
-
const fetchImpl = orgFakeFetch();
|
|
802
|
-
const res = await executeAction("messaging_send", { channelId: "C1", body: "hi" }, { accessLevel: "default" }, "s-org-3", { agentRoot: root, fetchImpl });
|
|
803
|
-
assert.equal(res.success, false);
|
|
804
|
-
assert.match(res.result, /permission/);
|
|
805
|
-
assert.equal(fetchImpl.calls.length, 0, "denied before any dispatch");
|
|
806
|
-
rmSync(root, { recursive: true, force: true });
|
|
807
|
-
});
|
|
808
|
-
|
|
809
|
-
test("org_rpc is denied below ceo; routed (and protocol-validated) at ceo", async () => {
|
|
810
|
-
const root = orgEnrolledRoot();
|
|
811
|
-
const fetchImpl = orgFakeFetch();
|
|
812
|
-
const lead = await executeAction("org_rpc", { method: "member.get", params: { memberId: "m" } }, { accessLevel: "leadership" }, "s-org-4", { agentRoot: root, fetchImpl });
|
|
813
|
-
assert.equal(lead.success, false);
|
|
814
|
-
assert.match(lead.result, /permission/, "escape hatch withheld below ceo");
|
|
815
|
-
assert.equal(fetchImpl.calls.length, 0);
|
|
816
|
-
|
|
817
|
-
const ceoBad = await executeAction("org_rpc", { method: "no.method" }, { accessLevel: "ceo" }, "s-org-5", { agentRoot: root, fetchImpl });
|
|
818
|
-
assert.equal(ceoBad.success, false);
|
|
819
|
-
assert.match(ceoBad.result, /NOT_FOUND/, "unknown method rejected against the protocol table");
|
|
820
|
-
assert.equal(fetchImpl.calls.length, 0, "no network for an invalid method");
|
|
821
|
-
|
|
822
|
-
const ceoOk = await executeAction("org_rpc", { method: "member.get", params: { memberId: "m" } }, { accessLevel: "ceo" }, "s-org-6", { agentRoot: root, fetchImpl });
|
|
823
|
-
assert.equal(ceoOk.success, true);
|
|
824
|
-
assert.match(fetchImpl.calls[0].url, /member\.get$/);
|
|
825
|
-
rmSync(root, { recursive: true, force: true });
|
|
826
|
-
});
|
|
827
|
-
|
|
828
|
-
test("send-gate invoked for messaging_send in the native plane (options.screenImpl)", async () => {
|
|
829
|
-
const root = orgEnrolledRoot();
|
|
830
|
-
const screened = [];
|
|
831
|
-
const fetchImpl = orgFakeFetch({ ok: true, result: { messageId: "m1" } });
|
|
832
|
-
const ok = await executeAction(
|
|
833
|
-
"messaging_send",
|
|
834
|
-
{ channelId: "C1", body: "shipping now" },
|
|
835
|
-
{ accessLevel: "leadership" },
|
|
836
|
-
"s-org-7",
|
|
837
|
-
{ agentRoot: root, fetchImpl, screenImpl: async (o) => { screened.push(o); return { allow: true }; } },
|
|
838
|
-
);
|
|
839
|
-
assert.equal(ok.success, true);
|
|
840
|
-
assert.equal(screened.length, 1, "screenOutbound ran before dispatch");
|
|
841
|
-
assert.equal(screened[0].recipient, "C1");
|
|
842
|
-
|
|
843
|
-
const blocked = await executeAction(
|
|
844
|
-
"messaging_send",
|
|
845
|
-
{ channelId: "C1", body: "As an AI…" },
|
|
846
|
-
{ accessLevel: "leadership" },
|
|
847
|
-
"s-org-8",
|
|
848
|
-
{ agentRoot: root, fetchImpl, screenImpl: async () => ({ allow: false, reason: "banned-phrase" }) },
|
|
849
|
-
);
|
|
850
|
-
assert.equal(blocked.success, false);
|
|
851
|
-
assert.match(blocked.result, /banned-phrase/);
|
|
852
|
-
assert.equal(fetchImpl.calls.length, 1, "blocked send never reached the wire");
|
|
853
|
-
rmSync(root, { recursive: true, force: true });
|
|
854
|
-
});
|
|
855
|
-
|
|
856
|
-
test("denied + completed org invocations write the native plane's audit rows", async () => {
|
|
857
|
-
const root = orgEnrolledRoot();
|
|
858
|
-
const fetchImpl = orgFakeFetch();
|
|
859
|
-
await executeAction("messaging_send", { channelId: "C1", body: "x" }, { accessLevel: "default" }, "s-audit-1", { agentRoot: root, fetchImpl });
|
|
860
|
-
await executeAction("org_describe", {}, { accessLevel: "default" }, "s-audit-2", { agentRoot: root, fetchImpl });
|
|
861
|
-
const dir = join(root, "logs", "audit");
|
|
862
|
-
const rows = [];
|
|
863
|
-
for (const f of fsExtra.readdirSync(dir)) {
|
|
864
|
-
for (const line of readFileSync(join(dir, f), "utf8").split("\n")) if (line.trim()) rows.push(JSON.parse(line));
|
|
865
|
-
}
|
|
866
|
-
assert.equal(rows.length, 2, "one audit row per invocation");
|
|
867
|
-
assert.equal(rows[0].denied, true, "denied write audited");
|
|
868
|
-
assert.equal(rows[1].denied, false);
|
|
869
|
-
assert.equal(rows[1].status, "completed");
|
|
870
|
-
rmSync(root, { recursive: true, force: true });
|
|
871
|
-
});
|