@selesai/code 0.2.1 → 0.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/config.d.ts +4 -15
- package/dist/config.d.ts.map +1 -1
- package/dist/config.js +5 -45
- package/dist/config.js.map +1 -1
- package/dist/core/agent-session.d.ts +1 -0
- package/dist/core/agent-session.d.ts.map +1 -1
- package/dist/core/agent-session.js +8 -0
- package/dist/core/agent-session.js.map +1 -1
- package/dist/core/diagnostics.d.ts +1 -0
- package/dist/core/diagnostics.d.ts.map +1 -1
- package/dist/core/diagnostics.js.map +1 -1
- package/dist/core/model-registry.d.ts +4 -1
- package/dist/core/model-registry.d.ts.map +1 -1
- package/dist/core/model-registry.js +101 -51
- package/dist/core/model-registry.js.map +1 -1
- package/dist/core/package-manager.d.ts +10 -0
- package/dist/core/package-manager.d.ts.map +1 -1
- package/dist/core/package-manager.js +93 -8
- package/dist/core/package-manager.js.map +1 -1
- package/dist/core/package-manager.test.d.ts +2 -0
- package/dist/core/package-manager.test.d.ts.map +1 -0
- package/dist/core/package-manager.test.js +20 -0
- package/dist/core/package-manager.test.js.map +1 -0
- package/dist/core/resource-loader.d.ts +3 -0
- package/dist/core/resource-loader.d.ts.map +1 -1
- package/dist/core/resource-loader.js +10 -0
- package/dist/core/resource-loader.js.map +1 -1
- package/dist/core/settings-manager.d.ts +1 -0
- package/dist/core/settings-manager.d.ts.map +1 -1
- package/dist/core/settings-manager.js.map +1 -1
- package/dist/defaults/models.json +64 -113
- package/dist/extensions/caveman/caveman-instructions.cjs +39 -0
- package/dist/extensions/caveman/index.js +116 -0
- package/dist/extensions/caveman/package.json +8 -0
- package/dist/extensions/caveman/test/extension.test.js +185 -0
- package/dist/extensions/caveman/test/helpers.test.js +67 -0
- package/dist/extensions/copy-turn.ts +2 -2
- package/dist/extensions/pi-intercom/broker/broker.ts +2 -3
- package/dist/extensions/pi-intercom/broker/paths.ts +10 -5
- package/dist/extensions/pi-intercom/broker/spawn.ts +2 -3
- package/dist/extensions/pi-intercom/config.ts +2 -2
- package/dist/extensions/pi-intercom/index.ts +2 -2
- package/dist/extensions/pi-intercom/package.json +2 -2
- package/dist/extensions/pi-intercom/ui/compose.ts +3 -3
- package/dist/extensions/pi-intercom/ui/inline-message.ts +3 -3
- package/dist/extensions/pi-intercom/ui/session-list.ts +3 -3
- package/dist/extensions/pi-powerline-footer/bash-mode/editor.ts +2 -2
- package/dist/extensions/pi-powerline-footer/bash-mode/history.ts +2 -1
- package/dist/extensions/pi-powerline-footer/index.ts +40 -47
- package/dist/extensions/pi-powerline-footer/package.json +2 -2
- package/dist/extensions/pi-powerline-footer/tests/bash-mode.test.ts +10 -10
- package/dist/extensions/pi-powerline-footer/tests/jump-shortcuts.test.ts +1 -1
- package/dist/extensions/pi-powerline-footer/tests/working-vibes.test.ts +2 -2
- package/dist/extensions/pi-powerline-footer/theme.ts +6 -6
- package/dist/extensions/pi-powerline-footer/tps.ts +1 -1
- package/dist/extensions/pi-powerline-footer/types.ts +7 -7
- package/dist/extensions/pi-powerline-footer/welcome.ts +23 -16
- package/dist/extensions/pi-powerline-footer/working-vibes.ts +52 -59
- package/dist/extensions/pi-rewind-hook/index.ts +1 -1
- package/dist/extensions/pi-rewind-hook/package.json +1 -1
- package/dist/extensions/pi-subagents/README.md +224 -52
- package/dist/extensions/pi-subagents/agents/architect.md +188 -0
- package/dist/extensions/pi-subagents/agents/builder.md +120 -0
- package/dist/extensions/pi-subagents/agents/commentator.md +131 -0
- package/dist/extensions/pi-subagents/agents/explorer.md +51 -0
- package/dist/extensions/pi-subagents/agents/recapper.md +21 -0
- package/dist/extensions/pi-subagents/install.mjs +93 -0
- package/dist/extensions/pi-subagents/package.json +4 -5
- package/dist/extensions/pi-subagents/prompts/parallel-context-build.md +1 -1
- package/dist/extensions/pi-subagents/prompts/parallel-handoff-plan.md +1 -1
- package/dist/extensions/pi-subagents/skills/pi-subagents/SKILL.md +76 -13
- package/dist/extensions/pi-subagents/src/agents/agent-management.ts +179 -2
- package/dist/extensions/pi-subagents/src/agents/agent-memory.ts +254 -0
- package/dist/extensions/pi-subagents/src/agents/agent-serializer.ts +11 -0
- package/dist/extensions/pi-subagents/src/agents/agents.ts +193 -19
- package/dist/extensions/pi-subagents/src/agents/chain-serializer.ts +27 -2
- package/dist/extensions/pi-subagents/src/extension/config.ts +27 -4
- package/dist/extensions/pi-subagents/src/extension/doctor.ts +1 -7
- package/dist/extensions/pi-subagents/src/extension/fanout-child.ts +3 -2
- package/dist/extensions/pi-subagents/src/extension/index.ts +73 -71
- package/dist/extensions/pi-subagents/src/extension/rpc.ts +369 -0
- package/dist/extensions/pi-subagents/src/extension/schemas.ts +54 -10
- package/dist/extensions/pi-subagents/src/extension/tool-description.ts +200 -0
- package/dist/extensions/pi-subagents/src/intercom/intercom-bridge.ts +21 -253
- package/dist/extensions/pi-subagents/src/intercom/native-supervisor-channel.ts +519 -0
- package/dist/extensions/pi-subagents/src/profiles/profiles.ts +1 -1
- package/dist/extensions/pi-subagents/src/runs/background/async-execution.ts +195 -105
- package/dist/extensions/pi-subagents/src/runs/background/async-job-tracker.ts +88 -2
- package/dist/extensions/pi-subagents/src/runs/background/async-status.ts +67 -10
- package/dist/extensions/pi-subagents/src/runs/background/chain-root-attachment.ts +34 -4
- package/dist/extensions/pi-subagents/src/runs/background/completion-batcher.ts +166 -0
- package/dist/extensions/pi-subagents/src/runs/background/control-channel.ts +156 -1
- package/dist/extensions/pi-subagents/src/runs/background/fleet-view.ts +515 -0
- package/dist/extensions/pi-subagents/src/runs/background/notify.ts +161 -44
- package/dist/extensions/pi-subagents/src/runs/background/result-watcher.ts +1 -2
- package/dist/extensions/pi-subagents/src/runs/background/run-id-resolver.ts +3 -2
- package/dist/extensions/pi-subagents/src/runs/background/run-status.ts +167 -6
- package/dist/extensions/pi-subagents/src/runs/background/scheduled-runs.ts +514 -0
- package/dist/extensions/pi-subagents/src/runs/background/stale-run-reconciler.ts +28 -1
- package/dist/extensions/pi-subagents/src/runs/background/subagent-runner.ts +873 -204
- package/dist/extensions/pi-subagents/src/runs/background/wait.ts +353 -0
- package/dist/extensions/pi-subagents/src/runs/foreground/chain-execution.ts +123 -27
- package/dist/extensions/pi-subagents/src/runs/foreground/execution.ts +174 -70
- package/dist/extensions/pi-subagents/src/runs/foreground/subagent-executor.ts +569 -81
- package/dist/extensions/pi-subagents/src/runs/shared/acceptance.ts +45 -22
- package/dist/extensions/pi-subagents/src/runs/shared/completion-guard.ts +3 -1
- package/dist/extensions/pi-subagents/src/runs/shared/dynamic-fanout.ts +2 -2
- package/dist/extensions/pi-subagents/src/runs/shared/model-fallback.ts +172 -21
- package/dist/extensions/pi-subagents/src/runs/shared/model-scope.ts +128 -0
- package/dist/extensions/pi-subagents/src/runs/shared/nested-events.ts +89 -0
- package/dist/extensions/pi-subagents/src/runs/shared/parallel-utils.ts +50 -1
- package/dist/extensions/pi-subagents/src/runs/shared/pi-args.ts +35 -4
- package/dist/extensions/pi-subagents/src/runs/shared/pi-spawn.ts +63 -45
- package/dist/extensions/pi-subagents/src/runs/shared/single-output.ts +2 -0
- package/dist/extensions/pi-subagents/src/runs/shared/subagent-prompt-runtime.ts +125 -5
- package/dist/extensions/pi-subagents/src/runs/shared/tool-budget.ts +74 -0
- package/dist/extensions/pi-subagents/src/runs/shared/turn-budget.ts +52 -0
- package/dist/extensions/pi-subagents/src/runs/shared/worktree.ts +28 -5
- package/dist/extensions/pi-subagents/src/shared/artifacts.ts +16 -1
- package/dist/extensions/pi-subagents/src/shared/atomic-json.ts +15 -2
- package/dist/extensions/pi-subagents/src/shared/child-transcript.ts +212 -0
- package/dist/extensions/pi-subagents/src/shared/fork-context.ts +133 -22
- package/dist/extensions/pi-subagents/src/shared/settings.ts +3 -1
- package/dist/extensions/pi-subagents/src/shared/types.ts +197 -4
- package/dist/extensions/pi-subagents/src/shared/utils.ts +108 -17
- package/dist/extensions/pi-subagents/src/slash/prompt-workflows.ts +330 -0
- package/dist/extensions/pi-subagents/src/slash/slash-commands.ts +135 -4
- package/dist/extensions/pi-subagents/src/tui/render.ts +32 -12
- package/dist/extensions/pi-subagents/test/e2e/real-session-subagent.test.ts +94 -0
- package/dist/extensions/pi-subagents/test/integration/async-execution.test.ts +2772 -0
- package/dist/extensions/pi-subagents/test/integration/async-job-tracker.test.ts +1107 -0
- package/dist/extensions/pi-subagents/test/integration/async-status.test.ts +463 -0
- package/dist/extensions/pi-subagents/test/integration/chain-clarify.test.ts +274 -0
- package/dist/extensions/pi-subagents/test/integration/chain-execution.test.ts +1535 -0
- package/dist/extensions/pi-subagents/test/integration/detect-error.test.ts +242 -0
- package/dist/extensions/pi-subagents/test/integration/doctor-executor.test.ts +128 -0
- package/dist/extensions/pi-subagents/test/integration/error-handling.test.ts +252 -0
- package/dist/extensions/pi-subagents/test/integration/foreground-result-size.test.ts +188 -0
- package/dist/extensions/pi-subagents/test/integration/fork-context-execution.test.ts +1306 -0
- package/dist/extensions/pi-subagents/test/integration/intercom-result-delivery.test.ts +1033 -0
- package/dist/extensions/pi-subagents/test/integration/parallel-execution.test.ts +397 -0
- package/dist/extensions/pi-subagents/test/integration/render-fork-badge.test.ts +583 -0
- package/dist/extensions/pi-subagents/test/integration/render-widget.test.ts +691 -0
- package/dist/extensions/pi-subagents/test/integration/result-watcher.test.ts +892 -0
- package/dist/extensions/pi-subagents/test/integration/session-tokens.test.ts +62 -0
- package/dist/extensions/pi-subagents/test/integration/single-execution.test.ts +1824 -0
- package/dist/extensions/pi-subagents/test/integration/slash-commands.test.ts +1284 -0
- package/dist/extensions/pi-subagents/test/integration/slash-live-state.test.ts +127 -0
- package/dist/extensions/pi-subagents/test/integration/template-resolution.test.ts +299 -0
- package/dist/extensions/pi-subagents/test/integration/top-level-async.test.ts +37 -0
- package/dist/extensions/pi-subagents/test/support/helpers.ts +189 -0
- package/dist/extensions/pi-subagents/test/support/mock-pi-script.mjs +335 -0
- package/dist/extensions/pi-subagents/test/support/mock-pi.ts +129 -0
- package/dist/extensions/pi-subagents/test/support/real-session-child-cli.mjs +184 -0
- package/dist/extensions/pi-subagents/test/support/real-session-runner.ts +285 -0
- package/dist/extensions/pi-subagents/test/support/register-loader.mjs +15 -0
- package/dist/extensions/pi-subagents/test/support/ts-loader.mjs +108 -0
- package/dist/extensions/pi-subagents/test/unit/acceptance.test.ts +421 -0
- package/dist/extensions/pi-subagents/test/unit/agent-disabled.test.ts +213 -0
- package/dist/extensions/pi-subagents/test/unit/agent-eject-disable.test.ts +343 -0
- package/dist/extensions/pi-subagents/test/unit/agent-frontmatter.test.ts +1199 -0
- package/dist/extensions/pi-subagents/test/unit/agent-management.test.ts +556 -0
- package/dist/extensions/pi-subagents/test/unit/agent-memory.test.ts +420 -0
- package/dist/extensions/pi-subagents/test/unit/agent-overrides.test.ts +579 -0
- package/dist/extensions/pi-subagents/test/unit/agent-scope.test.ts +20 -0
- package/dist/extensions/pi-subagents/test/unit/agent-selection.test.ts +88 -0
- package/dist/extensions/pi-subagents/test/unit/artifacts.test.ts +24 -0
- package/dist/extensions/pi-subagents/test/unit/async-execution.test.ts +86 -0
- package/dist/extensions/pi-subagents/test/unit/async-interrupt-action.test.ts +221 -0
- package/dist/extensions/pi-subagents/test/unit/async-permission-session.test.ts +98 -0
- package/dist/extensions/pi-subagents/test/unit/async-resume.test.ts +285 -0
- package/dist/extensions/pi-subagents/test/unit/atomic-json.test.ts +112 -0
- package/dist/extensions/pi-subagents/test/unit/chain-append.test.ts +206 -0
- package/dist/extensions/pi-subagents/test/unit/chain-root-attachment.test.ts +159 -0
- package/dist/extensions/pi-subagents/test/unit/chain-serializer.test.ts +229 -0
- package/dist/extensions/pi-subagents/test/unit/child-transcript.test.ts +225 -0
- package/dist/extensions/pi-subagents/test/unit/close-grace-timer.test.ts +101 -0
- package/dist/extensions/pi-subagents/test/unit/completion-batcher.test.ts +228 -0
- package/dist/extensions/pi-subagents/test/unit/completion-dedupe.test.ts +36 -0
- package/dist/extensions/pi-subagents/test/unit/completion-guard.test.ts +152 -0
- package/dist/extensions/pi-subagents/test/unit/config-dir-runtime.test.ts +72 -0
- package/dist/extensions/pi-subagents/test/unit/control-channel.test.ts +327 -0
- package/dist/extensions/pi-subagents/test/unit/control-notices.test.ts +161 -0
- package/dist/extensions/pi-subagents/test/unit/doctor.test.ts +159 -0
- package/dist/extensions/pi-subagents/test/unit/dynamic-fanout.test.ts +217 -0
- package/dist/extensions/pi-subagents/test/unit/extra-agent-dirs.test.ts +88 -0
- package/dist/extensions/pi-subagents/test/unit/file-coalescer.test.ts +68 -0
- package/dist/extensions/pi-subagents/test/unit/foreground-tool-call-compaction.test.ts +78 -0
- package/dist/extensions/pi-subagents/test/unit/fork-context.test.ts +415 -0
- package/dist/extensions/pi-subagents/test/unit/get-final-output.test.ts +126 -0
- package/dist/extensions/pi-subagents/test/unit/index-child-registration.test.ts +271 -0
- package/dist/extensions/pi-subagents/test/unit/intercom-bridge.test.ts +173 -0
- package/dist/extensions/pi-subagents/test/unit/jsonl-writer.test.ts +119 -0
- package/dist/extensions/pi-subagents/test/unit/live-async-resume.test.ts +170 -0
- package/dist/extensions/pi-subagents/test/unit/model-fallback.test.ts +346 -0
- package/dist/extensions/pi-subagents/test/unit/model-info.test.ts +62 -0
- package/dist/extensions/pi-subagents/test/unit/model-scope-settings.test.ts +87 -0
- package/dist/extensions/pi-subagents/test/unit/model-scope.test.ts +145 -0
- package/dist/extensions/pi-subagents/test/unit/native-supervisor-channel.test.ts +162 -0
- package/dist/extensions/pi-subagents/test/unit/nested-control.test.ts +470 -0
- package/dist/extensions/pi-subagents/test/unit/nested-events.test.ts +329 -0
- package/dist/extensions/pi-subagents/test/unit/notify.test.ts +319 -0
- package/dist/extensions/pi-subagents/test/unit/package-manifest.test.ts +70 -0
- package/dist/extensions/pi-subagents/test/unit/parallel-utils.test.ts +247 -0
- package/dist/extensions/pi-subagents/test/unit/path-handling.test.ts +113 -0
- package/dist/extensions/pi-subagents/test/unit/path-resolution.test.ts +94 -0
- package/dist/extensions/pi-subagents/test/unit/pi-args.test.ts +803 -0
- package/dist/extensions/pi-subagents/test/unit/pi-coding-agent-dir.test.ts +209 -0
- package/dist/extensions/pi-subagents/test/unit/pi-spawn.test.ts +238 -0
- package/dist/extensions/pi-subagents/test/unit/proactive-skills.test.ts +118 -0
- package/dist/extensions/pi-subagents/test/unit/profiles.test.ts +299 -0
- package/dist/extensions/pi-subagents/test/unit/prompt-template-bridge.test.ts +408 -0
- package/dist/extensions/pi-subagents/test/unit/prompt-workflows.test.ts +134 -0
- package/dist/extensions/pi-subagents/test/unit/recursion-guard.test.ts +257 -0
- package/dist/extensions/pi-subagents/test/unit/render-helpers.test.ts +209 -0
- package/dist/extensions/pi-subagents/test/unit/result-intercom.test.ts +226 -0
- package/dist/extensions/pi-subagents/test/unit/rpc.test.ts +294 -0
- package/dist/extensions/pi-subagents/test/unit/run-id-resolver.test.ts +147 -0
- package/dist/extensions/pi-subagents/test/unit/run-status.test.ts +997 -0
- package/dist/extensions/pi-subagents/test/unit/scheduled-runs.test.ts +478 -0
- package/dist/extensions/pi-subagents/test/unit/schemas.test.ts +562 -0
- package/dist/extensions/pi-subagents/test/unit/single-output.test.ts +213 -0
- package/dist/extensions/pi-subagents/test/unit/skills-fallback.test.ts +438 -0
- package/dist/extensions/pi-subagents/test/unit/slash-bridge.test.ts +54 -0
- package/dist/extensions/pi-subagents/test/unit/slash-chain-groups.test.ts +464 -0
- package/dist/extensions/pi-subagents/test/unit/stale-run-reconciler.test.ts +255 -0
- package/dist/extensions/pi-subagents/test/unit/status-format.test.ts +19 -0
- package/dist/extensions/pi-subagents/test/unit/subagent-control.test.ts +245 -0
- package/dist/extensions/pi-subagents/test/unit/subagent-prompt-runtime.test.ts +517 -0
- package/dist/extensions/pi-subagents/test/unit/temp-paths.test.ts +72 -0
- package/dist/extensions/pi-subagents/test/unit/tool-budget.test.ts +52 -0
- package/dist/extensions/pi-subagents/test/unit/tool-description.test.ts +237 -0
- package/dist/extensions/pi-subagents/test/unit/total-cost.test.ts +60 -0
- package/dist/extensions/pi-subagents/test/unit/ts-loader.test.ts +33 -0
- package/dist/extensions/pi-subagents/test/unit/turn-budget.test.ts +202 -0
- package/dist/extensions/pi-subagents/test/unit/types-fork-preamble.test.ts +23 -0
- package/dist/extensions/pi-subagents/test/unit/wait.test.ts +398 -0
- package/dist/extensions/pi-subagents/test/unit/widget-nested-render.test.ts +88 -0
- package/dist/extensions/pi-subagents/test/unit/windows-hide-spawn.test.ts +26 -0
- package/dist/extensions/pi-subagents/test/unit/workflow-graph.test.ts +134 -0
- package/dist/extensions/pi-subagents/test/unit/worktree.test.ts +479 -0
- package/dist/extensions/pi-web-agent/package.json +2 -2
- package/dist/extensions/pi-web-agent/src/backends/settings-reader.ts +2 -2
- package/dist/extensions/pi-web-agent/src/changelog-notice.ts +5 -2
- package/dist/extensions/pi-web-agent/src/commands/web-agent-config.ts +1 -1
- package/dist/extensions/pi-web-agent/src/extension.ts +1 -1
- package/dist/extensions/pi-web-agent/src/presentation/config-store.ts +6 -4
- package/dist/extensions/question/helpers.ts +3 -3
- package/dist/extensions/question/index.ts +2 -2
- package/dist/extensions/question/navigation.ts +2 -2
- package/dist/extensions/question/package.json +2 -2
- package/dist/extensions/question/question-list.ts +2 -2
- package/dist/extensions/question/tests/question-list.test.ts +2 -2
- package/dist/extensions/question/tui-adapter.ts +2 -2
- package/dist/extensions/question/types.ts +2 -2
- package/dist/extensions/question/ui-protocol.ts +2 -2
- package/dist/extensions/rtk.ts +2 -2
- package/dist/extensions/tokenin-onboarding.ts +2 -2
- package/dist/extensions/undo.ts +1 -1
- package/dist/extensions/web-agent-onboarding.ts +3 -3
- package/dist/extensions/workflow/adapter.ts +16 -8
- package/dist/extensions/workflow/extension.ts +1 -1
- package/dist/extensions/workflow/package.json +1 -1
- package/dist/main.d.ts.map +1 -1
- package/dist/main.js +4 -0
- package/dist/main.js.map +1 -1
- package/dist/modes/interactive/components/assistant-message.d.ts.map +1 -1
- package/dist/modes/interactive/components/assistant-message.js +2 -0
- package/dist/modes/interactive/components/assistant-message.js.map +1 -1
- package/dist/modes/interactive/components/first-time-setup.d.ts +1 -1
- package/dist/modes/interactive/components/first-time-setup.d.ts.map +1 -1
- package/dist/modes/interactive/components/first-time-setup.js +19 -5
- package/dist/modes/interactive/components/first-time-setup.js.map +1 -1
- package/dist/skills/caveman/SKILL.md +49 -0
- package/dist/skills/handoff-text/SKILL.md +13 -0
- package/dist/utils/thinking-tags.d.ts +7 -0
- package/dist/utils/thinking-tags.d.ts.map +1 -0
- package/dist/utils/thinking-tags.js +63 -0
- package/dist/utils/thinking-tags.js.map +1 -0
- package/docs/docs.json +4 -0
- package/docs/extensions.md +2 -0
- package/docs/settings.md +15 -0
- package/docs/shared-host-extensions.md +109 -0
- package/package.json +2 -2
|
@@ -0,0 +1,1824 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Integration tests for single (sync) agent execution.
|
|
3
|
+
*
|
|
4
|
+
* Uses the local createMockPi() helper to simulate the pi CLI.
|
|
5
|
+
* Tests the full spawn→parse→result pipeline in runSync without a real LLM.
|
|
6
|
+
*
|
|
7
|
+
* These tests require pi packages to be importable (they run inside a pi
|
|
8
|
+
* environment or with pi packages installed). If unavailable, tests skip
|
|
9
|
+
* gracefully.
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
import { describe, it, before, after, beforeEach, afterEach } from "node:test";
|
|
13
|
+
import assert from "node:assert/strict";
|
|
14
|
+
import * as fs from "node:fs";
|
|
15
|
+
import * as path from "node:path";
|
|
16
|
+
import type { MockPi } from "../support/helpers.ts";
|
|
17
|
+
import {
|
|
18
|
+
createMockPi,
|
|
19
|
+
createTempDir,
|
|
20
|
+
createEventBus,
|
|
21
|
+
removeTempDir,
|
|
22
|
+
makeAgentConfigs,
|
|
23
|
+
makeAgent,
|
|
24
|
+
makeMinimalCtx,
|
|
25
|
+
events,
|
|
26
|
+
tryImport,
|
|
27
|
+
} from "../support/helpers.ts";
|
|
28
|
+
import { INTERCOM_DETACH_REQUEST_EVENT, INTERCOM_DETACH_RESPONSE_EVENT } from "../../src/shared/types.ts";
|
|
29
|
+
import {
|
|
30
|
+
SUBAGENT_FANOUT_CHILD_ENV,
|
|
31
|
+
SUBAGENT_PARENT_CHILD_INDEX_ENV,
|
|
32
|
+
SUBAGENT_PARENT_CONTROL_INBOX_ENV,
|
|
33
|
+
SUBAGENT_PARENT_EVENT_SINK_ENV,
|
|
34
|
+
SUBAGENT_PARENT_RUN_ID_ENV,
|
|
35
|
+
} from "../../src/runs/shared/pi-args.ts";
|
|
36
|
+
|
|
37
|
+
interface ModelAttempt {
|
|
38
|
+
success?: boolean;
|
|
39
|
+
exitCode?: number;
|
|
40
|
+
error?: string;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
interface ProgressSummary {
|
|
44
|
+
agent: string;
|
|
45
|
+
index: number;
|
|
46
|
+
status: string;
|
|
47
|
+
activityState?: string;
|
|
48
|
+
lastActivityAt?: number;
|
|
49
|
+
currentTool?: string;
|
|
50
|
+
currentToolArgs?: string;
|
|
51
|
+
currentToolStartedAt?: number;
|
|
52
|
+
currentPath?: string;
|
|
53
|
+
turnCount?: number;
|
|
54
|
+
tokens?: number;
|
|
55
|
+
durationMs: number;
|
|
56
|
+
toolCount: number;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
interface ArtifactPaths {
|
|
60
|
+
outputPath: string;
|
|
61
|
+
transcriptPath?: string;
|
|
62
|
+
metadataPath?: string;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
interface RunSyncResult {
|
|
66
|
+
exitCode: number;
|
|
67
|
+
agent: string;
|
|
68
|
+
messages: unknown[];
|
|
69
|
+
error?: string;
|
|
70
|
+
model?: string;
|
|
71
|
+
skills?: string[];
|
|
72
|
+
skillsWarning?: string;
|
|
73
|
+
attemptedModels?: string[];
|
|
74
|
+
modelAttempts?: ModelAttempt[];
|
|
75
|
+
usage: { turns: number; input: number; output: number };
|
|
76
|
+
progress: ProgressSummary;
|
|
77
|
+
controlEvents?: Array<{ type?: string; message: string; reason?: string; turns?: number; tokens?: number; currentPath?: string; recentFailureSummary?: string }>;
|
|
78
|
+
artifactPaths?: ArtifactPaths;
|
|
79
|
+
transcriptPath?: string;
|
|
80
|
+
transcriptError?: string;
|
|
81
|
+
finalOutput?: string;
|
|
82
|
+
interrupted?: boolean;
|
|
83
|
+
timedOut?: boolean;
|
|
84
|
+
turnBudget?: { maxTurns: number; graceTurns: number; outcome: string; turnCount: number; wrapUpRequestedAtTurn?: number; exceededAtTurn?: number };
|
|
85
|
+
turnBudgetExceeded?: boolean;
|
|
86
|
+
wrapUpRequested?: boolean;
|
|
87
|
+
detached?: boolean;
|
|
88
|
+
detachedReason?: string;
|
|
89
|
+
savedOutputPath?: string;
|
|
90
|
+
outputMode?: "inline" | "file-only";
|
|
91
|
+
outputReference?: { path: string; bytes: number; lines: number; message: string };
|
|
92
|
+
outputSaveError?: string;
|
|
93
|
+
sessionFile?: string;
|
|
94
|
+
acceptance?: {
|
|
95
|
+
status?: string;
|
|
96
|
+
verifyRuns?: Array<{ status?: string }>;
|
|
97
|
+
runtimeChecks?: Array<{ id?: string; status?: string; message?: string }>;
|
|
98
|
+
};
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
interface MockPiCallRecord {
|
|
102
|
+
args?: string[];
|
|
103
|
+
systemPrompts?: Array<{ mode?: string; path?: string; text?: string; error?: string }>;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
function mockAssistantMessage(text: string, stopReason: "stop" | "tool_use" = "stop") {
|
|
107
|
+
return {
|
|
108
|
+
type: "message_end",
|
|
109
|
+
message: {
|
|
110
|
+
role: "assistant",
|
|
111
|
+
content: stopReason === "tool_use"
|
|
112
|
+
? [{ type: "text", text }, { type: "toolCall", name: "bash", arguments: { command: "echo test" } }]
|
|
113
|
+
: [{ type: "text", text }],
|
|
114
|
+
model: "mock/test-model",
|
|
115
|
+
stopReason,
|
|
116
|
+
usage: {
|
|
117
|
+
input: 10,
|
|
118
|
+
output: 5,
|
|
119
|
+
cacheRead: 0,
|
|
120
|
+
cacheWrite: 0,
|
|
121
|
+
cost: { total: 0.001 },
|
|
122
|
+
},
|
|
123
|
+
},
|
|
124
|
+
};
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
interface ExecutionModule {
|
|
128
|
+
runSync(
|
|
129
|
+
runtimeCwd: string,
|
|
130
|
+
agents: ReturnType<typeof makeAgentConfigs>,
|
|
131
|
+
agentName: string,
|
|
132
|
+
task: string,
|
|
133
|
+
options: Record<string, unknown>,
|
|
134
|
+
): Promise<RunSyncResult>;
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
interface UtilsModule {
|
|
138
|
+
getFinalOutput(messages: unknown[]): string;
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
interface ExecutorToolResult {
|
|
142
|
+
content: Array<{ text?: string }>;
|
|
143
|
+
isError?: boolean;
|
|
144
|
+
details?: {
|
|
145
|
+
totalCost?: { inputTokens: number; outputTokens: number; costUsd: number };
|
|
146
|
+
timeoutMs?: number;
|
|
147
|
+
};
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
interface ExecutorModule {
|
|
151
|
+
createSubagentExecutor?: (...args: unknown[]) => {
|
|
152
|
+
execute: (...args: unknown[]) => Promise<ExecutorToolResult>;
|
|
153
|
+
};
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
const execution = await tryImport<ExecutionModule>("./src/runs/foreground/execution.ts");
|
|
157
|
+
const utils = await tryImport<UtilsModule>("./src/shared/utils.ts");
|
|
158
|
+
const executorMod = await tryImport<ExecutorModule>("./src/runs/foreground/subagent-executor.ts");
|
|
159
|
+
const available = !!(execution && utils);
|
|
160
|
+
|
|
161
|
+
const runSync = execution?.runSync;
|
|
162
|
+
const getFinalOutput = utils?.getFinalOutput;
|
|
163
|
+
const createSubagentExecutor = executorMod?.createSubagentExecutor;
|
|
164
|
+
|
|
165
|
+
function escapeRegExp(value: string): string {
|
|
166
|
+
return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
function writePackageSkill(packageRoot: string, skillName: string): void {
|
|
170
|
+
const skillDir = path.join(packageRoot, "skills", skillName);
|
|
171
|
+
fs.mkdirSync(skillDir, { recursive: true });
|
|
172
|
+
fs.writeFileSync(
|
|
173
|
+
path.join(packageRoot, "package.json"),
|
|
174
|
+
JSON.stringify({ name: `${skillName}-pkg`, version: "1.0.0", pi: { skills: [`./skills/${skillName}`] } }, null, 2),
|
|
175
|
+
"utf-8",
|
|
176
|
+
);
|
|
177
|
+
fs.writeFileSync(
|
|
178
|
+
path.join(skillDir, "SKILL.md"),
|
|
179
|
+
`---\nname: ${skillName}\ndescription: test skill\n---\nbody\n`,
|
|
180
|
+
"utf-8",
|
|
181
|
+
);
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
describe("single sync execution", { skip: !available ? "pi packages not available" : undefined }, () => {
|
|
185
|
+
let tempDir: string;
|
|
186
|
+
let mockPi: MockPi;
|
|
187
|
+
|
|
188
|
+
before(() => {
|
|
189
|
+
mockPi = createMockPi();
|
|
190
|
+
mockPi.install();
|
|
191
|
+
});
|
|
192
|
+
|
|
193
|
+
after(() => {
|
|
194
|
+
mockPi.uninstall();
|
|
195
|
+
});
|
|
196
|
+
|
|
197
|
+
beforeEach(() => {
|
|
198
|
+
tempDir = createTempDir();
|
|
199
|
+
mockPi.reset();
|
|
200
|
+
});
|
|
201
|
+
|
|
202
|
+
afterEach(() => {
|
|
203
|
+
removeTempDir(tempDir);
|
|
204
|
+
});
|
|
205
|
+
|
|
206
|
+
function readCall(): { args: string[]; systemPrompts: NonNullable<MockPiCallRecord["systemPrompts"]> } {
|
|
207
|
+
const callFile = fs.readdirSync(mockPi.dir)
|
|
208
|
+
.filter((name) => name.startsWith("call-") && name.endsWith(".json"))
|
|
209
|
+
.sort()
|
|
210
|
+
.at(-1);
|
|
211
|
+
assert.ok(callFile, "expected a recorded mock pi call");
|
|
212
|
+
const payload = JSON.parse(fs.readFileSync(path.join(mockPi.dir, callFile), "utf-8")) as MockPiCallRecord;
|
|
213
|
+
assert.ok(Array.isArray(payload.args), "expected recorded args");
|
|
214
|
+
return { args: payload.args, systemPrompts: payload.systemPrompts ?? [] };
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
function readCallArgs(): string[] {
|
|
218
|
+
return readCall().args;
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
function makeExecutor(agents = [makeAgent("echo")], config: Record<string, unknown> = {}) {
|
|
222
|
+
return createSubagentExecutor!({
|
|
223
|
+
pi: { events: createEventBus(), getSessionName: () => undefined },
|
|
224
|
+
state: { baseCwd: tempDir, currentSessionId: null, asyncJobs: new Map(), foregroundControls: new Map(), lastForegroundControlId: null },
|
|
225
|
+
config,
|
|
226
|
+
asyncByDefault: false,
|
|
227
|
+
tempArtifactsDir: tempDir,
|
|
228
|
+
getSubagentSessionRoot: () => tempDir,
|
|
229
|
+
expandTilde: (value: string) => value,
|
|
230
|
+
discoverAgents: () => ({ agents }),
|
|
231
|
+
});
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
it("spawns agent and captures output", async () => {
|
|
235
|
+
mockPi.onCall({ output: "Hello from mock agent" });
|
|
236
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
237
|
+
|
|
238
|
+
const sessionFile = path.join(tempDir, "child-session.jsonl");
|
|
239
|
+
const result = await runSync(tempDir, agents, "echo", "Say hello", { sessionFile });
|
|
240
|
+
|
|
241
|
+
assert.equal(result.exitCode, 0);
|
|
242
|
+
assert.equal(result.agent, "echo");
|
|
243
|
+
assert.equal(result.sessionFile, sessionFile);
|
|
244
|
+
assert.ok(result.messages.length > 0, "should have messages");
|
|
245
|
+
|
|
246
|
+
const output = getFinalOutput(result.messages);
|
|
247
|
+
assert.equal(output, "Hello from mock agent");
|
|
248
|
+
});
|
|
249
|
+
|
|
250
|
+
it("treats action='single' with execution fields as single execution", { skip: !createSubagentExecutor ? "executor not importable" : undefined }, async () => {
|
|
251
|
+
mockPi.onCall({ output: "single alias finished" });
|
|
252
|
+
const executor = makeExecutor([makeAgent("echo")]);
|
|
253
|
+
|
|
254
|
+
const result = await executor.execute(
|
|
255
|
+
"single-alias",
|
|
256
|
+
{ action: "single", agent: "echo", task: "Run through alias" },
|
|
257
|
+
new AbortController().signal,
|
|
258
|
+
undefined,
|
|
259
|
+
makeMinimalCtx(tempDir),
|
|
260
|
+
);
|
|
261
|
+
|
|
262
|
+
assert.equal(result.isError, undefined);
|
|
263
|
+
assert.match(result.content[0]?.text ?? "", /single alias finished/);
|
|
264
|
+
});
|
|
265
|
+
|
|
266
|
+
it("rejects unknown action strings at runtime", { skip: !createSubagentExecutor ? "executor not importable" : undefined }, async () => {
|
|
267
|
+
const executor = makeExecutor([makeAgent("echo")]);
|
|
268
|
+
|
|
269
|
+
const result = await executor.execute(
|
|
270
|
+
"unknown-action",
|
|
271
|
+
{ action: "not-a-real-action" },
|
|
272
|
+
new AbortController().signal,
|
|
273
|
+
undefined,
|
|
274
|
+
makeMinimalCtx(tempDir),
|
|
275
|
+
);
|
|
276
|
+
|
|
277
|
+
assert.equal(result.isError, true);
|
|
278
|
+
assert.match(result.content[0]?.text ?? "", /Unknown action: not-a-real-action/);
|
|
279
|
+
assert.match(result.content[0]?.text ?? "", /Valid:/);
|
|
280
|
+
});
|
|
281
|
+
|
|
282
|
+
it("rejects duplicate concurrent subagent execution calls", async () => {
|
|
283
|
+
mockPi.onCall({ output: "first call completed", delay: 100 });
|
|
284
|
+
const executor = makeExecutor([makeAgent("echo")]);
|
|
285
|
+
const ctx = makeMinimalCtx(tempDir);
|
|
286
|
+
|
|
287
|
+
const first = executor.execute("first", { agent: "echo", task: "First call" }, new AbortController().signal, undefined, ctx);
|
|
288
|
+
const second = await executor.execute("second", { agent: "echo", task: "Duplicate call" }, new AbortController().signal, undefined, ctx);
|
|
289
|
+
const firstResult = await first;
|
|
290
|
+
|
|
291
|
+
assert.equal(firstResult.isError, undefined);
|
|
292
|
+
assert.equal(second.isError, true);
|
|
293
|
+
assert.match(second.content[0]?.text ?? "", /Issue exactly ONE subagent call per turn/);
|
|
294
|
+
assert.equal(mockPi.callCount(), 1);
|
|
295
|
+
});
|
|
296
|
+
|
|
297
|
+
it("blocks total subagent spawns after the per-session quota", async () => {
|
|
298
|
+
mockPi.onCall({ output: "first call completed" });
|
|
299
|
+
const executor = makeExecutor([makeAgent("echo")], { maxSubagentSpawnsPerSession: 1 });
|
|
300
|
+
const ctx = makeMinimalCtx(tempDir);
|
|
301
|
+
|
|
302
|
+
const first = await executor.execute("first", { agent: "echo", task: "First call" }, new AbortController().signal, undefined, ctx);
|
|
303
|
+
const second = await executor.execute("second", { agent: "echo", task: "Second call" }, new AbortController().signal, undefined, ctx);
|
|
304
|
+
|
|
305
|
+
assert.equal(first.isError, undefined);
|
|
306
|
+
assert.equal(second.isError, true);
|
|
307
|
+
assert.match(second.content[0]?.text ?? "", /Subagent spawn limit reached for this session \(1\/1 used, 1 requested\)/);
|
|
308
|
+
assert.equal(mockPi.callCount(), 1);
|
|
309
|
+
});
|
|
310
|
+
|
|
311
|
+
it("allows management actions while an execution call is in progress", async () => {
|
|
312
|
+
mockPi.onCall({ output: "first call completed", delay: 100 });
|
|
313
|
+
const executor = makeExecutor([makeAgent("echo")]);
|
|
314
|
+
const ctx = makeMinimalCtx(tempDir);
|
|
315
|
+
|
|
316
|
+
const first = executor.execute("first", { agent: "echo", task: "First call" }, new AbortController().signal, undefined, ctx);
|
|
317
|
+
const status = await executor.execute("status", { action: "status" }, new AbortController().signal, undefined, ctx);
|
|
318
|
+
const firstResult = await first;
|
|
319
|
+
|
|
320
|
+
assert.equal(firstResult.isError, undefined);
|
|
321
|
+
assert.equal(status.isError, undefined);
|
|
322
|
+
assert.doesNotMatch(status.content[0]?.text ?? "", /Rejected: a subagent call is already in progress/);
|
|
323
|
+
assert.equal(mockPi.callCount(), 1);
|
|
324
|
+
});
|
|
325
|
+
|
|
326
|
+
it("allows intentional parallel tasks inside one subagent execution call", async () => {
|
|
327
|
+
mockPi.onCall({ output: "first parallel result" });
|
|
328
|
+
mockPi.onCall({ output: "second parallel result" });
|
|
329
|
+
const executor = makeExecutor([makeAgent("echo"), makeAgent("second")]);
|
|
330
|
+
|
|
331
|
+
const result = await executor.execute(
|
|
332
|
+
"parallel",
|
|
333
|
+
{ tasks: [{ agent: "echo", task: "First task" }, { agent: "second", task: "Second task" }] },
|
|
334
|
+
new AbortController().signal,
|
|
335
|
+
undefined,
|
|
336
|
+
makeMinimalCtx(tempDir),
|
|
337
|
+
);
|
|
338
|
+
|
|
339
|
+
assert.equal(result.isError, undefined);
|
|
340
|
+
assert.equal(mockPi.callCount(), 2);
|
|
341
|
+
assert.deepEqual(result.details?.totalCost, { inputTokens: 200, outputTokens: 100, costUsd: 0.002 });
|
|
342
|
+
});
|
|
343
|
+
|
|
344
|
+
it("reports total cost for foreground single runs", { skip: !createSubagentExecutor ? "executor not importable" : undefined }, async () => {
|
|
345
|
+
mockPi.onCall({ output: "single result" });
|
|
346
|
+
const executor = makeExecutor([makeAgent("echo")]);
|
|
347
|
+
|
|
348
|
+
const result = await executor.execute(
|
|
349
|
+
"single-cost",
|
|
350
|
+
{ agent: "echo", task: "Single task" },
|
|
351
|
+
new AbortController().signal,
|
|
352
|
+
undefined,
|
|
353
|
+
makeMinimalCtx(tempDir),
|
|
354
|
+
);
|
|
355
|
+
|
|
356
|
+
assert.equal(result.isError, undefined);
|
|
357
|
+
assert.deepEqual(result.details?.totalCost, { inputTokens: 100, outputTokens: 50, costUsd: 0.001 });
|
|
358
|
+
});
|
|
359
|
+
|
|
360
|
+
it("fails implementation runs that complete without mutation attempts", async () => {
|
|
361
|
+
mockPi.onCall({ output: "Validation:\nlet rawFilename = params.filename.trim();" });
|
|
362
|
+
const agents = [makeAgent("worker")];
|
|
363
|
+
const controlEvents: Array<{ message: string }> = [];
|
|
364
|
+
|
|
365
|
+
const result = await runSync(tempDir, agents, "worker", "Implement the approved file changes", {
|
|
366
|
+
runId: "guard-run",
|
|
367
|
+
onControlEvent: (event: { message: string }) => controlEvents.push(event),
|
|
368
|
+
});
|
|
369
|
+
|
|
370
|
+
assert.equal(result.exitCode, 1);
|
|
371
|
+
assert.match(result.error ?? "", /completed without making edits/);
|
|
372
|
+
assert.equal(result.finalOutput, "Validation:\nlet rawFilename = params.filename.trim();");
|
|
373
|
+
assert.equal(result.progress.status, "failed");
|
|
374
|
+
assert.deepEqual(controlEvents.map((event) => event.message), [
|
|
375
|
+
"worker completed without making edits for an implementation task",
|
|
376
|
+
]);
|
|
377
|
+
assert.deepEqual(result.controlEvents?.map((event) => event.message), [
|
|
378
|
+
"worker completed without making edits for an implementation task",
|
|
379
|
+
]);
|
|
380
|
+
});
|
|
381
|
+
|
|
382
|
+
it("returns captured output when the foreground executor fails an implementation run", async () => {
|
|
383
|
+
mockPi.onCall({ output: "Oracle review:\n- finding one\n- finding two" });
|
|
384
|
+
const executor = makeExecutor([makeAgent("oracle")]);
|
|
385
|
+
|
|
386
|
+
const result = await executor.execute(
|
|
387
|
+
"failed-single-output",
|
|
388
|
+
{ agent: "oracle", task: "Implement the approved file changes" },
|
|
389
|
+
new AbortController().signal,
|
|
390
|
+
undefined,
|
|
391
|
+
makeMinimalCtx(tempDir),
|
|
392
|
+
);
|
|
393
|
+
|
|
394
|
+
const text = result.content[0]?.text ?? "";
|
|
395
|
+
assert.equal(result.isError, true);
|
|
396
|
+
assert.match(text, /completed without making edits/);
|
|
397
|
+
assert.match(text, /Output:\nOracle review:\n- finding one\n- finding two/);
|
|
398
|
+
assert.match(text, /Output artifact: /);
|
|
399
|
+
});
|
|
400
|
+
|
|
401
|
+
it("fails future-tense implementation summaries when no mutation attempt occurred", async () => {
|
|
402
|
+
mockPi.onCall({ output: "I’ll do that now and report back after implementing." });
|
|
403
|
+
const agents = [makeAgent("worker")];
|
|
404
|
+
|
|
405
|
+
const result = await runSync(tempDir, agents, "worker", "Implement the approved fixes", {
|
|
406
|
+
runId: "guard-future-tense",
|
|
407
|
+
});
|
|
408
|
+
|
|
409
|
+
assert.equal(result.exitCode, 1);
|
|
410
|
+
assert.match(result.error ?? "", /completed without making edits/);
|
|
411
|
+
});
|
|
412
|
+
|
|
413
|
+
it("allows declared read-only agents to mention implementation words without edits", async () => {
|
|
414
|
+
mockPi.onCall({ output: "Validation report after the patch" });
|
|
415
|
+
const agents = [makeAgent("architect", { tools: ["read", "grep", "find", "ls"] })];
|
|
416
|
+
|
|
417
|
+
const result = await runSync(tempDir, agents, "architect", "Produce a proposal that implements the approved fix", {
|
|
418
|
+
runId: "guard-readonly-tools",
|
|
419
|
+
});
|
|
420
|
+
|
|
421
|
+
assert.equal(result.exitCode, 0);
|
|
422
|
+
assert.equal(result.progress.status, "completed");
|
|
423
|
+
assert.equal(result.finalOutput, "Validation report after the patch");
|
|
424
|
+
});
|
|
425
|
+
|
|
426
|
+
it("keeps bash-enabled agents conservative unless completion guard is disabled", async () => {
|
|
427
|
+
mockPi.onCall({ output: "cold start test after patch" });
|
|
428
|
+
mockPi.onCall({ output: "cold start test after patch" });
|
|
429
|
+
const agents = [
|
|
430
|
+
makeAgent("test-runner", { tools: ["read", "grep", "bash", "ls"] }),
|
|
431
|
+
makeAgent("test-runner-optout", { tools: ["read", "grep", "bash", "ls"], completionGuard: false }),
|
|
432
|
+
];
|
|
433
|
+
|
|
434
|
+
const withoutOptOut = await runSync(tempDir, agents, "test-runner", "Run cold start test after patch", {
|
|
435
|
+
runId: "guard-bash-conservative",
|
|
436
|
+
});
|
|
437
|
+
assert.equal(withoutOptOut.exitCode, 1);
|
|
438
|
+
assert.match(withoutOptOut.error ?? "", /completed without making edits/);
|
|
439
|
+
|
|
440
|
+
const withOptOut = await runSync(tempDir, agents, "test-runner-optout", "Run cold start test after patch", {
|
|
441
|
+
runId: "guard-bash-optout",
|
|
442
|
+
});
|
|
443
|
+
assert.equal(withOptOut.exitCode, 0);
|
|
444
|
+
assert.equal(withOptOut.progress.status, "completed");
|
|
445
|
+
});
|
|
446
|
+
|
|
447
|
+
it("allows implementation runs when parsed messages include a real edit tool call", async () => {
|
|
448
|
+
mockPi.onCall({
|
|
449
|
+
jsonl: [
|
|
450
|
+
{
|
|
451
|
+
type: "message_end",
|
|
452
|
+
message: {
|
|
453
|
+
role: "assistant",
|
|
454
|
+
content: [{ type: "toolCall", name: "edit", arguments: { path: "src/file.ts", oldText: "a", newText: "b" } }],
|
|
455
|
+
model: "mock/test-model",
|
|
456
|
+
stopReason: "toolUse",
|
|
457
|
+
usage: { input: 100, output: 50, cacheRead: 0, cacheWrite: 0, cost: { total: 0.001 } },
|
|
458
|
+
},
|
|
459
|
+
},
|
|
460
|
+
events.assistantMessage("Applied edit"),
|
|
461
|
+
],
|
|
462
|
+
});
|
|
463
|
+
const agents = [makeAgent("worker")];
|
|
464
|
+
|
|
465
|
+
const result = await runSync(tempDir, agents, "worker", "Implement the approved file changes", {
|
|
466
|
+
runId: "guard-success",
|
|
467
|
+
});
|
|
468
|
+
|
|
469
|
+
assert.equal(result.exitCode, 0);
|
|
470
|
+
assert.equal(result.progress.status, "completed");
|
|
471
|
+
assert.equal(result.finalOutput, "Applied edit");
|
|
472
|
+
});
|
|
473
|
+
|
|
474
|
+
it("returns error for unknown agent", async () => {
|
|
475
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
476
|
+
const result = await runSync(tempDir, agents, "nonexistent", "Do something", {});
|
|
477
|
+
|
|
478
|
+
assert.equal(result.exitCode, 1);
|
|
479
|
+
assert.ok(result.error?.includes("Unknown agent"));
|
|
480
|
+
});
|
|
481
|
+
|
|
482
|
+
|
|
483
|
+
it("emits an active-long-running notice after the turn threshold", async () => {
|
|
484
|
+
mockPi.onCall({
|
|
485
|
+
jsonl: [
|
|
486
|
+
events.assistantMessage("first update"),
|
|
487
|
+
events.assistantMessage("second update"),
|
|
488
|
+
],
|
|
489
|
+
});
|
|
490
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
491
|
+
const controlEvents: NonNullable<RunSyncResult["controlEvents"]> = [];
|
|
492
|
+
|
|
493
|
+
const result = await runSync(tempDir, agents, "echo", "Investigate behavior", {
|
|
494
|
+
runId: "run-active",
|
|
495
|
+
controlConfig: { enabled: true, activeNoticeAfterTurns: 2, activeNoticeAfterMs: 999_999, activeNoticeAfterTokens: 999_999, notifyOn: ["active_long_running", "needs_attention"] },
|
|
496
|
+
onControlEvent: (event: NonNullable<RunSyncResult["controlEvents"]>[number]) => controlEvents.push(event),
|
|
497
|
+
});
|
|
498
|
+
|
|
499
|
+
assert.equal(result.exitCode, 0);
|
|
500
|
+
assert.equal(controlEvents.length, 1);
|
|
501
|
+
assert.equal(controlEvents[0]?.type, "active_long_running");
|
|
502
|
+
assert.equal(controlEvents[0]?.reason, "turn_threshold");
|
|
503
|
+
assert.equal(controlEvents[0]?.turns, 2);
|
|
504
|
+
assert.equal(result.controlEvents?.[0]?.type, "active_long_running");
|
|
505
|
+
assert.equal(result.progress.activityState, "active_long_running");
|
|
506
|
+
});
|
|
507
|
+
|
|
508
|
+
it("escalates repeated mutating tool failures to needs attention", async () => {
|
|
509
|
+
mockPi.onCall({
|
|
510
|
+
jsonl: [
|
|
511
|
+
events.toolStart("edit", { path: "src/runs/background/async-status.ts" }),
|
|
512
|
+
events.toolEnd("edit"),
|
|
513
|
+
events.toolResult("edit", "No exact match found for async-status.ts", true),
|
|
514
|
+
events.toolStart("edit", { path: "src/runs/background/async-status.ts" }),
|
|
515
|
+
events.toolEnd("edit"),
|
|
516
|
+
events.toolResult("edit", "No exact match found for async-status.ts", true),
|
|
517
|
+
events.toolStart("edit", { path: "src/runs/background/async-status.ts" }),
|
|
518
|
+
events.toolEnd("edit"),
|
|
519
|
+
events.toolResult("edit", "No exact match found for async-status.ts", true),
|
|
520
|
+
events.assistantMessage("I need to retry the same edit."),
|
|
521
|
+
],
|
|
522
|
+
});
|
|
523
|
+
const agents = [makeAgent("worker")];
|
|
524
|
+
const controlEvents: NonNullable<RunSyncResult["controlEvents"]> = [];
|
|
525
|
+
|
|
526
|
+
const result = await runSync(tempDir, agents, "worker", "Implement the approved fixes", {
|
|
527
|
+
runId: "run-failures",
|
|
528
|
+
controlConfig: { enabled: true, failedToolAttemptsBeforeAttention: 3, notifyOn: ["active_long_running", "needs_attention"] },
|
|
529
|
+
onControlEvent: (event: NonNullable<RunSyncResult["controlEvents"]>[number]) => controlEvents.push(event),
|
|
530
|
+
});
|
|
531
|
+
|
|
532
|
+
assert.equal(result.exitCode, 0);
|
|
533
|
+
const failureEvent = controlEvents.find((event) => event.reason === "tool_failures");
|
|
534
|
+
assert.equal(failureEvent?.type, "needs_attention");
|
|
535
|
+
assert.equal(failureEvent?.currentPath, "src/runs/background/async-status.ts");
|
|
536
|
+
assert.match(failureEvent?.recentFailureSummary ?? "", /No exact match/);
|
|
537
|
+
assert.equal(result.progress.activityState, "needs_attention");
|
|
538
|
+
});
|
|
539
|
+
|
|
540
|
+
it("does not surface control state or events when control is disabled", async () => {
|
|
541
|
+
mockPi.onCall({
|
|
542
|
+
jsonl: [
|
|
543
|
+
events.assistantMessage("first update"),
|
|
544
|
+
events.assistantMessage("second update"),
|
|
545
|
+
],
|
|
546
|
+
});
|
|
547
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
548
|
+
const controlEvents: NonNullable<RunSyncResult["controlEvents"]> = [];
|
|
549
|
+
|
|
550
|
+
const result = await runSync(tempDir, agents, "echo", "Investigate behavior", {
|
|
551
|
+
runId: "run-control-disabled",
|
|
552
|
+
controlConfig: { enabled: false, activeNoticeAfterTurns: 1, activeNoticeAfterMs: 1, activeNoticeAfterTokens: 1, notifyOn: ["active_long_running", "needs_attention"] },
|
|
553
|
+
onControlEvent: (event: NonNullable<RunSyncResult["controlEvents"]>[number]) => controlEvents.push(event),
|
|
554
|
+
});
|
|
555
|
+
|
|
556
|
+
assert.equal(result.exitCode, 0);
|
|
557
|
+
assert.equal(result.progress.activityState, undefined);
|
|
558
|
+
assert.equal(result.controlEvents, undefined);
|
|
559
|
+
assert.equal(controlEvents.length, 0);
|
|
560
|
+
});
|
|
561
|
+
|
|
562
|
+
it("captures non-zero exit code", async () => {
|
|
563
|
+
mockPi.onCall({ exitCode: 1, stderr: "Something went wrong" });
|
|
564
|
+
const agents = makeAgentConfigs(["fail"]);
|
|
565
|
+
|
|
566
|
+
const result = await runSync(tempDir, agents, "fail", "Do something", {});
|
|
567
|
+
|
|
568
|
+
assert.equal(result.exitCode, 1);
|
|
569
|
+
assert.ok(result.error?.includes("Something went wrong"));
|
|
570
|
+
});
|
|
571
|
+
|
|
572
|
+
it("handles long tasks via temp file (ENAMETOOLONG prevention)", async () => {
|
|
573
|
+
mockPi.onCall({ output: "Got it" });
|
|
574
|
+
const longTask = "Analyze ".repeat(2000); // ~16KB
|
|
575
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
576
|
+
|
|
577
|
+
const result = await runSync(tempDir, agents, "echo", longTask, {});
|
|
578
|
+
|
|
579
|
+
assert.equal(result.exitCode, 0);
|
|
580
|
+
const output = getFinalOutput(result.messages);
|
|
581
|
+
assert.equal(output, "Got it");
|
|
582
|
+
});
|
|
583
|
+
|
|
584
|
+
it("uses agent model config", async () => {
|
|
585
|
+
mockPi.onCall({ output: "Done" });
|
|
586
|
+
const agents = [makeAgent("echo", { model: "anthropic/claude-sonnet-4" })];
|
|
587
|
+
|
|
588
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {});
|
|
589
|
+
|
|
590
|
+
assert.equal(result.exitCode, 0);
|
|
591
|
+
// result.model is set from agent config via applyThinkingSuffix, then
|
|
592
|
+
// overwritten by the first message_end event only if result.model is unset.
|
|
593
|
+
// Since agent has model config, it stays as the configured value.
|
|
594
|
+
assert.equal(result.model, "anthropic/claude-sonnet-4");
|
|
595
|
+
});
|
|
596
|
+
|
|
597
|
+
it("model override from options takes precedence", async () => {
|
|
598
|
+
mockPi.onCall({ output: "Done" });
|
|
599
|
+
const agents = [makeAgent("echo", { model: "anthropic/claude-sonnet-4" })];
|
|
600
|
+
|
|
601
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
602
|
+
modelOverride: "openai/gpt-4o",
|
|
603
|
+
});
|
|
604
|
+
|
|
605
|
+
assert.equal(result.exitCode, 0);
|
|
606
|
+
assert.equal(result.model, "openai/gpt-4o");
|
|
607
|
+
});
|
|
608
|
+
|
|
609
|
+
it("prefers the parent session provider for ambiguous bare model ids", async () => {
|
|
610
|
+
mockPi.onCall({ output: "Done" });
|
|
611
|
+
const agents = [makeAgent("echo", { model: "gpt-5-mini" })];
|
|
612
|
+
|
|
613
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
614
|
+
availableModels: [
|
|
615
|
+
{ provider: "openai", id: "gpt-5-mini", fullId: "openai/gpt-5-mini" },
|
|
616
|
+
{ provider: "github-copilot", id: "gpt-5-mini", fullId: "github-copilot/gpt-5-mini" },
|
|
617
|
+
],
|
|
618
|
+
preferredModelProvider: "github-copilot",
|
|
619
|
+
});
|
|
620
|
+
|
|
621
|
+
assert.equal(result.exitCode, 0);
|
|
622
|
+
assert.equal(result.model, "github-copilot/gpt-5-mini");
|
|
623
|
+
assert.deepEqual(result.attemptedModels, ["github-copilot/gpt-5-mini"]);
|
|
624
|
+
});
|
|
625
|
+
|
|
626
|
+
it("tracks usage from message events", async () => {
|
|
627
|
+
mockPi.onCall({ output: "Done" });
|
|
628
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
629
|
+
|
|
630
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {});
|
|
631
|
+
|
|
632
|
+
assert.equal(result.usage.turns, 1);
|
|
633
|
+
assert.equal(result.usage.input, 100); // from mock
|
|
634
|
+
assert.equal(result.usage.output, 50); // from mock
|
|
635
|
+
});
|
|
636
|
+
|
|
637
|
+
it("retries with fallback models on retryable provider failures", async () => {
|
|
638
|
+
mockPi.onCall({
|
|
639
|
+
jsonl: [{
|
|
640
|
+
type: "message_end",
|
|
641
|
+
message: {
|
|
642
|
+
role: "assistant",
|
|
643
|
+
content: [{ type: "text", text: "temporary provider failure" }],
|
|
644
|
+
model: "openai/gpt-5-mini",
|
|
645
|
+
errorMessage: "rate limit exceeded",
|
|
646
|
+
usage: { input: 10, output: 5, cacheRead: 0, cacheWrite: 0, cost: { total: 0.01 } },
|
|
647
|
+
},
|
|
648
|
+
}],
|
|
649
|
+
exitCode: 1,
|
|
650
|
+
});
|
|
651
|
+
mockPi.onCall({ output: "Recovered on fallback" });
|
|
652
|
+
const agents = [makeAgent("echo", {
|
|
653
|
+
model: "openai/gpt-5-mini",
|
|
654
|
+
fallbackModels: ["anthropic/claude-sonnet-4"],
|
|
655
|
+
})];
|
|
656
|
+
|
|
657
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
658
|
+
runId: "fallback-sync",
|
|
659
|
+
});
|
|
660
|
+
|
|
661
|
+
assert.equal(result.exitCode, 0);
|
|
662
|
+
assert.equal(result.model, "anthropic/claude-sonnet-4");
|
|
663
|
+
assert.deepEqual(result.attemptedModels, ["openai/gpt-5-mini", "anthropic/claude-sonnet-4"]);
|
|
664
|
+
assert.equal(result.modelAttempts?.length, 2);
|
|
665
|
+
assert.equal(result.modelAttempts?.[0]?.success, false);
|
|
666
|
+
assert.equal(result.modelAttempts?.[1]?.success, true);
|
|
667
|
+
assert.equal(result.usage.turns, 2);
|
|
668
|
+
assert.equal(mockPi.callCount(), 2);
|
|
669
|
+
});
|
|
670
|
+
|
|
671
|
+
it("retries with fallback models when provider errors exit zero", async () => {
|
|
672
|
+
mockPi.onCall({
|
|
673
|
+
jsonl: [{
|
|
674
|
+
type: "message_end",
|
|
675
|
+
message: {
|
|
676
|
+
role: "assistant",
|
|
677
|
+
content: [{ type: "text", text: "weekly quota hit" }],
|
|
678
|
+
model: "openai/gpt-5-mini",
|
|
679
|
+
errorMessage: "429 you have reached your weekly usage limit / quota exceeded",
|
|
680
|
+
usage: { input: 10, output: 5, cacheRead: 0, cacheWrite: 0, cost: { total: 0.01 } },
|
|
681
|
+
},
|
|
682
|
+
}],
|
|
683
|
+
exitCode: 0,
|
|
684
|
+
});
|
|
685
|
+
mockPi.onCall({ output: "Recovered on fallback" });
|
|
686
|
+
const agents = [makeAgent("echo", {
|
|
687
|
+
model: "openai/gpt-5-mini",
|
|
688
|
+
fallbackModels: ["anthropic/claude-sonnet-4"],
|
|
689
|
+
})];
|
|
690
|
+
|
|
691
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
692
|
+
runId: "fallback-zero-exit-provider-error",
|
|
693
|
+
});
|
|
694
|
+
|
|
695
|
+
assert.equal(result.exitCode, 0);
|
|
696
|
+
assert.equal(result.model, "anthropic/claude-sonnet-4");
|
|
697
|
+
assert.deepEqual(result.modelAttempts?.map((attempt) => attempt.success), [false, true]);
|
|
698
|
+
});
|
|
699
|
+
|
|
700
|
+
it("retries with fallback models when a zero-exit attempt has empty output", async () => {
|
|
701
|
+
mockPi.onCall({
|
|
702
|
+
jsonl: [{
|
|
703
|
+
type: "message_end",
|
|
704
|
+
message: {
|
|
705
|
+
role: "assistant",
|
|
706
|
+
content: [{ type: "text", text: "" }],
|
|
707
|
+
model: "openai/gpt-5-mini",
|
|
708
|
+
stopReason: "error",
|
|
709
|
+
usage: { input: 10, output: 0, cacheRead: 0, cacheWrite: 0, cost: { total: 0.01 } },
|
|
710
|
+
},
|
|
711
|
+
}],
|
|
712
|
+
exitCode: 0,
|
|
713
|
+
});
|
|
714
|
+
mockPi.onCall({ output: "Recovered from empty output" });
|
|
715
|
+
const agents = [makeAgent("echo", {
|
|
716
|
+
model: "openai/gpt-5-mini",
|
|
717
|
+
fallbackModels: ["anthropic/claude-sonnet-4"],
|
|
718
|
+
})];
|
|
719
|
+
|
|
720
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
721
|
+
runId: "fallback-zero-exit-empty-output",
|
|
722
|
+
});
|
|
723
|
+
|
|
724
|
+
assert.equal(result.exitCode, 0);
|
|
725
|
+
assert.equal(result.model, "anthropic/claude-sonnet-4");
|
|
726
|
+
assert.equal(result.finalOutput, "Recovered from empty output");
|
|
727
|
+
assert.match(result.modelAttempts?.[0]?.error ?? "", /no output/i);
|
|
728
|
+
assert.deepEqual(result.modelAttempts?.map((attempt) => attempt.success), [false, true]);
|
|
729
|
+
assert.equal(mockPi.callCount(), 2);
|
|
730
|
+
});
|
|
731
|
+
|
|
732
|
+
it("fails zero-exit provider errors when no fallback succeeds", async () => {
|
|
733
|
+
mockPi.onCall({
|
|
734
|
+
jsonl: [{
|
|
735
|
+
type: "message_end",
|
|
736
|
+
message: {
|
|
737
|
+
role: "assistant",
|
|
738
|
+
content: [{ type: "text", text: "weekly quota hit" }],
|
|
739
|
+
model: "openai/gpt-5-mini",
|
|
740
|
+
errorMessage: "429 quota exceeded",
|
|
741
|
+
usage: { input: 10, output: 5, cacheRead: 0, cacheWrite: 0, cost: { total: 0.01 } },
|
|
742
|
+
},
|
|
743
|
+
}],
|
|
744
|
+
exitCode: 0,
|
|
745
|
+
});
|
|
746
|
+
const agents = [makeAgent("echo", { model: "openai/gpt-5-mini" })];
|
|
747
|
+
|
|
748
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
749
|
+
runId: "zero-exit-provider-error-no-fallback",
|
|
750
|
+
});
|
|
751
|
+
|
|
752
|
+
assert.equal(result.exitCode, 1);
|
|
753
|
+
assert.match(result.error ?? "", /429 quota exceeded/);
|
|
754
|
+
assert.deepEqual(result.modelAttempts?.map((attempt) => attempt.success), [false]);
|
|
755
|
+
});
|
|
756
|
+
|
|
757
|
+
it("treats recovered child tool errors as successful foreground runs", async () => {
|
|
758
|
+
mockPi.onCall({
|
|
759
|
+
jsonl: [
|
|
760
|
+
events.toolResult("read", "EISDIR: illegal operation on a directory", true),
|
|
761
|
+
events.assistantMessage("Done"),
|
|
762
|
+
],
|
|
763
|
+
});
|
|
764
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
765
|
+
|
|
766
|
+
const result = await runSync(tempDir, agents, "echo", "Inspect files", {
|
|
767
|
+
runId: "recovered-tool-error",
|
|
768
|
+
});
|
|
769
|
+
|
|
770
|
+
assert.equal(result.exitCode, 0);
|
|
771
|
+
assert.equal(result.error, undefined);
|
|
772
|
+
assert.equal(result.finalOutput, "Done");
|
|
773
|
+
assert.equal(getFinalOutput(result.messages), "Done");
|
|
774
|
+
assert.equal(result.progress.status, "completed");
|
|
775
|
+
});
|
|
776
|
+
|
|
777
|
+
it("treats recovered assistant provider errors as successful foreground runs", async () => {
|
|
778
|
+
mockPi.onCall({
|
|
779
|
+
jsonl: [
|
|
780
|
+
{
|
|
781
|
+
type: "message_end",
|
|
782
|
+
message: {
|
|
783
|
+
role: "assistant",
|
|
784
|
+
content: [{ type: "text", text: "temporary provider failure" }],
|
|
785
|
+
model: "openai/gpt-5-mini",
|
|
786
|
+
stopReason: "error",
|
|
787
|
+
errorMessage: "provider transport failed",
|
|
788
|
+
usage: { input: 10, output: 5, cacheRead: 0, cacheWrite: 0, cost: { total: 0.01 } },
|
|
789
|
+
},
|
|
790
|
+
},
|
|
791
|
+
events.assistantMessage("Recovered"),
|
|
792
|
+
],
|
|
793
|
+
});
|
|
794
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
795
|
+
|
|
796
|
+
const result = await runSync(tempDir, agents, "echo", "Recover from provider error", {
|
|
797
|
+
runId: "recovered-provider-error",
|
|
798
|
+
});
|
|
799
|
+
|
|
800
|
+
assert.equal(result.exitCode, 0);
|
|
801
|
+
assert.equal(result.error, undefined);
|
|
802
|
+
assert.equal(result.finalOutput, "Recovered");
|
|
803
|
+
assert.equal(getFinalOutput(result.messages), "Recovered");
|
|
804
|
+
assert.equal(result.progress.status, "completed");
|
|
805
|
+
});
|
|
806
|
+
|
|
807
|
+
it("keeps provider errors failed when followed only by empty assistant output", async () => {
|
|
808
|
+
mockPi.onCall({
|
|
809
|
+
jsonl: [
|
|
810
|
+
{
|
|
811
|
+
type: "message_end",
|
|
812
|
+
message: {
|
|
813
|
+
role: "assistant",
|
|
814
|
+
content: [{ type: "text", text: "temporary provider failure" }],
|
|
815
|
+
model: "openai/gpt-5-mini",
|
|
816
|
+
stopReason: "error",
|
|
817
|
+
errorMessage: "provider transport failed",
|
|
818
|
+
usage: { input: 10, output: 5, cacheRead: 0, cacheWrite: 0, cost: { total: 0.01 } },
|
|
819
|
+
},
|
|
820
|
+
},
|
|
821
|
+
events.assistantMessage(""),
|
|
822
|
+
],
|
|
823
|
+
});
|
|
824
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
825
|
+
|
|
826
|
+
const result = await runSync(tempDir, agents, "echo", "Recover from provider error", {
|
|
827
|
+
runId: "provider-error-empty-stop",
|
|
828
|
+
});
|
|
829
|
+
|
|
830
|
+
assert.equal(result.exitCode, 1);
|
|
831
|
+
assert.match(result.error ?? "", /provider transport failed/);
|
|
832
|
+
assert.equal(result.finalOutput, "");
|
|
833
|
+
assert.equal(result.progress.status, "failed");
|
|
834
|
+
});
|
|
835
|
+
|
|
836
|
+
it("fails when all fallback model attempts report provider errors", async () => {
|
|
837
|
+
for (const model of ["openai/gpt-5-mini", "anthropic/claude-sonnet-4"]) {
|
|
838
|
+
mockPi.onCall({
|
|
839
|
+
jsonl: [{
|
|
840
|
+
type: "message_end",
|
|
841
|
+
message: {
|
|
842
|
+
role: "assistant",
|
|
843
|
+
content: [{ type: "text", text: `${model} quota hit` }],
|
|
844
|
+
model,
|
|
845
|
+
errorMessage: "429 quota exceeded",
|
|
846
|
+
usage: { input: 10, output: 5, cacheRead: 0, cacheWrite: 0, cost: { total: 0.01 } },
|
|
847
|
+
},
|
|
848
|
+
}],
|
|
849
|
+
exitCode: 0,
|
|
850
|
+
});
|
|
851
|
+
}
|
|
852
|
+
const agents = [makeAgent("echo", {
|
|
853
|
+
model: "openai/gpt-5-mini",
|
|
854
|
+
fallbackModels: ["anthropic/claude-sonnet-4"],
|
|
855
|
+
})];
|
|
856
|
+
|
|
857
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
858
|
+
runId: "zero-exit-provider-error-all-fallbacks-fail",
|
|
859
|
+
});
|
|
860
|
+
|
|
861
|
+
assert.equal(result.exitCode, 1);
|
|
862
|
+
assert.deepEqual(result.modelAttempts?.map((attempt) => attempt.success), [false, false]);
|
|
863
|
+
assert.match(result.error ?? "", /429 quota exceeded/);
|
|
864
|
+
});
|
|
865
|
+
|
|
866
|
+
it("baselines output files per fallback attempt", async () => {
|
|
867
|
+
const outputPath = path.join(tempDir, "fallback-output.md");
|
|
868
|
+
mockPi.onCall({
|
|
869
|
+
jsonl: [{
|
|
870
|
+
type: "message_end",
|
|
871
|
+
message: {
|
|
872
|
+
role: "assistant",
|
|
873
|
+
content: [{ type: "text", text: "primary failed" }],
|
|
874
|
+
model: "openai/gpt-5-mini",
|
|
875
|
+
errorMessage: "429 quota exceeded",
|
|
876
|
+
usage: { input: 10, output: 5, cacheRead: 0, cacheWrite: 0, cost: { total: 0.01 } },
|
|
877
|
+
},
|
|
878
|
+
}],
|
|
879
|
+
exitCode: 0,
|
|
880
|
+
delay: 100,
|
|
881
|
+
});
|
|
882
|
+
mockPi.onCall({ output: "fallback assistant output" });
|
|
883
|
+
const agents = [makeAgent("echo", {
|
|
884
|
+
model: "openai/gpt-5-mini",
|
|
885
|
+
fallbackModels: ["anthropic/claude-sonnet-4"],
|
|
886
|
+
})];
|
|
887
|
+
|
|
888
|
+
const runPromise = runSync(tempDir, agents, "echo", "Task", {
|
|
889
|
+
runId: "fallback-output-per-attempt",
|
|
890
|
+
outputPath,
|
|
891
|
+
});
|
|
892
|
+
setTimeout(() => {
|
|
893
|
+
fs.writeFileSync(outputPath, "stale partial output from failed primary", "utf-8");
|
|
894
|
+
}, 20);
|
|
895
|
+
|
|
896
|
+
const result = await runPromise;
|
|
897
|
+
|
|
898
|
+
assert.equal(result.exitCode, 0);
|
|
899
|
+
assert.equal(fs.readFileSync(outputPath, "utf-8"), "fallback assistant output");
|
|
900
|
+
});
|
|
901
|
+
|
|
902
|
+
it("does not retry on ordinary task/tool failures", async () => {
|
|
903
|
+
mockPi.onCall({
|
|
904
|
+
jsonl: [events.toolResult("bash", "process exited with code 127")],
|
|
905
|
+
exitCode: 0,
|
|
906
|
+
});
|
|
907
|
+
const agents = [makeAgent("echo", {
|
|
908
|
+
model: "openai/gpt-5-mini",
|
|
909
|
+
fallbackModels: ["anthropic/claude-sonnet-4"],
|
|
910
|
+
})];
|
|
911
|
+
|
|
912
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
913
|
+
runId: "no-fallback-task-failure",
|
|
914
|
+
});
|
|
915
|
+
|
|
916
|
+
assert.equal(result.exitCode, 127);
|
|
917
|
+
assert.equal(result.modelAttempts?.length, 1);
|
|
918
|
+
assert.equal(mockPi.callCount(), 1);
|
|
919
|
+
});
|
|
920
|
+
|
|
921
|
+
it("tracks progress during execution", async () => {
|
|
922
|
+
mockPi.onCall({ output: "Done" });
|
|
923
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
924
|
+
|
|
925
|
+
const result = await runSync(tempDir, agents, "echo", "Task", { index: 3 });
|
|
926
|
+
|
|
927
|
+
assert.ok(result.progress, "should have progress");
|
|
928
|
+
assert.equal(result.progress.agent, "echo");
|
|
929
|
+
assert.equal(result.progress.index, 3);
|
|
930
|
+
assert.equal(result.progress.status, "completed");
|
|
931
|
+
assert.ok(result.progress.durationMs > 0, "should track duration");
|
|
932
|
+
});
|
|
933
|
+
|
|
934
|
+
it("tracks live activity updates and exposes artifact paths while running", async () => {
|
|
935
|
+
const updates: Array<{ details?: { results?: Array<{ artifactPaths?: ArtifactPaths }>; progress?: ProgressSummary[] } }> = [];
|
|
936
|
+
mockPi.onCall({
|
|
937
|
+
steps: [
|
|
938
|
+
{ jsonl: [events.toolStart("read", { path: "package.json" })], delay: 20 },
|
|
939
|
+
{ jsonl: [events.toolEnd("read"), events.toolResult("read", "{\"name\":\"pkg\"}")], delay: 20 },
|
|
940
|
+
{ jsonl: [events.assistantMessage("Done")] },
|
|
941
|
+
],
|
|
942
|
+
});
|
|
943
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
944
|
+
const artifactsDir = path.join(tempDir, "artifacts");
|
|
945
|
+
|
|
946
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
947
|
+
runId: "live-progress",
|
|
948
|
+
artifactsDir,
|
|
949
|
+
artifactConfig: { enabled: true, includeInput: true, includeOutput: true, includeMetadata: true },
|
|
950
|
+
onUpdate: (update: { details?: { results?: Array<{ artifactPaths?: ArtifactPaths }>; progress?: ProgressSummary[] } }) => {
|
|
951
|
+
updates.push(update);
|
|
952
|
+
},
|
|
953
|
+
});
|
|
954
|
+
|
|
955
|
+
assert.ok(updates.length > 0, "expected at least one live progress update");
|
|
956
|
+
assert.equal(
|
|
957
|
+
updates.some((update) => update.details?.results?.[0]?.artifactPaths?.outputPath.endsWith("_output.md") === true),
|
|
958
|
+
true,
|
|
959
|
+
);
|
|
960
|
+
const runningToolUpdate = updates.find((update) => update.details?.progress?.[0]?.currentTool === "read");
|
|
961
|
+
assert.ok(runningToolUpdate, "expected a live progress update for the running tool");
|
|
962
|
+
assert.equal(runningToolUpdate?.details?.progress?.[0]?.currentTool, "read");
|
|
963
|
+
assert.equal(typeof runningToolUpdate?.details?.progress?.[0]?.currentToolStartedAt, "number");
|
|
964
|
+
assert.equal(typeof result.progress.lastActivityAt, "number");
|
|
965
|
+
assert.equal(result.progress.currentToolStartedAt, undefined);
|
|
966
|
+
});
|
|
967
|
+
|
|
968
|
+
it("sets progress.status to failed on non-zero exit", async () => {
|
|
969
|
+
mockPi.onCall({ exitCode: 1 });
|
|
970
|
+
const agents = makeAgentConfigs(["fail"]);
|
|
971
|
+
|
|
972
|
+
const result = await runSync(tempDir, agents, "fail", "Task", {});
|
|
973
|
+
|
|
974
|
+
assert.equal(result.progress.status, "failed");
|
|
975
|
+
});
|
|
976
|
+
|
|
977
|
+
it("handles multi-turn conversation from JSONL", async () => {
|
|
978
|
+
mockPi.onCall({
|
|
979
|
+
jsonl: [
|
|
980
|
+
events.toolStart("bash", { command: "ls" }),
|
|
981
|
+
events.toolEnd("bash"),
|
|
982
|
+
events.toolResult("bash", "file1.txt\nfile2.txt"),
|
|
983
|
+
events.assistantMessage("Found 2 files: file1.txt and file2.txt"),
|
|
984
|
+
],
|
|
985
|
+
});
|
|
986
|
+
const agents = makeAgentConfigs(["scout"]);
|
|
987
|
+
|
|
988
|
+
const result = await runSync(tempDir, agents, "scout", "List files", {});
|
|
989
|
+
|
|
990
|
+
assert.equal(result.exitCode, 0);
|
|
991
|
+
const output = getFinalOutput(result.messages);
|
|
992
|
+
assert.ok(output.includes("file1.txt"), "should capture assistant text");
|
|
993
|
+
assert.equal(result.progress.toolCount, 1, "should count tool calls");
|
|
994
|
+
});
|
|
995
|
+
|
|
996
|
+
it("resolves skills from the effective task cwd", async () => {
|
|
997
|
+
const taskCwd = createTempDir("pi-subagent-task-cwd-");
|
|
998
|
+
try {
|
|
999
|
+
writePackageSkill(taskCwd, "task-cwd-skill");
|
|
1000
|
+
mockPi.onCall({ output: "Done" });
|
|
1001
|
+
const agents = [makeAgent("echo", { skills: ["task-cwd-skill"] })];
|
|
1002
|
+
|
|
1003
|
+
const result = await runSync(tempDir, agents, "echo", "Task", { cwd: taskCwd });
|
|
1004
|
+
|
|
1005
|
+
assert.equal(result.exitCode, 0);
|
|
1006
|
+
assert.deepEqual(result.skills, ["task-cwd-skill"]);
|
|
1007
|
+
assert.equal(result.skillsWarning, undefined);
|
|
1008
|
+
} finally {
|
|
1009
|
+
removeTempDir(taskCwd);
|
|
1010
|
+
}
|
|
1011
|
+
});
|
|
1012
|
+
|
|
1013
|
+
it("falls back to the runtime cwd when the task cwd lacks a skill", async () => {
|
|
1014
|
+
const taskCwd = path.join(tempDir, "nested");
|
|
1015
|
+
fs.mkdirSync(taskCwd, { recursive: true });
|
|
1016
|
+
writePackageSkill(tempDir, "runtime-fallback-skill");
|
|
1017
|
+
mockPi.onCall({ output: "Done" });
|
|
1018
|
+
const agents = [makeAgent("echo", { skills: ["runtime-fallback-skill"] })];
|
|
1019
|
+
|
|
1020
|
+
const result = await runSync(tempDir, agents, "echo", "Task", { cwd: taskCwd });
|
|
1021
|
+
|
|
1022
|
+
assert.equal(result.exitCode, 0);
|
|
1023
|
+
assert.deepEqual(result.skills, ["runtime-fallback-skill"]);
|
|
1024
|
+
assert.equal(result.skillsWarning, undefined);
|
|
1025
|
+
});
|
|
1026
|
+
|
|
1027
|
+
it("fails foreground runs on explicit unavailable pi-subagents skill requests without spawning", async () => {
|
|
1028
|
+
const agents = [makeAgent("worker")];
|
|
1029
|
+
|
|
1030
|
+
const result = await runSync(tempDir, agents, "worker", "Task", { skills: ["pi-subagents"] });
|
|
1031
|
+
|
|
1032
|
+
assert.equal(result.exitCode, 1);
|
|
1033
|
+
assert.equal(result.error, "Skills not found: pi-subagents");
|
|
1034
|
+
assert.equal(mockPi.callCount(), 0);
|
|
1035
|
+
});
|
|
1036
|
+
|
|
1037
|
+
it("fails foreground runs when an agent default requests pi-subagents skill", async () => {
|
|
1038
|
+
const agents = [makeAgent("worker", { skills: ["pi-subagents"] })];
|
|
1039
|
+
|
|
1040
|
+
const result = await runSync(tempDir, agents, "worker", "Task", {});
|
|
1041
|
+
|
|
1042
|
+
assert.equal(result.exitCode, 1);
|
|
1043
|
+
assert.equal(result.error, "Skills not found: pi-subagents");
|
|
1044
|
+
assert.equal(mockPi.callCount(), 0);
|
|
1045
|
+
});
|
|
1046
|
+
|
|
1047
|
+
it("writes artifacts when configured", async () => {
|
|
1048
|
+
mockPi.onCall({ output: "Result text" });
|
|
1049
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
1050
|
+
const artifactsDir = path.join(tempDir, "artifacts");
|
|
1051
|
+
|
|
1052
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
1053
|
+
runId: "test-run",
|
|
1054
|
+
artifactsDir,
|
|
1055
|
+
artifactConfig: { enabled: true, includeInput: true, includeOutput: true, includeMetadata: true },
|
|
1056
|
+
});
|
|
1057
|
+
|
|
1058
|
+
assert.equal(result.exitCode, 0);
|
|
1059
|
+
assert.ok(result.artifactPaths, "should have artifact paths");
|
|
1060
|
+
assert.ok(result.transcriptPath, "should expose transcript path on the result");
|
|
1061
|
+
assert.equal(result.transcriptPath, result.artifactPaths.transcriptPath);
|
|
1062
|
+
assert.ok(fs.existsSync(result.transcriptPath), "transcript should be written");
|
|
1063
|
+
const transcript = fs.readFileSync(result.transcriptPath, "utf-8").trim().split("\n").map((line) => JSON.parse(line) as { recordType?: string; source?: string; text?: string });
|
|
1064
|
+
assert.equal(transcript[0]?.recordType, "message");
|
|
1065
|
+
assert.equal(transcript[0]?.source, "foreground");
|
|
1066
|
+
assert.match(transcript.at(-1)?.text ?? "", /^Result text/);
|
|
1067
|
+
assert.equal(result.transcriptError, undefined);
|
|
1068
|
+
assert.ok(fs.existsSync(artifactsDir), "artifacts dir should exist");
|
|
1069
|
+
});
|
|
1070
|
+
|
|
1071
|
+
it("does not surface transcript paths when transcript artifacts are disabled", async () => {
|
|
1072
|
+
mockPi.onCall({ output: "Result text" });
|
|
1073
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
1074
|
+
const artifactsDir = path.join(tempDir, "artifacts-disabled-transcript");
|
|
1075
|
+
|
|
1076
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
1077
|
+
runId: "test-run-no-transcript",
|
|
1078
|
+
artifactsDir,
|
|
1079
|
+
artifactConfig: { enabled: true, includeInput: true, includeOutput: true, includeTranscript: false, includeMetadata: true },
|
|
1080
|
+
});
|
|
1081
|
+
|
|
1082
|
+
assert.equal(result.exitCode, 0);
|
|
1083
|
+
assert.equal(result.transcriptPath, undefined);
|
|
1084
|
+
assert.equal(result.transcriptError, undefined);
|
|
1085
|
+
assert.ok(result.artifactPaths?.metadataPath, "should have metadata path");
|
|
1086
|
+
const metadata = JSON.parse(fs.readFileSync(result.artifactPaths.metadataPath, "utf-8")) as { transcriptPath?: string; transcriptError?: string };
|
|
1087
|
+
assert.equal(metadata.transcriptPath, undefined);
|
|
1088
|
+
assert.equal(metadata.transcriptError, undefined);
|
|
1089
|
+
assert.equal(fs.existsSync(result.artifactPaths.transcriptPath!), false);
|
|
1090
|
+
});
|
|
1091
|
+
|
|
1092
|
+
it("preserves agent-written output files instead of overwriting them with the final receipt", async () => {
|
|
1093
|
+
const outputPath = path.join(tempDir, "report.md");
|
|
1094
|
+
const artifactsDir = path.join(tempDir, "artifacts");
|
|
1095
|
+
mockPi.onCall({ output: `Wrote to ${outputPath}`, delay: 100 });
|
|
1096
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
1097
|
+
|
|
1098
|
+
const runPromise = runSync(tempDir, agents, "echo", "Task", {
|
|
1099
|
+
runId: "output-file-preserved",
|
|
1100
|
+
outputPath,
|
|
1101
|
+
artifactsDir,
|
|
1102
|
+
artifactConfig: { enabled: true, includeInput: true, includeOutput: true, includeMetadata: true },
|
|
1103
|
+
});
|
|
1104
|
+
|
|
1105
|
+
setTimeout(() => {
|
|
1106
|
+
fs.writeFileSync(outputPath, "real file content", "utf-8");
|
|
1107
|
+
}, 20);
|
|
1108
|
+
|
|
1109
|
+
const result = await runPromise;
|
|
1110
|
+
assert.equal(result.exitCode, 0);
|
|
1111
|
+
assert.equal(result.finalOutput, "real file content");
|
|
1112
|
+
assert.equal(fs.readFileSync(outputPath, "utf-8"), "real file content");
|
|
1113
|
+
assert.ok(result.artifactPaths, "should have artifact paths");
|
|
1114
|
+
assert.equal(fs.readFileSync(result.artifactPaths.outputPath, "utf-8"), "real file content");
|
|
1115
|
+
});
|
|
1116
|
+
|
|
1117
|
+
it("falls back to persisting assistant output when the target file was not changed", async () => {
|
|
1118
|
+
const outputPath = path.join(tempDir, "report.md");
|
|
1119
|
+
fs.writeFileSync(outputPath, "stale content", "utf-8");
|
|
1120
|
+
mockPi.onCall({ output: "fresh assistant output" });
|
|
1121
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
1122
|
+
|
|
1123
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
1124
|
+
runId: "output-file-fallback",
|
|
1125
|
+
outputPath,
|
|
1126
|
+
});
|
|
1127
|
+
|
|
1128
|
+
assert.equal(result.exitCode, 0);
|
|
1129
|
+
assert.equal(result.finalOutput, "fresh assistant output");
|
|
1130
|
+
assert.equal(fs.readFileSync(outputPath, "utf-8"), "fresh assistant output");
|
|
1131
|
+
});
|
|
1132
|
+
|
|
1133
|
+
it("routes foreground single relative outputs to the run output artifact directory by default", { skip: !createSubagentExecutor ? "executor not importable" : undefined }, async () => {
|
|
1134
|
+
mockPi.onCall({ output: "default report" });
|
|
1135
|
+
const executor = makeExecutor([makeAgent("researcher", { output: "context.md" })]);
|
|
1136
|
+
|
|
1137
|
+
const result = await executor.execute(
|
|
1138
|
+
"single-default-output-base",
|
|
1139
|
+
{ agent: "researcher", task: "Write report" },
|
|
1140
|
+
new AbortController().signal,
|
|
1141
|
+
undefined,
|
|
1142
|
+
makeMinimalCtx(tempDir),
|
|
1143
|
+
);
|
|
1144
|
+
|
|
1145
|
+
const taskArg = readCallArgs().at(-1) ?? "";
|
|
1146
|
+
assert.equal(result.isError, undefined);
|
|
1147
|
+
assert.match(taskArg, new RegExp(`Write your findings to exactly this path: ${escapeRegExp(path.join(tempDir, ".pi-subagents", "artifacts", "outputs"))}.*context\\.md`));
|
|
1148
|
+
assert.equal(fs.existsSync(path.join(tempDir, "context.md")), false);
|
|
1149
|
+
});
|
|
1150
|
+
|
|
1151
|
+
it("routes foreground single relative outputs to configured singleRunOutputBaseDir", { skip: !createSubagentExecutor ? "executor not importable" : undefined }, async () => {
|
|
1152
|
+
mockPi.onCall({ output: "configured report" });
|
|
1153
|
+
const configuredBase = path.join(tempDir, "configured-outputs");
|
|
1154
|
+
const executor = makeExecutor(
|
|
1155
|
+
[makeAgent("researcher", { output: "context.md" })],
|
|
1156
|
+
{ singleRunOutputBaseDir: configuredBase },
|
|
1157
|
+
);
|
|
1158
|
+
|
|
1159
|
+
const result = await executor.execute(
|
|
1160
|
+
"single-configured-output-base",
|
|
1161
|
+
{ agent: "researcher", task: "Write report" },
|
|
1162
|
+
new AbortController().signal,
|
|
1163
|
+
undefined,
|
|
1164
|
+
makeMinimalCtx(tempDir),
|
|
1165
|
+
);
|
|
1166
|
+
|
|
1167
|
+
const expectedOutputPath = path.join(configuredBase, "context.md");
|
|
1168
|
+
const taskArg = readCallArgs().at(-1) ?? "";
|
|
1169
|
+
assert.equal(result.isError, undefined);
|
|
1170
|
+
assert.match(taskArg, new RegExp(`Write your findings to exactly this path: ${escapeRegExp(expectedOutputPath)}`));
|
|
1171
|
+
assert.equal(fs.readFileSync(expectedOutputPath, "utf-8"), "configured report");
|
|
1172
|
+
assert.equal(fs.existsSync(path.join(tempDir, "context.md")), false);
|
|
1173
|
+
});
|
|
1174
|
+
|
|
1175
|
+
it("makes task-level output overrides authoritative in the child system prompt", { skip: !createSubagentExecutor ? "executor not importable" : undefined }, async () => {
|
|
1176
|
+
mockPi.onCall({ output: "override report" });
|
|
1177
|
+
const overridePath = path.join(tempDir, "custom-report.md");
|
|
1178
|
+
const executor = makeExecutor([
|
|
1179
|
+
makeAgent("researcher", {
|
|
1180
|
+
output: "default-report.md",
|
|
1181
|
+
systemPrompt: "Output format (`default-report.md`):\n\nWrite the full report to default-report.md.",
|
|
1182
|
+
}),
|
|
1183
|
+
]);
|
|
1184
|
+
|
|
1185
|
+
const result = await executor.execute(
|
|
1186
|
+
"single-output-override-system-prompt",
|
|
1187
|
+
{ agent: "researcher", task: "Write report", output: overridePath },
|
|
1188
|
+
new AbortController().signal,
|
|
1189
|
+
undefined,
|
|
1190
|
+
makeMinimalCtx(tempDir),
|
|
1191
|
+
);
|
|
1192
|
+
|
|
1193
|
+
const call = readCall();
|
|
1194
|
+
const taskArg = call.args.at(-1) ?? "";
|
|
1195
|
+
const systemPrompt = call.systemPrompts[0]?.text ?? "";
|
|
1196
|
+
assert.equal(result.isError, undefined);
|
|
1197
|
+
assert.match(taskArg, new RegExp(`Write your findings to exactly this path: ${escapeRegExp(overridePath)}`));
|
|
1198
|
+
assert.match(systemPrompt, /Output format \(`default-report\.md`\):/);
|
|
1199
|
+
assert.match(systemPrompt, /Runtime output path override:/);
|
|
1200
|
+
assert.match(systemPrompt, new RegExp(`Write your findings to exactly this path: ${escapeRegExp(overridePath)}`));
|
|
1201
|
+
assert.match(systemPrompt, /Ignore any other output filename or output path mentioned elsewhere/);
|
|
1202
|
+
});
|
|
1203
|
+
|
|
1204
|
+
it("treats string false as disabled output in foreground single runs", { skip: !createSubagentExecutor ? "executor not importable" : undefined }, async () => {
|
|
1205
|
+
mockPi.onCall({ output: "inline report" });
|
|
1206
|
+
const executor = makeExecutor([makeAgent("echo", { output: "default-report.md" })]);
|
|
1207
|
+
|
|
1208
|
+
const result = await executor.execute(
|
|
1209
|
+
"single-string-false-output",
|
|
1210
|
+
{ agent: "echo", task: "Write report", output: "false" },
|
|
1211
|
+
new AbortController().signal,
|
|
1212
|
+
undefined,
|
|
1213
|
+
makeMinimalCtx(tempDir),
|
|
1214
|
+
);
|
|
1215
|
+
|
|
1216
|
+
assert.equal(result.isError, undefined);
|
|
1217
|
+
assert.match(result.content[0]?.text ?? "", /inline report/);
|
|
1218
|
+
assert.doesNotMatch(result.content[0]?.text ?? "", /Output saved to:/);
|
|
1219
|
+
assert.equal(fs.existsSync(path.join(tempDir, "false")), false);
|
|
1220
|
+
assert.equal(fs.existsSync(path.join(tempDir, "default-report.md")), false);
|
|
1221
|
+
assert.doesNotMatch(readCallArgs().at(-1) ?? "", /Write your findings to(?: exactly this path)?:/);
|
|
1222
|
+
});
|
|
1223
|
+
|
|
1224
|
+
it("rejects mismatched foreground timeout aliases before spawning", { skip: !createSubagentExecutor ? "executor not importable" : undefined }, async () => {
|
|
1225
|
+
const executor = makeExecutor();
|
|
1226
|
+
|
|
1227
|
+
const result = await executor.execute(
|
|
1228
|
+
"timeout-alias-validation",
|
|
1229
|
+
{ agent: "echo", task: "Task", timeoutMs: 100, maxRuntimeMs: 200 },
|
|
1230
|
+
new AbortController().signal,
|
|
1231
|
+
undefined,
|
|
1232
|
+
makeMinimalCtx(tempDir),
|
|
1233
|
+
);
|
|
1234
|
+
|
|
1235
|
+
assert.equal(result.isError, true);
|
|
1236
|
+
assert.match(result.content[0]?.text ?? "", /aliases/);
|
|
1237
|
+
assert.equal(mockPi.callCount(), 0);
|
|
1238
|
+
});
|
|
1239
|
+
|
|
1240
|
+
it("allows timeout settings for async runs before spawning", { skip: !createSubagentExecutor ? "executor not importable" : undefined }, async () => {
|
|
1241
|
+
const executor = makeExecutor();
|
|
1242
|
+
|
|
1243
|
+
const result = await executor.execute(
|
|
1244
|
+
"timeout-async-validation",
|
|
1245
|
+
{ agent: "echo", task: "Task", async: true, timeoutMs: 1_000 },
|
|
1246
|
+
new AbortController().signal,
|
|
1247
|
+
undefined,
|
|
1248
|
+
makeMinimalCtx(tempDir),
|
|
1249
|
+
);
|
|
1250
|
+
|
|
1251
|
+
assert.equal(result.isError, undefined);
|
|
1252
|
+
assert.match(result.content[0]?.text ?? "", /Async:/);
|
|
1253
|
+
assert.equal(result.details?.timeoutMs, 1_000);
|
|
1254
|
+
});
|
|
1255
|
+
|
|
1256
|
+
it("rejects file-only mode without an output path before spawning", async () => {
|
|
1257
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
1258
|
+
|
|
1259
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
1260
|
+
runId: "output-file-only-missing-path",
|
|
1261
|
+
outputMode: "file-only",
|
|
1262
|
+
});
|
|
1263
|
+
|
|
1264
|
+
assert.equal(result.exitCode, 1);
|
|
1265
|
+
assert.match(result.error ?? "", /outputMode: "file-only"/);
|
|
1266
|
+
assert.equal(mockPi.callCount(), 0);
|
|
1267
|
+
});
|
|
1268
|
+
|
|
1269
|
+
it("returns only a saved-output reference in file-only mode", async () => {
|
|
1270
|
+
const outputPath = path.join(tempDir, "file-only-report.md");
|
|
1271
|
+
const artifactsDir = path.join(tempDir, "file-only-artifacts");
|
|
1272
|
+
mockPi.onCall({ output: "full saved output\nwith details" });
|
|
1273
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
1274
|
+
|
|
1275
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
1276
|
+
runId: "output-file-only",
|
|
1277
|
+
outputPath,
|
|
1278
|
+
outputMode: "file-only",
|
|
1279
|
+
artifactsDir,
|
|
1280
|
+
});
|
|
1281
|
+
|
|
1282
|
+
assert.equal(result.exitCode, 0);
|
|
1283
|
+
assert.equal(result.outputMode, "file-only");
|
|
1284
|
+
assert.equal(result.savedOutputPath, outputPath);
|
|
1285
|
+
assert.equal(result.outputReference?.path, outputPath);
|
|
1286
|
+
assert.match(result.finalOutput ?? "", /^Output saved to:/);
|
|
1287
|
+
assert.match(result.finalOutput ?? "", /2 lines/);
|
|
1288
|
+
assert.doesNotMatch(result.finalOutput ?? "", /full saved output/);
|
|
1289
|
+
assert.equal(fs.readFileSync(outputPath, "utf-8"), "full saved output\nwith details");
|
|
1290
|
+
assert.ok(result.artifactPaths, "should have artifact paths");
|
|
1291
|
+
assert.equal(fs.readFileSync(result.artifactPaths.outputPath, "utf-8"), "full saved output\nwith details");
|
|
1292
|
+
});
|
|
1293
|
+
|
|
1294
|
+
it("passes maxSubagentDepth through to child execution env", async () => {
|
|
1295
|
+
mockPi.onCall({ echoEnv: ["SELESAI_SUBAGENT_DEPTH", "SELESAI_SUBAGENT_MAX_DEPTH"] });
|
|
1296
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
1297
|
+
const prevDepth = process.env.SELESAI_SUBAGENT_DEPTH;
|
|
1298
|
+
const prevMaxDepth = process.env.SELESAI_SUBAGENT_MAX_DEPTH;
|
|
1299
|
+
delete process.env.SELESAI_SUBAGENT_DEPTH;
|
|
1300
|
+
delete process.env.SELESAI_SUBAGENT_MAX_DEPTH;
|
|
1301
|
+
|
|
1302
|
+
try {
|
|
1303
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
1304
|
+
runId: "depth-env",
|
|
1305
|
+
maxSubagentDepth: 1,
|
|
1306
|
+
});
|
|
1307
|
+
|
|
1308
|
+
assert.equal(result.exitCode, 0);
|
|
1309
|
+
assert.deepEqual(JSON.parse(result.finalOutput ?? "{}"), {
|
|
1310
|
+
SELESAI_SUBAGENT_DEPTH: "1",
|
|
1311
|
+
SELESAI_SUBAGENT_MAX_DEPTH: "1",
|
|
1312
|
+
});
|
|
1313
|
+
} finally {
|
|
1314
|
+
if (prevDepth === undefined) delete process.env.SELESAI_SUBAGENT_DEPTH;
|
|
1315
|
+
else process.env.SELESAI_SUBAGENT_DEPTH = prevDepth;
|
|
1316
|
+
if (prevMaxDepth === undefined) delete process.env.SELESAI_SUBAGENT_MAX_DEPTH;
|
|
1317
|
+
else process.env.SELESAI_SUBAGENT_MAX_DEPTH = prevMaxDepth;
|
|
1318
|
+
}
|
|
1319
|
+
});
|
|
1320
|
+
|
|
1321
|
+
it("passes prompt inheritance env flags through to child execution", async () => {
|
|
1322
|
+
mockPi.onCall({ echoEnv: ["SELESAI_SUBAGENT_INHERIT_PROJECT_CONTEXT", "SELESAI_SUBAGENT_INHERIT_SKILLS"] });
|
|
1323
|
+
const agents = [makeAgent("echo", {
|
|
1324
|
+
systemPromptMode: "replace",
|
|
1325
|
+
inheritProjectContext: false,
|
|
1326
|
+
inheritSkills: false,
|
|
1327
|
+
})];
|
|
1328
|
+
|
|
1329
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
1330
|
+
runId: "prompt-inheritance-env",
|
|
1331
|
+
});
|
|
1332
|
+
|
|
1333
|
+
assert.equal(result.exitCode, 0);
|
|
1334
|
+
assert.deepEqual(JSON.parse(result.finalOutput ?? "{}"), {
|
|
1335
|
+
SELESAI_SUBAGENT_INHERIT_PROJECT_CONTEXT: "0",
|
|
1336
|
+
SELESAI_SUBAGENT_INHERIT_SKILLS: "0",
|
|
1337
|
+
});
|
|
1338
|
+
});
|
|
1339
|
+
|
|
1340
|
+
it("passes fanout routing env only when builtin subagent is declared", async () => {
|
|
1341
|
+
const envKeys = [
|
|
1342
|
+
SUBAGENT_FANOUT_CHILD_ENV,
|
|
1343
|
+
SUBAGENT_PARENT_EVENT_SINK_ENV,
|
|
1344
|
+
SUBAGENT_PARENT_CONTROL_INBOX_ENV,
|
|
1345
|
+
SUBAGENT_PARENT_RUN_ID_ENV,
|
|
1346
|
+
SUBAGENT_PARENT_CHILD_INDEX_ENV,
|
|
1347
|
+
];
|
|
1348
|
+
const saved = Object.fromEntries(envKeys.map((key) => [key, process.env[key]]));
|
|
1349
|
+
try {
|
|
1350
|
+
process.env[SUBAGENT_PARENT_EVENT_SINK_ENV] = "/tmp/inherited/events.jsonl";
|
|
1351
|
+
process.env[SUBAGENT_PARENT_CONTROL_INBOX_ENV] = "/tmp/inherited/control";
|
|
1352
|
+
process.env[SUBAGENT_PARENT_RUN_ID_ENV] = "inherited-run";
|
|
1353
|
+
process.env[SUBAGENT_PARENT_CHILD_INDEX_ENV] = "7";
|
|
1354
|
+
|
|
1355
|
+
mockPi.onCall({ echoEnv: envKeys });
|
|
1356
|
+
const fanoutAgents = [makeAgent("delegator", { tools: ["read", "subagent"] })];
|
|
1357
|
+
const fanout = await runSync(tempDir, fanoutAgents, "delegator", "Task", { runId: "fanout-run", index: 2 });
|
|
1358
|
+
assert.equal(fanout.exitCode, 0);
|
|
1359
|
+
assert.deepEqual(JSON.parse(fanout.finalOutput ?? "{}"), {
|
|
1360
|
+
SELESAI_SUBAGENT_FANOUT_CHILD: "1",
|
|
1361
|
+
SELESAI_SUBAGENT_PARENT_EVENT_SINK: "/tmp/inherited/events.jsonl",
|
|
1362
|
+
SELESAI_SUBAGENT_PARENT_CONTROL_INBOX: "/tmp/inherited/control",
|
|
1363
|
+
SELESAI_SUBAGENT_PARENT_RUN_ID: "fanout-run",
|
|
1364
|
+
SELESAI_SUBAGENT_PARENT_CHILD_INDEX: "2",
|
|
1365
|
+
});
|
|
1366
|
+
|
|
1367
|
+
mockPi.onCall({ echoEnv: envKeys });
|
|
1368
|
+
const nonFanoutAgents = [makeAgent("worker", { tools: ["read"] })];
|
|
1369
|
+
const nonFanout = await runSync(tempDir, nonFanoutAgents, "worker", "Task", { runId: "non-fanout-run" });
|
|
1370
|
+
assert.equal(nonFanout.exitCode, 0);
|
|
1371
|
+
assert.deepEqual(JSON.parse(nonFanout.finalOutput ?? "{}"), {
|
|
1372
|
+
SELESAI_SUBAGENT_FANOUT_CHILD: "0",
|
|
1373
|
+
SELESAI_SUBAGENT_PARENT_EVENT_SINK: "",
|
|
1374
|
+
SELESAI_SUBAGENT_PARENT_CONTROL_INBOX: "",
|
|
1375
|
+
SELESAI_SUBAGENT_PARENT_RUN_ID: "",
|
|
1376
|
+
SELESAI_SUBAGENT_PARENT_CHILD_INDEX: "",
|
|
1377
|
+
});
|
|
1378
|
+
} finally {
|
|
1379
|
+
for (const key of envKeys) {
|
|
1380
|
+
if (saved[key] === undefined) delete process.env[key];
|
|
1381
|
+
else process.env[key] = saved[key];
|
|
1382
|
+
}
|
|
1383
|
+
}
|
|
1384
|
+
});
|
|
1385
|
+
|
|
1386
|
+
it("passes supervisor metadata through to child execution", async () => {
|
|
1387
|
+
mockPi.onCall({ echoEnv: [
|
|
1388
|
+
"SELESAI_SUBAGENT_INTERCOM_SESSION_NAME",
|
|
1389
|
+
"SELESAI_SUBAGENT_ORCHESTRATOR_TARGET",
|
|
1390
|
+
"SELESAI_SUBAGENT_RUN_ID",
|
|
1391
|
+
"SELESAI_SUBAGENT_CHILD_AGENT",
|
|
1392
|
+
"SELESAI_SUBAGENT_CHILD_INDEX",
|
|
1393
|
+
] });
|
|
1394
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
1395
|
+
|
|
1396
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
1397
|
+
runId: "78f659a3",
|
|
1398
|
+
index: 2,
|
|
1399
|
+
intercomSessionName: "subagent-echo-78f659a3-3",
|
|
1400
|
+
orchestratorIntercomTarget: "subagent-chat-parent",
|
|
1401
|
+
});
|
|
1402
|
+
|
|
1403
|
+
assert.equal(result.exitCode, 0);
|
|
1404
|
+
assert.deepEqual(JSON.parse(result.finalOutput ?? "{}"), {
|
|
1405
|
+
SELESAI_SUBAGENT_INTERCOM_SESSION_NAME: "subagent-echo-78f659a3-3",
|
|
1406
|
+
SELESAI_SUBAGENT_ORCHESTRATOR_TARGET: "subagent-chat-parent",
|
|
1407
|
+
SELESAI_SUBAGENT_RUN_ID: "78f659a3",
|
|
1408
|
+
SELESAI_SUBAGENT_CHILD_AGENT: "echo",
|
|
1409
|
+
SELESAI_SUBAGENT_CHILD_INDEX: "2",
|
|
1410
|
+
});
|
|
1411
|
+
});
|
|
1412
|
+
|
|
1413
|
+
it("passes custom tool extensions through even when explicit extensions are allowlisted", { skip: process.platform === "win32" ? "extension path resolution intermittent on Windows CI" : undefined }, async () => {
|
|
1414
|
+
mockPi.onCall({ output: "Done" });
|
|
1415
|
+
const agents = [makeAgent("echo", {
|
|
1416
|
+
tools: ["read", "./custom-tool.ts"],
|
|
1417
|
+
extensions: ["./allowed-ext.ts"],
|
|
1418
|
+
})];
|
|
1419
|
+
|
|
1420
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
1421
|
+
runId: "tool-extension-allowlist",
|
|
1422
|
+
});
|
|
1423
|
+
|
|
1424
|
+
assert.equal(result.exitCode, 0);
|
|
1425
|
+
const args = readCallArgs();
|
|
1426
|
+
const extensionArgs = args.filter((arg, index) => args[index - 1] === "--extension");
|
|
1427
|
+
assert.ok(extensionArgs.some((arg) => arg.endsWith(path.join("src", "runs", "shared", "subagent-prompt-runtime.ts"))));
|
|
1428
|
+
assert.ok(extensionArgs.some((arg) => arg.replace(/\\/g, "/").endsWith("custom-tool.ts")));
|
|
1429
|
+
assert.ok(extensionArgs.some((arg) => arg.replace(/\\/g, "/").endsWith("allowed-ext.ts")));
|
|
1430
|
+
});
|
|
1431
|
+
|
|
1432
|
+
it("passes subagent-only extensions through to child execution", { skip: process.platform === "win32" ? "extension path resolution intermittent on Windows CI" : undefined }, async () => {
|
|
1433
|
+
mockPi.onCall({ output: "Done" });
|
|
1434
|
+
const agents = [makeAgent("echo", {
|
|
1435
|
+
tools: ["read"],
|
|
1436
|
+
subagentOnlyExtensions: ["./child-only-tool.ts"],
|
|
1437
|
+
})];
|
|
1438
|
+
|
|
1439
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
1440
|
+
runId: "subagent-only-extension",
|
|
1441
|
+
});
|
|
1442
|
+
|
|
1443
|
+
assert.equal(result.exitCode, 0);
|
|
1444
|
+
const args = readCallArgs();
|
|
1445
|
+
const extensionArgs = args.filter((arg, index) => args[index - 1] === "--extension");
|
|
1446
|
+
assert.ok(extensionArgs.some((arg) => arg.endsWith(path.join("src", "runs", "shared", "subagent-prompt-runtime.ts"))));
|
|
1447
|
+
assert.ok(extensionArgs.some((arg) => arg.replace(/\\/g, "/").endsWith("child-only-tool.ts")));
|
|
1448
|
+
});
|
|
1449
|
+
|
|
1450
|
+
it("treats forced drain after final assistant output as cleanup success", async () => {
|
|
1451
|
+
mockPi.onCall({
|
|
1452
|
+
jsonl: [events.assistantMessage("done-before-drain")],
|
|
1453
|
+
stderr: "Done after 1 turn(s). Ready for input.\n",
|
|
1454
|
+
keepAliveAfterFinalMessageMs: 10000,
|
|
1455
|
+
});
|
|
1456
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
1457
|
+
|
|
1458
|
+
const start = Date.now();
|
|
1459
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {});
|
|
1460
|
+
const elapsed = Date.now() - start;
|
|
1461
|
+
|
|
1462
|
+
assert.ok(elapsed < 4000, `should clean up shortly after terminal stop, took ${elapsed}ms`);
|
|
1463
|
+
assert.equal(result.exitCode, 0);
|
|
1464
|
+
assert.equal(result.error, undefined);
|
|
1465
|
+
assert.equal(result.finalOutput, "done-before-drain");
|
|
1466
|
+
assert.ok(!(result.progress?.recentOutput ?? []).some((line) => line.includes("Forcing termination")));
|
|
1467
|
+
});
|
|
1468
|
+
|
|
1469
|
+
it("treats forced drain after empty terminal assistant output as cleanup success", async () => {
|
|
1470
|
+
mockPi.onCall({
|
|
1471
|
+
jsonl: [{
|
|
1472
|
+
type: "message_end",
|
|
1473
|
+
message: {
|
|
1474
|
+
role: "assistant",
|
|
1475
|
+
content: [{ type: "text", text: "" }],
|
|
1476
|
+
model: "mock/test-model",
|
|
1477
|
+
stopReason: "stop",
|
|
1478
|
+
usage: { input: 100, output: 0, cacheRead: 0, cacheWrite: 0, cost: { total: 0.001 } },
|
|
1479
|
+
},
|
|
1480
|
+
}],
|
|
1481
|
+
keepAliveAfterFinalMessageMs: 10000,
|
|
1482
|
+
});
|
|
1483
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
1484
|
+
|
|
1485
|
+
const start = Date.now();
|
|
1486
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {});
|
|
1487
|
+
const elapsed = Date.now() - start;
|
|
1488
|
+
|
|
1489
|
+
assert.ok(elapsed < 4000, `should clean up shortly after empty terminal stop, took ${elapsed}ms`);
|
|
1490
|
+
assert.equal(result.exitCode, 0);
|
|
1491
|
+
assert.equal(result.error, undefined);
|
|
1492
|
+
assert.equal(result.finalOutput, "");
|
|
1493
|
+
assert.equal(result.progress.status, "completed");
|
|
1494
|
+
assert.ok(!(result.progress?.recentOutput ?? []).some((line) => line.includes("Forcing termination")));
|
|
1495
|
+
});
|
|
1496
|
+
|
|
1497
|
+
it("keeps explicit assistant errors as failures during final-drain cleanup", async () => {
|
|
1498
|
+
mockPi.onCall({
|
|
1499
|
+
jsonl: [{
|
|
1500
|
+
type: "message_end",
|
|
1501
|
+
message: {
|
|
1502
|
+
role: "assistant",
|
|
1503
|
+
content: [{ type: "text", text: "failed" }],
|
|
1504
|
+
model: "mock/test-model",
|
|
1505
|
+
stopReason: "stop",
|
|
1506
|
+
errorMessage: "provider exploded",
|
|
1507
|
+
usage: { input: 100, output: 0, cacheRead: 0, cacheWrite: 0, cost: { total: 0.001 } },
|
|
1508
|
+
},
|
|
1509
|
+
}],
|
|
1510
|
+
keepAliveAfterFinalMessageMs: 10000,
|
|
1511
|
+
});
|
|
1512
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
1513
|
+
|
|
1514
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {});
|
|
1515
|
+
|
|
1516
|
+
assert.equal(result.exitCode, 1);
|
|
1517
|
+
assert.equal(result.error, "provider exploded");
|
|
1518
|
+
assert.equal(result.progress.status, "failed");
|
|
1519
|
+
});
|
|
1520
|
+
|
|
1521
|
+
it("handles abort signal (completes faster than delay)", async () => {
|
|
1522
|
+
mockPi.onCall({ delay: 10000 }); // Long delay — process should be killed before this
|
|
1523
|
+
const agents = makeAgentConfigs(["slow"]);
|
|
1524
|
+
const controller = new AbortController();
|
|
1525
|
+
|
|
1526
|
+
const start = Date.now();
|
|
1527
|
+
setTimeout(() => controller.abort(), 200);
|
|
1528
|
+
|
|
1529
|
+
const result = await runSync(tempDir, agents, "slow", "Slow task", {
|
|
1530
|
+
signal: controller.signal,
|
|
1531
|
+
});
|
|
1532
|
+
const elapsed = Date.now() - start;
|
|
1533
|
+
|
|
1534
|
+
// The key assertion: the run should complete much faster than the 10s delay,
|
|
1535
|
+
// proving the abort signal terminated the process early.
|
|
1536
|
+
assert.ok(elapsed < 5000, `should abort early, took ${elapsed}ms`);
|
|
1537
|
+
// Exit code is platform-dependent (Windows: often 1 or 0, Linux: null/143)
|
|
1538
|
+
});
|
|
1539
|
+
|
|
1540
|
+
it("marks foreground runs that exceed timeoutMs as timed out", async () => {
|
|
1541
|
+
mockPi.onCall({ delay: 10000 });
|
|
1542
|
+
const agents = makeAgentConfigs(["slow"]);
|
|
1543
|
+
|
|
1544
|
+
const start = Date.now();
|
|
1545
|
+
const result = await runSync(tempDir, agents, "slow", "Slow task", {
|
|
1546
|
+
timeoutMs: 150,
|
|
1547
|
+
});
|
|
1548
|
+
const elapsed = Date.now() - start;
|
|
1549
|
+
|
|
1550
|
+
assert.ok(elapsed < 5000, `should time out early, took ${elapsed}ms`);
|
|
1551
|
+
assert.notEqual(result.exitCode, 0);
|
|
1552
|
+
assert.equal(result.timedOut, true);
|
|
1553
|
+
assert.equal(result.error, "Subagent timed out after 150ms.");
|
|
1554
|
+
assert.match(result.finalOutput ?? "", /Subagent timed out after 150ms\./);
|
|
1555
|
+
assert.equal(result.progress.status, "failed");
|
|
1556
|
+
});
|
|
1557
|
+
|
|
1558
|
+
it("allows a foreground run to finish on the final turn-budget grace turn", async () => {
|
|
1559
|
+
mockPi.onCall({
|
|
1560
|
+
jsonl: [
|
|
1561
|
+
mockAssistantMessage("working before wrap-up", "tool_use"),
|
|
1562
|
+
mockAssistantMessage("final wrapped output", "stop"),
|
|
1563
|
+
],
|
|
1564
|
+
});
|
|
1565
|
+
const agents = makeAgentConfigs(["worker"]);
|
|
1566
|
+
|
|
1567
|
+
const result = await runSync(tempDir, agents, "worker", "Use the final grace turn to wrap up.", {
|
|
1568
|
+
turnBudget: { maxTurns: 1, graceTurns: 1 },
|
|
1569
|
+
runId: "foreground-turn-budget-soft",
|
|
1570
|
+
});
|
|
1571
|
+
|
|
1572
|
+
assert.equal(result.exitCode, 0);
|
|
1573
|
+
assert.equal(result.turnBudgetExceeded, undefined);
|
|
1574
|
+
assert.equal(result.wrapUpRequested, true);
|
|
1575
|
+
assert.equal(result.turnBudget?.outcome, "wrap-up-requested");
|
|
1576
|
+
assert.equal(result.turnBudget?.turnCount, 2);
|
|
1577
|
+
assert.match(result.finalOutput ?? "", /Turn budget wrap-up was requested after 1 assistant turn/);
|
|
1578
|
+
assert.match(result.finalOutput ?? "", /final wrapped output/);
|
|
1579
|
+
});
|
|
1580
|
+
|
|
1581
|
+
it("does not run acceptance verification after a foreground timeout", async () => {
|
|
1582
|
+
const markerPath = path.join(tempDir, "verify-ran.txt");
|
|
1583
|
+
const report = [
|
|
1584
|
+
"done",
|
|
1585
|
+
"```acceptance-report",
|
|
1586
|
+
JSON.stringify({
|
|
1587
|
+
criteriaSatisfied: [{ id: "criterion-1", status: "satisfied", evidence: "integration test evidence" }],
|
|
1588
|
+
changedFiles: ["src/a.ts"],
|
|
1589
|
+
testsAddedOrUpdated: ["test/a.test.ts"],
|
|
1590
|
+
commandsRun: [{ command: "npm test", result: "passed", summary: "passed" }],
|
|
1591
|
+
validationOutput: ["validation passed"],
|
|
1592
|
+
residualRisks: [],
|
|
1593
|
+
noStagedFiles: true,
|
|
1594
|
+
notes: "complete",
|
|
1595
|
+
}),
|
|
1596
|
+
"```",
|
|
1597
|
+
].join("\n");
|
|
1598
|
+
mockPi.onCall({ jsonl: [events.assistantMessage(report)], keepAliveAfterFinalMessageMs: 10000 });
|
|
1599
|
+
const agents = makeAgentConfigs(["slow"]);
|
|
1600
|
+
|
|
1601
|
+
const result = await runSync(tempDir, agents, "slow", "Slow task", {
|
|
1602
|
+
timeoutMs: 150,
|
|
1603
|
+
acceptance: {
|
|
1604
|
+
level: "verified",
|
|
1605
|
+
verify: [{
|
|
1606
|
+
id: "marker",
|
|
1607
|
+
command: "node -e \"require('node:fs').writeFileSync(process.env.VERIFY_MARKER, 'ran')\"",
|
|
1608
|
+
env: { VERIFY_MARKER: markerPath },
|
|
1609
|
+
timeoutMs: 10_000,
|
|
1610
|
+
}],
|
|
1611
|
+
},
|
|
1612
|
+
});
|
|
1613
|
+
|
|
1614
|
+
assert.equal(result.timedOut, true);
|
|
1615
|
+
assert.equal(result.acceptance?.status, "rejected");
|
|
1616
|
+
assert.equal(result.acceptance?.runtimeChecks?.[0]?.id, "timeout");
|
|
1617
|
+
assert.equal(result.acceptance?.verifyRuns?.length, 0);
|
|
1618
|
+
assert.equal(fs.existsSync(markerPath), false);
|
|
1619
|
+
});
|
|
1620
|
+
|
|
1621
|
+
it("soft-interrupts the current turn and returns a paused result", async () => {
|
|
1622
|
+
mockPi.onCall({ delay: 10000 });
|
|
1623
|
+
const agents = makeAgentConfigs(["slow"]);
|
|
1624
|
+
const controller = new AbortController();
|
|
1625
|
+
const controlEvents: Array<{ type?: string; to?: string }> = [];
|
|
1626
|
+
|
|
1627
|
+
const start = Date.now();
|
|
1628
|
+
setTimeout(() => controller.abort(), 200);
|
|
1629
|
+
|
|
1630
|
+
const result = await runSync(tempDir, agents, "slow", "Slow task", {
|
|
1631
|
+
runId: "interrupt-run",
|
|
1632
|
+
interruptSignal: controller.signal,
|
|
1633
|
+
onControlEvent: (event: { type?: string; to?: string }) => {
|
|
1634
|
+
controlEvents.push(event);
|
|
1635
|
+
},
|
|
1636
|
+
});
|
|
1637
|
+
const elapsed = Date.now() - start;
|
|
1638
|
+
|
|
1639
|
+
assert.ok(elapsed < 5000, `should interrupt early, took ${elapsed}ms`);
|
|
1640
|
+
assert.equal(result.exitCode, 0);
|
|
1641
|
+
assert.equal(result.interrupted, true);
|
|
1642
|
+
assert.equal(result.progress.activityState, undefined);
|
|
1643
|
+
assert.deepEqual(controlEvents, []);
|
|
1644
|
+
assert.match(result.finalOutput ?? "", /Interrupted/);
|
|
1645
|
+
});
|
|
1646
|
+
|
|
1647
|
+
it("preserves manual interrupt semantics when a timeout is also configured", async () => {
|
|
1648
|
+
mockPi.onCall({ delay: 10000 });
|
|
1649
|
+
const agents = makeAgentConfigs(["slow"]);
|
|
1650
|
+
const controller = new AbortController();
|
|
1651
|
+
|
|
1652
|
+
setTimeout(() => controller.abort(), 100);
|
|
1653
|
+
const result = await runSync(tempDir, agents, "slow", "Slow task", {
|
|
1654
|
+
interruptSignal: controller.signal,
|
|
1655
|
+
timeoutMs: 500,
|
|
1656
|
+
});
|
|
1657
|
+
|
|
1658
|
+
assert.equal(result.exitCode, 0);
|
|
1659
|
+
assert.equal(result.interrupted, true);
|
|
1660
|
+
assert.equal(result.timedOut, undefined);
|
|
1661
|
+
assert.equal(result.error, undefined);
|
|
1662
|
+
assert.match(result.finalOutput ?? "", /Interrupted/);
|
|
1663
|
+
});
|
|
1664
|
+
|
|
1665
|
+
for (const toolName of ["intercom", "contact_supervisor"]) {
|
|
1666
|
+
it(`detaches cleanly on ${toolName} handoff without aborting the child process`, async () => {
|
|
1667
|
+
const eventBus = createEventBus();
|
|
1668
|
+
let accepted = false;
|
|
1669
|
+
eventBus.on(INTERCOM_DETACH_RESPONSE_EVENT, (payload) => {
|
|
1670
|
+
if (!payload || typeof payload !== "object") return;
|
|
1671
|
+
accepted = (payload as { accepted?: unknown }).accepted === true;
|
|
1672
|
+
});
|
|
1673
|
+
mockPi.onCall({
|
|
1674
|
+
steps: [
|
|
1675
|
+
{ jsonl: [events.toolStart(toolName, toolName === "intercom" ? { action: "ask", to: "orchestrator" } : { reason: "need_decision", message: "Need a decision" })] },
|
|
1676
|
+
{ delay: 1000, jsonl: [events.assistantMessage("received pong")] },
|
|
1677
|
+
],
|
|
1678
|
+
});
|
|
1679
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
1680
|
+
|
|
1681
|
+
// Emit the detach request the moment we observe the coordination tool start
|
|
1682
|
+
// in a progress update — this is the signal the parent has set
|
|
1683
|
+
// `intercomStarted=true`. Using a fixed delay here races the mock's
|
|
1684
|
+
// cold spawn and flakes under load.
|
|
1685
|
+
let detachEmitted = false;
|
|
1686
|
+
const runPromise = runSync(tempDir, agents, "echo", "Task", {
|
|
1687
|
+
runId: `${toolName}-detach`,
|
|
1688
|
+
allowIntercomDetach: true,
|
|
1689
|
+
intercomEvents: eventBus,
|
|
1690
|
+
onUpdate: (update) => {
|
|
1691
|
+
if (detachEmitted) return;
|
|
1692
|
+
const progress = (update as { details?: { progress?: Array<{ currentTool?: string }> } }).details?.progress;
|
|
1693
|
+
const sawCoordinationTool = Array.isArray(progress) && progress.some((p) => p?.currentTool === toolName);
|
|
1694
|
+
if (!sawCoordinationTool) return;
|
|
1695
|
+
detachEmitted = true;
|
|
1696
|
+
eventBus.emit(INTERCOM_DETACH_REQUEST_EVENT, { requestId: "test-request" });
|
|
1697
|
+
},
|
|
1698
|
+
});
|
|
1699
|
+
|
|
1700
|
+
const result = await runPromise;
|
|
1701
|
+
|
|
1702
|
+
assert.equal(result.exitCode, 0);
|
|
1703
|
+
assert.equal(result.detached, true);
|
|
1704
|
+
assert.equal(result.detachedReason, "intercom coordination");
|
|
1705
|
+
assert.equal(result.finalOutput, "Detached for intercom coordination.");
|
|
1706
|
+
assert.equal(result.progress?.status, "detached");
|
|
1707
|
+
assert.equal(accepted, true);
|
|
1708
|
+
});
|
|
1709
|
+
}
|
|
1710
|
+
|
|
1711
|
+
for (const testCase of [
|
|
1712
|
+
{ name: "intercom ask", toolName: "intercom", args: { action: "ask", to: "orchestrator" } },
|
|
1713
|
+
{ name: "contact_supervisor need_decision", toolName: "contact_supervisor", args: { reason: "need_decision", message: "Need a decision" } },
|
|
1714
|
+
{ name: "contact_supervisor interview_request", toolName: "contact_supervisor", args: { reason: "interview_request", message: "Need input", interview: { questions: [] } } },
|
|
1715
|
+
]) {
|
|
1716
|
+
it(`proactively detaches foreground children on blocking ${testCase.name}`, async () => {
|
|
1717
|
+
mockPi.onCall({
|
|
1718
|
+
steps: [
|
|
1719
|
+
{ jsonl: [events.toolStart(testCase.toolName, testCase.args)] },
|
|
1720
|
+
{ delay: 1000, jsonl: [events.assistantMessage("received pong")] },
|
|
1721
|
+
],
|
|
1722
|
+
});
|
|
1723
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
1724
|
+
|
|
1725
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
1726
|
+
runId: `${testCase.toolName}-blocking-detach`,
|
|
1727
|
+
allowIntercomDetach: true,
|
|
1728
|
+
});
|
|
1729
|
+
|
|
1730
|
+
assert.equal(result.exitCode, 0);
|
|
1731
|
+
assert.equal(result.detached, true);
|
|
1732
|
+
assert.equal(result.detachedReason, "intercom coordination");
|
|
1733
|
+
assert.equal(result.finalOutput, "Detached for intercom coordination.");
|
|
1734
|
+
assert.equal(result.progress?.status, "detached");
|
|
1735
|
+
});
|
|
1736
|
+
}
|
|
1737
|
+
|
|
1738
|
+
for (const testCase of [
|
|
1739
|
+
{ name: "intercom send", toolName: "intercom", args: { action: "send", to: "orchestrator", message: "FYI" } },
|
|
1740
|
+
{ name: "contact_supervisor progress_update", toolName: "contact_supervisor", args: { reason: "progress_update", message: "FYI" } },
|
|
1741
|
+
]) {
|
|
1742
|
+
it(`does not proactively detach foreground children on non-blocking ${testCase.name}`, async () => {
|
|
1743
|
+
mockPi.onCall({
|
|
1744
|
+
steps: [
|
|
1745
|
+
{ jsonl: [events.toolStart(testCase.toolName, testCase.args)] },
|
|
1746
|
+
{ jsonl: [events.toolEnd(testCase.toolName)] },
|
|
1747
|
+
{ jsonl: [events.assistantMessage("done")] },
|
|
1748
|
+
],
|
|
1749
|
+
});
|
|
1750
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
1751
|
+
|
|
1752
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {
|
|
1753
|
+
runId: `${testCase.toolName}-nonblocking`,
|
|
1754
|
+
allowIntercomDetach: true,
|
|
1755
|
+
});
|
|
1756
|
+
|
|
1757
|
+
assert.equal(result.exitCode, 0);
|
|
1758
|
+
assert.equal(result.detached, undefined);
|
|
1759
|
+
assert.equal(result.finalOutput, "done");
|
|
1760
|
+
assert.equal(result.progress?.status, "completed");
|
|
1761
|
+
});
|
|
1762
|
+
}
|
|
1763
|
+
|
|
1764
|
+
it("lets an active intercom child accept detach when another child is listening", async () => {
|
|
1765
|
+
const eventBus = createEventBus();
|
|
1766
|
+
let firstDetachResponse: boolean | undefined;
|
|
1767
|
+
eventBus.on(INTERCOM_DETACH_RESPONSE_EVENT, (payload) => {
|
|
1768
|
+
if (!payload || typeof payload !== "object") return;
|
|
1769
|
+
if ((payload as { requestId?: unknown }).requestId !== "parallel-request") return;
|
|
1770
|
+
firstDetachResponse ??= (payload as { accepted?: unknown }).accepted === true;
|
|
1771
|
+
});
|
|
1772
|
+
mockPi.onCall({ delay: 500, output: "quiet child done" });
|
|
1773
|
+
const agents = makeAgentConfigs(["quiet", "intercom"]);
|
|
1774
|
+
|
|
1775
|
+
const quietRun = runSync(tempDir, agents, "quiet", "Quiet task", {
|
|
1776
|
+
runId: "quiet-listener",
|
|
1777
|
+
allowIntercomDetach: true,
|
|
1778
|
+
intercomEvents: eventBus,
|
|
1779
|
+
});
|
|
1780
|
+
for (let attempt = 0; attempt < 50 && mockPi.callCount() < 1; attempt++) {
|
|
1781
|
+
await new Promise((resolve) => setTimeout(resolve, 10));
|
|
1782
|
+
}
|
|
1783
|
+
assert.equal(mockPi.callCount(), 1);
|
|
1784
|
+
mockPi.onCall({
|
|
1785
|
+
steps: [
|
|
1786
|
+
{ jsonl: [events.toolStart("intercom", { action: "send", to: "orchestrator" })] },
|
|
1787
|
+
{ delay: 500, jsonl: [events.assistantMessage("after intercom")] },
|
|
1788
|
+
],
|
|
1789
|
+
});
|
|
1790
|
+
|
|
1791
|
+
let detachEmitted = false;
|
|
1792
|
+
const intercomRun = runSync(tempDir, agents, "intercom", "Intercom task", {
|
|
1793
|
+
runId: "active-intercom",
|
|
1794
|
+
allowIntercomDetach: true,
|
|
1795
|
+
intercomEvents: eventBus,
|
|
1796
|
+
onUpdate: (update) => {
|
|
1797
|
+
if (detachEmitted) return;
|
|
1798
|
+
const progress = (update as { details?: { progress?: Array<{ currentTool?: string }> } }).details?.progress;
|
|
1799
|
+
const sawIntercom = Array.isArray(progress) && progress.some((p) => p?.currentTool === "intercom");
|
|
1800
|
+
if (!sawIntercom) return;
|
|
1801
|
+
detachEmitted = true;
|
|
1802
|
+
eventBus.emit(INTERCOM_DETACH_REQUEST_EVENT, { requestId: "parallel-request" });
|
|
1803
|
+
},
|
|
1804
|
+
});
|
|
1805
|
+
|
|
1806
|
+
const [quietResult, intercomResult] = await Promise.all([quietRun, intercomRun]);
|
|
1807
|
+
|
|
1808
|
+
assert.equal(quietResult.exitCode, 0);
|
|
1809
|
+
assert.equal(quietResult.detached, undefined);
|
|
1810
|
+
assert.equal(intercomResult.exitCode, 0);
|
|
1811
|
+
assert.equal(intercomResult.detached, true);
|
|
1812
|
+
assert.equal(firstDetachResponse, true);
|
|
1813
|
+
});
|
|
1814
|
+
|
|
1815
|
+
it("handles stderr without exit code as info (not error)", async () => {
|
|
1816
|
+
mockPi.onCall({ output: "Success", stderr: "Warning: something", exitCode: 0 });
|
|
1817
|
+
const agents = makeAgentConfigs(["echo"]);
|
|
1818
|
+
|
|
1819
|
+
const result = await runSync(tempDir, agents, "echo", "Task", {});
|
|
1820
|
+
|
|
1821
|
+
assert.equal(result.exitCode, 0);
|
|
1822
|
+
});
|
|
1823
|
+
|
|
1824
|
+
});
|