@selesai/code 0.8.5 → 0.8.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +28 -0
- package/dist/core/agent-session.js +13 -1
- package/dist/core/extensions/types.d.ts +2 -0
- package/dist/core/provider-composer.js +17 -4
- package/dist/core/settings-manager.d.ts +2 -0
- package/dist/core/settings-manager.js +3 -0
- package/dist/core/system-prompt.js +2 -8
- package/dist/core/system-prompt.test.js +6 -0
- package/dist/core/tools/index.d.ts +1 -0
- package/dist/core/tools/index.js +1 -0
- package/dist/core/tools/schema-prune.d.ts +16 -0
- package/dist/core/tools/schema-prune.js +39 -0
- package/dist/core/tools/schema-prune.test.d.ts +1 -0
- package/dist/core/tools/schema-prune.test.js +111 -0
- package/dist/defaults/models.json +5 -0
- package/dist/defaults/settings.json +1 -0
- package/dist/extensions/agent-browser.test.ts +178 -2
- package/dist/extensions/agent-browser.ts +2 -0
- package/dist/extensions/copy-turn.test.ts +229 -0
- package/dist/extensions/copy-turn.ts +11 -0
- package/dist/extensions/grep-app/index.test.ts +356 -1
- package/dist/extensions/grep-app/index.ts +6 -1
- package/dist/extensions/handoff-new.test.ts +351 -161
- package/dist/extensions/handoff-new.ts +3 -0
- package/dist/extensions/inline-skills.test.ts +93 -0
- package/dist/extensions/inline-skills.ts +3 -0
- package/dist/extensions/model-prompt-injector/config.json +10 -0
- package/dist/extensions/model-prompt-injector/index.test.ts +112 -0
- package/dist/extensions/model-prompt-injector/index.ts +151 -0
- package/dist/extensions/package.json +1 -0
- package/dist/extensions/pi-subagents/CHANGELOG.md +158 -90
- package/dist/extensions/pi-subagents/UPSTREAM-V0.50-MAPPING.md +172 -0
- package/dist/extensions/pi-subagents/agents/builder.md +5 -2
- package/dist/extensions/pi-subagents/agents/commentator.md +1 -1
- package/dist/extensions/pi-subagents/agents/oracle.md +7 -5
- package/dist/extensions/pi-subagents/agents/reviewer.md +2 -2
- package/dist/extensions/pi-subagents/agents/scout.md +1 -1
- package/dist/extensions/pi-subagents/agents/worker.md +1 -1
- package/dist/extensions/pi-subagents/docs/configuration.md +78 -11
- package/dist/extensions/pi-subagents/docs/extension-api.md +36 -0
- package/dist/extensions/pi-subagents/docs/missions.md +5 -3
- package/dist/extensions/pi-subagents/docs/models.md +1 -1
- package/dist/extensions/pi-subagents/docs/observability.md +42 -2
- package/dist/extensions/pi-subagents/docs/tool-reference.md +20 -3
- package/dist/extensions/pi-subagents/docs/workflows.md +4 -4
- package/dist/extensions/pi-subagents/install.mjs +2 -2
- package/dist/extensions/pi-subagents/package-lock.json +1470 -1472
- package/dist/extensions/pi-subagents/package.json +1 -1
- package/dist/extensions/pi-subagents/skills/pi-subagents/SKILL.md +1 -1
- package/dist/extensions/pi-subagents/skills/pi-subagents/references/constraints-and-recipes.md +5 -4
- package/dist/extensions/pi-subagents/skills/pi-subagents/references/execution-controls.md +16 -15
- package/dist/extensions/pi-subagents/skills/pi-subagents/references/management-authoring-rpc.md +5 -5
- package/dist/extensions/pi-subagents/skills/pi-subagents/references/prompting-and-roles.md +46 -26
- package/dist/extensions/pi-subagents/src/agents/agent-serializer.ts +2 -0
- package/dist/extensions/pi-subagents/src/agents/agents.ts +37 -12
- package/dist/extensions/pi-subagents/src/api/external-runs.ts +174 -84
- package/dist/extensions/pi-subagents/src/api/preflight.ts +13 -7
- package/dist/extensions/pi-subagents/src/extension/config.ts +59 -0
- package/dist/extensions/pi-subagents/src/extension/index.ts +146 -23
- package/dist/extensions/pi-subagents/src/extension/public-execution.ts +34 -6
- package/dist/extensions/pi-subagents/src/extension/rpc.ts +5 -1
- package/dist/extensions/pi-subagents/src/extension/schemas.ts +31 -30
- package/dist/extensions/pi-subagents/src/extension/tool-description.ts +10 -9
- package/dist/extensions/pi-subagents/src/inspectors/herdr/actions.ts +13 -8
- package/dist/extensions/pi-subagents/src/inspectors/herdr/inspector-runner.ts +16 -3
- package/dist/extensions/pi-subagents/src/inspectors/herdr/project-panes.ts +2 -6
- package/dist/extensions/pi-subagents/src/inspectors/herdr/shell-command.ts +16 -0
- package/dist/extensions/pi-subagents/src/intercom/intercom-bridge.ts +5 -4
- package/dist/extensions/pi-subagents/src/intercom/native-supervisor-channel.ts +19 -42
- package/dist/extensions/pi-subagents/src/missions/goal-driver.ts +3 -1
- package/dist/extensions/pi-subagents/src/missions/store.ts +8 -3
- package/dist/extensions/pi-subagents/src/runs/background/active-async-capacity.ts +82 -25
- package/dist/extensions/pi-subagents/src/runs/background/active-run-index.ts +71 -1
- package/dist/extensions/pi-subagents/src/runs/background/async-execution.ts +73 -43
- package/dist/extensions/pi-subagents/src/runs/background/async-job-tracker.ts +17 -6
- package/dist/extensions/pi-subagents/src/runs/background/async-resume.ts +14 -6
- package/dist/extensions/pi-subagents/src/runs/background/async-status-snapshot.ts +277 -0
- package/dist/extensions/pi-subagents/src/runs/background/async-status.ts +11 -3
- package/dist/extensions/pi-subagents/src/runs/background/chain-root-attachment.ts +2 -2
- package/dist/extensions/pi-subagents/src/runs/background/completion-replay.ts +11 -1
- package/dist/extensions/pi-subagents/src/runs/background/fleet-view.ts +21 -6
- package/dist/extensions/pi-subagents/src/runs/background/notify.ts +5 -2
- package/dist/extensions/pi-subagents/src/runs/background/result-files.ts +437 -0
- package/dist/extensions/pi-subagents/src/runs/background/result-watcher.ts +224 -42
- package/dist/extensions/pi-subagents/src/runs/background/resume-guidance.ts +27 -7
- package/dist/extensions/pi-subagents/src/runs/background/retained-children.ts +76 -19
- package/dist/extensions/pi-subagents/src/runs/background/run-id-resolver.ts +35 -26
- package/dist/extensions/pi-subagents/src/runs/background/run-status.ts +108 -9
- package/dist/extensions/pi-subagents/src/runs/background/scheduled-runs.ts +54 -28
- package/dist/extensions/pi-subagents/src/runs/background/stale-run-reconciler.ts +27 -13
- package/dist/extensions/pi-subagents/src/runs/background/subagent-runner.ts +298 -33
- package/dist/extensions/pi-subagents/src/runs/background/subagent-wait.ts +2 -0
- package/dist/extensions/pi-subagents/src/runs/background/wait-completions.ts +5 -2
- package/dist/extensions/pi-subagents/src/runs/background/wait-subscriptions.ts +2 -1
- package/dist/extensions/pi-subagents/src/runs/foreground/async-dismiss-action.ts +2 -1
- package/dist/extensions/pi-subagents/src/runs/foreground/chain-execution.ts +16 -0
- package/dist/extensions/pi-subagents/src/runs/foreground/execution.ts +219 -15
- package/dist/extensions/pi-subagents/src/runs/foreground/foreground-history.ts +7 -4
- package/dist/extensions/pi-subagents/src/runs/foreground/prompt-audit.ts +4 -3
- package/dist/extensions/pi-subagents/src/runs/foreground/subagent-executor.ts +368 -50
- package/dist/extensions/pi-subagents/src/runs/shared/completion-guard.ts +107 -1
- package/dist/extensions/pi-subagents/src/runs/shared/external-cli-runner.ts +4 -0
- package/dist/extensions/pi-subagents/src/runs/shared/llm-intent-arbiter.ts +39 -23
- package/dist/extensions/pi-subagents/src/runs/shared/model-fallback.ts +16 -2
- package/dist/extensions/pi-subagents/src/runs/shared/nested-events.ts +66 -62
- package/dist/extensions/pi-subagents/src/runs/shared/orca-progress-tabs.ts +376 -0
- package/dist/extensions/pi-subagents/src/runs/shared/parallel-utils.ts +2 -0
- package/dist/extensions/pi-subagents/src/runs/shared/subagent-control.ts +15 -0
- package/dist/extensions/pi-subagents/src/runs/shared/subagent-prompt-runtime.ts +1 -9
- package/dist/extensions/pi-subagents/src/runs/shared/subagent-startup-retry.ts +12 -0
- package/dist/extensions/pi-subagents/src/runs/shared/tool-timeout.ts +95 -0
- package/dist/extensions/pi-subagents/src/shared/agent-stream-options.ts +5 -0
- package/dist/extensions/pi-subagents/src/shared/artifacts.ts +0 -4
- package/dist/extensions/pi-subagents/src/shared/display-text.ts +50 -0
- package/dist/extensions/pi-subagents/src/shared/node-executable.ts +21 -0
- package/dist/extensions/pi-subagents/src/shared/session-lineage.ts +71 -0
- package/dist/extensions/pi-subagents/src/shared/types.ts +66 -4
- package/dist/extensions/pi-subagents/src/slash/slash-commands.ts +34 -25
- package/dist/extensions/pi-subagents/src/slash/slash-live-state.ts +4 -0
- package/dist/extensions/pi-subagents/src/tui/fleet-status.ts +160 -45
- package/dist/extensions/pi-subagents/src/tui/fleet-transcript.ts +1 -48
- package/dist/extensions/pi-subagents/src/tui/fleet.ts +129 -17
- package/dist/extensions/pi-subagents/src/tui/render.ts +122 -44
- package/dist/extensions/pi-subagents/src/watchdog/permission-arbiter.ts +2 -1
- package/dist/extensions/pi-subagents/src/watchdog/review.ts +4 -3
- package/dist/extensions/pi-subagents/src/workflows/chat-progress.ts +10 -2
- package/dist/extensions/pi-subagents/src/workflows/scripted-workflow.ts +272 -76
- package/dist/extensions/pi-subagents/src/workflows/workflow-auto-relaunch.ts +28 -0
- package/dist/extensions/pi-subagents/test/fixtures/pi-coding-agent-shim/dist/core/extensions/types.d.ts +2 -0
- package/dist/extensions/pi-subagents/test/integration/async-execution.test.ts +2 -2
- package/dist/extensions/pi-subagents/test/integration/async-job-tracker.test.ts +68 -9
- package/dist/extensions/pi-subagents/test/integration/async-status.test.ts +27 -1
- package/dist/extensions/pi-subagents/test/integration/chain-execution.test.ts +1 -1
- package/dist/extensions/pi-subagents/test/integration/error-handling.test.ts +49 -0
- package/dist/extensions/pi-subagents/test/integration/external-cli-runner.test.ts +1 -0
- package/dist/extensions/pi-subagents/test/integration/intercom-result-delivery.test.ts +26 -1
- package/dist/extensions/pi-subagents/test/integration/orca-progress-tabs.test.ts +144 -0
- package/dist/extensions/pi-subagents/test/integration/render-widget.test.ts +26 -0
- package/dist/extensions/pi-subagents/test/integration/result-watcher.test.ts +369 -98
- package/dist/extensions/pi-subagents/test/integration/single-execution.test.ts +111 -78
- package/dist/extensions/pi-subagents/test/integration/slash-commands.test.ts +49 -1
- package/dist/extensions/pi-subagents/test/integration/slash-live-state.test.ts +12 -0
- package/dist/extensions/pi-subagents/test/support/node-command.ts +15 -0
- package/dist/extensions/pi-subagents/test/unit/active-async-capacity.test.ts +50 -0
- package/dist/extensions/pi-subagents/test/unit/agent-frontmatter.test.ts +28 -0
- package/dist/extensions/pi-subagents/test/unit/agent-overrides.test.ts +61 -0
- package/dist/extensions/pi-subagents/test/unit/agent-stream-options.test.ts +14 -0
- package/dist/extensions/pi-subagents/test/unit/async-interrupt-action.test.ts +76 -0
- package/dist/extensions/pi-subagents/test/unit/async-resume.test.ts +2 -0
- package/dist/extensions/pi-subagents/test/unit/async-status-snapshot.test.ts +162 -0
- package/dist/extensions/pi-subagents/test/unit/completion-guard.test.ts +209 -10
- package/dist/extensions/pi-subagents/test/unit/completion-replay.test.ts +41 -1
- package/dist/extensions/pi-subagents/test/unit/external-runs.test.ts +135 -42
- package/dist/extensions/pi-subagents/test/unit/fleet-status.test.ts +252 -1
- package/dist/extensions/pi-subagents/test/unit/fleet.test.ts +192 -6
- package/dist/extensions/pi-subagents/test/unit/handoff-adoption.test.ts +103 -0
- package/dist/extensions/pi-subagents/test/unit/herdr-inspector-bootstrap.test.ts +32 -0
- package/dist/extensions/pi-subagents/test/unit/herdr-shell-command.test.ts +59 -0
- package/dist/extensions/pi-subagents/test/unit/index-child-registration.test.ts +91 -0
- package/dist/extensions/pi-subagents/test/unit/intercom-bridge.test.ts +23 -4
- package/dist/extensions/pi-subagents/test/unit/llm-intent-arbiter.test.ts +56 -1
- package/dist/extensions/pi-subagents/test/unit/mission-goal-driver.test.ts +67 -3
- package/dist/extensions/pi-subagents/test/unit/mission-lifecycle.test.ts +1 -1
- package/dist/extensions/pi-subagents/test/unit/mission-store.test.ts +14 -0
- package/dist/extensions/pi-subagents/test/unit/model-fallback.test.ts +18 -0
- package/dist/extensions/pi-subagents/test/unit/native-supervisor-channel.test.ts +2 -2
- package/dist/extensions/pi-subagents/test/unit/nested-events.test.ts +25 -2
- package/dist/extensions/pi-subagents/test/unit/node-executable.test.ts +26 -0
- package/dist/extensions/pi-subagents/test/unit/notify.test.ts +8 -0
- package/dist/extensions/pi-subagents/test/unit/orca-progress-tabs.test.ts +348 -0
- package/dist/extensions/pi-subagents/test/unit/package-manifest.test.ts +5 -2
- package/dist/extensions/pi-subagents/test/unit/pi-args.test.ts +1 -1
- package/dist/extensions/pi-subagents/test/unit/pi-coding-agent-dir.test.ts +34 -0
- package/dist/extensions/pi-subagents/test/unit/public-execution.test.ts +6 -1
- package/dist/extensions/pi-subagents/test/unit/render-helpers.test.ts +92 -0
- package/dist/extensions/pi-subagents/test/unit/result-files.test.ts +175 -0
- package/dist/extensions/pi-subagents/test/unit/retained-children.test.ts +127 -17
- package/dist/extensions/pi-subagents/test/unit/run-id-resolver.test.ts +44 -0
- package/dist/extensions/pi-subagents/test/unit/run-status.test.ts +247 -0
- package/dist/extensions/pi-subagents/test/unit/scheduled-runs.test.ts +107 -2
- package/dist/extensions/pi-subagents/test/unit/schemas.test.ts +5 -8
- package/dist/extensions/pi-subagents/test/unit/scripted-workflow.test.ts +456 -4
- package/dist/extensions/pi-subagents/test/unit/session-lineage.test.ts +73 -0
- package/dist/extensions/pi-subagents/test/unit/stale-run-reconciler.test.ts +128 -0
- package/dist/extensions/pi-subagents/test/unit/subagent-control.test.ts +28 -0
- package/dist/extensions/pi-subagents/test/unit/subagent-prompt-runtime.test.ts +16 -7
- package/dist/extensions/pi-subagents/test/unit/subagent-startup-retry.test.ts +15 -0
- package/dist/extensions/pi-subagents/test/unit/subagent-wait.test.ts +79 -15
- package/dist/extensions/pi-subagents/test/unit/tool-description.test.ts +12 -12
- package/dist/extensions/pi-subagents/test/unit/tool-timeout.test.ts +109 -0
- package/dist/extensions/pi-subagents/test/unit/wait-subscriptions.test.ts +34 -0
- package/dist/extensions/pi-subagents/test/unit/workflow-auto-relaunch.test.ts +45 -0
- package/dist/extensions/pi-subagents/test/unit/workflow-chat-progress.test.ts +94 -1
- package/dist/extensions/pi-subagents/test/unit/workflow-launch-params.test.ts +42 -2
- package/dist/extensions/ponytail/index.js +11 -0
- package/dist/extensions/ponytail/ponytail-config.cjs +2 -0
- package/dist/extensions/ponytail/ponytail-instructions.cjs +6 -0
- package/dist/extensions/ponytail/test/extension.test.js +274 -140
- package/dist/extensions/ponytail/test/helpers.test.js +280 -92
- package/dist/extensions/question/index.ts +33 -0
- package/dist/extensions/question/question-list.ts +28 -0
- package/dist/extensions/question/row-layout.ts +3 -0
- package/dist/extensions/question/tests/batch.test.ts +103 -65
- package/dist/extensions/question/tests/exp.test.ts +70 -0
- package/dist/extensions/question/tests/exp2.test.ts +59 -0
- package/dist/extensions/question/tests/exp3.test.ts +47 -0
- package/dist/extensions/question/tests/helpers.test.ts +75 -35
- package/dist/extensions/question/tests/question-list.test.ts +640 -198
- package/dist/extensions/question/tests/row-layout.test.ts +193 -114
- package/dist/extensions/question/tests/schemas.test.ts +42 -19
- package/dist/extensions/question/tests/selection-mode.test.ts +54 -0
- package/dist/extensions/question/tests/shortcuts.test.ts +106 -84
- package/dist/extensions/question/tests/wizard.test.ts +771 -50
- package/dist/extensions/question/tests/zz-ig3.test.ts +20 -0
- package/dist/extensions/question/tests/zz-probe.test.ts +29 -0
- package/dist/extensions/rtk.test.ts +180 -1
- package/dist/extensions/tokenin-onboarding.ts +3 -0
- package/dist/extensions/undo.test.ts +720 -0
- package/dist/extensions/undo.ts +6 -0
- package/dist/extensions/web-agent-onboarding.test.ts +415 -0
- package/dist/extensions/workflow/extension.ts +3 -3
- package/dist/extensions/workflow/modes.ts +41 -19
- package/dist/skills/pi-subagents/SKILL.md +1 -1
- package/dist/skills/pi-subagents/references/constraints-and-recipes.md +5 -4
- package/dist/skills/pi-subagents/references/execution-controls.md +16 -15
- package/dist/skills/pi-subagents/references/management-authoring-rpc.md +5 -5
- package/dist/skills/pi-subagents/references/prompting-and-roles.md +46 -26
- package/docs/plans/workflow-autoloop-reference.md +1 -2
- package/docs/plans/workflow-handoff-carryover.md +108 -0
- package/docs/settings.md +6 -0
- package/package.json +4 -1
|
@@ -3,6 +3,7 @@ import * as fs from "node:fs";
|
|
|
3
3
|
import * as os from "node:os";
|
|
4
4
|
import * as path from "node:path";
|
|
5
5
|
import { describe, it } from "node:test";
|
|
6
|
+
import { acquireActiveAsyncCapacity } from "../../src/runs/background/active-async-capacity.ts";
|
|
6
7
|
import { consumeSteerRequests, consumeSteerRequestsFromDir, stepSteerInboxDir, writeSteerAck } from "../../src/runs/background/control-channel.ts";
|
|
7
8
|
import { listAsyncRuns } from "../../src/runs/background/async-status.ts";
|
|
8
9
|
import { inspectSubagentStatus } from "../../src/runs/background/run-status.ts";
|
|
@@ -125,6 +126,74 @@ function text(result: { content: Array<{ type: string; text?: string }> }): stri
|
|
|
125
126
|
}
|
|
126
127
|
|
|
127
128
|
describe("async interrupt action", () => {
|
|
129
|
+
it("routes debug.run to async lifecycle debug, not live foreground status", async () => {
|
|
130
|
+
const state = createState();
|
|
131
|
+
state.currentSessionId = "session";
|
|
132
|
+
const runId = `debug-foreground-${Date.now().toString(36)}`;
|
|
133
|
+
const asyncDir = createRunningAsync(state, runId, { track: false, sessionId: "session" });
|
|
134
|
+
state.foregroundControls.set(runId, {
|
|
135
|
+
runId,
|
|
136
|
+
sessionId: "session",
|
|
137
|
+
mode: "single",
|
|
138
|
+
startedAt: 100,
|
|
139
|
+
updatedAt: 100,
|
|
140
|
+
cwd: os.tmpdir(),
|
|
141
|
+
agent: "worker",
|
|
142
|
+
status: "running",
|
|
143
|
+
controller: new AbortController(),
|
|
144
|
+
});
|
|
145
|
+
try {
|
|
146
|
+
const result = await executorWithKill(state, () => true)
|
|
147
|
+
.execute("debug.run", { action: "debug.run", id: runId }, new AbortController().signal, undefined, ctx());
|
|
148
|
+
const output = text(result);
|
|
149
|
+
|
|
150
|
+
assert.equal(result.isError, undefined);
|
|
151
|
+
assert.match(output, /Run lifecycle debug/);
|
|
152
|
+
assert.doesNotMatch(output, /Live foreground/);
|
|
153
|
+
} finally {
|
|
154
|
+
state.foregroundControls.delete(runId);
|
|
155
|
+
cleanup(runId, asyncDir);
|
|
156
|
+
}
|
|
157
|
+
});
|
|
158
|
+
|
|
159
|
+
it("renders run lifecycle debug without transcript content", () => {
|
|
160
|
+
const state = createState();
|
|
161
|
+
state.currentSessionId = "session";
|
|
162
|
+
const runId = `debug-run-${Date.now().toString(36)}`;
|
|
163
|
+
const activeCapacityRoot = fs.mkdtempSync(path.join(os.tmpdir(), "pi-debug-capacity-"));
|
|
164
|
+
const asyncDir = path.join(ASYNC_DIR, runId);
|
|
165
|
+
try {
|
|
166
|
+
const capacity = acquireActiveAsyncCapacity({ sessionId: "session", limit: 1, runId, kind: "workflow", asyncDir }, { rootDir: activeCapacityRoot });
|
|
167
|
+
assert.ok(capacity);
|
|
168
|
+
capacity.markWorkflowStarted();
|
|
169
|
+
writeJson(path.join(asyncDir, "status.json"), {
|
|
170
|
+
runId,
|
|
171
|
+
sessionId: "session",
|
|
172
|
+
mode: "workflow",
|
|
173
|
+
state: "complete",
|
|
174
|
+
startedAt: 100,
|
|
175
|
+
processTerminal: { version: 1, state: "pending", runId, runnerProcessInstanceId: "workflow-runner" },
|
|
176
|
+
steps: [{ agent: "worker", workflowKey: "review", status: "completed", async: false }],
|
|
177
|
+
});
|
|
178
|
+
fs.writeFileSync(path.join(asyncDir, "output-0.log"), "SECRET_TRANSCRIPT_TEXT", "utf-8");
|
|
179
|
+
|
|
180
|
+
const result = inspectSubagentStatus({ action: "debug.run", id: runId }, { state, activeCapacityRoot });
|
|
181
|
+
const output = text(result);
|
|
182
|
+
|
|
183
|
+
assert.match(output, /Run lifecycle debug/);
|
|
184
|
+
assert.match(output, new RegExp(`Run: ${runId}`));
|
|
185
|
+
assert.match(output, /Status process terminal: pending · runner workflow-runner/);
|
|
186
|
+
assert.match(output, /Sidecar process terminal: missing/);
|
|
187
|
+
assert.match(output, /Active capacity: releasable/);
|
|
188
|
+
assert.match(output, /Workflow children: 1/);
|
|
189
|
+
assert.match(output, /key review · worker · completed · async no/);
|
|
190
|
+
assert.doesNotMatch(output, /SECRET_TRANSCRIPT_TEXT/);
|
|
191
|
+
} finally {
|
|
192
|
+
fs.rmSync(asyncDir, { recursive: true, force: true });
|
|
193
|
+
fs.rmSync(activeCapacityRoot, { recursive: true, force: true });
|
|
194
|
+
}
|
|
195
|
+
});
|
|
196
|
+
|
|
128
197
|
it("steers a live workflow-owned foreground child by child id", async () => {
|
|
129
198
|
const state = createState();
|
|
130
199
|
const workflowRunId = `workflow-child-${Date.now().toString(36)}`;
|
|
@@ -489,6 +558,13 @@ describe("async interrupt action", () => {
|
|
|
489
558
|
assert.match(statusText, /State: display-dismissed/);
|
|
490
559
|
assert.match(statusText, /No running work was terminated/);
|
|
491
560
|
assert.doesNotMatch(statusText, /Steer/);
|
|
561
|
+
const debugResult = inspectSubagentStatus({ action: "debug.run", id: runId }, { state, kill: () => {
|
|
562
|
+
throw new Error("dismissed workflow debug must not inspect the pid");
|
|
563
|
+
} });
|
|
564
|
+
const debugText = text(debugResult);
|
|
565
|
+
assert.match(debugText, /Run lifecycle debug/);
|
|
566
|
+
assert.match(debugText, /State: running/);
|
|
567
|
+
assert.match(debugText, /Active capacity: not-owned/);
|
|
492
568
|
const transcriptResult = inspectSubagentStatus({ action: "status", id: runId, view: "transcript" }, { state, kill: () => {
|
|
493
569
|
throw new Error("dismissed workflow transcript must not inspect the pid");
|
|
494
570
|
} });
|
|
@@ -113,10 +113,12 @@ describe("async resume lookup", () => {
|
|
|
113
113
|
writeJson(path.join(asyncDir, "recovery-descriptor.json"), {
|
|
114
114
|
...descriptor,
|
|
115
115
|
launchContractDigest: "launch-contract-digest",
|
|
116
|
+
intercomBridge: { mode: "off" },
|
|
116
117
|
});
|
|
117
118
|
const valid = resolveAsyncResumeTarget({ id: "run-descriptor" }, { asyncDirRoot: asyncRoot, resultsDir });
|
|
118
119
|
assert.equal(valid.launchContractDigest, "launch-contract-digest");
|
|
119
120
|
assert.equal(valid.recoveryDescriptor?.launchContractDigest, "launch-contract-digest");
|
|
121
|
+
assert.deepEqual(valid.recoveryDescriptor?.intercomBridge, { mode: "off" });
|
|
120
122
|
|
|
121
123
|
writeJson(path.join(asyncDir, "recovery-descriptor.json"), { ...descriptor, sourceRunId: "another-run" });
|
|
122
124
|
assert.throws(() => resolveAsyncResumeTarget({ id: "run-descriptor" }, { asyncDirRoot: asyncRoot, resultsDir }), /different source run/);
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
import assert from "node:assert/strict";
|
|
2
|
+
import { describe, it } from "node:test";
|
|
3
|
+
import {
|
|
4
|
+
ASYNC_STATUS_SNAPSHOT_KIND,
|
|
5
|
+
ASYNC_STATUS_SNAPSHOT_VERSION,
|
|
6
|
+
ASYNC_STATUS_SNAPSHOT_WIDGET_PREFIX,
|
|
7
|
+
buildAsyncStatusSnapshot,
|
|
8
|
+
buildAsyncStatusSnapshotForState,
|
|
9
|
+
encodeAsyncStatusSnapshotWidget,
|
|
10
|
+
} from "../../src/runs/background/async-status-snapshot.ts";
|
|
11
|
+
|
|
12
|
+
const privateNeedle = "PRIVATE_LEAK_NEEDLE";
|
|
13
|
+
|
|
14
|
+
function json(value: unknown): string {
|
|
15
|
+
return JSON.stringify(value);
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
describe("async status snapshot", () => {
|
|
19
|
+
it("projects current async jobs with a versioned safe shape", () => {
|
|
20
|
+
const snapshot = buildAsyncStatusSnapshot([{
|
|
21
|
+
asyncId: "run-1",
|
|
22
|
+
asyncDir: `/tmp/${privateNeedle}/run`,
|
|
23
|
+
cwd: `/repo/${privateNeedle}`,
|
|
24
|
+
sessionRoot: `/sessions/${privateNeedle}`,
|
|
25
|
+
sessionDir: `/session-dir/${privateNeedle}`,
|
|
26
|
+
outputFile: `/output/${privateNeedle}.log`,
|
|
27
|
+
sessionFile: `/session-file/${privateNeedle}.jsonl`,
|
|
28
|
+
sessionId: "session-a",
|
|
29
|
+
status: "running",
|
|
30
|
+
mode: "parallel",
|
|
31
|
+
agents: ["worker"],
|
|
32
|
+
description: `task ${privateNeedle}`,
|
|
33
|
+
startedAt: 100,
|
|
34
|
+
updatedAt: 150,
|
|
35
|
+
currentTool: "bash\n\u001b]8;;bad\u0007",
|
|
36
|
+
currentPath: `/current/${privateNeedle}`,
|
|
37
|
+
steps: [{
|
|
38
|
+
index: 0,
|
|
39
|
+
agent: "worker",
|
|
40
|
+
status: "running",
|
|
41
|
+
startedAt: 110,
|
|
42
|
+
currentTool: "read",
|
|
43
|
+
currentToolArgs: `args ${privateNeedle}`,
|
|
44
|
+
recentOutput: [`output ${privateNeedle}`],
|
|
45
|
+
error: `error ${privateNeedle}`,
|
|
46
|
+
transcriptPath: `/transcript/${privateNeedle}.jsonl`,
|
|
47
|
+
}],
|
|
48
|
+
}], { generatedAt: 200 });
|
|
49
|
+
|
|
50
|
+
assert.equal(snapshot.kind, ASYNC_STATUS_SNAPSHOT_KIND);
|
|
51
|
+
assert.equal(snapshot.version, ASYNC_STATUS_SNAPSHOT_VERSION);
|
|
52
|
+
assert.equal(snapshot.generatedAt, 200);
|
|
53
|
+
assert.equal(snapshot.runs.length, 1);
|
|
54
|
+
assert.deepEqual(snapshot.runs[0], {
|
|
55
|
+
id: "run-1",
|
|
56
|
+
kind: "subagent",
|
|
57
|
+
label: "worker",
|
|
58
|
+
state: "running",
|
|
59
|
+
startedAt: 100,
|
|
60
|
+
updatedAt: 150,
|
|
61
|
+
activity: { currentTool: "bash" },
|
|
62
|
+
children: [{
|
|
63
|
+
id: "step:0",
|
|
64
|
+
kind: "step",
|
|
65
|
+
label: "worker",
|
|
66
|
+
state: "running",
|
|
67
|
+
startedAt: 110,
|
|
68
|
+
updatedAt: 110,
|
|
69
|
+
activity: { currentTool: "read" },
|
|
70
|
+
}],
|
|
71
|
+
});
|
|
72
|
+
assert.equal(json(snapshot).includes(privateNeedle), false);
|
|
73
|
+
assert.equal(json(snapshot).includes("currentPath"), false);
|
|
74
|
+
assert.equal(json(snapshot).includes("currentToolArgs"), false);
|
|
75
|
+
assert.equal(json(snapshot).includes("recentOutput"), false);
|
|
76
|
+
assert.equal(json(snapshot).includes("transcriptPath"), false);
|
|
77
|
+
});
|
|
78
|
+
|
|
79
|
+
it("normalizes pending child steps to queued", () => {
|
|
80
|
+
const snapshot = buildAsyncStatusSnapshot([{
|
|
81
|
+
asyncId: "run",
|
|
82
|
+
asyncDir: "/tmp/run",
|
|
83
|
+
status: "queued",
|
|
84
|
+
agents: ["worker"],
|
|
85
|
+
steps: [{ agent: "worker", status: "pending" }],
|
|
86
|
+
} as any], { generatedAt: 1 });
|
|
87
|
+
|
|
88
|
+
assert.equal(snapshot.runs[0]?.children?.[0]?.state, "queued");
|
|
89
|
+
});
|
|
90
|
+
|
|
91
|
+
it("applies run, child, depth, string, and byte caps", () => {
|
|
92
|
+
const jobs = Array.from({ length: 5 }, (_, runIndex) => ({
|
|
93
|
+
asyncId: `run-${runIndex}`,
|
|
94
|
+
asyncDir: `/tmp/run-${runIndex}`,
|
|
95
|
+
status: "running" as const,
|
|
96
|
+
mode: "workflow" as const,
|
|
97
|
+
agents: [`agent-${runIndex}-${"x".repeat(50)}`],
|
|
98
|
+
startedAt: runIndex,
|
|
99
|
+
updatedAt: runIndex,
|
|
100
|
+
steps: Array.from({ length: 5 }, (_, stepIndex) => ({
|
|
101
|
+
agent: `child-${stepIndex}-${"y".repeat(50)}`,
|
|
102
|
+
status: "running" as const,
|
|
103
|
+
children: [{
|
|
104
|
+
id: `nested-${stepIndex}`,
|
|
105
|
+
parentRunId: `run-${runIndex}`,
|
|
106
|
+
depth: 1,
|
|
107
|
+
path: [],
|
|
108
|
+
state: "running" as const,
|
|
109
|
+
agent: "nested",
|
|
110
|
+
}],
|
|
111
|
+
})),
|
|
112
|
+
}));
|
|
113
|
+
|
|
114
|
+
const snapshot = buildAsyncStatusSnapshot(jobs, {
|
|
115
|
+
generatedAt: 10,
|
|
116
|
+
maxRuns: 2,
|
|
117
|
+
maxChildrenPerNode: 2,
|
|
118
|
+
maxDepth: 1,
|
|
119
|
+
maxStringLength: 16,
|
|
120
|
+
maxSerializedBytes: 1200,
|
|
121
|
+
});
|
|
122
|
+
|
|
123
|
+
assert.equal(snapshot.runs.length, 2);
|
|
124
|
+
assert.equal(snapshot.omitted.runs, 3);
|
|
125
|
+
assert.equal(snapshot.runs[0]?.children?.length, 2);
|
|
126
|
+
assert.ok(snapshot.omitted.children >= 6);
|
|
127
|
+
assert.ok((snapshot.runs[0]?.label.length ?? 0) <= 16);
|
|
128
|
+
assert.ok(Buffer.byteLength(json(snapshot), "utf8") <= 1200);
|
|
129
|
+
|
|
130
|
+
const byteCapped = buildAsyncStatusSnapshot(jobs, { maxSerializedBytes: 512 });
|
|
131
|
+
assert.equal(byteCapped.omitted.byteLimitExceeded, true);
|
|
132
|
+
assert.ok(Buffer.byteLength(json(byteCapped), "utf8") <= 512);
|
|
133
|
+
});
|
|
134
|
+
|
|
135
|
+
it("uses current-session state and retained fleet jobs without rebuilding history", () => {
|
|
136
|
+
const state = {
|
|
137
|
+
currentSessionId: "session-a",
|
|
138
|
+
foregroundControls: new Map(),
|
|
139
|
+
asyncJobs: new Map([["active", { asyncId: "active", asyncDir: "/tmp/active", sessionId: "session-a", status: "running", agents: ["worker"] }]]),
|
|
140
|
+
fleetJobs: new Map([
|
|
141
|
+
["terminal", { asyncId: "terminal", asyncDir: "/tmp/terminal", sessionId: "session-a", status: "complete", agents: ["reviewer"], updatedAt: 300, outputFile: `/tmp/${privateNeedle}.log` }],
|
|
142
|
+
["foreign", { asyncId: "foreign", asyncDir: "/tmp/foreign", sessionId: "other", status: "running", agents: ["hidden"] }],
|
|
143
|
+
]),
|
|
144
|
+
} as any;
|
|
145
|
+
|
|
146
|
+
const snapshot = buildAsyncStatusSnapshotForState(state, "session-a", { generatedAt: 1 });
|
|
147
|
+
assert.deepEqual(snapshot.runs.map((run) => run.id).sort(), ["active", "terminal"]);
|
|
148
|
+
assert.equal(snapshot.runs.find((run) => run.id === "terminal")?.endedAt, 300);
|
|
149
|
+
assert.equal(json(snapshot).includes(privateNeedle), false);
|
|
150
|
+
assert.deepEqual(buildAsyncStatusSnapshotForState(state, "other").runs, []);
|
|
151
|
+
});
|
|
152
|
+
|
|
153
|
+
it("encodes RPC widget payloads as a string-array snapshot", () => {
|
|
154
|
+
const lines = encodeAsyncStatusSnapshotWidget([{ asyncId: "run", asyncDir: "/tmp/run", status: "queued", agents: ["planner"] } as any], { generatedAt: 5 });
|
|
155
|
+
assert.equal(lines.length, 1);
|
|
156
|
+
assert.ok(lines[0]?.startsWith(ASYNC_STATUS_SNAPSHOT_WIDGET_PREFIX));
|
|
157
|
+
const snapshot = JSON.parse(lines[0]!.slice(ASYNC_STATUS_SNAPSHOT_WIDGET_PREFIX.length));
|
|
158
|
+
assert.equal(snapshot.kind, ASYNC_STATUS_SNAPSHOT_KIND);
|
|
159
|
+
assert.equal(snapshot.version, 1);
|
|
160
|
+
assert.equal(snapshot.runs[0].id, "run");
|
|
161
|
+
});
|
|
162
|
+
});
|
|
@@ -25,17 +25,216 @@ function assistantText(text: string): Message {
|
|
|
25
25
|
}
|
|
26
26
|
|
|
27
27
|
test("implementation task with no mutation triggers the completion guard", () => {
|
|
28
|
-
const
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
28
|
+
for (const report of [
|
|
29
|
+
"No better current-scope change is needed.",
|
|
30
|
+
"Kept the current implementation. No new code or test changes were made in this challenge pass.",
|
|
31
|
+
]) {
|
|
32
|
+
const result = evaluateCompletionMutationGuard({
|
|
33
|
+
agent: "worker",
|
|
34
|
+
task: "Implement the approved fix",
|
|
35
|
+
messages: [assistantText(report)],
|
|
36
|
+
});
|
|
37
|
+
|
|
38
|
+
assert.deepEqual(result, {
|
|
39
|
+
expectedMutation: true,
|
|
40
|
+
attemptedMutation: false,
|
|
41
|
+
triggered: true,
|
|
42
|
+
});
|
|
43
|
+
}
|
|
44
|
+
});
|
|
45
|
+
|
|
46
|
+
function revivedTask(followUp: string): string {
|
|
47
|
+
return [
|
|
48
|
+
"You are reviving a previous subagent conversation.",
|
|
49
|
+
"",
|
|
50
|
+
"Original run: abc123",
|
|
51
|
+
"Original agent: worker",
|
|
52
|
+
"Original session file: /tmp/session.jsonl",
|
|
53
|
+
"",
|
|
54
|
+
"Use the stored session context as background. Answer the orchestrator's follow-up below. Do not assume the original child process is still alive.",
|
|
55
|
+
"",
|
|
56
|
+
"Follow-up:",
|
|
57
|
+
followUp,
|
|
58
|
+
].join("\n");
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
const implementationChallengeTask = revivedTask("Run implementation challenge pass two and implement any better current-scope change.");
|
|
62
|
+
|
|
63
|
+
test("implementation challenges may complete with explicit no-change reports", () => {
|
|
64
|
+
for (const report of [
|
|
65
|
+
"No better current-scope change is needed.",
|
|
66
|
+
[
|
|
67
|
+
"Kept the current implementation. No new code or test changes were made in this challenge pass.",
|
|
68
|
+
"Reason: the current candidate is the smallest correct shape.",
|
|
69
|
+
].join("\n\n"),
|
|
70
|
+
"The current implementation was kept. No code changes were made.",
|
|
71
|
+
"The current candidate was kept. No source changes were made.",
|
|
72
|
+
"The current shape was kept. No file or test changes were made.",
|
|
73
|
+
"Kept the current implementation. No code/source/file/test changes were made.",
|
|
74
|
+
"Kept the current implementation. No code, source, or test changes were made.",
|
|
75
|
+
"No better current-scope change is needed.\n\nReason: I cannot identify a smaller safe change.",
|
|
76
|
+
"No better current-scope change is needed because I did not identify a smaller safe change.",
|
|
77
|
+
"No better current-scope change is needed because no work remains.",
|
|
78
|
+
"No better current-scope change is needed. I haven't identified required changes.",
|
|
79
|
+
"No better current-scope change is needed. I haven’t identified required changes.",
|
|
80
|
+
"No better current-scope change is needed; I did not identify a smaller safe change.",
|
|
81
|
+
"No better current-scope change is needed, since I did not identify a smaller safe change.",
|
|
82
|
+
"No better current-scope change is needed, I did not identify a smaller safe change.",
|
|
83
|
+
"No better current-scope change is needed, the current implementation does not require further edits.",
|
|
84
|
+
"No better current-scope change is needed; the current implementation does not require further edits.",
|
|
85
|
+
"No better current-scope change is needed, but the current implementation does not require further edits.",
|
|
86
|
+
"Kept the current implementation. No code changes were made.\n\nReason: this does not need a broader rewrite.",
|
|
87
|
+
"Kept the current implementation; I did not identify a smaller safe change. No code changes were made.",
|
|
88
|
+
"Kept the current implementation. No code changes were made, since I did not identify a smaller safe change.",
|
|
89
|
+
"Kept the current implementation. No code changes were made, I did not identify a smaller safe change.",
|
|
90
|
+
"Kept the current implementation, the current candidate does not need more work. No code changes were made.",
|
|
91
|
+
"Kept the current implementation; the current candidate does not need more work. No code changes were made.",
|
|
92
|
+
"Kept the current implementation because I did not identify a smaller safe change. No code changes were made.",
|
|
93
|
+
"Kept the current implementation. No code changes were made because I did not identify a smaller safe change.",
|
|
94
|
+
]) {
|
|
95
|
+
const result = evaluateCompletionMutationGuard({
|
|
96
|
+
agent: "worker",
|
|
97
|
+
task: implementationChallengeTask,
|
|
98
|
+
messages: [assistantText(report)],
|
|
99
|
+
});
|
|
100
|
+
|
|
101
|
+
assert.deepEqual(result, {
|
|
102
|
+
expectedMutation: true,
|
|
103
|
+
attemptedMutation: false,
|
|
104
|
+
triggered: false,
|
|
105
|
+
});
|
|
106
|
+
}
|
|
107
|
+
});
|
|
108
|
+
|
|
109
|
+
test("implementation challenge reports require both a kept-current rationale and no-change statement", () => {
|
|
110
|
+
for (const report of [
|
|
111
|
+
"Kept the current implementation.",
|
|
112
|
+
"No new code or test changes were made in this challenge pass.",
|
|
113
|
+
"Kept the current implementation. No new code or test changes were made, but I am uncertain.",
|
|
114
|
+
]) {
|
|
115
|
+
assert.equal(evaluateCompletionMutationGuard({
|
|
116
|
+
agent: "worker",
|
|
117
|
+
task: implementationChallengeTask,
|
|
118
|
+
messages: [assistantText(report)],
|
|
119
|
+
}).triggered, true, report);
|
|
120
|
+
}
|
|
121
|
+
});
|
|
122
|
+
|
|
123
|
+
test("implementation challenge reports require current kept/no-change claims", () => {
|
|
124
|
+
for (const report of [
|
|
125
|
+
"The previous message said \"Kept the current implementation. No code changes were made\".",
|
|
126
|
+
"The prior report stated Kept the current implementation. No code changes were made.",
|
|
127
|
+
"The previous message said \"Kept the current implementation. No code changes were made\". Kept the current implementation.",
|
|
128
|
+
"'Kept the current implementation. No code changes were made.'",
|
|
129
|
+
]) {
|
|
130
|
+
assert.equal(evaluateCompletionMutationGuard({
|
|
131
|
+
agent: "worker",
|
|
132
|
+
task: implementationChallengeTask,
|
|
133
|
+
messages: [assistantText(report)],
|
|
134
|
+
}).triggered, true, report);
|
|
135
|
+
}
|
|
136
|
+
});
|
|
137
|
+
|
|
138
|
+
test("implementation challenge reports with later implementation retractions remain guarded", () => {
|
|
139
|
+
for (const report of [
|
|
140
|
+
"No better current-scope change is needed. I found a required code change.",
|
|
141
|
+
"Kept the current implementation. No code changes were made. Implementation work remains.",
|
|
142
|
+
"No better current-scope change is needed. A code change is needed.",
|
|
143
|
+
"Kept the current implementation. No code changes were made. I need to implement the fix.",
|
|
144
|
+
"No better current-scope change is needed, but I found a required code change.",
|
|
145
|
+
"No better current-scope change is needed\nI found a required code change.",
|
|
146
|
+
"No better current-scope change is needed. I found required changes.",
|
|
147
|
+
"No better current-scope change is needed. Code changes are needed.",
|
|
148
|
+
"No better current-scope change is needed. Changes are needed.",
|
|
149
|
+
"No better current-scope change is needed because no work remains. Code changes are needed.",
|
|
150
|
+
"No better current-scope change is needed. I need changes.",
|
|
151
|
+
"No better current-scope change is needed. We need edits.",
|
|
152
|
+
"Kept the current implementation. No code changes were made. I need changes.",
|
|
153
|
+
"Kept the current implementation. No code changes were made. We need patches.",
|
|
154
|
+
"No better current-scope change is needed. That claim is rejected.",
|
|
155
|
+
"No better current-scope change is needed. This report is retracted.",
|
|
156
|
+
"No better current-scope change is needed. Required changes.",
|
|
157
|
+
"No better current-scope change is needed. Need changes.",
|
|
158
|
+
"No better current-scope change is needed, I disagree.",
|
|
159
|
+
"No better current-scope change is needed; I disagree.",
|
|
160
|
+
"Kept the current implementation. No code changes were made, I reject.",
|
|
161
|
+
"Kept the current implementation. No code changes were made; I reject.",
|
|
162
|
+
"No better current-scope change is needed: I disagree.",
|
|
163
|
+
"No better current-scope change is needed — I disagree.",
|
|
164
|
+
"No better current-scope change is needed. \"I reject.\"",
|
|
165
|
+
"No better current-scope change is needed. 'I reject.'",
|
|
166
|
+
"No better current-scope change is needed. ‘I reject.’",
|
|
167
|
+
"Kept the current implementation. No code changes were made: I reject.",
|
|
168
|
+
"Kept the current implementation. No code changes were made. \"I retract.\"",
|
|
169
|
+
"No better current-scope change is needed. I disagree.",
|
|
170
|
+
"No better current-scope change is needed. I reject.",
|
|
171
|
+
"No better current-scope change is needed. I retract.",
|
|
172
|
+
"Kept the current implementation. No code changes were made. I disagree.",
|
|
173
|
+
"Kept the current implementation. No code changes were made. I reject.",
|
|
174
|
+
"Kept the current implementation. No code changes were made. I retract.",
|
|
175
|
+
"No better current-scope change is needed. I disagree with that.",
|
|
176
|
+
"No better current-scope change is needed. I retract that.",
|
|
177
|
+
"No better current-scope change is needed. I reject that.",
|
|
178
|
+
"Kept the current implementation. No code changes were made. I disagree with that.",
|
|
179
|
+
"Kept the current implementation. No code changes were made. I retract that.",
|
|
180
|
+
"Kept the current implementation. No code changes were made. I reject that.",
|
|
181
|
+
]) {
|
|
182
|
+
assert.equal(evaluateCompletionMutationGuard({
|
|
183
|
+
agent: "worker",
|
|
184
|
+
task: implementationChallengeTask,
|
|
185
|
+
messages: [assistantText(report)],
|
|
186
|
+
}).triggered, true, report);
|
|
187
|
+
}
|
|
188
|
+
});
|
|
189
|
+
|
|
190
|
+
test("revived implementation tasks that mention implementation challenge remain guarded", () => {
|
|
191
|
+
for (const followUp of [
|
|
192
|
+
"Fix the implementation challenge completion guard bug.",
|
|
193
|
+
"Implementation challenge pass 1. Implement the required fix.",
|
|
194
|
+
"Implementation challenge pass 1 and implement the fix.",
|
|
195
|
+
]) {
|
|
196
|
+
const result = evaluateCompletionMutationGuard({
|
|
197
|
+
agent: "worker",
|
|
198
|
+
task: revivedTask(followUp),
|
|
199
|
+
messages: [assistantText("No better current-scope change is needed.")],
|
|
200
|
+
});
|
|
201
|
+
|
|
202
|
+
assert.deepEqual(result, {
|
|
203
|
+
expectedMutation: true,
|
|
204
|
+
attemptedMutation: false,
|
|
205
|
+
triggered: true,
|
|
206
|
+
}, followUp);
|
|
207
|
+
}
|
|
208
|
+
});
|
|
209
|
+
|
|
210
|
+
test("implementation challenge reports with negated or uncertain no-better-change claims remain guarded", () => {
|
|
211
|
+
for (const report of [
|
|
212
|
+
"I cannot say no better current-scope change is needed.",
|
|
213
|
+
"The previous message said \"No better current-scope change is needed\", but I disagree.",
|
|
214
|
+
"The previous message said \"Kept the current implementation. No code changes were made\", but I disagree.",
|
|
215
|
+
"The previous message said \"Kept the current implementation. No code changes were made\", but that was wrong.",
|
|
216
|
+
"The previous message said \"Kept the current implementation. No code changes were made\", but that was false.",
|
|
217
|
+
"The previous message said \"Kept the current implementation. No code changes were made\", but I reject that.",
|
|
218
|
+
"The previous message said \"Kept the current implementation. No code changes were made\". I reject that.",
|
|
219
|
+
"The previous message said \"Kept the current implementation. No code changes were made\". I reject this.",
|
|
220
|
+
"The previous message said \"Kept the current implementation. No code changes were made\". I disagree.",
|
|
221
|
+
"The previous message said \"Kept the current implementation. No code changes were made\", but it was rejected.",
|
|
222
|
+
"The previous message said \"Kept the current implementation. No code changes were made\", but I am rejecting it.",
|
|
223
|
+
"The prior report stated no better current-scope change is needed.",
|
|
224
|
+
"I don't think no better current-scope change is needed.",
|
|
225
|
+
"I dont think no better current-scope change is needed.",
|
|
226
|
+
"I do not think no better current-scope change is needed.",
|
|
227
|
+
"I cant say no better current-scope change is needed.",
|
|
228
|
+
"It is unclear whether no better current-scope change is needed.",
|
|
229
|
+
"Maybe no better current-scope change is needed.",
|
|
230
|
+
]) {
|
|
231
|
+
assert.equal(evaluateCompletionMutationGuard({
|
|
232
|
+
agent: "worker",
|
|
233
|
+
task: implementationChallengeTask,
|
|
234
|
+
messages: [assistantText(report)],
|
|
235
|
+
}).triggered, true, report);
|
|
236
|
+
}
|
|
33
237
|
|
|
34
|
-
assert.deepEqual(result, {
|
|
35
|
-
expectedMutation: true,
|
|
36
|
-
attemptedMutation: false,
|
|
37
|
-
triggered: true,
|
|
38
|
-
});
|
|
39
238
|
});
|
|
40
239
|
|
|
41
240
|
test("declared read-only builtin tools suppress implementation-word false positives", () => {
|
|
@@ -4,7 +4,7 @@ import { createRequire, syncBuiltinESMExports } from "node:module";
|
|
|
4
4
|
import * as os from "node:os";
|
|
5
5
|
import * as path from "node:path";
|
|
6
6
|
import { describe, it } from "node:test";
|
|
7
|
-
import { cleanupCompletionReplay, completionArchivePath, readCompletionArchive, readCompletionReplay, writeCompletionArchive } from "../../src/runs/background/completion-replay.ts";
|
|
7
|
+
import { cleanupCompletionReplay, completionArchivePath, completionReplayPath, readCompletionArchive, readCompletionReplay, writeCompletionReplay, writeCompletionArchive } from "../../src/runs/background/completion-replay.ts";
|
|
8
8
|
import { utf8Tail } from "../../src/shared/utf8.ts";
|
|
9
9
|
import { collectWaitCompletions, recordWaitCompletion } from "../../src/runs/background/wait-completions.ts";
|
|
10
10
|
import type { AsyncRunSummary } from "../../src/runs/background/async-status.ts";
|
|
@@ -161,6 +161,46 @@ describe("completion replay", () => {
|
|
|
161
161
|
}
|
|
162
162
|
});
|
|
163
163
|
|
|
164
|
+
it("throttles cleanup while writing completion replay records", () => {
|
|
165
|
+
const root = fs.mkdtempSync(path.join(os.tmpdir(), "pi-completion-replay-throttle-"));
|
|
166
|
+
const originalReaddirSync = fsCjs.readdirSync;
|
|
167
|
+
let cleanupScans = 0;
|
|
168
|
+
try {
|
|
169
|
+
fsCjs.readdirSync = ((target: Parameters<typeof fs.readdirSync>[0], options?: Parameters<typeof fs.readdirSync>[1]) => {
|
|
170
|
+
if (String(target).includes(`${path.sep}completion-replay`)) cleanupScans += 1;
|
|
171
|
+
return originalReaddirSync(target, options as never);
|
|
172
|
+
}) as typeof fs.readdirSync;
|
|
173
|
+
syncBuiltinESMExports();
|
|
174
|
+
|
|
175
|
+
writeCompletionReplay({
|
|
176
|
+
resultsDir: root,
|
|
177
|
+
runId: "run-a",
|
|
178
|
+
sessionId: "session-a",
|
|
179
|
+
completion: { runId: "run-a" },
|
|
180
|
+
data: { summary: "done" },
|
|
181
|
+
now: 10_000,
|
|
182
|
+
ttlMs: 60_000,
|
|
183
|
+
});
|
|
184
|
+
writeCompletionReplay({
|
|
185
|
+
resultsDir: root,
|
|
186
|
+
runId: "run-b",
|
|
187
|
+
sessionId: "session-a",
|
|
188
|
+
completion: { runId: "run-b" },
|
|
189
|
+
data: { summary: "done" },
|
|
190
|
+
now: 10_001,
|
|
191
|
+
ttlMs: 60_000,
|
|
192
|
+
});
|
|
193
|
+
|
|
194
|
+
assert.equal(cleanupScans, 1);
|
|
195
|
+
assert.equal(fs.existsSync(completionReplayPath(root, "run-a")), true);
|
|
196
|
+
assert.equal(fs.existsSync(completionReplayPath(root, "run-b")), true);
|
|
197
|
+
} finally {
|
|
198
|
+
fsCjs.readdirSync = originalReaddirSync;
|
|
199
|
+
syncBuiltinESMExports();
|
|
200
|
+
fs.rmSync(root, { recursive: true, force: true });
|
|
201
|
+
}
|
|
202
|
+
});
|
|
203
|
+
|
|
164
204
|
it("prefers saved outputs and bounds fallback output tails", () => {
|
|
165
205
|
const root = fs.mkdtempSync(path.join(os.tmpdir(), "pi-completion-archive-"));
|
|
166
206
|
try {
|