gentle-pi 2.7.0 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +24 -6
- package/assets/agents/gentle-ai-worker.md +5 -1
- package/assets/agents/sdd-apply.md +9 -7
- package/assets/agents/sdd-archive.md +42 -23
- package/assets/agents/sdd-proposal.md +2 -2
- package/assets/agents/sdd-remediate.md +4 -4
- package/assets/agents/sdd-research.md +20 -48
- package/assets/agents/sdd-tasks.md +5 -5
- package/assets/agents/sdd-verify.md +6 -28
- package/assets/chains/sdd-full.chain.md +4 -22
- package/assets/chains/sdd-verify.chain.md +3 -12
- package/assets/orchestrator-delegation.md +33 -3
- package/assets/orchestrator-memory.md +20 -7
- package/assets/orchestrator.md +5 -3
- package/assets/sdd-orchestrator-workflow.md +25 -58
- package/assets/support/sdd-status-contract.md +9 -12
- package/docs/gentle-shell.md +14 -4
- package/docs/readme-reference.md +171 -32
- package/extensions/codegraph-tools.ts +2 -0
- package/extensions/gentle-agents.ts +281 -361
- package/extensions/gentle-ai.ts +588 -117
- package/extensions/gentle-shell.ts +74 -30
- package/extensions/pi-pretty.ts +63 -14
- package/extensions/quiet-tools.ts +1 -2
- package/extensions/startup-banner.ts +10 -9
- package/lib/agent-home.ts +8 -0
- package/lib/agent-profile-pin.ts +336 -0
- package/lib/agent-profiles.ts +28 -8
- package/lib/agents-config.ts +24 -2
- package/lib/agents-history.ts +3 -97
- package/lib/agents-keys.ts +27 -0
- package/lib/agents-protocol.ts +2 -15
- package/lib/agents-runner.ts +67 -111
- package/lib/agents-session-transport.ts +691 -0
- package/lib/command-palette-catalog.ts +87 -0
- package/lib/command-palette.ts +346 -0
- package/lib/native-choice-list.ts +5 -0
- package/lib/native-review-cli.ts +10 -97
- package/lib/review-publication-gate.ts +11 -1
- package/lib/review-repository.ts +1 -1
- package/lib/review-snapshot.ts +1 -0
- package/lib/review-transaction.ts +4 -2
- package/lib/sdd-preflight.ts +2 -1
- package/lib/sdd-research-capabilities.ts +18 -152
- package/lib/sdd-status.ts +7 -779
- package/lib/session-change-capture.ts +2 -1
- package/lib/session-changes.ts +8 -1
- package/lib/shell-bar.ts +21 -12
- package/lib/shell-card.ts +8 -12
- package/lib/shell-changes.ts +3 -2
- package/lib/shell-prompt.ts +25 -8
- package/lib/shell-sidebar-banner.ts +2 -2
- package/lib/shell-sidebar-layout.ts +5 -2
- package/lib/windows-session-transport.ts +877 -0
- package/package.json +3 -3
- package/runtime/native-review-cli.mjs +9 -96
- package/runtime/windows-session-transport.ps1 +791 -0
- package/scripts/test-packed-runner.mjs +1668 -20
- package/scripts/verify-package-files.mjs +0 -1
- package/tests/agent-home.test.ts +52 -0
- package/tests/agent-profiles.test.ts +30 -1
- package/tests/agents-config.test.ts +44 -0
- package/tests/agents-history.test.ts +12 -24
- package/tests/agents-runner.test.ts +307 -58
- package/tests/agents-session-transport-process.test.ts +249 -0
- package/tests/agents-session-transport.test.ts +823 -0
- package/tests/artifact-language.test.ts +10 -7
- package/tests/command-palette.test.ts +378 -0
- package/tests/delegated-key-learnings-contract.test.ts +2 -2
- package/tests/fixtures/agents-session-transport-process.mjs +108 -0
- package/tests/fixtures/legacy/sdd-research-v2.5.0.md +54 -0
- package/tests/fixtures/windows-session-bootstrap.ps1 +129 -0
- package/tests/fixtures/windows-session-compile.ps1 +110 -0
- package/tests/gentle-agents.test.ts +849 -356
- package/tests/gentle-ai.test.ts +472 -4
- package/tests/gentle-shell.test.ts +201 -8
- package/tests/native-choice-list.test.ts +13 -0
- package/tests/native-review-cli.test.ts +0 -33
- package/tests/odd-routing-contract.test.ts +208 -0
- package/tests/orchestrator-budget.test.ts +17 -2
- package/tests/package-manifest.test.ts +115 -27
- package/tests/persona-single-channel.test.ts +3 -3
- package/tests/pi-pretty.test.ts +45 -0
- package/tests/profile-pin.test.ts +370 -0
- package/tests/quiet-tool-rendering.test.ts +32 -5
- package/tests/review-contract-prompt.test.ts +9 -0
- package/tests/review-controller.test.ts +0 -44
- package/tests/review-session-standing-permission-ipc.test.ts +427 -13
- package/tests/runtime-harness.mjs +4 -4
- package/tests/sdd-agent-tools.test.ts +15 -36
- package/tests/sdd-archive-replay.test.ts +82 -0
- package/tests/sdd-classical-continuation.test.ts +74 -0
- package/tests/sdd-execution-routing-contract.test.ts +18 -2
- package/tests/sdd-managed-runtime-settlement.test.ts +42 -330
- package/tests/sdd-native-managed-uptake.test.ts +11 -21
- package/tests/sdd-no-attempts-contract.test.ts +15 -0
- package/tests/sdd-odd-integration.test.ts +33 -0
- package/tests/sdd-optional-research.test.ts +124 -0
- package/tests/sdd-planning-routing-contract.test.ts +1 -1
- package/tests/sdd-preflight-rpc-input.test.ts +125 -0
- package/tests/sdd-preflight.test.ts +1 -1
- package/tests/sdd-research-capabilities.test.ts +20 -162
- package/tests/sdd-selection-transport.test.ts +180 -88
- package/tests/sdd-status.test.ts +5 -778
- package/tests/sdd-task-truth.test.ts +43 -0
- package/tests/session-change-capture.test.ts +20 -2
- package/tests/session-changes.test.ts +11 -0
- package/tests/shell-bar.test.ts +21 -0
- package/tests/shell-card.test.ts +8 -6
- package/tests/shell-changes.test.ts +8 -0
- package/tests/shell-prompt.test.ts +41 -7
- package/tests/shell-sidebar-banner.test.ts +4 -4
- package/tests/shell-sidebar-layout.test.ts +97 -13
- package/tests/startup-banner.test.ts +55 -2
- package/tests/windows-hidden-processes.test.ts +303 -0
- package/tests/windows-session-bootstrap.test.ts +1772 -0
- package/tests/windows-session-compile.test.ts +170 -0
- package/tests/windows-session-transport.test.ts +754 -0
- package/assets/agents/sdd-sync.md +0 -146
- package/lib/openspec-guardrails.ts +0 -99
- package/tests/native-sdd-attempt-authority.test.ts +0 -240
- package/tests/openspec-guardrails.test.ts +0 -71
|
@@ -2,8 +2,8 @@ import assert from "node:assert/strict";
|
|
|
2
2
|
import test from "node:test";
|
|
3
3
|
import { PassThrough } from "node:stream";
|
|
4
4
|
import { AGENT_MODE, parseAgentsConfig, resolveAgentProfile, type AgentDefinition } from "../lib/agents-config.ts";
|
|
5
|
-
import { TASK_STATUS, TaskStore, type
|
|
6
|
-
import { AgentRunner, childArguments, JsonLines, piCommand, abortReasonText, type
|
|
5
|
+
import { TASK_STATUS, TaskStore, type TaskRecord } from "../lib/agents-protocol.ts";
|
|
6
|
+
import { AgentRunner, childArguments, JsonLines, piCommand, abortReasonText, type ChildLike, type RunnerDeps, type RunnerHooks, type TaskRequest } from "../lib/agents-runner.ts";
|
|
7
7
|
import { fakeChild, type FakeChild } from "./agents-fake-child.ts";
|
|
8
8
|
|
|
9
9
|
// Gentle Agents runner: every subagent is a child `pi --mode rpc` process.
|
|
@@ -26,7 +26,7 @@ interface Harness {
|
|
|
26
26
|
spawnOptions: Array<{ env: NodeJS.ProcessEnv; stdio?: string[] }>;
|
|
27
27
|
}
|
|
28
28
|
|
|
29
|
-
function harness(options: { pid?: number; maxConcurrency?: number; stallTimeoutMs?: number; answer?: Record<string, unknown>; exitOnKill?: boolean; state?: Record<string, unknown>; stateSuccess?: boolean; onNotification?: RunnerHooks["onNotification"]; onSuccessfulMutation?: RunnerHooks["onSuccessfulMutation"]; onFinish?: RunnerHooks["onFinish"] } = {}): Harness {
|
|
29
|
+
function harness(options: { failStart?: boolean; process?: RunnerDeps["process"]; pid?: number; maxConcurrency?: number; stallTimeoutMs?: number; toolStallTimeoutMs?: number; answer?: Record<string, unknown>; exitOnKill?: boolean; state?: Record<string, unknown>; stateSuccess?: boolean; onNotification?: RunnerHooks["onNotification"]; onSuccessfulMutation?: RunnerHooks["onSuccessfulMutation"]; onFinish?: RunnerHooks["onFinish"] } = {}): Harness {
|
|
30
30
|
const children: FakeChild[] = [];
|
|
31
31
|
const timers: Harness["timers"] = [];
|
|
32
32
|
const asks: Harness["asks"] = [];
|
|
@@ -34,7 +34,9 @@ function harness(options: { pid?: number; maxConcurrency?: number; stallTimeoutM
|
|
|
34
34
|
const spawnOptions: Harness["spawnOptions"] = [];
|
|
35
35
|
let clock = 1000;
|
|
36
36
|
const deps: RunnerDeps = {
|
|
37
|
+
process: options.process,
|
|
37
38
|
spawn: (_command, _args, launchOptions) => {
|
|
39
|
+
if (options.failStart) throw new Error("fixture spawn failed");
|
|
38
40
|
spawnOptions.push({ env: launchOptions.env, stdio: launchOptions.stdio });
|
|
39
41
|
const fake = fakeChild({ exitOnKill: options.exitOnKill, pid: options.pid });
|
|
40
42
|
if (options.state !== undefined) {
|
|
@@ -60,7 +62,7 @@ function harness(options: { pid?: number; maxConcurrency?: number; stallTimeoutM
|
|
|
60
62
|
pi: { command: "pi", args: [] },
|
|
61
63
|
};
|
|
62
64
|
const store = new TaskStore();
|
|
63
|
-
const runner = new AgentRunner(store, { maxConcurrency: options.maxConcurrency ?? 2, stallTimeoutMs: options.stallTimeoutMs ?? 10_000 }, deps, {
|
|
65
|
+
const runner = new AgentRunner(store, { maxConcurrency: options.maxConcurrency ?? 2, stallTimeoutMs: options.stallTimeoutMs ?? 10_000, toolStallTimeoutMs: options.toolStallTimeoutMs }, deps, {
|
|
64
66
|
askUser: async (taskId, ask) => {
|
|
65
67
|
asks.push({ taskId, method: ask.method });
|
|
66
68
|
return options.answer ?? { value: "yes" };
|
|
@@ -127,7 +129,38 @@ test("stall after get_state and prompt responses records the prompt accepted sta
|
|
|
127
129
|
assert.ok(stall);
|
|
128
130
|
stall!.fn();
|
|
129
131
|
await tick();
|
|
130
|
-
assert.equal(h.store.get(task.id)?.error, "stalled for 4 min after: prompt accepted");
|
|
132
|
+
assert.equal(h.store.get(task.id)?.error, "stalled for 4 min after: prompt accepted; no first run event received for model: openai-codex/gpt-5.6-terra");
|
|
133
|
+
});
|
|
134
|
+
|
|
135
|
+
test("text progress after prompt acceptance retains the generic idle timeout diagnostic", async () => {
|
|
136
|
+
const h = harness({ stallTimeoutMs: FOUR_MIN_MS });
|
|
137
|
+
const task = h.runner.run(request());
|
|
138
|
+
await tick();
|
|
139
|
+
assert.equal(h.store.get(task.id)?.lastStep, "prompt accepted");
|
|
140
|
+
h.children[0].emit({ type: "message_update", assistantMessageEvent: { type: "text_delta", delta: "progress" } });
|
|
141
|
+
await tick();
|
|
142
|
+
assert.equal(h.store.get(task.id)?.lastStep, "prompt accepted");
|
|
143
|
+
const stall = h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).at(-1);
|
|
144
|
+
assert.ok(stall);
|
|
145
|
+
stall.fn();
|
|
146
|
+
await tick();
|
|
147
|
+
const error = h.store.get(task.id)?.error;
|
|
148
|
+
assert.doesNotMatch(error!, /no first run event received/);
|
|
149
|
+
assert.equal(error, "stalled for 4 min after: prompt accepted");
|
|
150
|
+
});
|
|
151
|
+
|
|
152
|
+
test("prompt-accepted stall names the resolved model and preserves the stderr suffix", async () => {
|
|
153
|
+
const h = harness({ stallTimeoutMs: FOUR_MIN_MS, state: { model: { provider: "resolved-provider", id: "resolved-model" } } });
|
|
154
|
+
const task = h.runner.run(request());
|
|
155
|
+
await tick();
|
|
156
|
+
assert.equal(h.store.get(task.id)?.model, "resolved-provider/resolved-model");
|
|
157
|
+
(h.children[0].child.stderr as unknown as PassThrough).write("child diagnostic\n");
|
|
158
|
+
await tick();
|
|
159
|
+
const stall = h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).at(-1);
|
|
160
|
+
assert.ok(stall);
|
|
161
|
+
stall.fn();
|
|
162
|
+
await tick();
|
|
163
|
+
assert.equal(h.store.get(task.id)?.error, "stalled for 4 min after: prompt accepted; no first run event received for model: resolved-provider/resolved-model; stderr: child diagnostic");
|
|
131
164
|
});
|
|
132
165
|
|
|
133
166
|
test("stderr tail bounds the child's raw output to 512 characters before stripping ANSI escapes", async () => {
|
|
@@ -773,20 +806,108 @@ for (const [platform, detached] of [["win32", false], ["linux", true]] as const)
|
|
|
773
806
|
assert.equal((await runner.waitFor(task.id)).status, TASK_STATUS.COMPLETED);
|
|
774
807
|
});
|
|
775
808
|
|
|
809
|
+
function ipcCleanupHarness(connected: boolean | undefined) {
|
|
810
|
+
const child = fakeChild({ exitOnKill: false });
|
|
811
|
+
const disconnectListeners: Array<(...args: unknown[]) => void> = [];
|
|
812
|
+
const originalOn = child.child.on as unknown as (event: string, listener: (...args: unknown[]) => void) => unknown;
|
|
813
|
+
child.child.on = ((event: string, listener: (...args: unknown[]) => void) => {
|
|
814
|
+
if (event === "disconnect") disconnectListeners.push(listener);
|
|
815
|
+
return originalOn(event, listener);
|
|
816
|
+
}) as ChildLike["on"];
|
|
817
|
+
child.child.connected = connected;
|
|
818
|
+
let resolveAnswer!: (answer: { cancelled: true }) => void;
|
|
819
|
+
const answer = new Promise<{ cancelled: true }>((resolve) => { resolveAnswer = resolve; });
|
|
820
|
+
const runner = new AgentRunner(new TaskStore(), { maxConcurrency: 1, stallTimeoutMs: 1_000 }, {
|
|
821
|
+
spawn: () => child.child,
|
|
822
|
+
now: () => 1,
|
|
823
|
+
schedule: () => () => {},
|
|
824
|
+
pi: { command: "pi", args: [] },
|
|
825
|
+
}, { askUser: async () => answer });
|
|
826
|
+
return {
|
|
827
|
+
child,
|
|
828
|
+
runner,
|
|
829
|
+
resolveAnswer,
|
|
830
|
+
emitNativeDisconnect: () => {
|
|
831
|
+
assert.equal(disconnectListeners.length, 1, "the runner listens for the native disconnect event");
|
|
832
|
+
disconnectListeners[0]();
|
|
833
|
+
},
|
|
834
|
+
};
|
|
835
|
+
}
|
|
836
|
+
|
|
837
|
+
test("AgentRunner primary IPC cleanup respects native connection state", async () => {
|
|
838
|
+
for (const scenario of [
|
|
839
|
+
{ name: "connected=false finalize", connected: false, ending: "finalize", expectedDisconnects: 0, reentrant: false },
|
|
840
|
+
{ name: "connected=false cancel", connected: false, ending: "cancel", expectedDisconnects: 0, reentrant: false },
|
|
841
|
+
{ name: "connected=true reentrant cleanup", connected: true, ending: "cancel", expectedDisconnects: 1, reentrant: true },
|
|
842
|
+
{ name: "partial fake without connected", connected: undefined, ending: "cancel", expectedDisconnects: 1, reentrant: false },
|
|
843
|
+
] as const) {
|
|
844
|
+
const h = ipcCleanupHarness(scenario.connected);
|
|
845
|
+
const task = h.runner.run(request());
|
|
846
|
+
await tick();
|
|
847
|
+
h.child.emit({ type: "extension_ui_request", id: "pending", method: "confirm", title: "Pending?" });
|
|
848
|
+
await tick();
|
|
849
|
+
h.emitNativeDisconnect();
|
|
850
|
+
if (scenario.reentrant) h.emitNativeDisconnect();
|
|
851
|
+
assert.equal(h.child.disconnects, scenario.expectedDisconnects, `${scenario.name}: native disconnect does not duplicate the physical close`);
|
|
852
|
+
h.resolveAnswer({ cancelled: true });
|
|
853
|
+
await tick();
|
|
854
|
+
assert.equal(h.child.written.filter((command) => command.type === "extension_ui_response").length, 1, `${scenario.name}: IPC closure does not suppress the independent live RPC UI response`);
|
|
855
|
+
if (scenario.ending === "finalize") {
|
|
856
|
+
h.child.emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "done" }], stopReason: "stop" }] });
|
|
857
|
+
h.child.emit({ type: "agent_settled" });
|
|
858
|
+
h.child.exit(0);
|
|
859
|
+
assert.equal((await h.runner.waitFor(task.id)).status, TASK_STATUS.COMPLETED, `${scenario.name}: later finalization remains intact`);
|
|
860
|
+
} else {
|
|
861
|
+
h.runner.cancel(task.id);
|
|
862
|
+
h.child.exit(0);
|
|
863
|
+
assert.equal((await h.runner.waitFor(task.id)).status, TASK_STATUS.CANCELLED, `${scenario.name}: later cancellation remains intact`);
|
|
864
|
+
}
|
|
865
|
+
assert.equal(h.child.disconnects, scenario.expectedDisconnects, `${scenario.name}: later cleanup remains idempotent`);
|
|
866
|
+
}
|
|
867
|
+
});
|
|
868
|
+
|
|
776
869
|
test("AgentRunner retains permission broker fd3 and assigns messaging IPC to fd4", async () => {
|
|
777
870
|
const { runner, children, spawnOptions } = harness();
|
|
778
871
|
const task = runner.run(request({ authorizeParentStandingReviewPermission: () => true }));
|
|
779
872
|
await tick();
|
|
780
873
|
const launch = spawnOptions[0];
|
|
874
|
+
const permissionChannelStdio = process.platform === "win32" ? "overlapped" : "pipe";
|
|
781
875
|
assert.match(launch?.env.GENTLE_PI_AGENTS_OWNED_IPC ?? "", /^\d+-[a-z0-9]+$/, "the owned-IPC marker has the runner's opaque shape");
|
|
782
876
|
assert.deepEqual(launch?.env, { GENTLE_PI_AGENTS_CHILD: "1", GENTLE_PI_AGENTS_OWNED_IPC: launch?.env.GENTLE_PI_AGENTS_OWNED_IPC, GENTLE_PI_AGENTS_PARENT_PERMISSION_FD: "3" });
|
|
783
|
-
assert.deepEqual(launch?.stdio, ["pipe", "pipe", "pipe",
|
|
877
|
+
assert.deepEqual(launch?.stdio, ["pipe", "pipe", "pipe", permissionChannelStdio, "ipc"]);
|
|
784
878
|
assert.equal(launch?.stdio?.length, 5);
|
|
785
879
|
children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "channel checked" }], stopReason: "stop" }] });
|
|
786
880
|
children[0].emit({ type: "agent_settled" });
|
|
787
881
|
assert.equal((await runner.waitFor(task.id)).status, TASK_STATUS.COMPLETED);
|
|
788
882
|
});
|
|
789
883
|
|
|
884
|
+
test("AgentRunner platform matrix scopes permission fd3 transport", async () => {
|
|
885
|
+
for (const platform of ["win32", "linux", "darwin"] as const) {
|
|
886
|
+
for (const eligible of [false, true]) {
|
|
887
|
+
const launches: Array<Parameters<RunnerDeps["spawn"]>[2]> = [];
|
|
888
|
+
const child = fakeChild();
|
|
889
|
+
const runner = new AgentRunner(new TaskStore(), { maxConcurrency: 1, stallTimeoutMs: 1_000 }, {
|
|
890
|
+
spawn: (_command, _args, options) => {
|
|
891
|
+
launches.push(options);
|
|
892
|
+
return child.child;
|
|
893
|
+
},
|
|
894
|
+
now: () => 1,
|
|
895
|
+
schedule: () => () => {},
|
|
896
|
+
pi: { command: "pi-fixture", args: [] },
|
|
897
|
+
process: { platform, kill: () => {} },
|
|
898
|
+
}, { askUser: async () => ({ cancelled: true }) });
|
|
899
|
+
const task = runner.run(request({ authorizeParentStandingReviewPermission: eligible ? () => true : undefined }));
|
|
900
|
+
await tick();
|
|
901
|
+
const launch = launches[0];
|
|
902
|
+
assert.ok(launch, `${platform} ${eligible ? "eligible" : "ineligible"} child launches`);
|
|
903
|
+
assert.equal(launch.env.GENTLE_PI_AGENTS_PARENT_PERMISSION_FD, eligible ? "3" : undefined, "only eligible children receive the fd3 marker");
|
|
904
|
+
assert.deepEqual(launch.stdio, eligible ? ["pipe", "pipe", "pipe", platform === "win32" ? "overlapped" : "pipe", "ipc"] : ["pipe", "pipe", "pipe", "ipc"]);
|
|
905
|
+
assert.equal(launch.stdio?.indexOf("ipc"), eligible ? 4 : 3, "messaging IPC follows fd3 only for eligible children");
|
|
906
|
+
runner.cancel(task.id);
|
|
907
|
+
}
|
|
908
|
+
}
|
|
909
|
+
});
|
|
910
|
+
|
|
790
911
|
test("AgentRunner answers dialogs through askUser in task mode and cancels them in background mode", async () => {
|
|
791
912
|
const { store, runner, children, asks } = harness({ answer: { confirmed: true } });
|
|
792
913
|
const task = runner.run(request());
|
|
@@ -829,8 +950,11 @@ test("AgentRunner has no total-duration watchdog but keeps active work alive and
|
|
|
829
950
|
children[0].emit({ type: "response", id: "r1", success: true });
|
|
830
951
|
await tick();
|
|
831
952
|
assert.equal(initialStall.cancelled, true, "every child RPC event, including a response, re-arms the inactivity watchdog");
|
|
953
|
+
const afterResponse = timers.filter((timer) => timer.ms === 10_000 && !timer.cancelled).at(-1);
|
|
954
|
+
assert.ok(afterResponse);
|
|
832
955
|
children[0].emit({ type: "message_update", assistantMessageEvent: { type: "text_delta", delta: "still working" } });
|
|
833
956
|
await tick();
|
|
957
|
+
assert.equal(afterResponse!.cancelled, true, "normalized task progress re-arms the inactivity watchdog");
|
|
834
958
|
assert.equal(store.get(task.id)?.status, TASK_STATUS.RUNNING, "ongoing RPC activity keeps a long-running task active");
|
|
835
959
|
const stall = timers.filter((timer) => timer.ms === 10_000 && !timer.cancelled).at(-1);
|
|
836
960
|
assert.ok(stall);
|
|
@@ -840,6 +964,104 @@ test("AgentRunner has no total-duration watchdog but keeps active work alive and
|
|
|
840
964
|
assert.match(store.get(task.id)?.error ?? "", /stalled/);
|
|
841
965
|
});
|
|
842
966
|
|
|
967
|
+
test("an announced tool call in flight arms the tool ceiling and names the tool when it fires", async () => {
|
|
968
|
+
const h = harness({ stallTimeoutMs: FOUR_MIN_MS, toolStallTimeoutMs: 30 * 60_000 });
|
|
969
|
+
const task = h.runner.run(request());
|
|
970
|
+
await tick();
|
|
971
|
+
h.children[0].emit({ type: "tool_execution_start", toolCallId: "t1", toolName: "bash", args: { command: "pnpm test" } });
|
|
972
|
+
await tick();
|
|
973
|
+
assert.equal(h.store.get(task.id)?.lastStep, "bash");
|
|
974
|
+
assert.equal(h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).length, 0, "the idle budget no longer bounds a task with a tool in flight");
|
|
975
|
+
const toolTimer = h.timers.filter((timer) => timer.ms === 30 * 60_000 && !timer.cancelled).at(-1);
|
|
976
|
+
assert.ok(toolTimer, "an in-flight tool call arms the tool ceiling");
|
|
977
|
+
toolTimer!.fn();
|
|
978
|
+
await tick();
|
|
979
|
+
assert.equal(h.store.get(task.id)?.status, TASK_STATUS.TIMED_OUT);
|
|
980
|
+
assert.equal(h.store.get(task.id)?.error, 'stalled for 30 min with tool "bash" still running after: bash');
|
|
981
|
+
});
|
|
982
|
+
|
|
983
|
+
test("a finished tool call returns the task to the idle silence budget", async () => {
|
|
984
|
+
const h = harness({ stallTimeoutMs: FOUR_MIN_MS, toolStallTimeoutMs: 30 * 60_000 });
|
|
985
|
+
const task = h.runner.run(request());
|
|
986
|
+
await tick();
|
|
987
|
+
h.children[0].emit({ type: "tool_execution_start", toolCallId: "t1", toolName: "bash", args: { command: "pnpm test" } });
|
|
988
|
+
await tick();
|
|
989
|
+
h.children[0].emit({ type: "tool_execution_end", toolCallId: "t1", isError: false });
|
|
990
|
+
await tick();
|
|
991
|
+
assert.equal(h.timers.filter((timer) => timer.ms === 30 * 60_000 && !timer.cancelled).length, 0, "a finished tool is back on the idle budget");
|
|
992
|
+
const idle = h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).at(-1);
|
|
993
|
+
assert.ok(idle, "tool_end re-arms the idle budget");
|
|
994
|
+
idle!.fn();
|
|
995
|
+
await tick();
|
|
996
|
+
assert.equal(h.store.get(task.id)?.status, TASK_STATUS.TIMED_OUT);
|
|
997
|
+
assert.equal(h.store.get(task.id)?.error, "stalled for 4 min after: bash");
|
|
998
|
+
});
|
|
999
|
+
|
|
1000
|
+
test("ignored non-dialog UI traffic does not renew the idle silence budget", async () => {
|
|
1001
|
+
const h = harness({ stallTimeoutMs: FOUR_MIN_MS, toolStallTimeoutMs: 30 * 60_000 });
|
|
1002
|
+
const task = h.runner.run(request());
|
|
1003
|
+
await tick();
|
|
1004
|
+
const armed = h.timers.at(-1);
|
|
1005
|
+
assert.ok(armed, "launch arms the idle silence budget");
|
|
1006
|
+
assert.equal(armed!.ms, FOUR_MIN_MS);
|
|
1007
|
+
// Fire-and-forget UI notifications normalize to zero task events and prove
|
|
1008
|
+
// only that the transport is alive; they must not postpone the silence bound.
|
|
1009
|
+
h.children[0].emit({ type: "extension_ui_request", id: "u1", method: "setStatus", statusKey: "fixture", statusText: "idle" });
|
|
1010
|
+
h.children[0].emit({ type: "extension_ui_request", id: "u2", method: "notify", message: "still here" });
|
|
1011
|
+
await tick();
|
|
1012
|
+
assert.equal(armed!.cancelled, false, "ignored UI traffic must not cancel the armed silence budget");
|
|
1013
|
+
assert.equal(h.timers.filter((timer) => !timer.cancelled && timer.ms === FOUR_MIN_MS).length, 1, "no replacement timer is scheduled for ignored UI traffic");
|
|
1014
|
+
assert.equal(h.timers.filter((timer) => timer.ms === 30 * 60_000).length, 0, "ignored UI traffic never earns the tool ceiling");
|
|
1015
|
+
armed!.fn();
|
|
1016
|
+
await tick();
|
|
1017
|
+
assert.equal(h.store.get(task.id)?.status, TASK_STATUS.TIMED_OUT);
|
|
1018
|
+
assert.equal(h.store.get(task.id)?.error, "stalled for 4 min after: prompt accepted; no first run event received for model: openai-codex/gpt-5.6-terra");
|
|
1019
|
+
});
|
|
1020
|
+
|
|
1021
|
+
test("an unrecognized RPC object does not renew the idle silence budget", async () => {
|
|
1022
|
+
const h = harness({ stallTimeoutMs: FOUR_MIN_MS });
|
|
1023
|
+
const task = h.runner.run(request());
|
|
1024
|
+
await tick();
|
|
1025
|
+
const armed = h.timers.at(-1);
|
|
1026
|
+
assert.ok(armed);
|
|
1027
|
+
h.children[0].emit({ type: "some_future_event", payload: { nested: true } });
|
|
1028
|
+
await tick();
|
|
1029
|
+
assert.equal(armed!.cancelled, false, "an unknown object is not progress");
|
|
1030
|
+
assert.equal(h.timers.filter((timer) => !timer.cancelled).length, 1);
|
|
1031
|
+
armed!.fn();
|
|
1032
|
+
await tick();
|
|
1033
|
+
assert.equal(h.store.get(task.id)?.status, TASK_STATUS.TIMED_OUT);
|
|
1034
|
+
});
|
|
1035
|
+
|
|
1036
|
+
test("a blocking child dialog still re-arms the idle silence budget", async () => {
|
|
1037
|
+
const h = harness({ stallTimeoutMs: FOUR_MIN_MS, answer: { confirmed: true } });
|
|
1038
|
+
h.runner.run(request());
|
|
1039
|
+
await tick();
|
|
1040
|
+
const armed = h.timers.at(-1);
|
|
1041
|
+
assert.ok(armed);
|
|
1042
|
+
h.children[0].emit({ type: "extension_ui_request", id: "u1", method: "confirm", title: "Continue?" });
|
|
1043
|
+
await tick();
|
|
1044
|
+
assert.equal(armed!.cancelled, true, "a dialog the parent must answer is meaningful activity");
|
|
1045
|
+
assert.equal(h.asks.length, 1);
|
|
1046
|
+
});
|
|
1047
|
+
|
|
1048
|
+
test("the tool ceiling holds while any announced tool call is still in flight", async () => {
|
|
1049
|
+
const h = harness({ stallTimeoutMs: FOUR_MIN_MS, toolStallTimeoutMs: 30 * 60_000 });
|
|
1050
|
+
h.runner.run(request());
|
|
1051
|
+
await tick();
|
|
1052
|
+
h.children[0].emit({ type: "tool_execution_start", toolCallId: "t1", toolName: "bash", args: { command: "pnpm test" } });
|
|
1053
|
+
await tick();
|
|
1054
|
+
h.children[0].emit({ type: "tool_execution_start", toolCallId: "t2", toolName: "bash", args: { command: "pnpm run typecheck" } });
|
|
1055
|
+
await tick();
|
|
1056
|
+
h.children[0].emit({ type: "tool_execution_end", toolCallId: "t1", isError: false });
|
|
1057
|
+
await tick();
|
|
1058
|
+
assert.ok(h.timers.filter((timer) => timer.ms === 30 * 60_000 && !timer.cancelled).length > 0, "a second tool still in flight keeps the tool ceiling");
|
|
1059
|
+
assert.equal(h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).length, 0);
|
|
1060
|
+
h.children[0].emit({ type: "tool_execution_end", toolCallId: "t2", isError: false });
|
|
1061
|
+
await tick();
|
|
1062
|
+
assert.ok(h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).length > 0, "ending the last tool returns to the idle budget");
|
|
1063
|
+
});
|
|
1064
|
+
|
|
843
1065
|
test("AgentRunner.cancelAll stops every queued and running task", async () => {
|
|
844
1066
|
const { store, runner, children } = harness({ maxConcurrency: 1 });
|
|
845
1067
|
const running = runner.run(request());
|
|
@@ -1012,8 +1234,9 @@ test("an unprobeable process group quarantines at its deadline and still records
|
|
|
1012
1234
|
const finishes: string[] = [];
|
|
1013
1235
|
let now = 1_000;
|
|
1014
1236
|
let launches = 0;
|
|
1237
|
+
let child: FakeChild;
|
|
1015
1238
|
const runner = new AgentRunner(store, { maxConcurrency: 1, stallTimeoutMs: 10_000 }, {
|
|
1016
|
-
spawn: () => fakeChild({ exitOnKill: false, pid: 90 + (launches += 1) }).child, now: () => now,
|
|
1239
|
+
spawn: () => (child = fakeChild({ exitOnKill: false, pid: 90 + (launches += 1) })).child, now: () => now,
|
|
1017
1240
|
schedule: (fn, ms) => {
|
|
1018
1241
|
const timer = { fn, ms, cancelled: false };
|
|
1019
1242
|
timers.push(timer);
|
|
@@ -1022,7 +1245,7 @@ test("an unprobeable process group quarantines at its deadline and still records
|
|
|
1022
1245
|
pi: { command: "pi", args: [] },
|
|
1023
1246
|
process: { platform: "win32", kill: () => {} },
|
|
1024
1247
|
}, { askUser: async () => ({ value: "yes" }), onFinish: (task) => { finishes.push(task.id); } });
|
|
1025
|
-
const first = runner.run(
|
|
1248
|
+
const first = runner.run(managedRequest());
|
|
1026
1249
|
const second = runner.run(request({ prompt: "queued" }));
|
|
1027
1250
|
await tick();
|
|
1028
1251
|
runner.cancel(first.id);
|
|
@@ -1039,6 +1262,12 @@ test("an unprobeable process group quarantines at its deadline and still records
|
|
|
1039
1262
|
assert.equal(finishes.length, 1, "the run is recorded exactly once");
|
|
1040
1263
|
assert.equal(store.get(second.id)?.status, TASK_STATUS.QUEUED, "an unconfirmed exit retains its capacity");
|
|
1041
1264
|
assert.equal(launches, 1, "no further launch happens while the slot is quarantined");
|
|
1265
|
+
assert.throws(() => runner.run(managedRequest()), /Remediation already queued or running/, "a failed record does not release its quarantined child");
|
|
1266
|
+
child!.exit(0);
|
|
1267
|
+
await tick();
|
|
1268
|
+
assert.doesNotThrow(() => runner.run(managedRequest()), "confirmed cleanup releases the managed workspace");
|
|
1269
|
+
runner.cancelAll();
|
|
1270
|
+
child!.exit(0);
|
|
1042
1271
|
});
|
|
1043
1272
|
|
|
1044
1273
|
test("abortReasonText renders an Error, a string, and nothing for unknown reasons", () => {
|
|
@@ -1053,17 +1282,12 @@ test("research narrowing transport keeps exact argv paths and replaces inherited
|
|
|
1053
1282
|
const h = harness();
|
|
1054
1283
|
const selection = { documentation: { tools: ["fetch_content"], extensions: { fetch_content: "/installed/docs tools.ts" } } };
|
|
1055
1284
|
for (const researchSelection of [selection, undefined]) {
|
|
1056
|
-
const
|
|
1057
|
-
|
|
1058
|
-
const launch = request({ researchSelection, researchArtifact: artifact, extensionPaths: researchSelection ? ["/installed/docs tools.ts"] : [],
|
|
1059
|
-
env: { PATH: "/bin", GENTLE_PI_RESEARCH_SELECTION: "stale broad selection", GENTLE_PI_RESEARCH_ARTIFACT: "stale broader scope" } });
|
|
1285
|
+
const launch = request({ researchSelection, extensionPaths: researchSelection ? ["/installed/docs tools.ts"] : [],
|
|
1286
|
+
env: { PATH: "/bin", GENTLE_PI_RESEARCH_SELECTION: "stale broad selection" } });
|
|
1060
1287
|
const argv = childArguments(launch);
|
|
1061
1288
|
assert.deepEqual(argv.filter((_, i) => argv[i - 1] === "--extension"), launch.extensionPaths);
|
|
1062
1289
|
const task = h.runner.run(launch);
|
|
1063
|
-
artifact.worktree = "/wrong";
|
|
1064
1290
|
await tick();
|
|
1065
|
-
assert.deepEqual(JSON.parse(h.spawnOptions.at(-1)!.env.GENTLE_PI_RESEARCH_ARTIFACT!), expected);
|
|
1066
|
-
assert.deepEqual("researchArtifact" in task ? task.researchArtifact : undefined, expected);
|
|
1067
1291
|
assert.deepEqual(JSON.parse(h.spawnOptions.at(-1)!.env.GENTLE_PI_RESEARCH_SELECTION!), researchSelection ?? null);
|
|
1068
1292
|
assert.equal(h.spawnOptions.at(-1)!.env.PATH, "/bin");
|
|
1069
1293
|
h.runner.cancel(task.id);
|
|
@@ -1071,55 +1295,80 @@ test("research narrowing transport keeps exact argv paths and replaces inherited
|
|
|
1071
1295
|
}
|
|
1072
1296
|
});
|
|
1073
1297
|
|
|
1298
|
+
function managedRequest(cwd = "/repo"): TaskRequest {
|
|
1299
|
+
return request({ agent: { ...explorer, name: "sdd-remediate" }, cwd, sddRemediation: {
|
|
1300
|
+
failedEvidenceRevision: "failed-revision",
|
|
1301
|
+
plan: { cwd, commands: ["pnpm test"], runtimeHarness: { naReason: "Not applicable because this tests runner admission." }, rollback: { boundary: "fixture", command: "git diff --check" } },
|
|
1302
|
+
scope: { cwd, editPaths: [], commands: ["pnpm test", "git diff --check"], allowedEditRoots: [cwd] },
|
|
1303
|
+
} });
|
|
1304
|
+
}
|
|
1074
1305
|
|
|
1075
|
-
test(
|
|
1076
|
-
const h = harness({ pid: 123 });
|
|
1077
|
-
|
|
1078
|
-
|
|
1079
|
-
|
|
1080
|
-
|
|
1081
|
-
|
|
1082
|
-
|
|
1083
|
-
|
|
1084
|
-
|
|
1085
|
-
|
|
1086
|
-
h.children[0].emit({ type: "tool_execution_end", toolName: "bash", toolCallId: command, isError: false, result: { content: [{ type: "text", text: "command output" }], details: { remediationCommand: { command, toolCallId: command, cwd: "/repo", exitCode: 0 } } } });
|
|
1087
|
-
}
|
|
1088
|
-
h.children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "done" }], stopReason: "stop" }] });
|
|
1089
|
-
h.children[0].emit({ type: "agent_settled" });
|
|
1090
|
-
await tick();
|
|
1091
|
-
assert.ok(finalized);
|
|
1092
|
-
assert.equal(finalized.record.sddRemediation?.observations.length, 2);
|
|
1093
|
-
const prompt = h.children[0].written.find(value => value.type === "prompt").message;
|
|
1094
|
-
assert.ok(typeof prompt === "string");
|
|
1095
|
-
assert.match(prompt, /pnpm test/);
|
|
1096
|
-
assert.doesNotMatch(prompt, /opaque/);
|
|
1097
|
-
assert.equal(finalized.facts.cleanupConfirmed, true);
|
|
1098
|
-
assert.equal(h.finishes.length, 0);
|
|
1099
|
-
release(); await h.runner.waitFor(task.id);
|
|
1100
|
-
assert.equal(h.finishes.length, 1);
|
|
1101
|
-
assert.equal(remediation.observations.length, 0);
|
|
1306
|
+
for (const queued of [true, false]) test(`managed exclusion covers ${queued ? "queued" : "running"} same-workspace actors`, async () => {
|
|
1307
|
+
const h = harness({ pid: 123, process: { platform: "win32", kill() {} } });
|
|
1308
|
+
const first = h.runner.run(managedRequest());
|
|
1309
|
+
if (!queued) await tick();
|
|
1310
|
+
try {
|
|
1311
|
+
assert.throws(() => h.runner.run(managedRequest()), /Remediation already queued or running/);
|
|
1312
|
+
assert.equal(h.store.list().length, 1, "rejection creates no task or queue entry");
|
|
1313
|
+
await tick();
|
|
1314
|
+
assert.equal(h.children.length, 1);
|
|
1315
|
+
assert.equal(h.store.get(first.id)?.status, TASK_STATUS.RUNNING);
|
|
1316
|
+
} finally { h.runner.cancelAll(); await tick(); }
|
|
1102
1317
|
});
|
|
1103
1318
|
|
|
1319
|
+
test("managed exclusion does not serialize other workspaces or ordinary tasks", async () => {
|
|
1320
|
+
const h = harness({ maxConcurrency: 3, pid: 123, process: { platform: "win32", kill() {} } });
|
|
1321
|
+
h.runner.run(managedRequest());
|
|
1322
|
+
h.runner.run(managedRequest("/other"));
|
|
1323
|
+
h.runner.run(request());
|
|
1324
|
+
await tick();
|
|
1325
|
+
assert.equal(h.children.length, 3);
|
|
1326
|
+
h.runner.cancelAll();
|
|
1327
|
+
await tick();
|
|
1328
|
+
});
|
|
1104
1329
|
|
|
1105
|
-
|
|
1106
|
-
const h = harness()
|
|
1107
|
-
const
|
|
1330
|
+
for (const ending of ["complete", "failure", "cancel", "queued-cancel"] as const) test(`managed exclusion releases after ${ending}`, async () => {
|
|
1331
|
+
const h = harness({ pid: 123, process: { platform: "win32", kill() {} } });
|
|
1332
|
+
const first = h.runner.run(managedRequest());
|
|
1333
|
+
if (ending === "queued-cancel") h.runner.cancel(first.id);
|
|
1334
|
+
else {
|
|
1335
|
+
await tick();
|
|
1336
|
+
if (ending === "complete") {
|
|
1337
|
+
h.children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "Finished" }], stopReason: "stop" }] });
|
|
1338
|
+
h.children[0].emit({ type: "agent_settled" });
|
|
1339
|
+
} else if (ending === "failure") h.children[0].exit(1);
|
|
1340
|
+
else h.runner.cancel(first.id);
|
|
1341
|
+
}
|
|
1342
|
+
await h.runner.waitFor(first.id);
|
|
1343
|
+
const next = h.runner.run(managedRequest());
|
|
1344
|
+
await tick();
|
|
1345
|
+
assert.equal(h.store.get(next.id)?.status, TASK_STATUS.RUNNING);
|
|
1346
|
+
h.runner.cancelAll();
|
|
1108
1347
|
await tick();
|
|
1109
|
-
assert.equal(h.children[0].written.some(value => value.type === "prompt"), false);
|
|
1110
|
-
await h.runner.waitFor(task.id);
|
|
1111
|
-
assert.equal(facts.spawned, false);
|
|
1112
|
-
assert.equal(facts.exited, false);
|
|
1113
|
-
assert.equal(facts.cleanupConfirmed, false);
|
|
1114
1348
|
});
|
|
1115
1349
|
|
|
1350
|
+
test("managed exclusion lasts until child cleanup is confirmed", async () => {
|
|
1351
|
+
const h = harness({ pid: 123, exitOnKill: false, process: { platform: "win32", kill() {} } });
|
|
1352
|
+
const first = h.runner.run(managedRequest());
|
|
1353
|
+
await tick();
|
|
1354
|
+
h.runner.cancel(first.id);
|
|
1355
|
+
try { assert.throws(() => h.runner.run(managedRequest()), /Remediation already queued or running/); }
|
|
1356
|
+
finally { h.children[0].exit(0); }
|
|
1357
|
+
await h.runner.waitFor(first.id);
|
|
1358
|
+
const next = h.runner.run(managedRequest());
|
|
1359
|
+
await tick();
|
|
1360
|
+
assert.equal(h.store.get(next.id)?.status, TASK_STATUS.RUNNING);
|
|
1361
|
+
h.runner.cancelAll();
|
|
1362
|
+
for (const child of h.children) child.exit(0);
|
|
1363
|
+
await tick();
|
|
1364
|
+
});
|
|
1116
1365
|
|
|
1117
|
-
test("
|
|
1118
|
-
|
|
1119
|
-
const
|
|
1120
|
-
|
|
1121
|
-
|
|
1122
|
-
|
|
1123
|
-
assert.
|
|
1124
|
-
assert.
|
|
1366
|
+
test("managed exclusion releases failed startup and ignores historical-only tasks", async () => {
|
|
1367
|
+
const h = harness({ failStart: true });
|
|
1368
|
+
const first = h.runner.run(managedRequest());
|
|
1369
|
+
assert.equal((await h.runner.waitFor(first.id)).status, TASK_STATUS.FAILED);
|
|
1370
|
+
h.store.add({ ...h.store.get(first.id)!, id: "historical-only", status: TASK_STATUS.RUNNING });
|
|
1371
|
+
const next = h.runner.run(managedRequest());
|
|
1372
|
+
assert.equal((await h.runner.waitFor(next.id)).status, TASK_STATUS.FAILED, "the next actor reaches spawn, not a historical admission lock");
|
|
1373
|
+
assert.match(h.store.get(next.id)?.error ?? "", /fixture spawn failed/);
|
|
1125
1374
|
});
|