gentle-pi 2.6.4 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +25 -7
- package/assets/agents/gentle-ai-worker.md +5 -1
- package/assets/agents/sdd-apply.md +9 -7
- package/assets/agents/sdd-archive.md +42 -23
- package/assets/agents/sdd-proposal.md +2 -2
- package/assets/agents/sdd-remediate.md +4 -4
- package/assets/agents/sdd-research.md +20 -48
- package/assets/agents/sdd-tasks.md +5 -5
- package/assets/agents/sdd-verify.md +6 -28
- package/assets/chains/sdd-full.chain.md +4 -22
- package/assets/chains/sdd-verify.chain.md +3 -12
- package/assets/orchestrator-delegation.md +33 -3
- package/assets/orchestrator-memory.md +20 -7
- package/assets/orchestrator.md +5 -3
- package/assets/sdd-orchestrator-workflow.md +25 -58
- package/assets/support/sdd-status-contract.md +9 -12
- package/docs/gentle-shell.md +41 -19
- package/docs/readme-reference.md +175 -36
- package/extensions/codegraph-tools.ts +2 -0
- package/extensions/gentle-agents.ts +289 -361
- package/extensions/gentle-ai.ts +588 -117
- package/extensions/gentle-shell.ts +123 -97
- package/extensions/pi-pretty.ts +63 -14
- package/extensions/quiet-tools.ts +1 -2
- package/extensions/startup-banner.ts +10 -9
- package/lib/agent-home.ts +8 -0
- package/lib/agent-profile-pin.ts +336 -0
- package/lib/agent-profiles.ts +28 -8
- package/lib/agents-config.ts +24 -2
- package/lib/agents-history.ts +3 -97
- package/lib/agents-keys.ts +27 -0
- package/lib/agents-protocol.ts +2 -15
- package/lib/agents-runner.ts +74 -113
- package/lib/agents-session-transport.ts +691 -0
- package/lib/command-palette-catalog.ts +87 -0
- package/lib/command-palette.ts +346 -0
- package/lib/native-choice-list.ts +5 -0
- package/lib/native-review-cli.ts +19 -97
- package/lib/review-publication-gate.ts +11 -1
- package/lib/review-repository.ts +1 -1
- package/lib/review-snapshot.ts +1 -0
- package/lib/review-transaction.ts +4 -2
- package/lib/sdd-preflight.ts +2 -1
- package/lib/sdd-research-capabilities.ts +18 -152
- package/lib/sdd-status.ts +7 -779
- package/lib/session-change-capture.ts +88 -0
- package/lib/session-changes.ts +147 -0
- package/lib/shell-bar.ts +24 -10
- package/lib/shell-card.ts +8 -12
- package/lib/shell-changes-view.ts +2 -1
- package/lib/shell-changes.ts +5 -2
- package/lib/shell-prompt.ts +25 -8
- package/lib/shell-sidebar-banner.ts +2 -2
- package/lib/shell-sidebar-layout.ts +5 -2
- package/lib/windows-session-transport.ts +877 -0
- package/package.json +3 -3
- package/runtime/native-review-cli.mjs +18 -96
- package/runtime/windows-session-transport.ps1 +791 -0
- package/scripts/gentle-ai-installer.mjs +10 -10
- package/scripts/test-packed-runner.mjs +1668 -20
- package/scripts/verify-package-files.mjs +2 -3
- package/tests/agent-home.test.ts +52 -0
- package/tests/agent-profiles.test.ts +30 -1
- package/tests/agents-config.test.ts +44 -0
- package/tests/agents-history.test.ts +12 -24
- package/tests/agents-runner.test.ts +321 -58
- package/tests/agents-session-transport-process.test.ts +249 -0
- package/tests/agents-session-transport.test.ts +823 -0
- package/tests/artifact-language.test.ts +10 -7
- package/tests/command-palette.test.ts +378 -0
- package/tests/delegated-key-learnings-contract.test.ts +2 -2
- package/tests/fixtures/agents-session-transport-process.mjs +108 -0
- package/tests/fixtures/legacy/sdd-research-v2.5.0.md +54 -0
- package/tests/fixtures/windows-session-bootstrap.ps1 +129 -0
- package/tests/fixtures/windows-session-compile.ps1 +110 -0
- package/tests/gentle-agents.test.ts +883 -356
- package/tests/gentle-ai-binary.test.ts +1 -1
- package/tests/gentle-ai-installer.test.ts +47 -47
- package/tests/gentle-ai.test.ts +472 -4
- package/tests/gentle-shell.test.ts +338 -205
- package/tests/native-choice-list.test.ts +13 -0
- package/tests/native-review-capability-contract.test.ts +15 -1
- package/tests/native-review-cli.test.ts +0 -33
- package/tests/odd-routing-contract.test.ts +208 -0
- package/tests/orchestrator-budget.test.ts +17 -2
- package/tests/package-manifest.test.ts +119 -31
- package/tests/persona-single-channel.test.ts +3 -3
- package/tests/pi-pretty.test.ts +45 -0
- package/tests/profile-pin.test.ts +370 -0
- package/tests/quiet-tool-rendering.test.ts +32 -5
- package/tests/review-contract-prompt.test.ts +9 -0
- package/tests/review-controller.test.ts +0 -44
- package/tests/review-session-standing-permission-ipc.test.ts +427 -13
- package/tests/runtime-harness.mjs +4 -4
- package/tests/sdd-agent-tools.test.ts +15 -36
- package/tests/sdd-archive-replay.test.ts +82 -0
- package/tests/sdd-classical-continuation.test.ts +74 -0
- package/tests/sdd-execution-routing-contract.test.ts +18 -2
- package/tests/sdd-managed-runtime-settlement.test.ts +42 -330
- package/tests/sdd-native-managed-uptake.test.ts +11 -21
- package/tests/sdd-no-attempts-contract.test.ts +15 -0
- package/tests/sdd-odd-integration.test.ts +33 -0
- package/tests/sdd-optional-research.test.ts +124 -0
- package/tests/sdd-planning-routing-contract.test.ts +1 -1
- package/tests/sdd-preflight-rpc-input.test.ts +125 -0
- package/tests/sdd-preflight.test.ts +1 -1
- package/tests/sdd-research-capabilities.test.ts +20 -162
- package/tests/sdd-selection-transport.test.ts +180 -88
- package/tests/sdd-status.test.ts +5 -778
- package/tests/sdd-task-truth.test.ts +43 -0
- package/tests/session-change-capture.test.ts +86 -0
- package/tests/session-changes-shell.test.ts +38 -0
- package/tests/session-changes.test.ts +114 -0
- package/tests/shell-bar.test.ts +35 -0
- package/tests/shell-card.test.ts +8 -6
- package/tests/shell-changes.test.ts +8 -0
- package/tests/shell-prompt.test.ts +41 -7
- package/tests/shell-sidebar-banner.test.ts +4 -4
- package/tests/shell-sidebar-layout.test.ts +97 -13
- package/tests/startup-banner.test.ts +55 -2
- package/tests/windows-hidden-processes.test.ts +303 -0
- package/tests/windows-session-bootstrap.test.ts +1772 -0
- package/tests/windows-session-compile.test.ts +170 -0
- package/tests/windows-session-transport.test.ts +754 -0
- package/assets/agents/sdd-sync.md +0 -146
- package/lib/openspec-guardrails.ts +0 -99
- package/tests/native-sdd-attempt-authority.test.ts +0 -240
- package/tests/openspec-guardrails.test.ts +0 -71
|
@@ -2,8 +2,8 @@ import assert from "node:assert/strict";
|
|
|
2
2
|
import test from "node:test";
|
|
3
3
|
import { PassThrough } from "node:stream";
|
|
4
4
|
import { AGENT_MODE, parseAgentsConfig, resolveAgentProfile, type AgentDefinition } from "../lib/agents-config.ts";
|
|
5
|
-
import { TASK_STATUS, TaskStore, type
|
|
6
|
-
import { AgentRunner, childArguments, JsonLines, piCommand, abortReasonText, type
|
|
5
|
+
import { TASK_STATUS, TaskStore, type TaskRecord } from "../lib/agents-protocol.ts";
|
|
6
|
+
import { AgentRunner, childArguments, JsonLines, piCommand, abortReasonText, type ChildLike, type RunnerDeps, type RunnerHooks, type TaskRequest } from "../lib/agents-runner.ts";
|
|
7
7
|
import { fakeChild, type FakeChild } from "./agents-fake-child.ts";
|
|
8
8
|
|
|
9
9
|
// Gentle Agents runner: every subagent is a child `pi --mode rpc` process.
|
|
@@ -26,7 +26,7 @@ interface Harness {
|
|
|
26
26
|
spawnOptions: Array<{ env: NodeJS.ProcessEnv; stdio?: string[] }>;
|
|
27
27
|
}
|
|
28
28
|
|
|
29
|
-
function harness(options: { pid?: number; maxConcurrency?: number; stallTimeoutMs?: number; answer?: Record<string, unknown>; exitOnKill?: boolean; state?: Record<string, unknown>; stateSuccess?: boolean; onNotification?: RunnerHooks["onNotification"]; onSuccessfulMutation?: RunnerHooks["onSuccessfulMutation"]; onFinish?: RunnerHooks["onFinish"] } = {}): Harness {
|
|
29
|
+
function harness(options: { failStart?: boolean; process?: RunnerDeps["process"]; pid?: number; maxConcurrency?: number; stallTimeoutMs?: number; toolStallTimeoutMs?: number; answer?: Record<string, unknown>; exitOnKill?: boolean; state?: Record<string, unknown>; stateSuccess?: boolean; onNotification?: RunnerHooks["onNotification"]; onSuccessfulMutation?: RunnerHooks["onSuccessfulMutation"]; onFinish?: RunnerHooks["onFinish"] } = {}): Harness {
|
|
30
30
|
const children: FakeChild[] = [];
|
|
31
31
|
const timers: Harness["timers"] = [];
|
|
32
32
|
const asks: Harness["asks"] = [];
|
|
@@ -34,7 +34,9 @@ function harness(options: { pid?: number; maxConcurrency?: number; stallTimeoutM
|
|
|
34
34
|
const spawnOptions: Harness["spawnOptions"] = [];
|
|
35
35
|
let clock = 1000;
|
|
36
36
|
const deps: RunnerDeps = {
|
|
37
|
+
process: options.process,
|
|
37
38
|
spawn: (_command, _args, launchOptions) => {
|
|
39
|
+
if (options.failStart) throw new Error("fixture spawn failed");
|
|
38
40
|
spawnOptions.push({ env: launchOptions.env, stdio: launchOptions.stdio });
|
|
39
41
|
const fake = fakeChild({ exitOnKill: options.exitOnKill, pid: options.pid });
|
|
40
42
|
if (options.state !== undefined) {
|
|
@@ -60,7 +62,7 @@ function harness(options: { pid?: number; maxConcurrency?: number; stallTimeoutM
|
|
|
60
62
|
pi: { command: "pi", args: [] },
|
|
61
63
|
};
|
|
62
64
|
const store = new TaskStore();
|
|
63
|
-
const runner = new AgentRunner(store, { maxConcurrency: options.maxConcurrency ?? 2, stallTimeoutMs: options.stallTimeoutMs ?? 10_000 }, deps, {
|
|
65
|
+
const runner = new AgentRunner(store, { maxConcurrency: options.maxConcurrency ?? 2, stallTimeoutMs: options.stallTimeoutMs ?? 10_000, toolStallTimeoutMs: options.toolStallTimeoutMs }, deps, {
|
|
64
66
|
askUser: async (taskId, ask) => {
|
|
65
67
|
asks.push({ taskId, method: ask.method });
|
|
66
68
|
return options.answer ?? { value: "yes" };
|
|
@@ -127,7 +129,38 @@ test("stall after get_state and prompt responses records the prompt accepted sta
|
|
|
127
129
|
assert.ok(stall);
|
|
128
130
|
stall!.fn();
|
|
129
131
|
await tick();
|
|
130
|
-
assert.equal(h.store.get(task.id)?.error, "stalled for 4 min after: prompt accepted");
|
|
132
|
+
assert.equal(h.store.get(task.id)?.error, "stalled for 4 min after: prompt accepted; no first run event received for model: openai-codex/gpt-5.6-terra");
|
|
133
|
+
});
|
|
134
|
+
|
|
135
|
+
test("text progress after prompt acceptance retains the generic idle timeout diagnostic", async () => {
|
|
136
|
+
const h = harness({ stallTimeoutMs: FOUR_MIN_MS });
|
|
137
|
+
const task = h.runner.run(request());
|
|
138
|
+
await tick();
|
|
139
|
+
assert.equal(h.store.get(task.id)?.lastStep, "prompt accepted");
|
|
140
|
+
h.children[0].emit({ type: "message_update", assistantMessageEvent: { type: "text_delta", delta: "progress" } });
|
|
141
|
+
await tick();
|
|
142
|
+
assert.equal(h.store.get(task.id)?.lastStep, "prompt accepted");
|
|
143
|
+
const stall = h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).at(-1);
|
|
144
|
+
assert.ok(stall);
|
|
145
|
+
stall.fn();
|
|
146
|
+
await tick();
|
|
147
|
+
const error = h.store.get(task.id)?.error;
|
|
148
|
+
assert.doesNotMatch(error!, /no first run event received/);
|
|
149
|
+
assert.equal(error, "stalled for 4 min after: prompt accepted");
|
|
150
|
+
});
|
|
151
|
+
|
|
152
|
+
test("prompt-accepted stall names the resolved model and preserves the stderr suffix", async () => {
|
|
153
|
+
const h = harness({ stallTimeoutMs: FOUR_MIN_MS, state: { model: { provider: "resolved-provider", id: "resolved-model" } } });
|
|
154
|
+
const task = h.runner.run(request());
|
|
155
|
+
await tick();
|
|
156
|
+
assert.equal(h.store.get(task.id)?.model, "resolved-provider/resolved-model");
|
|
157
|
+
(h.children[0].child.stderr as unknown as PassThrough).write("child diagnostic\n");
|
|
158
|
+
await tick();
|
|
159
|
+
const stall = h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).at(-1);
|
|
160
|
+
assert.ok(stall);
|
|
161
|
+
stall.fn();
|
|
162
|
+
await tick();
|
|
163
|
+
assert.equal(h.store.get(task.id)?.error, "stalled for 4 min after: prompt accepted; no first run event received for model: resolved-provider/resolved-model; stderr: child diagnostic");
|
|
131
164
|
});
|
|
132
165
|
|
|
133
166
|
test("stderr tail bounds the child's raw output to 512 characters before stripping ANSI escapes", async () => {
|
|
@@ -321,6 +354,20 @@ test("child observation guard is never consulted when collection is default-off"
|
|
|
321
354
|
assert.equal(calls, 0);
|
|
322
355
|
});
|
|
323
356
|
|
|
357
|
+
test("child session diff evidence travels only with a paired successful tool outcome", async () => {
|
|
358
|
+
const observed: any[] = [];
|
|
359
|
+
const h = harness({ onSuccessfulMutation: (_task, tool) => { observed.push(tool); } });
|
|
360
|
+
const task = h.runner.run(request()); await tick();
|
|
361
|
+
const child = h.children[0];
|
|
362
|
+
const evidence = {id:"w",root:"/repo",path:"src/file.ts",before:{kind:"absent"},after:{kind:"text",text:"agent\n"}};
|
|
363
|
+
child.emit({type:"tool_execution_start",toolCallId:"w",toolName:"write",args:{path:"src/file.ts"}});
|
|
364
|
+
child.emit({type:"tool_execution_end",toolCallId:"w",isError:false,result:{content:[],details:{gentleSessionChange:evidence}}});
|
|
365
|
+
assert.deepEqual(observed[0].evidence,evidence);
|
|
366
|
+
child.emit({type:"tool_execution_end",toolCallId:"w",isError:false,result:{content:[],details:{gentleSessionChange:evidence}}});
|
|
367
|
+
assert.equal(observed.length,1);
|
|
368
|
+
h.runner.cancel(task.id); await tick();
|
|
369
|
+
});
|
|
370
|
+
|
|
324
371
|
for (const ending of ["cancel", "failure", "hook-error", "hook-async-error"] as const) {
|
|
325
372
|
test(`successful child mutations require paired RPC events and survive ${ending}`, async () => {
|
|
326
373
|
const mutations: unknown[] = [];
|
|
@@ -759,20 +806,108 @@ for (const [platform, detached] of [["win32", false], ["linux", true]] as const)
|
|
|
759
806
|
assert.equal((await runner.waitFor(task.id)).status, TASK_STATUS.COMPLETED);
|
|
760
807
|
});
|
|
761
808
|
|
|
809
|
+
function ipcCleanupHarness(connected: boolean | undefined) {
|
|
810
|
+
const child = fakeChild({ exitOnKill: false });
|
|
811
|
+
const disconnectListeners: Array<(...args: unknown[]) => void> = [];
|
|
812
|
+
const originalOn = child.child.on as unknown as (event: string, listener: (...args: unknown[]) => void) => unknown;
|
|
813
|
+
child.child.on = ((event: string, listener: (...args: unknown[]) => void) => {
|
|
814
|
+
if (event === "disconnect") disconnectListeners.push(listener);
|
|
815
|
+
return originalOn(event, listener);
|
|
816
|
+
}) as ChildLike["on"];
|
|
817
|
+
child.child.connected = connected;
|
|
818
|
+
let resolveAnswer!: (answer: { cancelled: true }) => void;
|
|
819
|
+
const answer = new Promise<{ cancelled: true }>((resolve) => { resolveAnswer = resolve; });
|
|
820
|
+
const runner = new AgentRunner(new TaskStore(), { maxConcurrency: 1, stallTimeoutMs: 1_000 }, {
|
|
821
|
+
spawn: () => child.child,
|
|
822
|
+
now: () => 1,
|
|
823
|
+
schedule: () => () => {},
|
|
824
|
+
pi: { command: "pi", args: [] },
|
|
825
|
+
}, { askUser: async () => answer });
|
|
826
|
+
return {
|
|
827
|
+
child,
|
|
828
|
+
runner,
|
|
829
|
+
resolveAnswer,
|
|
830
|
+
emitNativeDisconnect: () => {
|
|
831
|
+
assert.equal(disconnectListeners.length, 1, "the runner listens for the native disconnect event");
|
|
832
|
+
disconnectListeners[0]();
|
|
833
|
+
},
|
|
834
|
+
};
|
|
835
|
+
}
|
|
836
|
+
|
|
837
|
+
test("AgentRunner primary IPC cleanup respects native connection state", async () => {
|
|
838
|
+
for (const scenario of [
|
|
839
|
+
{ name: "connected=false finalize", connected: false, ending: "finalize", expectedDisconnects: 0, reentrant: false },
|
|
840
|
+
{ name: "connected=false cancel", connected: false, ending: "cancel", expectedDisconnects: 0, reentrant: false },
|
|
841
|
+
{ name: "connected=true reentrant cleanup", connected: true, ending: "cancel", expectedDisconnects: 1, reentrant: true },
|
|
842
|
+
{ name: "partial fake without connected", connected: undefined, ending: "cancel", expectedDisconnects: 1, reentrant: false },
|
|
843
|
+
] as const) {
|
|
844
|
+
const h = ipcCleanupHarness(scenario.connected);
|
|
845
|
+
const task = h.runner.run(request());
|
|
846
|
+
await tick();
|
|
847
|
+
h.child.emit({ type: "extension_ui_request", id: "pending", method: "confirm", title: "Pending?" });
|
|
848
|
+
await tick();
|
|
849
|
+
h.emitNativeDisconnect();
|
|
850
|
+
if (scenario.reentrant) h.emitNativeDisconnect();
|
|
851
|
+
assert.equal(h.child.disconnects, scenario.expectedDisconnects, `${scenario.name}: native disconnect does not duplicate the physical close`);
|
|
852
|
+
h.resolveAnswer({ cancelled: true });
|
|
853
|
+
await tick();
|
|
854
|
+
assert.equal(h.child.written.filter((command) => command.type === "extension_ui_response").length, 1, `${scenario.name}: IPC closure does not suppress the independent live RPC UI response`);
|
|
855
|
+
if (scenario.ending === "finalize") {
|
|
856
|
+
h.child.emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "done" }], stopReason: "stop" }] });
|
|
857
|
+
h.child.emit({ type: "agent_settled" });
|
|
858
|
+
h.child.exit(0);
|
|
859
|
+
assert.equal((await h.runner.waitFor(task.id)).status, TASK_STATUS.COMPLETED, `${scenario.name}: later finalization remains intact`);
|
|
860
|
+
} else {
|
|
861
|
+
h.runner.cancel(task.id);
|
|
862
|
+
h.child.exit(0);
|
|
863
|
+
assert.equal((await h.runner.waitFor(task.id)).status, TASK_STATUS.CANCELLED, `${scenario.name}: later cancellation remains intact`);
|
|
864
|
+
}
|
|
865
|
+
assert.equal(h.child.disconnects, scenario.expectedDisconnects, `${scenario.name}: later cleanup remains idempotent`);
|
|
866
|
+
}
|
|
867
|
+
});
|
|
868
|
+
|
|
762
869
|
test("AgentRunner retains permission broker fd3 and assigns messaging IPC to fd4", async () => {
|
|
763
870
|
const { runner, children, spawnOptions } = harness();
|
|
764
871
|
const task = runner.run(request({ authorizeParentStandingReviewPermission: () => true }));
|
|
765
872
|
await tick();
|
|
766
873
|
const launch = spawnOptions[0];
|
|
874
|
+
const permissionChannelStdio = process.platform === "win32" ? "overlapped" : "pipe";
|
|
767
875
|
assert.match(launch?.env.GENTLE_PI_AGENTS_OWNED_IPC ?? "", /^\d+-[a-z0-9]+$/, "the owned-IPC marker has the runner's opaque shape");
|
|
768
876
|
assert.deepEqual(launch?.env, { GENTLE_PI_AGENTS_CHILD: "1", GENTLE_PI_AGENTS_OWNED_IPC: launch?.env.GENTLE_PI_AGENTS_OWNED_IPC, GENTLE_PI_AGENTS_PARENT_PERMISSION_FD: "3" });
|
|
769
|
-
assert.deepEqual(launch?.stdio, ["pipe", "pipe", "pipe",
|
|
877
|
+
assert.deepEqual(launch?.stdio, ["pipe", "pipe", "pipe", permissionChannelStdio, "ipc"]);
|
|
770
878
|
assert.equal(launch?.stdio?.length, 5);
|
|
771
879
|
children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "channel checked" }], stopReason: "stop" }] });
|
|
772
880
|
children[0].emit({ type: "agent_settled" });
|
|
773
881
|
assert.equal((await runner.waitFor(task.id)).status, TASK_STATUS.COMPLETED);
|
|
774
882
|
});
|
|
775
883
|
|
|
884
|
+
test("AgentRunner platform matrix scopes permission fd3 transport", async () => {
|
|
885
|
+
for (const platform of ["win32", "linux", "darwin"] as const) {
|
|
886
|
+
for (const eligible of [false, true]) {
|
|
887
|
+
const launches: Array<Parameters<RunnerDeps["spawn"]>[2]> = [];
|
|
888
|
+
const child = fakeChild();
|
|
889
|
+
const runner = new AgentRunner(new TaskStore(), { maxConcurrency: 1, stallTimeoutMs: 1_000 }, {
|
|
890
|
+
spawn: (_command, _args, options) => {
|
|
891
|
+
launches.push(options);
|
|
892
|
+
return child.child;
|
|
893
|
+
},
|
|
894
|
+
now: () => 1,
|
|
895
|
+
schedule: () => () => {},
|
|
896
|
+
pi: { command: "pi-fixture", args: [] },
|
|
897
|
+
process: { platform, kill: () => {} },
|
|
898
|
+
}, { askUser: async () => ({ cancelled: true }) });
|
|
899
|
+
const task = runner.run(request({ authorizeParentStandingReviewPermission: eligible ? () => true : undefined }));
|
|
900
|
+
await tick();
|
|
901
|
+
const launch = launches[0];
|
|
902
|
+
assert.ok(launch, `${platform} ${eligible ? "eligible" : "ineligible"} child launches`);
|
|
903
|
+
assert.equal(launch.env.GENTLE_PI_AGENTS_PARENT_PERMISSION_FD, eligible ? "3" : undefined, "only eligible children receive the fd3 marker");
|
|
904
|
+
assert.deepEqual(launch.stdio, eligible ? ["pipe", "pipe", "pipe", platform === "win32" ? "overlapped" : "pipe", "ipc"] : ["pipe", "pipe", "pipe", "ipc"]);
|
|
905
|
+
assert.equal(launch.stdio?.indexOf("ipc"), eligible ? 4 : 3, "messaging IPC follows fd3 only for eligible children");
|
|
906
|
+
runner.cancel(task.id);
|
|
907
|
+
}
|
|
908
|
+
}
|
|
909
|
+
});
|
|
910
|
+
|
|
776
911
|
test("AgentRunner answers dialogs through askUser in task mode and cancels them in background mode", async () => {
|
|
777
912
|
const { store, runner, children, asks } = harness({ answer: { confirmed: true } });
|
|
778
913
|
const task = runner.run(request());
|
|
@@ -815,8 +950,11 @@ test("AgentRunner has no total-duration watchdog but keeps active work alive and
|
|
|
815
950
|
children[0].emit({ type: "response", id: "r1", success: true });
|
|
816
951
|
await tick();
|
|
817
952
|
assert.equal(initialStall.cancelled, true, "every child RPC event, including a response, re-arms the inactivity watchdog");
|
|
953
|
+
const afterResponse = timers.filter((timer) => timer.ms === 10_000 && !timer.cancelled).at(-1);
|
|
954
|
+
assert.ok(afterResponse);
|
|
818
955
|
children[0].emit({ type: "message_update", assistantMessageEvent: { type: "text_delta", delta: "still working" } });
|
|
819
956
|
await tick();
|
|
957
|
+
assert.equal(afterResponse!.cancelled, true, "normalized task progress re-arms the inactivity watchdog");
|
|
820
958
|
assert.equal(store.get(task.id)?.status, TASK_STATUS.RUNNING, "ongoing RPC activity keeps a long-running task active");
|
|
821
959
|
const stall = timers.filter((timer) => timer.ms === 10_000 && !timer.cancelled).at(-1);
|
|
822
960
|
assert.ok(stall);
|
|
@@ -826,6 +964,104 @@ test("AgentRunner has no total-duration watchdog but keeps active work alive and
|
|
|
826
964
|
assert.match(store.get(task.id)?.error ?? "", /stalled/);
|
|
827
965
|
});
|
|
828
966
|
|
|
967
|
+
test("an announced tool call in flight arms the tool ceiling and names the tool when it fires", async () => {
|
|
968
|
+
const h = harness({ stallTimeoutMs: FOUR_MIN_MS, toolStallTimeoutMs: 30 * 60_000 });
|
|
969
|
+
const task = h.runner.run(request());
|
|
970
|
+
await tick();
|
|
971
|
+
h.children[0].emit({ type: "tool_execution_start", toolCallId: "t1", toolName: "bash", args: { command: "pnpm test" } });
|
|
972
|
+
await tick();
|
|
973
|
+
assert.equal(h.store.get(task.id)?.lastStep, "bash");
|
|
974
|
+
assert.equal(h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).length, 0, "the idle budget no longer bounds a task with a tool in flight");
|
|
975
|
+
const toolTimer = h.timers.filter((timer) => timer.ms === 30 * 60_000 && !timer.cancelled).at(-1);
|
|
976
|
+
assert.ok(toolTimer, "an in-flight tool call arms the tool ceiling");
|
|
977
|
+
toolTimer!.fn();
|
|
978
|
+
await tick();
|
|
979
|
+
assert.equal(h.store.get(task.id)?.status, TASK_STATUS.TIMED_OUT);
|
|
980
|
+
assert.equal(h.store.get(task.id)?.error, 'stalled for 30 min with tool "bash" still running after: bash');
|
|
981
|
+
});
|
|
982
|
+
|
|
983
|
+
test("a finished tool call returns the task to the idle silence budget", async () => {
|
|
984
|
+
const h = harness({ stallTimeoutMs: FOUR_MIN_MS, toolStallTimeoutMs: 30 * 60_000 });
|
|
985
|
+
const task = h.runner.run(request());
|
|
986
|
+
await tick();
|
|
987
|
+
h.children[0].emit({ type: "tool_execution_start", toolCallId: "t1", toolName: "bash", args: { command: "pnpm test" } });
|
|
988
|
+
await tick();
|
|
989
|
+
h.children[0].emit({ type: "tool_execution_end", toolCallId: "t1", isError: false });
|
|
990
|
+
await tick();
|
|
991
|
+
assert.equal(h.timers.filter((timer) => timer.ms === 30 * 60_000 && !timer.cancelled).length, 0, "a finished tool is back on the idle budget");
|
|
992
|
+
const idle = h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).at(-1);
|
|
993
|
+
assert.ok(idle, "tool_end re-arms the idle budget");
|
|
994
|
+
idle!.fn();
|
|
995
|
+
await tick();
|
|
996
|
+
assert.equal(h.store.get(task.id)?.status, TASK_STATUS.TIMED_OUT);
|
|
997
|
+
assert.equal(h.store.get(task.id)?.error, "stalled for 4 min after: bash");
|
|
998
|
+
});
|
|
999
|
+
|
|
1000
|
+
test("ignored non-dialog UI traffic does not renew the idle silence budget", async () => {
|
|
1001
|
+
const h = harness({ stallTimeoutMs: FOUR_MIN_MS, toolStallTimeoutMs: 30 * 60_000 });
|
|
1002
|
+
const task = h.runner.run(request());
|
|
1003
|
+
await tick();
|
|
1004
|
+
const armed = h.timers.at(-1);
|
|
1005
|
+
assert.ok(armed, "launch arms the idle silence budget");
|
|
1006
|
+
assert.equal(armed!.ms, FOUR_MIN_MS);
|
|
1007
|
+
// Fire-and-forget UI notifications normalize to zero task events and prove
|
|
1008
|
+
// only that the transport is alive; they must not postpone the silence bound.
|
|
1009
|
+
h.children[0].emit({ type: "extension_ui_request", id: "u1", method: "setStatus", statusKey: "fixture", statusText: "idle" });
|
|
1010
|
+
h.children[0].emit({ type: "extension_ui_request", id: "u2", method: "notify", message: "still here" });
|
|
1011
|
+
await tick();
|
|
1012
|
+
assert.equal(armed!.cancelled, false, "ignored UI traffic must not cancel the armed silence budget");
|
|
1013
|
+
assert.equal(h.timers.filter((timer) => !timer.cancelled && timer.ms === FOUR_MIN_MS).length, 1, "no replacement timer is scheduled for ignored UI traffic");
|
|
1014
|
+
assert.equal(h.timers.filter((timer) => timer.ms === 30 * 60_000).length, 0, "ignored UI traffic never earns the tool ceiling");
|
|
1015
|
+
armed!.fn();
|
|
1016
|
+
await tick();
|
|
1017
|
+
assert.equal(h.store.get(task.id)?.status, TASK_STATUS.TIMED_OUT);
|
|
1018
|
+
assert.equal(h.store.get(task.id)?.error, "stalled for 4 min after: prompt accepted; no first run event received for model: openai-codex/gpt-5.6-terra");
|
|
1019
|
+
});
|
|
1020
|
+
|
|
1021
|
+
test("an unrecognized RPC object does not renew the idle silence budget", async () => {
|
|
1022
|
+
const h = harness({ stallTimeoutMs: FOUR_MIN_MS });
|
|
1023
|
+
const task = h.runner.run(request());
|
|
1024
|
+
await tick();
|
|
1025
|
+
const armed = h.timers.at(-1);
|
|
1026
|
+
assert.ok(armed);
|
|
1027
|
+
h.children[0].emit({ type: "some_future_event", payload: { nested: true } });
|
|
1028
|
+
await tick();
|
|
1029
|
+
assert.equal(armed!.cancelled, false, "an unknown object is not progress");
|
|
1030
|
+
assert.equal(h.timers.filter((timer) => !timer.cancelled).length, 1);
|
|
1031
|
+
armed!.fn();
|
|
1032
|
+
await tick();
|
|
1033
|
+
assert.equal(h.store.get(task.id)?.status, TASK_STATUS.TIMED_OUT);
|
|
1034
|
+
});
|
|
1035
|
+
|
|
1036
|
+
test("a blocking child dialog still re-arms the idle silence budget", async () => {
|
|
1037
|
+
const h = harness({ stallTimeoutMs: FOUR_MIN_MS, answer: { confirmed: true } });
|
|
1038
|
+
h.runner.run(request());
|
|
1039
|
+
await tick();
|
|
1040
|
+
const armed = h.timers.at(-1);
|
|
1041
|
+
assert.ok(armed);
|
|
1042
|
+
h.children[0].emit({ type: "extension_ui_request", id: "u1", method: "confirm", title: "Continue?" });
|
|
1043
|
+
await tick();
|
|
1044
|
+
assert.equal(armed!.cancelled, true, "a dialog the parent must answer is meaningful activity");
|
|
1045
|
+
assert.equal(h.asks.length, 1);
|
|
1046
|
+
});
|
|
1047
|
+
|
|
1048
|
+
test("the tool ceiling holds while any announced tool call is still in flight", async () => {
|
|
1049
|
+
const h = harness({ stallTimeoutMs: FOUR_MIN_MS, toolStallTimeoutMs: 30 * 60_000 });
|
|
1050
|
+
h.runner.run(request());
|
|
1051
|
+
await tick();
|
|
1052
|
+
h.children[0].emit({ type: "tool_execution_start", toolCallId: "t1", toolName: "bash", args: { command: "pnpm test" } });
|
|
1053
|
+
await tick();
|
|
1054
|
+
h.children[0].emit({ type: "tool_execution_start", toolCallId: "t2", toolName: "bash", args: { command: "pnpm run typecheck" } });
|
|
1055
|
+
await tick();
|
|
1056
|
+
h.children[0].emit({ type: "tool_execution_end", toolCallId: "t1", isError: false });
|
|
1057
|
+
await tick();
|
|
1058
|
+
assert.ok(h.timers.filter((timer) => timer.ms === 30 * 60_000 && !timer.cancelled).length > 0, "a second tool still in flight keeps the tool ceiling");
|
|
1059
|
+
assert.equal(h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).length, 0);
|
|
1060
|
+
h.children[0].emit({ type: "tool_execution_end", toolCallId: "t2", isError: false });
|
|
1061
|
+
await tick();
|
|
1062
|
+
assert.ok(h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).length > 0, "ending the last tool returns to the idle budget");
|
|
1063
|
+
});
|
|
1064
|
+
|
|
829
1065
|
test("AgentRunner.cancelAll stops every queued and running task", async () => {
|
|
830
1066
|
const { store, runner, children } = harness({ maxConcurrency: 1 });
|
|
831
1067
|
const running = runner.run(request());
|
|
@@ -998,8 +1234,9 @@ test("an unprobeable process group quarantines at its deadline and still records
|
|
|
998
1234
|
const finishes: string[] = [];
|
|
999
1235
|
let now = 1_000;
|
|
1000
1236
|
let launches = 0;
|
|
1237
|
+
let child: FakeChild;
|
|
1001
1238
|
const runner = new AgentRunner(store, { maxConcurrency: 1, stallTimeoutMs: 10_000 }, {
|
|
1002
|
-
spawn: () => fakeChild({ exitOnKill: false, pid: 90 + (launches += 1) }).child, now: () => now,
|
|
1239
|
+
spawn: () => (child = fakeChild({ exitOnKill: false, pid: 90 + (launches += 1) })).child, now: () => now,
|
|
1003
1240
|
schedule: (fn, ms) => {
|
|
1004
1241
|
const timer = { fn, ms, cancelled: false };
|
|
1005
1242
|
timers.push(timer);
|
|
@@ -1008,7 +1245,7 @@ test("an unprobeable process group quarantines at its deadline and still records
|
|
|
1008
1245
|
pi: { command: "pi", args: [] },
|
|
1009
1246
|
process: { platform: "win32", kill: () => {} },
|
|
1010
1247
|
}, { askUser: async () => ({ value: "yes" }), onFinish: (task) => { finishes.push(task.id); } });
|
|
1011
|
-
const first = runner.run(
|
|
1248
|
+
const first = runner.run(managedRequest());
|
|
1012
1249
|
const second = runner.run(request({ prompt: "queued" }));
|
|
1013
1250
|
await tick();
|
|
1014
1251
|
runner.cancel(first.id);
|
|
@@ -1025,6 +1262,12 @@ test("an unprobeable process group quarantines at its deadline and still records
|
|
|
1025
1262
|
assert.equal(finishes.length, 1, "the run is recorded exactly once");
|
|
1026
1263
|
assert.equal(store.get(second.id)?.status, TASK_STATUS.QUEUED, "an unconfirmed exit retains its capacity");
|
|
1027
1264
|
assert.equal(launches, 1, "no further launch happens while the slot is quarantined");
|
|
1265
|
+
assert.throws(() => runner.run(managedRequest()), /Remediation already queued or running/, "a failed record does not release its quarantined child");
|
|
1266
|
+
child!.exit(0);
|
|
1267
|
+
await tick();
|
|
1268
|
+
assert.doesNotThrow(() => runner.run(managedRequest()), "confirmed cleanup releases the managed workspace");
|
|
1269
|
+
runner.cancelAll();
|
|
1270
|
+
child!.exit(0);
|
|
1028
1271
|
});
|
|
1029
1272
|
|
|
1030
1273
|
test("abortReasonText renders an Error, a string, and nothing for unknown reasons", () => {
|
|
@@ -1039,17 +1282,12 @@ test("research narrowing transport keeps exact argv paths and replaces inherited
|
|
|
1039
1282
|
const h = harness();
|
|
1040
1283
|
const selection = { documentation: { tools: ["fetch_content"], extensions: { fetch_content: "/installed/docs tools.ts" } } };
|
|
1041
1284
|
for (const researchSelection of [selection, undefined]) {
|
|
1042
|
-
const
|
|
1043
|
-
|
|
1044
|
-
const launch = request({ researchSelection, researchArtifact: artifact, extensionPaths: researchSelection ? ["/installed/docs tools.ts"] : [],
|
|
1045
|
-
env: { PATH: "/bin", GENTLE_PI_RESEARCH_SELECTION: "stale broad selection", GENTLE_PI_RESEARCH_ARTIFACT: "stale broader scope" } });
|
|
1285
|
+
const launch = request({ researchSelection, extensionPaths: researchSelection ? ["/installed/docs tools.ts"] : [],
|
|
1286
|
+
env: { PATH: "/bin", GENTLE_PI_RESEARCH_SELECTION: "stale broad selection" } });
|
|
1046
1287
|
const argv = childArguments(launch);
|
|
1047
1288
|
assert.deepEqual(argv.filter((_, i) => argv[i - 1] === "--extension"), launch.extensionPaths);
|
|
1048
1289
|
const task = h.runner.run(launch);
|
|
1049
|
-
artifact.worktree = "/wrong";
|
|
1050
1290
|
await tick();
|
|
1051
|
-
assert.deepEqual(JSON.parse(h.spawnOptions.at(-1)!.env.GENTLE_PI_RESEARCH_ARTIFACT!), expected);
|
|
1052
|
-
assert.deepEqual("researchArtifact" in task ? task.researchArtifact : undefined, expected);
|
|
1053
1291
|
assert.deepEqual(JSON.parse(h.spawnOptions.at(-1)!.env.GENTLE_PI_RESEARCH_SELECTION!), researchSelection ?? null);
|
|
1054
1292
|
assert.equal(h.spawnOptions.at(-1)!.env.PATH, "/bin");
|
|
1055
1293
|
h.runner.cancel(task.id);
|
|
@@ -1057,55 +1295,80 @@ test("research narrowing transport keeps exact argv paths and replaces inherited
|
|
|
1057
1295
|
}
|
|
1058
1296
|
});
|
|
1059
1297
|
|
|
1298
|
+
function managedRequest(cwd = "/repo"): TaskRequest {
|
|
1299
|
+
return request({ agent: { ...explorer, name: "sdd-remediate" }, cwd, sddRemediation: {
|
|
1300
|
+
failedEvidenceRevision: "failed-revision",
|
|
1301
|
+
plan: { cwd, commands: ["pnpm test"], runtimeHarness: { naReason: "Not applicable because this tests runner admission." }, rollback: { boundary: "fixture", command: "git diff --check" } },
|
|
1302
|
+
scope: { cwd, editPaths: [], commands: ["pnpm test", "git diff --check"], allowedEditRoots: [cwd] },
|
|
1303
|
+
} });
|
|
1304
|
+
}
|
|
1060
1305
|
|
|
1061
|
-
test(
|
|
1062
|
-
const h = harness({ pid: 123 });
|
|
1063
|
-
|
|
1064
|
-
|
|
1065
|
-
|
|
1066
|
-
|
|
1067
|
-
|
|
1068
|
-
|
|
1069
|
-
|
|
1070
|
-
|
|
1071
|
-
|
|
1072
|
-
h.children[0].emit({ type: "tool_execution_end", toolName: "bash", toolCallId: command, isError: false, result: { content: [{ type: "text", text: "command output" }], details: { remediationCommand: { command, toolCallId: command, cwd: "/repo", exitCode: 0 } } } });
|
|
1073
|
-
}
|
|
1074
|
-
h.children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "done" }], stopReason: "stop" }] });
|
|
1075
|
-
h.children[0].emit({ type: "agent_settled" });
|
|
1076
|
-
await tick();
|
|
1077
|
-
assert.ok(finalized);
|
|
1078
|
-
assert.equal(finalized.record.sddRemediation?.observations.length, 2);
|
|
1079
|
-
const prompt = h.children[0].written.find(value => value.type === "prompt").message;
|
|
1080
|
-
assert.ok(typeof prompt === "string");
|
|
1081
|
-
assert.match(prompt, /pnpm test/);
|
|
1082
|
-
assert.doesNotMatch(prompt, /opaque/);
|
|
1083
|
-
assert.equal(finalized.facts.cleanupConfirmed, true);
|
|
1084
|
-
assert.equal(h.finishes.length, 0);
|
|
1085
|
-
release(); await h.runner.waitFor(task.id);
|
|
1086
|
-
assert.equal(h.finishes.length, 1);
|
|
1087
|
-
assert.equal(remediation.observations.length, 0);
|
|
1306
|
+
for (const queued of [true, false]) test(`managed exclusion covers ${queued ? "queued" : "running"} same-workspace actors`, async () => {
|
|
1307
|
+
const h = harness({ pid: 123, process: { platform: "win32", kill() {} } });
|
|
1308
|
+
const first = h.runner.run(managedRequest());
|
|
1309
|
+
if (!queued) await tick();
|
|
1310
|
+
try {
|
|
1311
|
+
assert.throws(() => h.runner.run(managedRequest()), /Remediation already queued or running/);
|
|
1312
|
+
assert.equal(h.store.list().length, 1, "rejection creates no task or queue entry");
|
|
1313
|
+
await tick();
|
|
1314
|
+
assert.equal(h.children.length, 1);
|
|
1315
|
+
assert.equal(h.store.get(first.id)?.status, TASK_STATUS.RUNNING);
|
|
1316
|
+
} finally { h.runner.cancelAll(); await tick(); }
|
|
1088
1317
|
});
|
|
1089
1318
|
|
|
1319
|
+
test("managed exclusion does not serialize other workspaces or ordinary tasks", async () => {
|
|
1320
|
+
const h = harness({ maxConcurrency: 3, pid: 123, process: { platform: "win32", kill() {} } });
|
|
1321
|
+
h.runner.run(managedRequest());
|
|
1322
|
+
h.runner.run(managedRequest("/other"));
|
|
1323
|
+
h.runner.run(request());
|
|
1324
|
+
await tick();
|
|
1325
|
+
assert.equal(h.children.length, 3);
|
|
1326
|
+
h.runner.cancelAll();
|
|
1327
|
+
await tick();
|
|
1328
|
+
});
|
|
1090
1329
|
|
|
1091
|
-
|
|
1092
|
-
const h = harness()
|
|
1093
|
-
const
|
|
1330
|
+
for (const ending of ["complete", "failure", "cancel", "queued-cancel"] as const) test(`managed exclusion releases after ${ending}`, async () => {
|
|
1331
|
+
const h = harness({ pid: 123, process: { platform: "win32", kill() {} } });
|
|
1332
|
+
const first = h.runner.run(managedRequest());
|
|
1333
|
+
if (ending === "queued-cancel") h.runner.cancel(first.id);
|
|
1334
|
+
else {
|
|
1335
|
+
await tick();
|
|
1336
|
+
if (ending === "complete") {
|
|
1337
|
+
h.children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "Finished" }], stopReason: "stop" }] });
|
|
1338
|
+
h.children[0].emit({ type: "agent_settled" });
|
|
1339
|
+
} else if (ending === "failure") h.children[0].exit(1);
|
|
1340
|
+
else h.runner.cancel(first.id);
|
|
1341
|
+
}
|
|
1342
|
+
await h.runner.waitFor(first.id);
|
|
1343
|
+
const next = h.runner.run(managedRequest());
|
|
1344
|
+
await tick();
|
|
1345
|
+
assert.equal(h.store.get(next.id)?.status, TASK_STATUS.RUNNING);
|
|
1346
|
+
h.runner.cancelAll();
|
|
1094
1347
|
await tick();
|
|
1095
|
-
assert.equal(h.children[0].written.some(value => value.type === "prompt"), false);
|
|
1096
|
-
await h.runner.waitFor(task.id);
|
|
1097
|
-
assert.equal(facts.spawned, false);
|
|
1098
|
-
assert.equal(facts.exited, false);
|
|
1099
|
-
assert.equal(facts.cleanupConfirmed, false);
|
|
1100
1348
|
});
|
|
1101
1349
|
|
|
1350
|
+
test("managed exclusion lasts until child cleanup is confirmed", async () => {
|
|
1351
|
+
const h = harness({ pid: 123, exitOnKill: false, process: { platform: "win32", kill() {} } });
|
|
1352
|
+
const first = h.runner.run(managedRequest());
|
|
1353
|
+
await tick();
|
|
1354
|
+
h.runner.cancel(first.id);
|
|
1355
|
+
try { assert.throws(() => h.runner.run(managedRequest()), /Remediation already queued or running/); }
|
|
1356
|
+
finally { h.children[0].exit(0); }
|
|
1357
|
+
await h.runner.waitFor(first.id);
|
|
1358
|
+
const next = h.runner.run(managedRequest());
|
|
1359
|
+
await tick();
|
|
1360
|
+
assert.equal(h.store.get(next.id)?.status, TASK_STATUS.RUNNING);
|
|
1361
|
+
h.runner.cancelAll();
|
|
1362
|
+
for (const child of h.children) child.exit(0);
|
|
1363
|
+
await tick();
|
|
1364
|
+
});
|
|
1102
1365
|
|
|
1103
|
-
test("
|
|
1104
|
-
|
|
1105
|
-
const
|
|
1106
|
-
|
|
1107
|
-
|
|
1108
|
-
|
|
1109
|
-
assert.
|
|
1110
|
-
assert.
|
|
1366
|
+
test("managed exclusion releases failed startup and ignores historical-only tasks", async () => {
|
|
1367
|
+
const h = harness({ failStart: true });
|
|
1368
|
+
const first = h.runner.run(managedRequest());
|
|
1369
|
+
assert.equal((await h.runner.waitFor(first.id)).status, TASK_STATUS.FAILED);
|
|
1370
|
+
h.store.add({ ...h.store.get(first.id)!, id: "historical-only", status: TASK_STATUS.RUNNING });
|
|
1371
|
+
const next = h.runner.run(managedRequest());
|
|
1372
|
+
assert.equal((await h.runner.waitFor(next.id)).status, TASK_STATUS.FAILED, "the next actor reaches spawn, not a historical admission lock");
|
|
1373
|
+
assert.match(h.store.get(next.id)?.error ?? "", /fixture spawn failed/);
|
|
1111
1374
|
});
|