gentle-pi 2.4.0 → 2.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +292 -25
- package/assets/agents/gentle-ai-worker.md +13 -0
- package/assets/agents/jd-fix-agent.md +18 -0
- package/assets/agents/jd-judge-a.md +1 -1
- package/assets/agents/jd-judge-b.md +1 -1
- package/assets/agents/sdd-apply.md +7 -5
- package/assets/agents/sdd-archive.md +5 -3
- package/assets/agents/sdd-design.md +4 -0
- package/assets/agents/sdd-explore.md +4 -0
- package/assets/agents/sdd-init.md +4 -0
- package/assets/agents/sdd-onboard.md +4 -0
- package/assets/agents/sdd-proposal.md +4 -0
- package/assets/agents/sdd-remediate.md +37 -0
- package/assets/agents/sdd-research.md +26 -3
- package/assets/agents/sdd-spec.md +4 -0
- package/assets/agents/sdd-status.md +9 -75
- package/assets/agents/sdd-sync.md +4 -0
- package/assets/agents/sdd-tasks.md +4 -0
- package/assets/agents/sdd-verify.md +5 -3
- package/assets/chains/sdd-full.chain.md +4 -0
- package/assets/chains/sdd-plan.chain.md +4 -0
- package/assets/chains/sdd-verify.chain.md +4 -0
- package/assets/migrations/managed-assets-v2.5.0.json +7 -0
- package/assets/orchestrator-delegation.md +39 -11
- package/assets/orchestrator.md +5 -5
- package/assets/sdd-orchestrator-workflow.md +54 -21
- package/assets/support/sdd-status-contract.md +34 -90
- package/contracts/telemetry/runtime-aggregate-v1.schema.json +65 -0
- package/docs/delegated-verification.md +25 -0
- package/docs/telemetry.md +94 -0
- package/docs/windows-startup-console-visibility.md +18 -0
- package/extensions/ask-user-choice.ts +159 -25
- package/extensions/codegraph-tools.ts +95 -5
- package/extensions/gentle-agents.ts +1337 -0
- package/extensions/gentle-ai.ts +2916 -386
- package/extensions/gentle-shell.ts +650 -0
- package/extensions/gentle-todo.ts +234 -0
- package/extensions/quiet-tools.ts +2 -1
- package/extensions/runtime-metrics.ts +130 -0
- package/extensions/sdd-init.ts +2 -2
- package/extensions/startup-banner.ts +52 -75
- package/lib/agent-profiles.ts +550 -0
- package/lib/agents-completion-delivery.ts +72 -0
- package/lib/agents-config.ts +315 -0
- package/lib/agents-history.ts +88 -0
- package/lib/agents-messaging.ts +187 -0
- package/lib/agents-protocol.ts +501 -0
- package/lib/agents-runner.ts +1012 -0
- package/lib/agents-thread-view.ts +57 -0
- package/lib/agents-transcript.ts +87 -0
- package/lib/agents-view-layout.ts +40 -0
- package/lib/agents-view.ts +914 -0
- package/lib/agents-widget.ts +241 -0
- package/lib/gentle-ai-binary.ts +3 -1
- package/lib/gentle-ai-renderer.ts +143 -25
- package/lib/native-choice-list.ts +194 -0
- package/lib/native-fullscreen-interaction.ts +47 -0
- package/lib/native-pointer-region.ts +164 -0
- package/lib/native-review-cli.ts +371 -13
- package/lib/orchestrator-presence.ts +337 -0
- package/lib/profiles-orchestrator.ts +203 -0
- package/lib/review-candidate-view-owner.ts +427 -0
- package/lib/review-candidate-view.ts +150 -48
- package/lib/review-consent-component.ts +247 -0
- package/lib/review-consent-ui.ts +110 -0
- package/lib/review-host-relay.ts +28 -0
- package/lib/review-integration-v2.ts +243 -11
- package/lib/review-last-event-controller.ts +8 -4
- package/lib/review-relay-contract.ts +11 -0
- package/lib/review-reminder-receipt.ts +74 -0
- package/lib/review-repository.ts +2 -2
- package/lib/review-risk-assessment.ts +339 -0
- package/lib/review-session-standing-permission-ipc.ts +309 -0
- package/lib/review-session-standing-permission.ts +240 -0
- package/lib/runtime-metrics-children.ts +199 -0
- package/lib/runtime-metrics-delivery.ts +68 -0
- package/lib/runtime-metrics-native.ts +166 -0
- package/lib/runtime-metrics-pi-identity.ts +113 -0
- package/lib/runtime-metrics-policy.ts +51 -0
- package/lib/runtime-metrics.ts +255 -0
- package/lib/sdd-preflight.ts +362 -81
- package/lib/sdd-research-capabilities.ts +228 -0
- package/lib/sdd-status.ts +29 -7
- package/lib/session-worktree-registry.ts +118 -0
- package/lib/shell-bar.ts +184 -0
- package/lib/shell-card.ts +133 -0
- package/lib/shell-changes-view.ts +530 -0
- package/lib/shell-changes.ts +290 -0
- package/lib/shell-gauge.ts +40 -0
- package/lib/shell-prompt.ts +115 -0
- package/lib/shell-sidebar-banner.ts +11 -0
- package/lib/shell-sidebar-layout.ts +213 -0
- package/lib/shell-sidebar.ts +41 -0
- package/lib/shell-todo.ts +297 -0
- package/lib/shell-usage-view.ts +76 -0
- package/lib/shell-usage.ts +246 -0
- package/lib/telemetry-trigger.ts +153 -0
- package/package.json +8 -5
- package/runtime/gentle-ai-binary.mjs +3 -1
- package/runtime/native-review-cli.mjs +370 -12
- package/runtime/review-integration-v2.mjs +243 -11
- package/runtime/review-relay-contract.mjs +11 -0
- package/runtime/review-risk-assessment.mjs +340 -0
- package/runtime/telemetry-trigger.mjs +154 -0
- package/scripts/build-runtime-modules.mjs +11 -1
- package/scripts/check-types.mjs +125 -0
- package/scripts/gentle-ai-installer.mjs +10 -10
- package/scripts/install-gentle-ai.mjs +12 -0
- package/scripts/install-tui-mode-setting.mjs +114 -0
- package/scripts/test-packed-runner.mjs +38 -2
- package/scripts/types-baseline.json +99 -0
- package/scripts/verify-package-files.mjs +8 -2
- package/skills/_shared/review-ledger-contract.md +20 -2
- package/skills/issue-creation/SKILL.md +3 -3
- package/skills/judgment-day/SKILL.md +17 -3
- package/skills/judgment-day/references/prompts-and-formats.md +14 -3
- package/tests/agent-profiles.test.ts +722 -0
- package/tests/agents-completion-delivery.test.ts +94 -0
- package/tests/agents-config.test.ts +205 -0
- package/tests/agents-fake-child.ts +66 -0
- package/tests/agents-grouping.test.ts +179 -0
- package/tests/agents-history.test.ts +54 -0
- package/tests/agents-integration.test.ts +100 -0
- package/tests/agents-messaging.test.ts +94 -0
- package/tests/agents-protocol.test.ts +198 -0
- package/tests/agents-queries.test.ts +190 -0
- package/tests/agents-responsive.test.ts +43 -0
- package/tests/agents-runner-process.test.ts +111 -0
- package/tests/agents-runner.test.ts +959 -0
- package/tests/agents-thread-view.test.ts +45 -0
- package/tests/agents-transcript.test.ts +30 -0
- package/tests/agents-view.test.ts +685 -0
- package/tests/agents-widget.test.ts +141 -0
- package/tests/artifact-language.test.ts +25 -2
- package/tests/ask-user-choice.test.ts +325 -5
- package/tests/asset-installation-runtime.test.ts +108 -0
- package/tests/autonomous-guard.test.ts +116 -1
- package/tests/codegraph-tools.test.ts +112 -2
- package/tests/delegated-key-learnings-contract.test.ts +1 -1
- package/tests/devbinary/native-review-parity.devtest.ts +110 -0
- package/tests/feature-request-form.test.ts +67 -0
- package/tests/fixtures/agents-messaging-child.mjs +5 -0
- package/tests/fixtures/agents-process-child.mjs +23 -0
- package/tests/fixtures/runtime-metrics-native-batches.json +6 -0
- package/tests/gentle-agents.test.ts +2168 -0
- package/tests/gentle-ai-binary.test.ts +7 -2
- package/tests/gentle-ai-installer.test.ts +47 -47
- package/tests/gentle-ai-renderer.test.ts +103 -0
- package/tests/gentle-ai.test.ts +971 -15
- package/tests/gentle-card-text.ts +35 -0
- package/tests/gentle-shell.test.ts +818 -0
- package/tests/gentle-todo.test.ts +226 -0
- package/tests/install-tui-mode-setting.test.ts +324 -0
- package/tests/issue-creation-skill.test.ts +22 -0
- package/tests/model-routing-authority.test.ts +12 -0
- package/tests/native-choice-list.test.ts +202 -0
- package/tests/native-fullscreen-interaction.test.ts +125 -0
- package/tests/native-pointer-region.test.ts +245 -0
- package/tests/native-review-capability-contract.test.ts +27 -1
- package/tests/native-review-cli.test.ts +317 -3
- package/tests/native-review-consent.test.ts +91 -0
- package/tests/native-review-parity-runtime.test.ts +8 -2
- package/tests/native-review-parity.test.ts +43 -29
- package/tests/native-sdd-attempt-authority.test.ts +7 -2
- package/tests/orchestrator-budget.test.ts +69 -0
- package/tests/orchestrator-presence.test.ts +389 -0
- package/tests/orchestrator-rdd-ownership.test.ts +9 -0
- package/tests/package-manifest.test.ts +243 -7
- package/tests/profiles-orchestrator.test.ts +208 -0
- package/tests/quiet-tool-rendering.test.ts +97 -37
- package/tests/rdd-aware-verification-contract.test.ts +226 -0
- package/tests/rdd-status-line.test.ts +286 -0
- package/tests/review-agent-end-preflight.test.ts +332 -24
- package/tests/review-candidate-view.test.ts +751 -7
- package/tests/review-consent-ui.test.ts +352 -0
- package/tests/review-contract-prompt.test.ts +17 -0
- package/tests/review-controller-native-recovery.test.ts +29 -4
- package/tests/review-controller-native-routing.test.ts +884 -7
- package/tests/review-controller-workspace-root.test.ts +45 -2
- package/tests/review-controller.test.ts +26 -1
- package/tests/review-host-relay-restart-parity.test.ts +142 -1
- package/tests/review-host-relay-routing.test.ts +384 -8
- package/tests/review-host-relay.test.ts +29 -0
- package/tests/review-integration-v2-forward.test.ts +44 -0
- package/tests/review-integration-v2.test.ts +276 -0
- package/tests/review-last-event-closure.test.ts +112 -3
- package/tests/review-ledger-contract.test.ts +61 -6
- package/tests/review-relay-contract.test.ts +26 -0
- package/tests/review-reminder-receipt.test.ts +62 -0
- package/tests/review-repository.test.ts +28 -1
- package/tests/review-risk-assessment.test.ts +626 -0
- package/tests/review-session-standing-permission-controller.test.ts +656 -0
- package/tests/review-session-standing-permission-ipc.test.ts +233 -0
- package/tests/review-session-standing-permission-runtime.test.ts +212 -0
- package/tests/review-session-standing-permission.test.ts +156 -0
- package/tests/runtime-harness.mjs +447 -39
- package/tests/runtime-metrics-children.test.ts +206 -0
- package/tests/runtime-metrics-delivery.test.ts +85 -0
- package/tests/runtime-metrics-extension.test.ts +187 -0
- package/tests/runtime-metrics-native.test.ts +209 -0
- package/tests/runtime-metrics-pi-identity.test.ts +113 -0
- package/tests/runtime-metrics-policy.test.ts +62 -0
- package/tests/runtime-metrics.test.ts +184 -0
- package/tests/sdd-agent-tools.test.ts +10 -1
- package/tests/sdd-execution-routing-contract.test.ts +28 -0
- package/tests/sdd-managed-runtime-settlement.test.ts +331 -0
- package/tests/sdd-native-managed-uptake.test.ts +253 -0
- package/tests/sdd-planning-routing-contract.test.ts +45 -0
- package/tests/sdd-preflight.test.ts +252 -8
- package/tests/sdd-research-capabilities.test.ts +256 -0
- package/tests/sdd-research-live.test.ts +241 -0
- package/tests/sdd-selection-transport.test.ts +504 -0
- package/tests/sdd-status.test.ts +51 -0
- package/tests/session-worktree-registry.test.ts +135 -0
- package/tests/shell-bar.test.ts +176 -0
- package/tests/shell-card.test.ts +139 -0
- package/tests/shell-changes-view.test.ts +609 -0
- package/tests/shell-changes.test.ts +350 -0
- package/tests/shell-prompt.test.ts +140 -0
- package/tests/shell-sidebar-banner.test.ts +23 -0
- package/tests/shell-sidebar-layout.test.ts +387 -0
- package/tests/shell-sidebar.test.ts +50 -0
- package/tests/shell-todo.test.ts +259 -0
- package/tests/shell-usage-view.test.ts +62 -0
- package/tests/shell-usage.test.ts +197 -0
- package/tests/startup-banner.test.ts +126 -0
- package/tests/telemetry-trigger.test.ts +351 -0
|
@@ -0,0 +1,959 @@
|
|
|
1
|
+
import assert from "node:assert/strict";
|
|
2
|
+
import test from "node:test";
|
|
3
|
+
import { AGENT_MODE, parseAgentsConfig, resolveAgentProfile, type AgentDefinition } from "../lib/agents-config.ts";
|
|
4
|
+
import { TASK_STATUS, TaskStore, type RemediationTaskState, type TaskRecord } from "../lib/agents-protocol.ts";
|
|
5
|
+
import { AgentRunner, childArguments, JsonLines, piCommand, abortReasonText, type RemediationPlan, type RemediationTerminalFacts, type RunnerDeps, type RunnerHooks, type TaskRequest } from "../lib/agents-runner.ts";
|
|
6
|
+
import { fakeChild, type FakeChild } from "./agents-fake-child.ts";
|
|
7
|
+
|
|
8
|
+
// Gentle Agents runner: every subagent is a child `pi --mode rpc` process.
|
|
9
|
+
// The host only parses JSON lines, applies deltas to the store, answers
|
|
10
|
+
// dialogs, and enforces its inactivity watchdog. These tests drive a fake child.
|
|
11
|
+
|
|
12
|
+
const explorer: AgentDefinition = { name: "explore", description: "maps", filePath: "/a/explore.md", scope: "global", instructions: "You map things.", model: undefined, thinking: undefined, mode: undefined, tools: ["read", "grep"] };
|
|
13
|
+
|
|
14
|
+
function request(overrides: Partial<TaskRequest> = {}): TaskRequest {
|
|
15
|
+
return { agent: explorer, prompt: "Map the repo", label: undefined, context: undefined, mode: AGENT_MODE.TASK, cwd: "/repo", parentSessionId: "s1", model: { provider: "openai-codex", id: "gpt-5.6-terra" }, thinking: "high", sessionDir: "/sessions", resumeSessionPath: undefined, env: {}, ...overrides };
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
interface Harness {
|
|
19
|
+
store: TaskStore;
|
|
20
|
+
runner: AgentRunner;
|
|
21
|
+
children: FakeChild[];
|
|
22
|
+
timers: Array<{ fn: () => void; ms: number; cancelled: boolean }>;
|
|
23
|
+
asks: Array<{ taskId: string; method: string }>;
|
|
24
|
+
finishes: string[];
|
|
25
|
+
spawnOptions: Array<{ env: NodeJS.ProcessEnv; stdio?: string[] }>;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
function harness(options: { pid?: number; maxConcurrency?: number; answer?: Record<string, unknown>; exitOnKill?: boolean; state?: Record<string, unknown>; stateSuccess?: boolean; onNotification?: RunnerHooks["onNotification"]; onSuccessfulMutation?: RunnerHooks["onSuccessfulMutation"]; onFinish?: RunnerHooks["onFinish"] } = {}): Harness {
|
|
29
|
+
const children: FakeChild[] = [];
|
|
30
|
+
const timers: Harness["timers"] = [];
|
|
31
|
+
const asks: Harness["asks"] = [];
|
|
32
|
+
const finishes: string[] = [];
|
|
33
|
+
const spawnOptions: Harness["spawnOptions"] = [];
|
|
34
|
+
let clock = 1000;
|
|
35
|
+
const deps: RunnerDeps = {
|
|
36
|
+
spawn: (_command, _args, launchOptions) => {
|
|
37
|
+
spawnOptions.push({ env: launchOptions.env, stdio: launchOptions.stdio });
|
|
38
|
+
const fake = fakeChild({ exitOnKill: options.exitOnKill, pid: options.pid });
|
|
39
|
+
if (options.state !== undefined) {
|
|
40
|
+
fake.child.stdin.removeAllListeners("data");
|
|
41
|
+
fake.child.stdin.on("data", (chunk) => {
|
|
42
|
+
const command = JSON.parse(String(chunk));
|
|
43
|
+
fake.written.push(command);
|
|
44
|
+
fake.emit({ type: "response", id: command.id, success: command.type !== "get_state" || options.stateSuccess !== false,
|
|
45
|
+
data: command.type === "get_state" ? options.state : undefined });
|
|
46
|
+
});
|
|
47
|
+
}
|
|
48
|
+
children.push(fake);
|
|
49
|
+
return fake.child;
|
|
50
|
+
},
|
|
51
|
+
now: () => (clock += 1),
|
|
52
|
+
schedule: (fn, ms) => {
|
|
53
|
+
const timer = { fn, ms, cancelled: false };
|
|
54
|
+
timers.push(timer);
|
|
55
|
+
return () => {
|
|
56
|
+
timer.cancelled = true;
|
|
57
|
+
};
|
|
58
|
+
},
|
|
59
|
+
pi: { command: "pi", args: [] },
|
|
60
|
+
};
|
|
61
|
+
const store = new TaskStore();
|
|
62
|
+
const runner = new AgentRunner(store, { maxConcurrency: options.maxConcurrency ?? 2, stallTimeoutMs: 10_000 }, deps, {
|
|
63
|
+
askUser: async (taskId, ask) => {
|
|
64
|
+
asks.push({ taskId, method: ask.method });
|
|
65
|
+
return options.answer ?? { value: "yes" };
|
|
66
|
+
},
|
|
67
|
+
onFinish: (task, observations) => { finishes.push(task.id); options.onFinish?.(task, observations); },
|
|
68
|
+
onNotification: options.onNotification,
|
|
69
|
+
onSuccessfulMutation: options.onSuccessfulMutation,
|
|
70
|
+
});
|
|
71
|
+
return { store, runner, children, timers, asks, finishes, spawnOptions };
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
const tick = () => new Promise((resolve) => setImmediate(resolve));
|
|
75
|
+
|
|
76
|
+
test("synchronous cancellation before dequeue never invokes the policy callback", async () => {
|
|
77
|
+
const h = harness(); let checks = 0;
|
|
78
|
+
const task = h.runner.run(request({ prepareResponseObservations: async () => { checks++; return true; } }));
|
|
79
|
+
h.runner.cancel(task.id);
|
|
80
|
+
await tick();
|
|
81
|
+
assert.equal(checks, 0);
|
|
82
|
+
assert.equal(h.children.length, 0);
|
|
83
|
+
});
|
|
84
|
+
|
|
85
|
+
test("hanging preparation never blocks spawn or queued core work; late grants are dropped", async () => {
|
|
86
|
+
const h = harness({ maxConcurrency: 1 });
|
|
87
|
+
let grant!: (value: boolean) => void;
|
|
88
|
+
let checks = 0;
|
|
89
|
+
const first = h.runner.run(request({ prepareResponseObservations: () => { checks++; return new Promise(resolve => { grant = resolve; }); } }));
|
|
90
|
+
const second = h.runner.run(request());
|
|
91
|
+
await tick();
|
|
92
|
+
assert.equal(checks, 1);
|
|
93
|
+
assert.equal(h.children.length, 1);
|
|
94
|
+
assert.equal(h.store.get(second.id)?.status, TASK_STATUS.QUEUED);
|
|
95
|
+
assert.equal(h.runner.cancel(first.id), true);
|
|
96
|
+
await tick();
|
|
97
|
+
assert.equal(h.children.length, 2);
|
|
98
|
+
grant(true);
|
|
99
|
+
await tick();
|
|
100
|
+
assert.equal(h.children.length, 2);
|
|
101
|
+
assert.equal((await h.runner.waitFor(first.id)).status, TASK_STATUS.CANCELLED);
|
|
102
|
+
h.runner.cancel(second.id);
|
|
103
|
+
});
|
|
104
|
+
|
|
105
|
+
for (const outcome of ["ready", "late", "reject", "throw"] as const) {
|
|
106
|
+
test(`parallel preparation ${outcome} cannot delay execution or revive dropped observations`, async () => {
|
|
107
|
+
const snapshots: Parameters<NonNullable<RunnerHooks["onFinish"]>>[1][] = [];
|
|
108
|
+
const h = harness({ onFinish: (_task, snapshot) => snapshots.push(snapshot) });
|
|
109
|
+
let grant!: (value: boolean) => void;
|
|
110
|
+
const task = h.runner.run(request({ prepareResponseObservations: () => {
|
|
111
|
+
if (outcome === "throw") throw new Error("preparation failed");
|
|
112
|
+
if (outcome === "reject") return Promise.reject(new Error("preparation failed"));
|
|
113
|
+
return new Promise(resolve => { grant = resolve; });
|
|
114
|
+
} }));
|
|
115
|
+
await tick();
|
|
116
|
+
assert.equal(h.children.length, 1);
|
|
117
|
+
if (outcome === "ready") { grant(true); await tick(); }
|
|
118
|
+
const message = { type: "message_end", message: { role: "assistant", provider: "openai", model: "gpt-4o", stopReason: "stop", content: [{ type: "text", text: "done" }] } };
|
|
119
|
+
h.children[0].emit(message);
|
|
120
|
+
if (outcome === "late") { grant(true); await tick(); }
|
|
121
|
+
h.children[0].emit(message);
|
|
122
|
+
h.children[0].emit({ type: "agent_end" });
|
|
123
|
+
h.children[0].emit({ type: "agent_settled" });
|
|
124
|
+
await h.runner.waitFor(task.id);
|
|
125
|
+
assert.equal(snapshots.length, 1);
|
|
126
|
+
assert.equal(snapshots[0]?.responses.length, outcome === "ready" ? 2 : undefined);
|
|
127
|
+
});
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
for (const checkpoint of ["launch", "stream", "finish", "throw"] as const) {
|
|
131
|
+
test(`child observation guard discards permanently at ${checkpoint} without changing task execution`, async () => {
|
|
132
|
+
let allowed = checkpoint !== "launch";
|
|
133
|
+
let calls = 0;
|
|
134
|
+
const snapshots: Parameters<NonNullable<RunnerHooks["onFinish"]>>[1][] = [];
|
|
135
|
+
const h = harness({ onFinish: (_task, snapshot) => snapshots.push(snapshot) });
|
|
136
|
+
const task = h.runner.run(request({ collectResponseObservations: true,
|
|
137
|
+
canCollectResponseObservations: () => {
|
|
138
|
+
calls++;
|
|
139
|
+
if (checkpoint === "throw") throw new Error("private policy failure");
|
|
140
|
+
return allowed;
|
|
141
|
+
} }));
|
|
142
|
+
await tick();
|
|
143
|
+
const child = h.children[0];
|
|
144
|
+
const response = { type: "message_end", message: { role: "assistant", stopReason: "stop", usage: { input: 3 } } };
|
|
145
|
+
child.emit(response);
|
|
146
|
+
if (checkpoint === "stream") {
|
|
147
|
+
allowed = false;
|
|
148
|
+
child.emit({ type: "message_update", assistantMessageEvent: { type: "text_delta", delta: "progress" } });
|
|
149
|
+
allowed = true;
|
|
150
|
+
child.emit(response);
|
|
151
|
+
}
|
|
152
|
+
if (checkpoint === "finish") allowed = false;
|
|
153
|
+
h.runner.cancel(task.id);
|
|
154
|
+
await tick();
|
|
155
|
+
assert.equal((await h.runner.waitFor(task.id)).status, TASK_STATUS.CANCELLED);
|
|
156
|
+
assert.deepEqual(snapshots, [undefined]);
|
|
157
|
+
assert.ok(calls > 0);
|
|
158
|
+
});
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
test("child observation guard is never consulted when collection is default-off", async () => {
|
|
162
|
+
let calls = 0;
|
|
163
|
+
const h = harness();
|
|
164
|
+
const task = h.runner.run(request({ canCollectResponseObservations: () => { calls++; return true; } }));
|
|
165
|
+
await tick();
|
|
166
|
+
h.children[0].emit({ type: "message_end", message: { role: "assistant", stopReason: "stop" } });
|
|
167
|
+
h.runner.cancel(task.id);
|
|
168
|
+
await tick();
|
|
169
|
+
assert.equal(calls, 0);
|
|
170
|
+
});
|
|
171
|
+
|
|
172
|
+
for (const ending of ["cancel", "failure", "hook-error", "hook-async-error"] as const) {
|
|
173
|
+
test(`successful child mutations require paired RPC events and survive ${ending}`, async () => {
|
|
174
|
+
const mutations: unknown[] = [];
|
|
175
|
+
const h = harness({ onSuccessfulMutation: (task, tool) => {
|
|
176
|
+
mutations.push({ taskId: task.id, parent: task.parentSessionId, ...tool });
|
|
177
|
+
if (ending === "hook-error") throw new Error("receipt append unavailable");
|
|
178
|
+
if (ending === "hook-async-error") return Promise.reject(new Error("async receipt append unavailable"));
|
|
179
|
+
} });
|
|
180
|
+
const task = h.runner.run(request());
|
|
181
|
+
await tick();
|
|
182
|
+
const child = h.children[0];
|
|
183
|
+
const start = (id: string, toolName: string) => child.emit({ type: "tool_execution_start", toolCallId: id, toolName, args: { path: "src/file.ts" } });
|
|
184
|
+
const end = (id: string, isError: unknown = false) => child.emit({ type: "tool_execution_end", toolCallId: id, isError, result: { content: [] } });
|
|
185
|
+
assert.deepEqual(mutations, [], "spawn is not mutation evidence");
|
|
186
|
+
end("missing");
|
|
187
|
+
for (const name of ["read", "bash", "subagent_run"]) { start(name, name); end(name); }
|
|
188
|
+
start("failed", "write"); end("failed", true);
|
|
189
|
+
start("unknown", "edit"); end("unknown", null);
|
|
190
|
+
child.emit({ type: "message_update", assistantMessageEvent: { type: "text_delta", delta: "I edited files" } });
|
|
191
|
+
assert.deepEqual(mutations, []);
|
|
192
|
+
for (const name of ["write", "edit"]) { start(name, name); end(name); end(name); }
|
|
193
|
+
assert.deepEqual(mutations, ["write", "edit"].map((toolName) => ({ taskId: task.id, parent: "s1", toolName, toolCallId: toolName, path: "src/file.ts" })));
|
|
194
|
+
start("unfinished", "write");
|
|
195
|
+
if (ending === "failure") child.fail("later failure");
|
|
196
|
+
else h.runner.cancel(task.id);
|
|
197
|
+
await tick();
|
|
198
|
+
end("unfinished"); start("late", "write"); end("late");
|
|
199
|
+
assert.equal(mutations.length, 2, "terminal cleanup rejects late events without retracting successful writes");
|
|
200
|
+
});
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
test("launch registration waits for actual spawn, including queued launches, and ignores failed spawns", async () => {
|
|
204
|
+
const launches: string[] = [];
|
|
205
|
+
const spawns: Array<() => void> = [];
|
|
206
|
+
const children: FakeChild[] = [];
|
|
207
|
+
const cwds: string[] = [];
|
|
208
|
+
const store = new TaskStore();
|
|
209
|
+
const runner = new AgentRunner(store, { maxConcurrency: 1, stallTimeoutMs: 1000 }, {
|
|
210
|
+
spawn: (_command, _args, options) => {
|
|
211
|
+
cwds.push(options.cwd);
|
|
212
|
+
if (options.cwd === "/throws") throw new Error("missing executable");
|
|
213
|
+
const fake = fakeChild();
|
|
214
|
+
const on = fake.child.on.bind(fake.child);
|
|
215
|
+
fake.child.on = ((event: string, listener: () => void) => {
|
|
216
|
+
if (event === "spawn") spawns.push(listener);
|
|
217
|
+
else on(event as "exit", listener);
|
|
218
|
+
return fake.child;
|
|
219
|
+
}) as typeof fake.child.on;
|
|
220
|
+
children.push(fake);
|
|
221
|
+
return fake.child;
|
|
222
|
+
},
|
|
223
|
+
now: () => 1000, schedule: () => () => {}, pi: { command: "pi", args: [] },
|
|
224
|
+
}, { askUser: async () => ({ cancelled: true }) });
|
|
225
|
+
const first = runner.run(request({ cwd: "/child", onLaunch: () => launches.push("s1:/child") }));
|
|
226
|
+
const second = runner.run(request({ cwd: "/queued", onLaunch: () => launches.push("s1:/queued") }));
|
|
227
|
+
assert.deepEqual(launches, []);
|
|
228
|
+
await tick();
|
|
229
|
+
assert.deepEqual(launches, [], "returning a child handle is not successful spawn");
|
|
230
|
+
assert.equal(typeof spawns[0], "function");
|
|
231
|
+
spawns[0]();
|
|
232
|
+
assert.deepEqual(launches, ["s1:/child"]);
|
|
233
|
+
runner.cancel(first.id);
|
|
234
|
+
await tick();
|
|
235
|
+
assert.equal(store.get(second.id)?.cwd, "/queued");
|
|
236
|
+
spawns[1]();
|
|
237
|
+
assert.deepEqual(launches, ["s1:/child", "s1:/queued"]);
|
|
238
|
+
runner.cancel(second.id);
|
|
239
|
+
await tick();
|
|
240
|
+
const failed = runner.run(request({ cwd: "/missing", onLaunch: () => launches.push("bad") }));
|
|
241
|
+
await tick();
|
|
242
|
+
children[2].fail("ENOENT");
|
|
243
|
+
await tick();
|
|
244
|
+
assert.equal(store.get(failed.id)?.status, TASK_STATUS.FAILED);
|
|
245
|
+
const thrown = runner.run(request({ cwd: "/throws", onLaunch: () => launches.push("bad") }));
|
|
246
|
+
await tick();
|
|
247
|
+
assert.equal(store.get(thrown.id)?.status, TASK_STATUS.FAILED);
|
|
248
|
+
assert.deepEqual(launches, ["s1:/child", "s1:/queued"]);
|
|
249
|
+
assert.deepEqual(cwds, ["/child", "/queued", "/missing", "/throws"]);
|
|
250
|
+
});
|
|
251
|
+
|
|
252
|
+
test("runner captures resolved model and effort, retaining omitted launch values", async () => {
|
|
253
|
+
for (const scenario of [
|
|
254
|
+
{ state: { model: { provider: "anthropic", id: "resolved-model" }, thinkingLevel: "off" }, model: "anthropic/resolved-model", thinking: "off" },
|
|
255
|
+
{ state: { thinkingLevel: "max" }, model: "openai-codex/gpt-5.6-terra", thinking: "max" },
|
|
256
|
+
{ state: {}, model: "openai-codex/gpt-5.6-terra", thinking: "high" },
|
|
257
|
+
{ state: { model: null }, model: "default", thinking: "high" },
|
|
258
|
+
{ state: { model: { id: 7 }, thinkingLevel: 7 }, model: "openai-codex/gpt-5.6-terra", thinking: "high" },
|
|
259
|
+
]) {
|
|
260
|
+
const h = harness({ state: scenario.state });
|
|
261
|
+
const task = h.runner.run(request());
|
|
262
|
+
await tick();
|
|
263
|
+
assert.equal(h.store.get(task.id)?.model, scenario.model);
|
|
264
|
+
assert.equal(h.store.get(task.id)?.thinking, scenario.thinking);
|
|
265
|
+
h.runner.cancel(task.id);
|
|
266
|
+
}
|
|
267
|
+
const h = harness({ state: { model: { provider: "wrong", id: "wrong" }, thinkingLevel: "low" }, stateSuccess: false });
|
|
268
|
+
const task = h.runner.run(request({ model: undefined, thinking: undefined }));
|
|
269
|
+
await tick();
|
|
270
|
+
assert.equal(h.store.get(task.id)?.model, "default");
|
|
271
|
+
assert.equal(h.store.get(task.id)?.thinking, undefined);
|
|
272
|
+
h.runner.cancel(task.id);
|
|
273
|
+
});
|
|
274
|
+
|
|
275
|
+
test("runner delivers each response combination once at finish, never attributing launch selection", async () => {
|
|
276
|
+
const snapshots: NonNullable<Parameters<NonNullable<RunnerHooks["onFinish"]>>[1]>[] = [];
|
|
277
|
+
const h = harness({ exitOnKill: false, state: { model: { provider: "anthropic", id: "launch" }, thinkingLevel: "max" },
|
|
278
|
+
onFinish: (_task, snapshot) => { assert.ok(snapshot); snapshots.push(snapshot); } });
|
|
279
|
+
const task = h.runner.run(request({ collectResponseObservations: true }));
|
|
280
|
+
await tick();
|
|
281
|
+
const child = h.children[0];
|
|
282
|
+
const responses = [
|
|
283
|
+
{ provider: "openai", model: "gpt-4o", providerThinkingLevel: "low", stopReason: "error" },
|
|
284
|
+
{ provider: "anthropic", model: "claude-sonnet-4", providerThinkingLevel: "high", stopReason: "toolUse" },
|
|
285
|
+
{ provider: "openai", model: "gpt-4o", providerThinkingLevel: "high", stopReason: "stop" },
|
|
286
|
+
];
|
|
287
|
+
for (const response of responses) {
|
|
288
|
+
const message = { role: "assistant", ...response, usage: { input: 10, totalTokens: 10, cost: { total: 0.1 } }, content: [{ type: "text", text: "private report" }] };
|
|
289
|
+
child.emit({ type: "message_start", message });
|
|
290
|
+
child.emit({ type: "message_end", message });
|
|
291
|
+
child.emit({ type: "turn_end", message });
|
|
292
|
+
child.emit({ type: "agent_end", messages: [message] });
|
|
293
|
+
}
|
|
294
|
+
assert.deepEqual(snapshots, [], "agent_end is not settlement");
|
|
295
|
+
child.emit({ type: "agent_settled" });
|
|
296
|
+
assert.deepEqual(snapshots, [], "settlement still waits for process cleanup");
|
|
297
|
+
child.exit(0);
|
|
298
|
+
await tick();
|
|
299
|
+
assert.equal(snapshots.length, 1);
|
|
300
|
+
const snapshot = snapshots[0];
|
|
301
|
+
assert.equal(snapshot.agentSettled, true);
|
|
302
|
+
assert.equal(snapshot.droppedResponses, 0);
|
|
303
|
+
assert.equal(snapshot.coverage, "final_assistant_messages_only");
|
|
304
|
+
assert.deepEqual(snapshot.responses.map((response) => [response.provider, response.model, response.providerThinkingLevel]),
|
|
305
|
+
responses.map((response) => [response.provider, response.model, response.providerThinkingLevel].map((value) => ({ state: "observed", value }))));
|
|
306
|
+
assert.ok(snapshot.responses.every((response) => Object.values(response.selected).every((field) => field.state === "unavailable")));
|
|
307
|
+
assert.equal(h.store.get(task.id)?.tokens, 30);
|
|
308
|
+
assert.equal(h.store.get(task.id)?.cost, 0.1 + 0.1 + 0.1);
|
|
309
|
+
assert.equal(h.store.get(task.id)?.model, "anthropic/launch");
|
|
310
|
+
assert.deepEqual(child.written.map((command) => command.type), ["get_state", "prompt"]);
|
|
311
|
+
assert.doesNotMatch(JSON.stringify(snapshot), /private|launch|s1|modelVersion/);
|
|
312
|
+
assert.ok(Object.isFrozen(snapshot) && Object.isFrozen(snapshot.responses) && Object.isFrozen(snapshot.responses[0].tokens.input));
|
|
313
|
+
child.emit({ type: "agent_settled" }); child.exit(0);
|
|
314
|
+
assert.equal(snapshots.length, 1);
|
|
315
|
+
});
|
|
316
|
+
|
|
317
|
+
for (const ending of ["cancel", "exit", "error", "timeout", "settled-error"] as const) test(`bounded response coverage survives ${ending} honestly`, async () => {
|
|
318
|
+
let snapshot: Parameters<NonNullable<RunnerHooks["onFinish"]>>[1];
|
|
319
|
+
const h = harness({ onFinish: (_task, observations) => { snapshot = observations; } });
|
|
320
|
+
const task = h.runner.run(request({ collectResponseObservations: true }));
|
|
321
|
+
await tick();
|
|
322
|
+
const child = h.children[0];
|
|
323
|
+
for (let index = 0; index < 130; index++) child.emit({ type: "message_end", message: {
|
|
324
|
+
role: "assistant", model: `model-${index}`, stopReason: "error", usage: { totalTokens: 1 } } });
|
|
325
|
+
if (ending === "cancel") h.runner.cancel(task.id);
|
|
326
|
+
else if (ending === "exit") child.exit(1);
|
|
327
|
+
else if (ending === "error") child.fail("private process error");
|
|
328
|
+
else if (ending === "timeout") h.timers.filter((timer) => !timer.cancelled && timer.ms === 10_000).at(-1)!.fn();
|
|
329
|
+
else { child.emit({ type: "agent_end", messages: [{ role: "assistant", stopReason: "error" }] }); child.emit({ type: "agent_settled" }); }
|
|
330
|
+
await h.runner.waitFor(task.id);
|
|
331
|
+
assert.ok(snapshot);
|
|
332
|
+
assert.equal(snapshot.responses.length, 128);
|
|
333
|
+
assert.equal(snapshot.droppedResponses, 2);
|
|
334
|
+
assert.equal(snapshot.agentSettled, ending === "settled-error");
|
|
335
|
+
assert.equal(h.store.get(task.id)?.tokens, 130, "buffer cap never caps existing UI totals");
|
|
336
|
+
assert.notEqual(h.store.get(task.id)?.status, TASK_STATUS.COMPLETED);
|
|
337
|
+
child.emit({ type: "message_end", message: { role: "assistant", stopReason: "stop" } });
|
|
338
|
+
assert.equal(snapshot.responses.length, 128);
|
|
339
|
+
assert.equal(h.finishes.length, 1);
|
|
340
|
+
});
|
|
341
|
+
|
|
342
|
+
test("response buffering is disabled by default and absent for tasks cancelled before launch", async () => {
|
|
343
|
+
const snapshots: unknown[] = [];
|
|
344
|
+
const h = harness({ maxConcurrency: 1, onFinish: (_task, snapshot) => snapshots.push(snapshot) });
|
|
345
|
+
const task = h.runner.run(request());
|
|
346
|
+
const queued = h.runner.run(request({ collectResponseObservations: true }));
|
|
347
|
+
await tick();
|
|
348
|
+
h.children[0].emit({ type: "message_end", message: { role: "assistant", stopReason: "stop", usage: { totalTokens: 7 } } });
|
|
349
|
+
h.runner.cancel(queued.id); h.runner.cancel(task.id);
|
|
350
|
+
await tick();
|
|
351
|
+
assert.deepEqual(snapshots, [undefined, undefined]);
|
|
352
|
+
assert.equal(h.store.get(task.id)?.tokens, 7);
|
|
353
|
+
});
|
|
354
|
+
|
|
355
|
+
test("childArguments builds an rpc launch with model, thinking, tools, session dir, and instructions", () => {
|
|
356
|
+
const args = childArguments(request());
|
|
357
|
+
assert.deepEqual(args.slice(0, 2), ["--mode", "rpc"]);
|
|
358
|
+
assert.ok(args.includes("--session-dir") && args[args.indexOf("--session-dir") + 1] === "/sessions");
|
|
359
|
+
assert.equal(args[args.indexOf("--model") + 1], "openai-codex/gpt-5.6-terra:high");
|
|
360
|
+
assert.equal(args[args.indexOf("--tools") + 1], "read,grep,subagent_parent_message");
|
|
361
|
+
assert.equal(args[args.indexOf("--append-system-prompt") + 1], "You map things.");
|
|
362
|
+
assert.ok(!args.includes("--session"));
|
|
363
|
+
const resumed = childArguments(request({ resumeSessionPath: "/sessions/old.jsonl", model: undefined, thinking: undefined, agent: { ...explorer, tools: [] } }));
|
|
364
|
+
assert.equal(resumed[resumed.indexOf("--session") + 1], "/sessions/old.jsonl");
|
|
365
|
+
assert.ok(!resumed.includes("--model") && !resumed.includes("--tools"));
|
|
366
|
+
});
|
|
367
|
+
|
|
368
|
+
test("childArguments preserves a max profile instead of the definition's medium effort", () => {
|
|
369
|
+
const agent: AgentDefinition = { ...explorer, name: "worker", thinking: "medium" };
|
|
370
|
+
const config = parseAgentsConfig({ model_profiles: { worker: { model: "openai-codex/gpt-5.6-luna", effort: "max" } } }, undefined);
|
|
371
|
+
const profile = resolveAgentProfile(agent, config);
|
|
372
|
+
const args = childArguments(request({ agent, model: profile.model, thinking: profile.thinking }));
|
|
373
|
+
assert.equal(args[args.indexOf("--model") + 1], "openai-codex/gpt-5.6-luna:max");
|
|
374
|
+
});
|
|
375
|
+
|
|
376
|
+
test("childArguments preserves a max default without a selected model", () => {
|
|
377
|
+
const profile = resolveAgentProfile(explorer, parseAgentsConfig({ default_effort: "max" }, undefined));
|
|
378
|
+
const args = childArguments(request({ model: profile.model, thinking: profile.thinking }));
|
|
379
|
+
assert.ok(!args.includes("--model"));
|
|
380
|
+
assert.equal(args[args.indexOf("--thinking") + 1], "max");
|
|
381
|
+
});
|
|
382
|
+
|
|
383
|
+
test("childArguments grants every child the notification-only parent message tool", () => {
|
|
384
|
+
const args = childArguments(request());
|
|
385
|
+
assert.equal(args[args.indexOf("--tools") + 1], "read,grep,subagent_parent_message");
|
|
386
|
+
});
|
|
387
|
+
|
|
388
|
+
test("AgentRunner admits strict live notifications once and closes IPC before Stop", async () => {
|
|
389
|
+
const notifications: string[] = [];
|
|
390
|
+
const { runner, children, spawnOptions } = harness({ onNotification: (task, message) => task.parentSessionId === "s1" && (notifications.push(message), true) });
|
|
391
|
+
const task = runner.run(request());
|
|
392
|
+
await tick();
|
|
393
|
+
children[0].message({ id: "n1", kind: "notification", message: "checkpoint" });
|
|
394
|
+
children[0].message({ id: "n1", kind: "notification", message: "checkpoint" });
|
|
395
|
+
children[0].message({ id: "n2", kind: "notification", message: "x".repeat(8 * 1024 + 1) });
|
|
396
|
+
children[0].message({ id: "q3", kind: "query", message: "unsupported" });
|
|
397
|
+
children[0].message({ id: "n4", kind: "notification", message: "\uD800" });
|
|
398
|
+
children[0].message({ id: "n5", kind: "notification", message: "forged field", sender: "forged" });
|
|
399
|
+
children[0].message({ id: "n0", kind: "notification", message: "invalid correlation" });
|
|
400
|
+
children[0].message({ id: `n${"1".repeat(1_000)}`, kind: "notification", message: "invalid correlation" });
|
|
401
|
+
await tick();
|
|
402
|
+
assert.deepEqual(spawnOptions[0]?.stdio, ["pipe", "pipe", "pipe", "ipc"]);
|
|
403
|
+
assert.deepEqual(notifications, ["checkpoint"]);
|
|
404
|
+
assert.deepEqual(children[0].sent, [
|
|
405
|
+
{ id: "n1", kind: "ack", accepted: true },
|
|
406
|
+
{ id: "n2", kind: "ack", accepted: false, error: "invalid child IPC message" },
|
|
407
|
+
{ id: "q3", kind: "reply", error: "task parent cannot accept queries" },
|
|
408
|
+
{ id: "n4", kind: "ack", accepted: false, error: "invalid child IPC message" },
|
|
409
|
+
{ id: "n5", kind: "ack", accepted: false, error: "invalid child IPC frame" },
|
|
410
|
+
]);
|
|
411
|
+
runner.cancel(task.id);
|
|
412
|
+
children[0].message({ id: "after-stop", kind: "notification", message: "ignored" });
|
|
413
|
+
await tick();
|
|
414
|
+
assert.equal(children[0].sent.length, 5);
|
|
415
|
+
assert.ok(children[0].disconnects > 0);
|
|
416
|
+
});
|
|
417
|
+
|
|
418
|
+
test("AgentRunner rejects notifications from an inactive parent session with a static acknowledgement", async () => {
|
|
419
|
+
const { runner, children } = harness({ onNotification: () => false });
|
|
420
|
+
runner.run(request());
|
|
421
|
+
await tick();
|
|
422
|
+
children[0].message({ id: "n1", kind: "notification", message: "not active" });
|
|
423
|
+
await tick();
|
|
424
|
+
assert.deepEqual(children[0].sent, [{ id: "n1", kind: "ack", accepted: false, error: "task parent is not the active host session" }]);
|
|
425
|
+
});
|
|
426
|
+
|
|
427
|
+
test("AgentRunner retains only a 64-notification duplicate window", async () => {
|
|
428
|
+
const notifications: string[] = [];
|
|
429
|
+
const { runner, children } = harness({ onNotification: (_task, message) => { notifications.push(message); } });
|
|
430
|
+
runner.run(request());
|
|
431
|
+
await tick();
|
|
432
|
+
for (let index = 1; index <= 65; index += 1) children[0].message({ id: `n${index}`, kind: "notification", message: `message ${index}` });
|
|
433
|
+
children[0].message({ id: "n1", kind: "notification", message: "message 1 again" });
|
|
434
|
+
await tick();
|
|
435
|
+
assert.equal(notifications.length, 66, "an ID evicted from the recent 64-ack window can be admitted again");
|
|
436
|
+
});
|
|
437
|
+
|
|
438
|
+
test("piCommand reuses the running pi entry point and honors the override", () => {
|
|
439
|
+
assert.deepEqual(piCommand({ execPath: "/bin/node", argv: ["/bin/node", "/x/dist/cli.js"], env: {} }), { command: "/bin/node", args: ["/x/dist/cli.js"] });
|
|
440
|
+
assert.deepEqual(piCommand({ execPath: "/bin/node", argv: ["/bin/node", "/x/other.js"], env: {} }), { command: "pi", args: [] });
|
|
441
|
+
assert.deepEqual(piCommand({ execPath: "/bin/node", argv: [], env: { GENTLE_PI_AGENTS_PI: "/opt/pi --flag" } }), { command: "/opt/pi", args: ["--flag"] });
|
|
442
|
+
});
|
|
443
|
+
|
|
444
|
+
test("JsonLines splits on LF only, tolerates CRLF, and skips lines that are not JSON", () => {
|
|
445
|
+
const seen: unknown[] = [];
|
|
446
|
+
const lines = new JsonLines((value) => seen.push(value));
|
|
447
|
+
lines.push('{"a":1}\r\n{"b":"x
y"}\nnot json\n{"c":');
|
|
448
|
+
lines.push("3}\n");
|
|
449
|
+
assert.deepEqual(seen, [{ a: 1 }, { b: "x
y" }, { c: 3 }]);
|
|
450
|
+
});
|
|
451
|
+
|
|
452
|
+
test("AgentRunner runs a task end to end: prompt, deltas into the store, completion with the last answer", async () => {
|
|
453
|
+
const { store, runner, children } = harness();
|
|
454
|
+
const task = runner.run(request());
|
|
455
|
+
assert.equal(task.status, TASK_STATUS.QUEUED);
|
|
456
|
+
await tick();
|
|
457
|
+
assert.equal(store.get(task.id)?.status, TASK_STATUS.RUNNING);
|
|
458
|
+
const [child] = children;
|
|
459
|
+
await tick();
|
|
460
|
+
assert.deepEqual(children[0].written.map((command) => command.type), ["get_state", "prompt"]);
|
|
461
|
+
assert.equal(children[0].written[1].message, "Map the repo");
|
|
462
|
+
child.emit({ type: "tool_execution_start", toolCallId: "c1", toolName: "grep", args: {} });
|
|
463
|
+
child.emit({ type: "message_update", assistantMessageEvent: { type: "text_delta", delta: "Found it" } });
|
|
464
|
+
child.emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "Found it" }] }] });
|
|
465
|
+
await tick();
|
|
466
|
+
assert.equal(store.get(task.id)?.status, TASK_STATUS.RUNNING, "agent_end retains the latest answer while queued follow-up may still run");
|
|
467
|
+
assert.equal(children[0].killed.length, 0, "the child remains available until Pi reports settlement");
|
|
468
|
+
child.emit({ type: "agent_settled" });
|
|
469
|
+
await tick();
|
|
470
|
+
const finished = store.get(task.id);
|
|
471
|
+
assert.equal(finished?.status, TASK_STATUS.COMPLETED);
|
|
472
|
+
assert.equal(finished?.result, "Found it");
|
|
473
|
+
assert.equal(finished?.toolCalls, 1);
|
|
474
|
+
assert.equal(finished?.sessionPath, "/sessions/child.jsonl");
|
|
475
|
+
assert.equal(finished?.label, "Map the repo");
|
|
476
|
+
assert.ok(children[0].killed.length > 0, "the child is stopped once the answer is in");
|
|
477
|
+
assert.equal(store.thread(task.id).items.length, 2);
|
|
478
|
+
assert.equal((await runner.waitFor(task.id)).status, TASK_STATUS.COMPLETED);
|
|
479
|
+
});
|
|
480
|
+
|
|
481
|
+
test("AgentRunner waits for child exit after settlement before releasing its queue slot or finishing twice", async () => {
|
|
482
|
+
const { store, runner, children, finishes } = harness({ maxConcurrency: 1, exitOnKill: false });
|
|
483
|
+
const first = runner.run(request());
|
|
484
|
+
const second = runner.run(request({ prompt: "Second" }));
|
|
485
|
+
await tick();
|
|
486
|
+
assert.equal(children.length, 1);
|
|
487
|
+
children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "Answer" }] }] });
|
|
488
|
+
await tick();
|
|
489
|
+
assert.equal(store.get(first.id)?.status, TASK_STATUS.RUNNING, "agent_end ends one run, not the session");
|
|
490
|
+
assert.equal(store.get(first.id)?.result, "Answer", "agent_end retains the final run output");
|
|
491
|
+
assert.equal(store.get(second.id)?.status, TASK_STATUS.QUEUED, "the slot stays occupied until settlement");
|
|
492
|
+
assert.deepEqual(finishes, []);
|
|
493
|
+
children[0].emit({ type: "agent_settled" });
|
|
494
|
+
await tick();
|
|
495
|
+
assert.equal(store.get(first.id)?.status, TASK_STATUS.RUNNING, "terminal RPC state does not release a live process");
|
|
496
|
+
assert.equal(store.get(second.id)?.status, TASK_STATUS.QUEUED);
|
|
497
|
+
children[0].exit(0);
|
|
498
|
+
await tick();
|
|
499
|
+
await tick();
|
|
500
|
+
assert.equal(store.get(first.id)?.status, TASK_STATUS.COMPLETED);
|
|
501
|
+
assert.deepEqual(finishes, [first.id], "settlement delivers completion once");
|
|
502
|
+
assert.equal(children.length, 2, "child exit releases the queue slot");
|
|
503
|
+
children[0].emit({ type: "agent_settled" });
|
|
504
|
+
await tick();
|
|
505
|
+
assert.deepEqual(finishes, [first.id], "duplicate terminal events do not finalize twice");
|
|
506
|
+
});
|
|
507
|
+
|
|
508
|
+
test("AgentRunner queues beyond max concurrency and starts the next task when one finishes", async () => {
|
|
509
|
+
const { store, runner, children } = harness({ maxConcurrency: 1 });
|
|
510
|
+
const first = runner.run(request());
|
|
511
|
+
const second = runner.run(request({ prompt: "Second" }));
|
|
512
|
+
await tick();
|
|
513
|
+
assert.equal(children.length, 1);
|
|
514
|
+
assert.equal(store.get(second.id)?.status, TASK_STATUS.QUEUED);
|
|
515
|
+
children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "First complete." }], stopReason: "stop" }] });
|
|
516
|
+
await tick();
|
|
517
|
+
assert.equal(store.get(first.id)?.status, TASK_STATUS.RUNNING, "the concurrency slot remains held through a queued follow-up");
|
|
518
|
+
children[0].emit({ type: "agent_settled" });
|
|
519
|
+
await tick();
|
|
520
|
+
await tick();
|
|
521
|
+
assert.equal(store.get(first.id)?.status, TASK_STATUS.COMPLETED);
|
|
522
|
+
assert.equal(children.length, 2);
|
|
523
|
+
assert.equal(store.get(second.id)?.status, TASK_STATUS.RUNNING);
|
|
524
|
+
});
|
|
525
|
+
|
|
526
|
+
test("AgentRunner classifies terminal assistant outcomes only after settlement", async () => {
|
|
527
|
+
const scenarios = [
|
|
528
|
+
{ name: "error", messages: [{ role: "assistant", content: [], stopReason: "error", errorMessage: "WebSocket error: secret=never-copy" }], status: TASK_STATUS.FAILED, error: /assistant reported an error/ },
|
|
529
|
+
{ name: "aborted", messages: [{ role: "assistant", content: [], stopReason: "aborted" }], status: TASK_STATUS.FAILED, error: /assistant aborted/ },
|
|
530
|
+
{ name: "empty", messages: [{ role: "assistant", content: [], stopReason: "stop" }], status: TASK_STATUS.FAILED, error: /no final report/ },
|
|
531
|
+
{ name: "success", messages: [{ role: "assistant", content: [{ type: "text", text: "final report" }], stopReason: "stop" }], status: TASK_STATUS.COMPLETED, error: null },
|
|
532
|
+
] as const;
|
|
533
|
+
for (const scenario of scenarios) {
|
|
534
|
+
const { store, runner, children } = harness();
|
|
535
|
+
const task = runner.run(request());
|
|
536
|
+
await tick();
|
|
537
|
+
children[0].emit({ type: "agent_end", messages: scenario.messages });
|
|
538
|
+
assert.equal(store.get(task.id)?.status, TASK_STATUS.RUNNING, `${scenario.name} stays running until settlement`);
|
|
539
|
+
children[0].emit({ type: "agent_settled" });
|
|
540
|
+
const finished = await runner.waitFor(task.id);
|
|
541
|
+
assert.equal(finished.status, scenario.status, scenario.name);
|
|
542
|
+
if (scenario.error) assert.match(finished.error ?? "", scenario.error);
|
|
543
|
+
else assert.equal(finished.result, "final report");
|
|
544
|
+
}
|
|
545
|
+
});
|
|
546
|
+
|
|
547
|
+
test("AgentRunner clears an earlier answer after a later error, but permits a successful retry before settlement", async () => {
|
|
548
|
+
const first = harness();
|
|
549
|
+
const failedTask = first.runner.run(request());
|
|
550
|
+
await tick();
|
|
551
|
+
first.children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "stale success" }], stopReason: "stop" }] });
|
|
552
|
+
first.children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [], stopReason: "error", errorMessage: "provider detail must not persist" }] });
|
|
553
|
+
first.children[0].emit({ type: "agent_settled" });
|
|
554
|
+
const failed = await first.runner.waitFor(failedTask.id);
|
|
555
|
+
assert.equal(failed.status, TASK_STATUS.FAILED);
|
|
556
|
+
assert.equal(failed.result, null, "a later error must not report stale successful text");
|
|
557
|
+
|
|
558
|
+
const retry = harness();
|
|
559
|
+
const retryTask = retry.runner.run(request());
|
|
560
|
+
await tick();
|
|
561
|
+
retry.children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [], stopReason: "error" }] });
|
|
562
|
+
retry.children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "retry report" }], stopReason: "stop" }] });
|
|
563
|
+
retry.children[0].emit({ type: "agent_settled" });
|
|
564
|
+
const recovered = await retry.runner.waitFor(retryTask.id);
|
|
565
|
+
assert.equal(recovered.status, TASK_STATUS.COMPLETED);
|
|
566
|
+
assert.equal(recovered.result, "retry report");
|
|
567
|
+
});
|
|
568
|
+
|
|
569
|
+
test("AgentRunner fails if the child exits after agent_end but before agent_settled", async () => {
|
|
570
|
+
const { store, runner, children } = harness();
|
|
571
|
+
const task = runner.run(request());
|
|
572
|
+
await tick();
|
|
573
|
+
children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "partial answer" }] }] });
|
|
574
|
+
await tick();
|
|
575
|
+
children[0].exit(0);
|
|
576
|
+
await tick();
|
|
577
|
+
assert.equal(store.get(task.id)?.status, TASK_STATUS.FAILED);
|
|
578
|
+
assert.match(store.get(task.id)?.error ?? "", /before agent_settled/);
|
|
579
|
+
assert.equal(store.get(task.id)?.result, "partial answer", "the final observed answer remains available for diagnostics");
|
|
580
|
+
});
|
|
581
|
+
|
|
582
|
+
for (const [platform, detached] of [["win32", false], ["linux", true]] as const) test(`AgentRunner selects detached=${detached} for ${platform} without changing the launch contract`, async () => {
|
|
583
|
+
const store = new TaskStore();
|
|
584
|
+
const launches: Array<{ command: string; args: string[]; options: Parameters<RunnerDeps["spawn"]>[2] }> = [];
|
|
585
|
+
const child = fakeChild();
|
|
586
|
+
const runner = new AgentRunner(store, { maxConcurrency: 1, stallTimeoutMs: 1_000 }, {
|
|
587
|
+
spawn: (command, args, options) => {
|
|
588
|
+
launches.push({ command, args, options });
|
|
589
|
+
return child.child;
|
|
590
|
+
},
|
|
591
|
+
now: () => 1,
|
|
592
|
+
schedule: () => () => {},
|
|
593
|
+
pi: { command: "pi-fixture", args: ["--from-host"] },
|
|
594
|
+
process: { platform, kill: () => {} },
|
|
595
|
+
}, { askUser: async () => ({ cancelled: true }) });
|
|
596
|
+
const task = runner.run(request({ env: { PATH: "/fixture", KEEP: "yes" } }));
|
|
597
|
+
await tick();
|
|
598
|
+
const ownedIpc = launches[0]?.options.env.GENTLE_PI_AGENTS_OWNED_IPC;
|
|
599
|
+
assert.match(ownedIpc ?? "", /^\d+-[a-z0-9]+$/, "the runner creates an opaque owned-IPC marker");
|
|
600
|
+
assert.deepEqual(launches, [{
|
|
601
|
+
command: "pi-fixture",
|
|
602
|
+
args: ["--from-host", "--mode", "rpc", "--session-dir", "/sessions", "--model", "openai-codex/gpt-5.6-terra:high", "--tools", "read,grep,subagent_parent_message", "--append-system-prompt", "You map things."],
|
|
603
|
+
options: { cwd: "/repo", env: { PATH: "/fixture", KEEP: "yes", GENTLE_PI_AGENTS_CHILD: "1", GENTLE_PI_AGENTS_OWNED_IPC: ownedIpc }, detached, stdio: ["pipe", "pipe", "pipe", "ipc"] },
|
|
604
|
+
}]);
|
|
605
|
+
child.emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "platform checked" }], stopReason: "stop" }] });
|
|
606
|
+
child.emit({ type: "agent_settled" });
|
|
607
|
+
assert.equal((await runner.waitFor(task.id)).status, TASK_STATUS.COMPLETED);
|
|
608
|
+
});
|
|
609
|
+
|
|
610
|
+
test("AgentRunner retains permission broker fd3 and assigns messaging IPC to fd4", async () => {
|
|
611
|
+
const { runner, children, spawnOptions } = harness();
|
|
612
|
+
const task = runner.run(request({ authorizeParentStandingReviewPermission: () => true }));
|
|
613
|
+
await tick();
|
|
614
|
+
const launch = spawnOptions[0];
|
|
615
|
+
assert.match(launch?.env.GENTLE_PI_AGENTS_OWNED_IPC ?? "", /^\d+-[a-z0-9]+$/, "the owned-IPC marker has the runner's opaque shape");
|
|
616
|
+
assert.deepEqual(launch?.env, { GENTLE_PI_AGENTS_CHILD: "1", GENTLE_PI_AGENTS_OWNED_IPC: launch?.env.GENTLE_PI_AGENTS_OWNED_IPC, GENTLE_PI_AGENTS_PARENT_PERMISSION_FD: "3" });
|
|
617
|
+
assert.deepEqual(launch?.stdio, ["pipe", "pipe", "pipe", "pipe", "ipc"]);
|
|
618
|
+
assert.equal(launch?.stdio?.length, 5);
|
|
619
|
+
children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "channel checked" }], stopReason: "stop" }] });
|
|
620
|
+
children[0].emit({ type: "agent_settled" });
|
|
621
|
+
assert.equal((await runner.waitFor(task.id)).status, TASK_STATUS.COMPLETED);
|
|
622
|
+
});
|
|
623
|
+
|
|
624
|
+
test("AgentRunner answers dialogs through askUser in task mode and cancels them in background mode", async () => {
|
|
625
|
+
const { store, runner, children, asks } = harness({ answer: { confirmed: true } });
|
|
626
|
+
const task = runner.run(request());
|
|
627
|
+
const background = runner.run(request({ mode: AGENT_MODE.BACKGROUND }));
|
|
628
|
+
await tick();
|
|
629
|
+
children[0].emit({ type: "extension_ui_request", id: "u1", method: "confirm", title: "Delete?" });
|
|
630
|
+
children[1].emit({ type: "extension_ui_request", id: "u2", method: "select", title: "Pick", options: ["a"] });
|
|
631
|
+
children[1].emit({ type: "extension_ui_request", id: "u3", method: "notify", message: "hi" });
|
|
632
|
+
await tick();
|
|
633
|
+
await tick();
|
|
634
|
+
assert.deepEqual(asks, [{ taskId: task.id, method: "confirm" }]);
|
|
635
|
+
assert.deepEqual(children[0].written.at(-1), { type: "extension_ui_response", id: "u1", confirmed: true });
|
|
636
|
+
assert.deepEqual(children[1].written.at(-1), { type: "extension_ui_response", id: "u2", cancelled: true });
|
|
637
|
+
assert.equal(store.get(background.id)?.status, TASK_STATUS.RUNNING);
|
|
638
|
+
assert.equal(store.get(task.id)?.status, TASK_STATUS.RUNNING, "answered questions do not leave the task waiting");
|
|
639
|
+
});
|
|
640
|
+
|
|
641
|
+
test("AgentRunner cancels and fails when the child exits early", async () => {
|
|
642
|
+
const { store, runner, children } = harness({ maxConcurrency: 3 });
|
|
643
|
+
const cancelled = runner.run(request());
|
|
644
|
+
const crashed = runner.run(request());
|
|
645
|
+
await tick();
|
|
646
|
+
runner.cancel(cancelled.id);
|
|
647
|
+
await tick();
|
|
648
|
+
assert.equal(store.get(cancelled.id)?.status, TASK_STATUS.CANCELLED);
|
|
649
|
+
assert.ok(children[0].written.some((command) => command.type === "abort"));
|
|
650
|
+
children[1].exit(1);
|
|
651
|
+
await tick();
|
|
652
|
+
assert.equal(store.get(crashed.id)?.status, TASK_STATUS.FAILED);
|
|
653
|
+
assert.match(store.get(crashed.id)?.error ?? "", /exited with code 1/);
|
|
654
|
+
assert.ok(runner.steer(cancelled.id, "x") === false, "a finished task cannot be steered");
|
|
655
|
+
});
|
|
656
|
+
|
|
657
|
+
test("AgentRunner has no total-duration watchdog but keeps active work alive and times out true silence", async () => {
|
|
658
|
+
const { store, runner, children, timers } = harness();
|
|
659
|
+
const task = runner.run(request({ mode: AGENT_MODE.BACKGROUND }));
|
|
660
|
+
await tick();
|
|
661
|
+
assert.deepEqual(timers.filter((timer) => !timer.cancelled).map((timer) => timer.ms), [10_000], "only the inactivity watchdog is scheduled");
|
|
662
|
+
const initialStall = timers[0];
|
|
663
|
+
children[0].emit({ type: "response", id: "r1", success: true });
|
|
664
|
+
await tick();
|
|
665
|
+
assert.equal(initialStall.cancelled, true, "every child RPC event, including a response, re-arms the inactivity watchdog");
|
|
666
|
+
children[0].emit({ type: "message_update", assistantMessageEvent: { type: "text_delta", delta: "still working" } });
|
|
667
|
+
await tick();
|
|
668
|
+
assert.equal(store.get(task.id)?.status, TASK_STATUS.RUNNING, "ongoing RPC activity keeps a long-running task active");
|
|
669
|
+
const stall = timers.filter((timer) => timer.ms === 10_000 && !timer.cancelled).at(-1);
|
|
670
|
+
assert.ok(stall);
|
|
671
|
+
stall.fn();
|
|
672
|
+
await tick();
|
|
673
|
+
assert.equal(store.get(task.id)?.status, TASK_STATUS.TIMED_OUT);
|
|
674
|
+
assert.match(store.get(task.id)?.error ?? "", /stalled/);
|
|
675
|
+
});
|
|
676
|
+
|
|
677
|
+
test("AgentRunner.cancelAll stops every queued and running task", async () => {
|
|
678
|
+
const { store, runner, children } = harness({ maxConcurrency: 1 });
|
|
679
|
+
const running = runner.run(request());
|
|
680
|
+
const queued = runner.run(request());
|
|
681
|
+
await tick();
|
|
682
|
+
assert.equal(runner.cancelAll(), 2);
|
|
683
|
+
await tick();
|
|
684
|
+
assert.equal(store.get(running.id)?.status, TASK_STATUS.CANCELLED);
|
|
685
|
+
assert.equal(store.get(queued.id)?.status, TASK_STATUS.CANCELLED);
|
|
686
|
+
assert.deepEqual(children[0].killed, ["SIGTERM"]);
|
|
687
|
+
assert.equal(children.length, 1, "nothing else starts after cancelAll");
|
|
688
|
+
});
|
|
689
|
+
|
|
690
|
+
test("AgentRunner fails only the task when the child cannot start, and the queue moves on", async () => {
|
|
691
|
+
const { store, runner, children, timers } = harness({ maxConcurrency: 1 });
|
|
692
|
+
const broken = runner.run(request());
|
|
693
|
+
const next = runner.run(request({ prompt: "After" }));
|
|
694
|
+
await tick();
|
|
695
|
+
children[0].fail("spawn pi ENOENT");
|
|
696
|
+
await tick();
|
|
697
|
+
assert.equal(store.get(broken.id)?.status, TASK_STATUS.FAILED);
|
|
698
|
+
assert.match(store.get(broken.id)?.error ?? "", /could not start pi: spawn pi ENOENT/);
|
|
699
|
+
assert.equal((await runner.waitFor(broken.id)).status, TASK_STATUS.FAILED, "waiters settle");
|
|
700
|
+
await tick();
|
|
701
|
+
assert.equal(children.length, 2, "the next queued task starts");
|
|
702
|
+
assert.equal(store.get(next.id)?.status, TASK_STATUS.RUNNING);
|
|
703
|
+
assert.ok(timers.filter((timer) => timer.ms === 10_000).some((timer) => timer.cancelled), "the failed task's inactivity watchdog is cancelled");
|
|
704
|
+
});
|
|
705
|
+
|
|
706
|
+
test("AgentRunner turns a synchronous spawn exception into a failed task", async () => {
|
|
707
|
+
const store = new TaskStore();
|
|
708
|
+
const runner = new AgentRunner(store, { maxConcurrency: 1, stallTimeoutMs: 1000 }, {
|
|
709
|
+
spawn: () => {
|
|
710
|
+
throw new Error("ENOENT: pi not found");
|
|
711
|
+
},
|
|
712
|
+
now: () => 1,
|
|
713
|
+
schedule: () => () => {},
|
|
714
|
+
pi: { command: "missing-pi", args: [] },
|
|
715
|
+
}, { askUser: async () => ({ cancelled: true }) });
|
|
716
|
+
const task = runner.run(request());
|
|
717
|
+
const finished = await runner.waitFor(task.id);
|
|
718
|
+
assert.equal(finished.status, TASK_STATUS.FAILED);
|
|
719
|
+
assert.match(finished.error ?? "", /could not start pi: ENOENT/);
|
|
720
|
+
});
|
|
721
|
+
|
|
722
|
+
for (const lateEvents of [false, true]) test(`AgentRunner releases quarantined capacity only on proven exit (late events: ${lateEvents})`, async () => {
|
|
723
|
+
const store = new TaskStore();
|
|
724
|
+
const timers: Array<{ fn: () => void; ms: number; cancelled: boolean }> = [];
|
|
725
|
+
let now = 0;
|
|
726
|
+
let groupGone = false;
|
|
727
|
+
let launches = 0;
|
|
728
|
+
let asks = 0;
|
|
729
|
+
const finishes: string[] = [];
|
|
730
|
+
const observations: Parameters<NonNullable<RunnerHooks["onFinish"]>>[1][] = [];
|
|
731
|
+
let resolveAnswer!: (answer: { value: string }) => void;
|
|
732
|
+
const answer = new Promise<{ value: string }>((resolve) => { resolveAnswer = resolve; });
|
|
733
|
+
const child = fakeChild({ exitOnKill: false, pid: 71 });
|
|
734
|
+
const runner = new AgentRunner(store, { maxConcurrency: 1, stallTimeoutMs: 10_000 }, {
|
|
735
|
+
spawn: () => { launches += 1; return launches === 1 ? child.child : fakeChild().child; },
|
|
736
|
+
now: () => now,
|
|
737
|
+
schedule: (fn, ms) => {
|
|
738
|
+
const timer = { fn, ms, cancelled: false };
|
|
739
|
+
timers.push(timer);
|
|
740
|
+
return () => { timer.cancelled = true; };
|
|
741
|
+
},
|
|
742
|
+
pi: { command: "pi", args: [] },
|
|
743
|
+
process: { platform: "linux", kill: (_pid, signal) => {
|
|
744
|
+
if (signal === 0) throw Object.assign(new Error("group probe"), { code: groupGone ? "ESRCH" : "EPERM" });
|
|
745
|
+
} },
|
|
746
|
+
}, { askUser: async () => { asks += 1; return answer; }, onFinish: (task, snapshot) => { finishes.push(task.id); observations.push(snapshot); } });
|
|
747
|
+
const first = runner.run(request({ collectResponseObservations: true }));
|
|
748
|
+
const second = runner.run(request({ prompt: "queued" }));
|
|
749
|
+
await tick();
|
|
750
|
+
const waiter = runner.waitFor(first.id);
|
|
751
|
+
child.emit({ type: "message_end", message: { role: "assistant", stopReason: "aborted", usage: { input: 3 } } });
|
|
752
|
+
if (lateEvents) child.emit({ type: "extension_ui_request", id: "early", method: "input", title: "Pending?" });
|
|
753
|
+
runner.cancel(first.id);
|
|
754
|
+
const grace = timers.find((timer) => timer.ms === 250);
|
|
755
|
+
assert.ok(grace);
|
|
756
|
+
grace.fn();
|
|
757
|
+
now = 2_000;
|
|
758
|
+
const check = timers.filter((timer) => timer.ms === 25).at(-1);
|
|
759
|
+
assert.ok(check);
|
|
760
|
+
check.fn();
|
|
761
|
+
await tick();
|
|
762
|
+
assert.equal(store.get(first.id)?.status, TASK_STATUS.FAILED);
|
|
763
|
+
assert.equal((await waiter).status, TASK_STATUS.FAILED);
|
|
764
|
+
assert.match(store.get(first.id)?.error ?? "", /cleanup unconfirmed/);
|
|
765
|
+
assert.equal(observations.length, 1);
|
|
766
|
+
assert.equal(observations[0]?.agentSettled, false);
|
|
767
|
+
assert.equal(observations[0]?.responses.length, 1);
|
|
768
|
+
assert.equal(store.get(second.id)?.status, TASK_STATUS.QUEUED, "the unconfirmed group retains its capacity");
|
|
769
|
+
assert.equal(timers.filter((timer) => timer.ms === 25 && !timer.cancelled).length, 0, "confirmation polling stops at its deadline");
|
|
770
|
+
const finished = structuredClone(store.get(first.id));
|
|
771
|
+
if (lateEvents) {
|
|
772
|
+
const thread = structuredClone(store.thread(first.id));
|
|
773
|
+
const timerCount = timers.length;
|
|
774
|
+
const writes = child.written.length;
|
|
775
|
+
resolveAnswer({ value: "too late" });
|
|
776
|
+
await tick();
|
|
777
|
+
child.emit({ type: "extension_ui_request", id: "late", method: "input", title: "Reopen?" });
|
|
778
|
+
child.emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "late result" }] }] });
|
|
779
|
+
child.emit({ type: "message_update", assistantMessageEvent: { type: "text_delta", delta: "late" } });
|
|
780
|
+
child.emit({ type: "agent_settled" });
|
|
781
|
+
await tick();
|
|
782
|
+
assert.equal(asks, 1, "late dialogs must not reopen");
|
|
783
|
+
assert.equal(child.written.length, writes, "pending answers must not reach a terminal child");
|
|
784
|
+
assert.equal(timers.length, timerCount, "late activity must not rearm the stall watchdog");
|
|
785
|
+
assert.deepEqual(store.get(first.id), finished);
|
|
786
|
+
assert.deepEqual(store.thread(first.id), thread);
|
|
787
|
+
assert.equal(launches, 1, "late events are not process-exit proof");
|
|
788
|
+
}
|
|
789
|
+
groupGone = true;
|
|
790
|
+
child.exit(0);
|
|
791
|
+
await tick();
|
|
792
|
+
assert.equal(launches, 2, "proven late exit must pump queued work");
|
|
793
|
+
assert.equal(store.get(second.id)?.status, TASK_STATUS.RUNNING);
|
|
794
|
+
child.exit(0);
|
|
795
|
+
await tick();
|
|
796
|
+
assert.deepEqual(finishes, [first.id], "cleanup must not finish the quarantined task twice");
|
|
797
|
+
assert.equal(observations.length, 1, "late cleanup does not redeliver observations");
|
|
798
|
+
assert.deepEqual(store.get(first.id), finished);
|
|
799
|
+
});
|
|
800
|
+
|
|
801
|
+
test("a confirmed-gone group completes the exit and frees its slot without an observed exit", async () => {
|
|
802
|
+
// The group probe reports ESRCH (the group is gone) while the child never emits
|
|
803
|
+
// its exit event. Returning without finishing left the task terminal in memory
|
|
804
|
+
// with no record and no retry; finishing without releasing the live entry would
|
|
805
|
+
// keep the concurrency slot occupied and never pump queued work.
|
|
806
|
+
const store = new TaskStore();
|
|
807
|
+
const timers: Array<{ fn: () => void; ms: number; cancelled: boolean }> = [];
|
|
808
|
+
const finishes: string[] = [];
|
|
809
|
+
let now = 1_000;
|
|
810
|
+
let launches = 0;
|
|
811
|
+
const runner = new AgentRunner(store, { maxConcurrency: 1, stallTimeoutMs: 10_000 }, {
|
|
812
|
+
spawn: () => fakeChild({ exitOnKill: false, pid: 90 + (launches += 1) }).child,
|
|
813
|
+
now: () => now,
|
|
814
|
+
schedule: (fn, ms) => {
|
|
815
|
+
const timer = { fn, ms, cancelled: false };
|
|
816
|
+
timers.push(timer);
|
|
817
|
+
return () => { timer.cancelled = true; };
|
|
818
|
+
},
|
|
819
|
+
pi: { command: "pi", args: [] },
|
|
820
|
+
process: { platform: "linux", kill: (_pid, signal) => {
|
|
821
|
+
if (signal === 0) throw Object.assign(new Error("group probe"), { code: "ESRCH" });
|
|
822
|
+
} },
|
|
823
|
+
}, { askUser: async () => ({ value: "yes" }), onFinish: (task) => { finishes.push(task.id); } });
|
|
824
|
+
const first = runner.run(request());
|
|
825
|
+
const second = runner.run(request({ prompt: "queued" }));
|
|
826
|
+
await tick();
|
|
827
|
+
const waiter = runner.waitFor(first.id);
|
|
828
|
+
runner.cancel(first.id);
|
|
829
|
+
const grace = timers.find((timer) => timer.ms === 250);
|
|
830
|
+
assert.ok(grace, "termination grace is scheduled");
|
|
831
|
+
grace.fn();
|
|
832
|
+
await tick();
|
|
833
|
+
assert.equal(store.get(first.id)?.status, TASK_STATUS.CANCELLED);
|
|
834
|
+
assert.equal((await waiter).status, TASK_STATUS.CANCELLED, "the waiter receives the recorded outcome");
|
|
835
|
+
assert.equal(finishes.length, 1, "the run is recorded exactly once");
|
|
836
|
+
assert.equal(store.get(second.id)?.status, TASK_STATUS.RUNNING, "the freed slot starts queued work");
|
|
837
|
+
});
|
|
838
|
+
|
|
839
|
+
test("an unprobeable process group quarantines at its deadline and still records the run", async () => {
|
|
840
|
+
// On win32 the child is not detached, so there is no process group to probe and
|
|
841
|
+
// an observed exit is the only confirmation available. With no exit event the
|
|
842
|
+
// run must still be recorded at the deadline, and its slot must be retained
|
|
843
|
+
// rather than freed on an unproven assumption.
|
|
844
|
+
const store = new TaskStore();
|
|
845
|
+
const timers: Array<{ fn: () => void; ms: number; cancelled: boolean }> = [];
|
|
846
|
+
const finishes: string[] = [];
|
|
847
|
+
let now = 1_000;
|
|
848
|
+
let launches = 0;
|
|
849
|
+
const runner = new AgentRunner(store, { maxConcurrency: 1, stallTimeoutMs: 10_000 }, {
|
|
850
|
+
spawn: () => fakeChild({ exitOnKill: false, pid: 90 + (launches += 1) }).child, now: () => now,
|
|
851
|
+
schedule: (fn, ms) => {
|
|
852
|
+
const timer = { fn, ms, cancelled: false };
|
|
853
|
+
timers.push(timer);
|
|
854
|
+
return () => { timer.cancelled = true; };
|
|
855
|
+
},
|
|
856
|
+
pi: { command: "pi", args: [] },
|
|
857
|
+
process: { platform: "win32", kill: () => {} },
|
|
858
|
+
}, { askUser: async () => ({ value: "yes" }), onFinish: (task) => { finishes.push(task.id); } });
|
|
859
|
+
const first = runner.run(request());
|
|
860
|
+
const second = runner.run(request({ prompt: "queued" }));
|
|
861
|
+
await tick();
|
|
862
|
+
runner.cancel(first.id);
|
|
863
|
+
const grace = timers.find((timer) => timer.ms === 250);
|
|
864
|
+
assert.ok(grace, "termination grace is scheduled");
|
|
865
|
+
grace.fn();
|
|
866
|
+
now = 5_000;
|
|
867
|
+
const check = timers.filter((timer) => timer.ms === 25).at(-1);
|
|
868
|
+
assert.ok(check, "an unprobeable group must keep polling instead of stopping silently");
|
|
869
|
+
check.fn();
|
|
870
|
+
await tick();
|
|
871
|
+
assert.equal(store.get(first.id)?.status, TASK_STATUS.FAILED);
|
|
872
|
+
assert.match(store.get(first.id)?.error ?? "", /capacity quarantined/);
|
|
873
|
+
assert.equal(finishes.length, 1, "the run is recorded exactly once");
|
|
874
|
+
assert.equal(store.get(second.id)?.status, TASK_STATUS.QUEUED, "an unconfirmed exit retains its capacity");
|
|
875
|
+
assert.equal(launches, 1, "no further launch happens while the slot is quarantined");
|
|
876
|
+
});
|
|
877
|
+
|
|
878
|
+
test("abortReasonText renders an Error, a string, and nothing for unknown reasons", () => {
|
|
879
|
+
assert.equal(abortReasonText(undefined), "");
|
|
880
|
+
assert.equal(abortReasonText(new Error("interrupted by user")), " (interrupted by user)");
|
|
881
|
+
assert.equal(abortReasonText("host timeout"), " (host timeout)");
|
|
882
|
+
assert.equal(abortReasonText(new Error("")), "");
|
|
883
|
+
assert.equal(abortReasonText(42), "");
|
|
884
|
+
});
|
|
885
|
+
|
|
886
|
+
test("research narrowing transport keeps exact argv paths and replaces inherited selection", async () => {
|
|
887
|
+
const h = harness();
|
|
888
|
+
const selection = { documentation: { tools: ["fetch_content"], extensions: { fetch_content: "/installed/docs tools.ts" } } };
|
|
889
|
+
for (const researchSelection of [selection, undefined]) {
|
|
890
|
+
const artifact = { store: "none" as const, worktree: "/work", changeName: "demo", retainedIntent: "denied questions", locators: [] };
|
|
891
|
+
const expected = structuredClone(artifact);
|
|
892
|
+
const launch = request({ researchSelection, researchArtifact: artifact, extensionPaths: researchSelection ? ["/installed/docs tools.ts"] : [],
|
|
893
|
+
env: { PATH: "/bin", GENTLE_PI_RESEARCH_SELECTION: "stale broad selection", GENTLE_PI_RESEARCH_ARTIFACT: "stale broader scope" } });
|
|
894
|
+
const argv = childArguments(launch);
|
|
895
|
+
assert.deepEqual(argv.filter((_, i) => argv[i - 1] === "--extension"), launch.extensionPaths);
|
|
896
|
+
const task = h.runner.run(launch);
|
|
897
|
+
artifact.worktree = "/wrong";
|
|
898
|
+
await tick();
|
|
899
|
+
assert.deepEqual(JSON.parse(h.spawnOptions.at(-1)!.env.GENTLE_PI_RESEARCH_ARTIFACT!), expected);
|
|
900
|
+
assert.deepEqual("researchArtifact" in task ? task.researchArtifact : undefined, expected);
|
|
901
|
+
assert.deepEqual(JSON.parse(h.spawnOptions.at(-1)!.env.GENTLE_PI_RESEARCH_SELECTION!), researchSelection ?? null);
|
|
902
|
+
assert.equal(h.spawnOptions.at(-1)!.env.PATH, "/bin");
|
|
903
|
+
h.runner.cancel(task.id);
|
|
904
|
+
assert.equal((await h.runner.waitFor(task.id)).status, TASK_STATUS.CANCELLED);
|
|
905
|
+
}
|
|
906
|
+
});
|
|
907
|
+
|
|
908
|
+
|
|
909
|
+
test("runner retains remediation observations and awaits terminal settlement", async () => {
|
|
910
|
+
const h = harness({ pid: 123 });
|
|
911
|
+
let finalized: { record: TaskRecord; facts: RemediationTerminalFacts } | undefined;
|
|
912
|
+
let release: () => void;
|
|
913
|
+
const gate = new Promise<void>(resolve => { release = resolve; });
|
|
914
|
+
const remediationPlan: RemediationPlan = { cwd: "/repo", commands: ["pnpm test"], runtimeHarness: { naReason: "Not applicable because the fixture has no runtime boundary." }, rollback: { boundary: "Revert fixture", command: "git diff --check" } };
|
|
915
|
+
const remediation: RemediationTaskState = { failedEvidenceRevision: `sha256:${"a".repeat(64)}`, plan: remediationPlan, pending: {}, observations: [], invalid: false, token: "opaque", acquire: { workspaceRoot: "/repo", changeName: "demo", requestId: "fixture", workUnit: "correct", evidenceGoal: "Observed correction" } };
|
|
916
|
+
const task = h.runner.run(request({ sddRemediation: remediation, finalizeRemediation: async (record, facts) => { finalized = { record, facts }; await gate; } }));
|
|
917
|
+
await tick();
|
|
918
|
+
for (const command of ["pnpm test", "git diff --check"]) {
|
|
919
|
+
h.children[0].emit({ type: "tool_execution_start", toolName: "bash", toolCallId: command, args: { command } });
|
|
920
|
+
h.children[0].emit({ type: "tool_execution_end", toolName: "bash", toolCallId: command, isError: false, result: { content: [{ type: "text", text: "command output" }], details: { remediationCommand: { command, toolCallId: command, cwd: "/repo", exitCode: 0 } } } });
|
|
921
|
+
}
|
|
922
|
+
h.children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "done" }], stopReason: "stop" }] });
|
|
923
|
+
h.children[0].emit({ type: "agent_settled" });
|
|
924
|
+
await tick();
|
|
925
|
+
assert.ok(finalized);
|
|
926
|
+
assert.equal(finalized.record.sddRemediation?.observations.length, 2);
|
|
927
|
+
const prompt = h.children[0].written.find(value => value.type === "prompt").message;
|
|
928
|
+
assert.ok(typeof prompt === "string");
|
|
929
|
+
assert.match(prompt, /pnpm test/);
|
|
930
|
+
assert.doesNotMatch(prompt, /opaque/);
|
|
931
|
+
assert.equal(finalized.facts.cleanupConfirmed, true);
|
|
932
|
+
assert.equal(h.finishes.length, 0);
|
|
933
|
+
release(); await h.runner.waitFor(task.id);
|
|
934
|
+
assert.equal(h.finishes.length, 1);
|
|
935
|
+
assert.equal(remediation.observations.length, 0);
|
|
936
|
+
});
|
|
937
|
+
|
|
938
|
+
|
|
939
|
+
test("admitted no-PID failure settles interrupted before sending a prompt", async () => {
|
|
940
|
+
const h = harness(); let facts;
|
|
941
|
+
const task = h.runner.run(request({ sddRemediation: { plan: {} } as unknown as RemediationTaskState, finalizeRemediation: async (_task, observed) => { facts = observed; } }));
|
|
942
|
+
await tick();
|
|
943
|
+
assert.equal(h.children[0].written.some(value => value.type === "prompt"), false);
|
|
944
|
+
await h.runner.waitFor(task.id);
|
|
945
|
+
assert.equal(facts.spawned, false);
|
|
946
|
+
assert.equal(facts.exited, false);
|
|
947
|
+
assert.equal(facts.cleanupConfirmed, false);
|
|
948
|
+
});
|
|
949
|
+
|
|
950
|
+
|
|
951
|
+
test("admitted synchronous spawn failure finalizes without launching another actor", async () => {
|
|
952
|
+
let spawns = 0, facts;
|
|
953
|
+
const store = new TaskStore();
|
|
954
|
+
const runner = new AgentRunner(store, { maxConcurrency: 1, stallTimeoutMs: 100 }, { spawn: () => { spawns++; throw new Error("spawn refused"); }, pi: { command: "pi", args: [] }, now: () => 1, schedule: () => () => {} }, { askUser: async () => ({}) });
|
|
955
|
+
const task = runner.run(request({ sddRemediation: { plan: {} } as unknown as RemediationTaskState, finalizeRemediation: async (_task, value) => { facts = value; } }));
|
|
956
|
+
await runner.waitFor(task.id);
|
|
957
|
+
assert.deepEqual(facts, { spawned: false, exited: false, cleanupConfirmed: true });
|
|
958
|
+
assert.equal(spawns, 1);
|
|
959
|
+
});
|