gentle-pi 2.4.0 → 2.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. package/README.md +170 -12
  2. package/assets/agents/gentle-ai-worker.md +9 -0
  3. package/assets/orchestrator-delegation.md +19 -9
  4. package/assets/orchestrator.md +5 -5
  5. package/docs/delegated-verification.md +25 -0
  6. package/docs/telemetry.md +38 -0
  7. package/extensions/ask-user-choice.ts +26 -20
  8. package/extensions/codegraph-tools.ts +94 -5
  9. package/extensions/gentle-agents.ts +588 -0
  10. package/extensions/gentle-ai.ts +898 -79
  11. package/extensions/gentle-shell.ts +547 -0
  12. package/extensions/gentle-todo.ts +199 -0
  13. package/extensions/quiet-tools.ts +1 -1
  14. package/lib/agents-config.ts +318 -0
  15. package/lib/agents-history.ts +80 -0
  16. package/lib/agents-protocol.ts +429 -0
  17. package/lib/agents-runner.ts +490 -0
  18. package/lib/agents-transcript.ts +87 -0
  19. package/lib/agents-view.ts +557 -0
  20. package/lib/agents-widget.ts +222 -0
  21. package/lib/gentle-ai-renderer.ts +142 -26
  22. package/lib/native-choice-list.ts +194 -0
  23. package/lib/native-fullscreen-interaction.ts +47 -0
  24. package/lib/native-pointer-region.ts +164 -0
  25. package/lib/native-review-cli.ts +88 -12
  26. package/lib/review-candidate-view-owner.ts +177 -0
  27. package/lib/review-candidate-view.ts +127 -35
  28. package/lib/review-consent-ui.ts +65 -0
  29. package/lib/review-integration-v2.ts +58 -8
  30. package/lib/review-last-event-controller.ts +1 -0
  31. package/lib/review-relay-contract.ts +11 -0
  32. package/lib/review-repository.ts +2 -2
  33. package/lib/review-risk-assessment.ts +339 -0
  34. package/lib/review-session-standing-permission-ipc.ts +309 -0
  35. package/lib/review-session-standing-permission.ts +219 -0
  36. package/lib/shell-bar.ts +138 -0
  37. package/lib/shell-card.ts +136 -0
  38. package/lib/shell-changes-view.ts +205 -0
  39. package/lib/shell-changes.ts +210 -0
  40. package/lib/shell-gauge.ts +40 -0
  41. package/lib/shell-prompt.ts +119 -0
  42. package/lib/shell-todo.ts +280 -0
  43. package/lib/shell-usage-view.ts +76 -0
  44. package/lib/shell-usage.ts +246 -0
  45. package/lib/telemetry-trigger.ts +151 -0
  46. package/package.json +4 -4
  47. package/runtime/native-review-cli.mjs +87 -11
  48. package/runtime/review-integration-v2.mjs +58 -8
  49. package/runtime/review-relay-contract.mjs +11 -0
  50. package/runtime/review-risk-assessment.mjs +340 -0
  51. package/runtime/telemetry-trigger.mjs +152 -0
  52. package/scripts/build-runtime-modules.mjs +2 -0
  53. package/scripts/gentle-ai-installer.mjs +10 -10
  54. package/scripts/test-packed-runner.mjs +22 -0
  55. package/scripts/verify-package-files.mjs +6 -2
  56. package/skills/_shared/review-ledger-contract.md +3 -1
  57. package/tests/agents-config.test.ts +143 -0
  58. package/tests/agents-fake-child.ts +52 -0
  59. package/tests/agents-history.test.ts +54 -0
  60. package/tests/agents-protocol.test.ts +153 -0
  61. package/tests/agents-runner-process.test.ts +111 -0
  62. package/tests/agents-runner.test.ts +402 -0
  63. package/tests/agents-transcript.test.ts +30 -0
  64. package/tests/agents-view.test.ts +274 -0
  65. package/tests/agents-widget.test.ts +111 -0
  66. package/tests/ask-user-choice.test.ts +157 -3
  67. package/tests/codegraph-tools.test.ts +110 -1
  68. package/tests/devbinary/native-review-parity.devtest.ts +108 -0
  69. package/tests/fixtures/agents-process-child.mjs +23 -0
  70. package/tests/gentle-agents.test.ts +741 -0
  71. package/tests/gentle-ai-binary.test.ts +1 -1
  72. package/tests/gentle-ai-installer.test.ts +47 -47
  73. package/tests/gentle-ai-renderer.test.ts +65 -0
  74. package/tests/gentle-ai.test.ts +28 -12
  75. package/tests/gentle-card-text.ts +35 -0
  76. package/tests/gentle-shell.test.ts +527 -0
  77. package/tests/gentle-todo.test.ts +182 -0
  78. package/tests/native-choice-list.test.ts +202 -0
  79. package/tests/native-fullscreen-interaction.test.ts +125 -0
  80. package/tests/native-pointer-region.test.ts +245 -0
  81. package/tests/native-review-capability-contract.test.ts +16 -1
  82. package/tests/native-review-cli.test.ts +40 -0
  83. package/tests/native-review-consent.test.ts +91 -0
  84. package/tests/native-review-parity-runtime.test.ts +8 -2
  85. package/tests/native-review-parity.test.ts +29 -22
  86. package/tests/orchestrator-budget.test.ts +69 -0
  87. package/tests/orchestrator-rdd-ownership.test.ts +9 -0
  88. package/tests/package-manifest.test.ts +17 -6
  89. package/tests/quiet-tool-rendering.test.ts +96 -37
  90. package/tests/rdd-aware-verification-contract.test.ts +216 -0
  91. package/tests/rdd-status-line.test.ts +286 -0
  92. package/tests/review-candidate-view.test.ts +452 -6
  93. package/tests/review-contract-prompt.test.ts +3 -0
  94. package/tests/review-controller-native-recovery.test.ts +29 -4
  95. package/tests/review-controller-native-routing.test.ts +321 -4
  96. package/tests/review-controller-workspace-root.test.ts +45 -2
  97. package/tests/review-controller.test.ts +25 -0
  98. package/tests/review-host-relay-routing.test.ts +20 -4
  99. package/tests/review-integration-v2.test.ts +112 -0
  100. package/tests/review-last-event-closure.test.ts +7 -2
  101. package/tests/review-relay-contract.test.ts +26 -0
  102. package/tests/review-repository.test.ts +28 -1
  103. package/tests/review-risk-assessment.test.ts +626 -0
  104. package/tests/review-session-standing-permission-controller.test.ts +608 -0
  105. package/tests/review-session-standing-permission-ipc.test.ts +233 -0
  106. package/tests/review-session-standing-permission-runtime.test.ts +212 -0
  107. package/tests/review-session-standing-permission.test.ts +126 -0
  108. package/tests/shell-bar.test.ts +176 -0
  109. package/tests/shell-card.test.ts +118 -0
  110. package/tests/shell-changes-view.test.ts +146 -0
  111. package/tests/shell-changes.test.ts +182 -0
  112. package/tests/shell-prompt.test.ts +118 -0
  113. package/tests/shell-todo.test.ts +170 -0
  114. package/tests/shell-usage-view.test.ts +62 -0
  115. package/tests/shell-usage.test.ts +197 -0
  116. package/tests/telemetry-trigger.test.ts +349 -0
@@ -0,0 +1,143 @@
1
+ import assert from "node:assert/strict";
2
+ import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs";
3
+ import { tmpdir } from "node:os";
4
+ import { join } from "node:path";
5
+ import test, { after } from "node:test";
6
+ import {
7
+ AGENT_MODE,
8
+ agentDirectories,
9
+ discoverAgents,
10
+ loadAgentsConfig,
11
+ parseAgentDefinition,
12
+ parseAgentsConfig,
13
+ parseFrontmatter,
14
+ parseModelRef,
15
+ resolveAgentProfile,
16
+ } from "../lib/agents-config.ts";
17
+
18
+ // Gentle Agents configuration: markdown agent definitions (the same files
19
+ // gentle-ai installs) and subagents.json, both parsed without touching pi.
20
+
21
+ const root = mkdtempSync(join(tmpdir(), "gentle-agents-config-"));
22
+ after(() => rmSync(root, { recursive: true, force: true }));
23
+
24
+ const EXPLORER = `---
25
+ name: gentle-ai-explore
26
+ description: Read-only exploration and mapping.
27
+ model: openai-codex/gpt-5.6-terra
28
+ thinking: high
29
+ tools:
30
+ - read
31
+ - grep
32
+ - codegraph
33
+ ---
34
+
35
+ You are the read-only explorer.
36
+ Map files and return a compressed handoff.
37
+ `;
38
+
39
+ test("parseFrontmatter reads scalars, inline lists, and block lists, and keeps the body", () => {
40
+ const { data, body } = parseFrontmatter("---\nname: a\ntools: [read, grep]\nlist:\n - one\n - two\nquoted: \"x: y\"\n---\nBody here\n");
41
+ assert.deepEqual(data, { name: "a", tools: ["read", "grep"], list: ["one", "two"], quoted: "x: y" });
42
+ assert.equal(body, "Body here");
43
+ assert.deepEqual(parseFrontmatter("no frontmatter"), { data: {}, body: "no frontmatter" });
44
+ });
45
+
46
+ test("parseModelRef splits provider/id and accepts a bare id", () => {
47
+ assert.deepEqual(parseModelRef("openai-codex/gpt-5.6-terra"), { provider: "openai-codex", id: "gpt-5.6-terra" });
48
+ assert.deepEqual(parseModelRef("sonnet"), { provider: undefined, id: "sonnet" });
49
+ assert.equal(parseModelRef(" "), undefined);
50
+ });
51
+
52
+ test("parseAgentDefinition builds a definition from the gentle-ai agent format", () => {
53
+ const agent = parseAgentDefinition(EXPLORER, "/home/x/.pi/agent/agents/gentle-ai-explore.md", "global");
54
+ assert.ok(!("error" in agent));
55
+ assert.equal(agent.name, "gentle-ai-explore");
56
+ assert.equal(agent.description, "Read-only exploration and mapping.");
57
+ assert.deepEqual(agent.model, { provider: "openai-codex", id: "gpt-5.6-terra" });
58
+ assert.equal(agent.thinking, "high");
59
+ assert.deepEqual(agent.tools, ["read", "grep", "codegraph"]);
60
+ assert.equal(agent.mode, undefined);
61
+ assert.equal(agent.scope, "global");
62
+ assert.match(agent.instructions, /^You are the read-only explorer\./);
63
+ });
64
+
65
+ test("parseAgentDefinition accepts effort and subagent_mode aliases, csv tools, and names from the file", () => {
66
+ const agent = parseAgentDefinition("---\ndescription: d\neffort: low\nsubagent_mode: background\ntools: read, bash\n---\nbody", "/p/.pi/agents/worker.md", "project");
67
+ assert.ok(!("error" in agent));
68
+ assert.equal(agent.name, "worker");
69
+ assert.equal(agent.thinking, "low");
70
+ assert.equal(agent.mode, AGENT_MODE.BACKGROUND);
71
+ assert.deepEqual(agent.tools, ["read", "bash"]);
72
+ });
73
+
74
+ test("parseAgentDefinition rejects unknown thinking levels, modes, and empty bodies", () => {
75
+ assert.match((parseAgentDefinition("---\nname: a\nthinking: extreme\n---\nbody", "/a.md", "global") as { error: string }).error, /thinking "extreme"/);
76
+ assert.match((parseAgentDefinition("---\nname: a\nsubagent_mode: forever\n---\nbody", "/a.md", "global") as { error: string }).error, /mode "forever"/);
77
+ assert.match((parseAgentDefinition("---\nname: a\n---\n \n", "/a.md", "global") as { error: string }).error, /no instructions/);
78
+ });
79
+
80
+ test("discoverAgents merges the four directories with project over global and subagents over agents", () => {
81
+ const home = join(root, "home");
82
+ const cwd = join(root, "project");
83
+ for (const dir of [".pi/agent/agents", ".pi/agent/subagents"]) mkdirSync(join(home, dir), { recursive: true });
84
+ for (const dir of [".pi/agents", ".pi/subagents"]) mkdirSync(join(cwd, dir), { recursive: true });
85
+ writeFileSync(join(home, ".pi/agent/agents/explore.md"), EXPLORER);
86
+ writeFileSync(join(home, ".pi/agent/agents/shared.md"), "---\ndescription: global agents\n---\nglobal");
87
+ writeFileSync(join(home, ".pi/agent/subagents/shared.md"), "---\ndescription: global subagents\n---\nglobal sub");
88
+ writeFileSync(join(cwd, ".pi/agents/shared.md"), "---\ndescription: project agents\n---\nproject");
89
+ writeFileSync(join(cwd, ".pi/agents/broken.md"), "---\nthinking: nope\n---\nx");
90
+ writeFileSync(join(cwd, ".pi/agents/notes.txt"), "ignored");
91
+ const { agents, errors } = discoverAgents({ cwd, home });
92
+ assert.deepEqual(agents.map((agent) => `${agent.name}:${agent.description}:${agent.scope}`), ["gentle-ai-explore:Read-only exploration and mapping.:global", "shared:project agents:project"]);
93
+ assert.equal(errors.length, 1);
94
+ assert.match(errors[0], /broken\.md/);
95
+ assert.deepEqual(discoverAgents({ cwd: join(root, "empty"), home: join(root, "nohome") }), { agents: [], errors: [] });
96
+ });
97
+
98
+ test("profile agent roots isolate global definitions and subagents.json while explicit homes keep their fallback", () => {
99
+ const cwd = join(root, "profile-project");
100
+ const principal = join(root, "pi-principal", "agent");
101
+ const lab = join(root, "pi-lab", "agent");
102
+ for (const [agentHome, name, model] of [[principal, "principal", "openai/principal"], [lab, "lab", "openai/lab"]] as const) {
103
+ mkdirSync(join(agentHome, "agents"), { recursive: true });
104
+ writeFileSync(join(agentHome, "agents", `${name}.md`), `---\ndescription: ${name}\n---\n${name}`);
105
+ writeFileSync(join(agentHome, "subagents.json"), JSON.stringify({ default_model: model }));
106
+ }
107
+ assert.deepEqual(agentDirectories({ cwd, home: join(root, "legacy-home"), agentHome: principal }).slice(0, 2).map(({ dir }) => dir), [join(principal, "agents"), join(principal, "subagents")]);
108
+ assert.deepEqual(discoverAgents({ cwd, home: join(root, "legacy-home"), agentHome: principal }).agents.map((agent) => agent.name), ["principal"]);
109
+ assert.deepEqual(discoverAgents({ cwd, home: join(root, "legacy-home"), agentHome: lab }).agents.map((agent) => agent.name), ["lab"]);
110
+ assert.equal(loadAgentsConfig({ cwd, home: join(root, "legacy-home"), agentHome: principal }).defaultModel?.id, "principal");
111
+ assert.equal(loadAgentsConfig({ cwd, home: join(root, "legacy-home"), agentHome: lab }).defaultModel?.id, "lab");
112
+ assert.equal(agentDirectories({ cwd, home: join(root, "legacy-home") })[0].dir, join(root, "legacy-home", ".pi", "agent", "agents"));
113
+ });
114
+
115
+ test("parseAgentsConfig applies defaults, validates values, and silently ignores the retired total-timeout key", () => {
116
+ const config = parseAgentsConfig({ default_model: "openai-codex/gpt-6-astra", default_effort: "medium", max_concurrency: 3, timeout_ms: 1000, stall_timeout_ms: 12_000, model_profiles: { explore: { model: "openai-codex/gpt-5.6-terra", effort: "high" } } }, { max_concurrency: 2, timeout_ms: 500, model_profiles: { explore: { effort: "low" }, worker: { model: "anthropic/claude-sonnet-5" } } });
117
+ assert.deepEqual(config.defaultModel, { provider: "openai-codex", id: "gpt-6-astra" });
118
+ assert.equal(config.defaultThinking, "medium");
119
+ assert.equal(config.maxConcurrency, 2);
120
+ assert.equal(config.stallTimeoutMs, 12_000);
121
+ assert.equal("timeoutMs" in config, false, "legacy timeout_ms must not become an active runtime setting");
122
+ assert.deepEqual(config.modelProfiles.explore, { model: { provider: "openai-codex", id: "gpt-5.6-terra" }, thinking: "low" });
123
+ assert.deepEqual(config.modelProfiles.worker, { model: { provider: "anthropic", id: "claude-sonnet-5" }, thinking: undefined });
124
+ const defaults = parseAgentsConfig(undefined, undefined);
125
+ assert.equal(defaults.maxConcurrency, 5);
126
+ assert.equal("timeoutMs" in defaults, false);
127
+ assert.equal(defaults.stallTimeoutMs, 4 * 60_000);
128
+ assert.equal(defaults.defaultMode, AGENT_MODE.TASK);
129
+ assert.equal(defaults.historyMaxTasks, 200);
130
+ assert.equal(parseAgentsConfig({ max_concurrency: "many", default_effort: "wild", default_mode: "background" }, undefined).maxConcurrency, 5);
131
+ assert.equal(parseAgentsConfig({ default_mode: "background" }, undefined).defaultMode, AGENT_MODE.BACKGROUND);
132
+ });
133
+
134
+ test("resolveAgentProfile prefers the profile, then the definition, then the defaults", () => {
135
+ const config = parseAgentsConfig({ default_model: "openai-codex/gpt-6-astra", default_effort: "medium", model_profiles: { "gentle-ai-explore": { effort: "high" } } }, undefined);
136
+ const explore = parseAgentDefinition(EXPLORER, "/x/explore.md", "global");
137
+ assert.ok(!("error" in explore));
138
+ assert.deepEqual(resolveAgentProfile(explore, config), { model: { provider: "openai-codex", id: "gpt-5.6-terra" }, thinking: "high", source: { model: "definition", thinking: "profile" } });
139
+ const bare = parseAgentDefinition("---\nname: bare\n---\nbody", "/x/bare.md", "global");
140
+ assert.ok(!("error" in bare));
141
+ assert.deepEqual(resolveAgentProfile(bare, config), { model: { provider: "openai-codex", id: "gpt-6-astra" }, thinking: "medium", source: { model: "default", thinking: "default" } });
142
+ assert.deepEqual(resolveAgentProfile(bare, parseAgentsConfig(undefined, undefined)).source, { model: "unresolved", thinking: "unresolved" });
143
+ });
@@ -0,0 +1,52 @@
1
+ import { EventEmitter } from "node:events";
2
+ import { PassThrough } from "node:stream";
3
+ import type { ChildLike } from "../lib/agents-runner.ts";
4
+
5
+ // A fake `pi --mode rpc` child: answers every command with a success
6
+ // response, records what the host wrote, and lets tests emit events.
7
+
8
+ export interface FakeChild {
9
+ child: ChildLike;
10
+ written: Array<Record<string, unknown>>;
11
+ emit(event: Record<string, unknown>): void;
12
+ exit(code: number): void;
13
+ fail(message: string): void;
14
+ killed: string[];
15
+ }
16
+
17
+ export function fakeChild(options: { exitOnKill?: boolean; pid?: number } = {}): FakeChild {
18
+ const emitter = new EventEmitter();
19
+ const stdin = new PassThrough();
20
+ const stdout = new PassThrough();
21
+ const written: Array<Record<string, unknown>> = [];
22
+ const killed: string[] = [];
23
+ let buffer = "";
24
+ stdin.on("data", (chunk: Buffer) => {
25
+ buffer += chunk.toString();
26
+ const lines = buffer.split("\n");
27
+ buffer = lines.pop() ?? "";
28
+ for (const line of lines) {
29
+ const command = JSON.parse(line) as Record<string, unknown>;
30
+ written.push(command);
31
+ if (command.type === "extension_ui_response") continue;
32
+ const data = command.type === "get_state" ? { sessionFile: "/sessions/child.jsonl" } : undefined;
33
+ stdout.write(`${JSON.stringify({ type: "response", id: command.id, command: command.type, success: true, data })}\n`);
34
+ }
35
+ });
36
+ const child: ChildLike = {
37
+ pid: options.pid,
38
+ stdin,
39
+ stdout,
40
+ stderr: new PassThrough(),
41
+ kill: (signal) => {
42
+ killed.push(String(signal ?? "SIGTERM"));
43
+ if (options.exitOnKill !== false) queueMicrotask(() => emitter.emit("exit", 0, signal ?? "SIGTERM"));
44
+ return true;
45
+ },
46
+ on: (event, listener) => {
47
+ emitter.on(event, listener);
48
+ return child;
49
+ },
50
+ };
51
+ return { child, written, killed, emit: (event) => stdout.write(`${JSON.stringify(event)}\n`), exit: (code) => emitter.emit("exit", code, null), fail: (message) => emitter.emit("error", new Error(message)) };
52
+ }
@@ -0,0 +1,54 @@
1
+ import assert from "node:assert/strict";
2
+ import { mkdtempSync, readdirSync, rmSync, writeFileSync } from "node:fs";
3
+ import { tmpdir } from "node:os";
4
+ import { join } from "node:path";
5
+ import test, { after } from "node:test";
6
+ import { historyDir, loadHistory, loadStoredTask, pruneHistory, saveTask } from "../lib/agents-history.ts";
7
+ import { applyTaskEvent, emptyThread, TASK_EVENT, TASK_STATUS, TaskStore, type TaskRecord } from "../lib/agents-protocol.ts";
8
+
9
+ // Gentle Agents history: JSON per task, async, lazy, pruned by count.
10
+
11
+ const root = mkdtempSync(join(tmpdir(), "gentle-agents-history-"));
12
+ after(() => rmSync(root, { recursive: true, force: true }));
13
+ const dir = join(root, "tasks");
14
+
15
+ function task(id: string, createdAt: number): TaskRecord {
16
+ return { id, agent: "explore", mode: "task", prompt: "p", label: "p", cwd: "/r", parentSessionId: "s", status: TASK_STATUS.COMPLETED, createdAt, startedAt: createdAt, endedAt: createdAt + 5, model: "m", thinking: undefined, sessionPath: null, error: null, result: "ok", lastStep: "done", lastActivityAt: createdAt, turns: 1, toolCalls: 0, tokens: 10, cost: 0.01 };
17
+ }
18
+
19
+ test("historyDir follows an isolated agent profile while explicit homes retain the default fallback", () => {
20
+ assert.equal(historyDir("/home/x", "/profiles/pi-principal/agent"), join("/profiles/pi-principal/agent", "gentle-agents", "tasks"));
21
+ assert.equal(historyDir("/home/x", "/profiles/pi-lab/agent"), join("/profiles/pi-lab/agent", "gentle-agents", "tasks"));
22
+ assert.equal(historyDir("/home/x"), join("/home/x", ".pi", "agent", "gentle-agents", "tasks"));
23
+ });
24
+
25
+ test("saveTask writes a task with its thread and loadStoredTask reads it back", async () => {
26
+ const thread = applyTaskEvent(emptyThread(), { type: TASK_EVENT.TEXT, text: "hello" });
27
+ await saveTask(dir, task("a1", 1000), thread);
28
+ const stored = await loadStoredTask(dir, "a1");
29
+ assert.equal(stored?.task.result, "ok");
30
+ assert.deepEqual(stored?.thread.items, [{ kind: "text", text: "hello" }]);
31
+ assert.equal(await loadStoredTask(dir, "missing"), undefined);
32
+ assert.equal(await loadStoredTask(dir, "../etc/passwd"), undefined);
33
+ assert.deepEqual(readdirSync(dir), ["a1.json"], "no temp file is left behind");
34
+ });
35
+
36
+ test("loadHistory skips broken files, sorts newest first, and pruneHistory keeps the newest N", async () => {
37
+ await saveTask(dir, task("b2", 3000), emptyThread());
38
+ await saveTask(dir, task("c3", 2000), emptyThread());
39
+ writeFileSync(join(dir, "junk.json"), "{not json");
40
+ writeFileSync(join(dir, "shape.json"), JSON.stringify({ task: { id: 1 } }));
41
+ assert.deepEqual((await loadHistory(dir)).map((entry) => entry.task.id), ["b2", "c3", "a1"]);
42
+ assert.equal(await pruneHistory(dir, 2), 1);
43
+ assert.deepEqual((await loadHistory(dir)).map((entry) => entry.task.id), ["b2", "c3"]);
44
+ assert.deepEqual(await loadHistory(join(root, "nowhere")), []);
45
+ });
46
+
47
+ test("TaskStore.restore adds a stored task without clobbering a live one", () => {
48
+ const store = new TaskStore();
49
+ const thread = applyTaskEvent(emptyThread(), { type: TASK_EVENT.NOTE, text: "restored" });
50
+ assert.equal(store.restore(task("r1", 1000), thread), true);
51
+ assert.equal(store.thread("r1").items.length, 1);
52
+ assert.equal(store.restore({ ...task("r1", 1000), result: "other" }, emptyThread()), false);
53
+ assert.equal(store.get("r1")?.result, "ok");
54
+ });
@@ -0,0 +1,153 @@
1
+ import assert from "node:assert/strict";
2
+ import test from "node:test";
3
+ import {
4
+ applyTaskEvent,
5
+ emptyThread,
6
+ normalizeRpcEvent,
7
+ TASK_EVENT,
8
+ TASK_STATUS,
9
+ taskLabel,
10
+ TaskStore,
11
+ THREAD_ITEM,
12
+ type TaskRecord,
13
+ } from "../lib/agents-protocol.ts";
14
+
15
+ // Gentle Agents protocol: the child pi process streams RPC events; the host
16
+ // normalizes them into small typed deltas, applies them to an append-only
17
+ // thread, and notifies only the listeners of the task that changed.
18
+
19
+ function record(overrides: Partial<TaskRecord> = {}): TaskRecord {
20
+ return {
21
+ id: "t1",
22
+ agent: "gentle-ai-explore",
23
+ mode: "task",
24
+ prompt: "Map the repo",
25
+ label: "map the repo",
26
+ cwd: "/repo",
27
+ parentSessionId: "s1",
28
+ status: TASK_STATUS.QUEUED,
29
+ createdAt: 1000,
30
+ startedAt: null,
31
+ endedAt: null,
32
+ model: "openai-codex/gpt-5.6-terra",
33
+ thinking: "high",
34
+ sessionPath: null,
35
+ error: null,
36
+ result: null,
37
+ lastStep: "queued",
38
+ lastActivityAt: 1000,
39
+ turns: 0,
40
+ toolCalls: 0,
41
+ tokens: 0,
42
+ cost: 0,
43
+ ...overrides,
44
+ };
45
+ }
46
+
47
+ test("normalizeRpcEvent maps pi RPC events to task deltas and ignores the rest", () => {
48
+ assert.deepEqual(normalizeRpcEvent({ type: "message_update", assistantMessageEvent: { type: "text_delta", delta: "Hi" } }), [{ type: TASK_EVENT.TEXT, text: "Hi" }]);
49
+ assert.deepEqual(normalizeRpcEvent({ type: "message_update", assistantMessageEvent: { type: "thinking_delta", delta: "hmm" } }), [{ type: TASK_EVENT.THINKING, text: "hmm" }]);
50
+ assert.deepEqual(normalizeRpcEvent({ type: "tool_execution_start", toolCallId: "c1", toolName: "bash", args: { command: "ls" } }), [{ type: TASK_EVENT.TOOL_START, callId: "c1", name: "bash", args: { command: "ls" } }]);
51
+ assert.deepEqual(normalizeRpcEvent({ type: "tool_execution_update", toolCallId: "c1", toolName: "bash", partialResult: { content: [{ type: "text", text: "a\nb" }] } }), [{ type: TASK_EVENT.TOOL_UPDATE, callId: "c1", output: "a\nb" }]);
52
+ assert.deepEqual(normalizeRpcEvent({ type: "tool_execution_end", toolCallId: "c1", toolName: "bash", isError: true, result: { content: [{ type: "text", text: "boom" }] } }), [{ type: TASK_EVENT.TOOL_END, callId: "c1", output: "boom", isError: true }]);
53
+ assert.deepEqual(normalizeRpcEvent({ type: "turn_end" }), [{ type: TASK_EVENT.TURN_END }]);
54
+ assert.deepEqual(normalizeRpcEvent({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "done." }], stopReason: "stop" }] }), [{ type: TASK_EVENT.AGENT_END, text: "done.", outcome: "success" }]);
55
+ assert.deepEqual(normalizeRpcEvent({ type: "agent_end", messages: [{ role: "assistant", content: [], stopReason: "error", errorMessage: "WebSocket error with untrusted provider payload" }] }), [{ type: TASK_EVENT.AGENT_END, text: "", outcome: "error", diagnostic: "assistant reported an error" }]);
56
+ assert.deepEqual(normalizeRpcEvent({ type: "agent_end", messages: [{ role: "assistant", content: [], stopReason: "aborted" }] }), [{ type: TASK_EVENT.AGENT_END, text: "", outcome: "aborted", diagnostic: "assistant aborted" }]);
57
+ assert.deepEqual(normalizeRpcEvent({ type: "agent_end", messages: [{ role: "assistant", content: [], stopReason: "stop" }] }), [{ type: TASK_EVENT.AGENT_END, text: "", outcome: "empty", diagnostic: "assistant returned no final report" }]);
58
+ assert.deepEqual(normalizeRpcEvent({ type: "agent_settled" }), [{ type: TASK_EVENT.AGENT_SETTLED }]);
59
+ assert.deepEqual(normalizeRpcEvent({ type: "message_update", assistantMessageEvent: { type: "error", reason: "error", error: { message: "rate limited" } } }), [{ type: TASK_EVENT.ERROR, message: "rate limited" }]);
60
+ assert.deepEqual(normalizeRpcEvent({ type: "extension_ui_request", id: "u1", method: "confirm", title: "Delete?" }), [{ type: TASK_EVENT.ASK, request: { id: "u1", method: "confirm", title: "Delete?" } }]);
61
+ assert.deepEqual(normalizeRpcEvent({ type: "extension_ui_request", id: "u2", method: "setStatus", statusKey: "mcp" }), [], "fire-and-forget UI requests never count as questions");
62
+ assert.deepEqual(normalizeRpcEvent({ type: "extension_ui_request", id: "u3", method: "notify", message: "hi" }), []);
63
+ assert.deepEqual(normalizeRpcEvent({ type: "auto_retry_start", attempt: 1, maxAttempts: 3 }), [{ type: TASK_EVENT.NOTE, text: "retrying (1/3)" }]);
64
+ assert.deepEqual(normalizeRpcEvent({ type: "message_end", message: { role: "assistant", usage: { totalTokens: 9621, cost: { total: 0.0193 } } } }), [{ type: TASK_EVENT.USAGE, tokens: 9621, cost: 0.0193 }]);
65
+ assert.deepEqual(normalizeRpcEvent({ type: "message_end", message: { role: "user" } }), []);
66
+ assert.deepEqual(normalizeRpcEvent({ type: "queue_update" }), []);
67
+ assert.deepEqual(normalizeRpcEvent("garbage"), []);
68
+ });
69
+
70
+ test("taskLabel prefers an explicit label and otherwise takes the prompt's first sentence", () => {
71
+ assert.equal(taskLabel("Map the repo. Then report.", " map the repo "), "map the repo");
72
+ assert.equal(taskLabel("Repeat a fresh read-only exploration of lib. Own runtime architecture: extensions, hooks."), "Repeat a fresh read-only exploration of lib");
73
+ assert.equal(taskLabel("\n\nFirst line here:\nsecond"), "First line here");
74
+ assert.equal(taskLabel(`${"x".repeat(100)} tail`).length, 72);
75
+ assert.equal(taskLabel("\x1b[31mred\x1b[0m"), "red");
76
+ });
77
+
78
+ test("applyTaskEvent appends incrementally: text deltas merge, tool output is replaced, items stay bounded", () => {
79
+ let thread = emptyThread({ maxItems: 3, maxOutputChars: 9 });
80
+ thread = applyTaskEvent(thread, { type: TASK_EVENT.TEXT, text: "Hel" });
81
+ thread = applyTaskEvent(thread, { type: TASK_EVENT.TEXT, text: "lo" });
82
+ assert.deepEqual(thread.items, [{ kind: THREAD_ITEM.TEXT, text: "Hello" }]);
83
+ thread = applyTaskEvent(thread, { type: TASK_EVENT.TOOL_START, callId: "c1", name: "bash", args: { command: "ls" } });
84
+ thread = applyTaskEvent(thread, { type: TASK_EVENT.TOOL_UPDATE, callId: "c1", output: "line one\nline two" });
85
+ assert.equal(thread.items[1].kind, THREAD_ITEM.TOOL);
86
+ assert.equal((thread.items[1] as { output: string }).output, "…line two", "tool output keeps the tail");
87
+ thread = applyTaskEvent(thread, { type: TASK_EVENT.TOOL_END, callId: "c1", output: "ok", isError: false });
88
+ assert.deepEqual(thread.items[1], { kind: THREAD_ITEM.TOOL, callId: "c1", name: "bash", args: { command: "ls" }, output: "ok", running: false, isError: false });
89
+ thread = applyTaskEvent(thread, { type: TASK_EVENT.NOTE, text: "retrying (1/3)" });
90
+ thread = applyTaskEvent(thread, { type: TASK_EVENT.TEXT, text: "After" });
91
+ assert.equal(thread.items.length, 3);
92
+ assert.equal(thread.dropped, 1, "the oldest item made room");
93
+ assert.equal(thread.items[0].kind, THREAD_ITEM.TOOL);
94
+ assert.equal(thread.version, 7);
95
+ });
96
+
97
+ test("applyTaskEvent sanitizes text, ignores updates for unknown tools, and records asks as notes", () => {
98
+ let thread = emptyThread();
99
+ thread = applyTaskEvent(thread, { type: TASK_EVENT.TEXT, text: "\x1b[31mred\x1b[0m" });
100
+ assert.deepEqual(thread.items, [{ kind: THREAD_ITEM.TEXT, text: "red" }]);
101
+ const before = thread;
102
+ thread = applyTaskEvent(thread, { type: TASK_EVENT.TOOL_UPDATE, callId: "missing", output: "x" });
103
+ assert.strictEqual(thread, before);
104
+ thread = applyTaskEvent(thread, { type: TASK_EVENT.ASK, request: { id: "u1", method: "confirm", title: "Delete?" } });
105
+ assert.deepEqual(thread.items[1], { kind: THREAD_ITEM.NOTE, text: "asked: Delete?" });
106
+ });
107
+
108
+ test("TaskStore notifies only the listeners of the task that changed and keeps summaries cheap", () => {
109
+ const store = new TaskStore();
110
+ store.add(record({ id: "a" }));
111
+ store.add(record({ id: "b", agent: "worker" }));
112
+ const seenA: string[] = [];
113
+ const seenB: string[] = [];
114
+ const summaries: string[] = [];
115
+ const offA = store.subscribe("a", (task) => seenA.push(`${task.status}:${task.lastStep}`));
116
+ store.subscribe("b", (task) => seenB.push(task.status));
117
+ store.subscribeSummary((summary) => summaries.push(`${summary.running}/${summary.queued}/${summary.finished}`));
118
+ store.update("a", { status: TASK_STATUS.RUNNING, startedAt: 2000, lastStep: "starting" });
119
+ store.apply("a", { type: TASK_EVENT.TOOL_START, callId: "c1", name: "bash", args: {} }, 2500);
120
+ assert.deepEqual(seenA, ["running:starting", "running:bash"]);
121
+ assert.deepEqual(seenB, []);
122
+ assert.deepEqual(summaries, ["1/1/0"], "an event inside a task never touches the summary");
123
+ assert.equal(store.get("a")?.toolCalls, 1);
124
+ assert.equal(store.get("a")?.lastActivityAt, 2500);
125
+ assert.equal(store.thread("a").items.length, 1);
126
+ offA();
127
+ store.update("a", { status: TASK_STATUS.COMPLETED, endedAt: 3000 });
128
+ assert.equal(seenA.length, 2, "unsubscribed listener stays quiet");
129
+ assert.deepEqual(summaries, ["1/1/0", "0/1/1"]);
130
+ assert.deepEqual(store.list("s1").map((task) => task.id), ["a", "b"], "newest activity first");
131
+ assert.equal(store.get("missing"), undefined);
132
+ assert.deepEqual(store.thread("missing"), emptyThread());
133
+ });
134
+
135
+ test("TaskStore.apply moves the task to waiting on ask, back to running on any later event, and counts turns", () => {
136
+ const store = new TaskStore();
137
+ store.add(record({ id: "a", status: TASK_STATUS.RUNNING }));
138
+ store.apply("a", { type: TASK_EVENT.ASK, request: { id: "u1", method: "input", title: "Name?" } }, 1);
139
+ assert.equal(store.get("a")?.status, TASK_STATUS.WAITING);
140
+ assert.equal(store.get("a")?.lastStep, "asked: Name?");
141
+ store.apply("a", { type: TASK_EVENT.TEXT, text: "thanks" }, 2);
142
+ assert.equal(store.get("a")?.status, TASK_STATUS.RUNNING);
143
+ store.apply("a", { type: TASK_EVENT.TURN_END }, 3);
144
+ store.apply("a", { type: TASK_EVENT.TURN_END }, 4);
145
+ assert.equal(store.get("a")?.turns, 2);
146
+ store.apply("a", { type: TASK_EVENT.USAGE, tokens: 100, cost: 0.5 }, 4);
147
+ store.apply("a", { type: TASK_EVENT.USAGE, tokens: 50, cost: 0.25 }, 4);
148
+ assert.equal(store.get("a")?.tokens, 150);
149
+ assert.equal(store.get("a")?.cost, 0.75);
150
+ store.apply("a", { type: TASK_EVENT.AGENT_END, text: "final answer", outcome: "success" }, 5);
151
+ assert.equal(store.get("a")?.result, "final answer");
152
+ assert.equal(store.get("a")?.status, TASK_STATUS.RUNNING, "agent_end alone does not finish: the runner decides");
153
+ });
@@ -0,0 +1,111 @@
1
+ import assert from "node:assert/strict";
2
+ import { spawn as nodeSpawn } from "node:child_process";
3
+ import { fileURLToPath } from "node:url";
4
+ import test from "node:test";
5
+ import { AGENT_MODE, type AgentDefinition } from "../lib/agents-config.ts";
6
+ import { AgentRunner, type ChildLike, type RunnerDeps, type TaskRequest } from "../lib/agents-runner.ts";
7
+ import { TASK_STATUS, TaskStore } from "../lib/agents-protocol.ts";
8
+
9
+ const fixture = fileURLToPath(new URL("./fixtures/agents-process-child.mjs", import.meta.url));
10
+ const agent: AgentDefinition = { name: "process", description: "test", filePath: "/test.md", scope: "global", instructions: "", model: undefined, thinking: undefined, mode: undefined, tools: [] };
11
+ const request = (prompt: string): TaskRequest => ({ agent, prompt, label: undefined, context: undefined, mode: AGENT_MODE.BACKGROUND, cwd: process.cwd(), parentSessionId: "test", model: undefined, thinking: undefined, sessionDir: "/tmp", resumeSessionPath: undefined, env: {} });
12
+
13
+ const waitFor = async (predicate: () => boolean, timeoutMs = 10_000): Promise<void> => {
14
+ const deadline = Date.now() + timeoutMs;
15
+ while (!predicate()) {
16
+ if (Date.now() >= deadline) throw new Error(`condition was not met within ${timeoutMs}ms`);
17
+ await new Promise((resolve) => setTimeout(resolve, 10));
18
+ }
19
+ };
20
+
21
+ test("POSIX cleanup retains queue slots when a leader exits but its TERM-resisting descendant remains", { skip: process.platform === "win32" }, async () => {
22
+ const store = new TaskStore();
23
+ let launches = 0;
24
+ let firstPid: number | undefined;
25
+ const descendantPids: Array<number | undefined> = [];
26
+ let firstDetached = false;
27
+ const ownedPids: number[] = [];
28
+ const runtimeTimers: Array<{ fn: () => void; timer: ReturnType<typeof setTimeout> | undefined; cancelled: boolean }> = [];
29
+ const armRuntimeTimeout = (index: number) => {
30
+ const timer = runtimeTimers[index];
31
+ if (timer && !timer.cancelled && timer.timer === undefined) timer.timer = setTimeout(timer.fn, 500);
32
+ };
33
+ const deps: RunnerDeps = {
34
+ spawn: (_command, _args, options): ChildLike => {
35
+ launches += 1;
36
+ const env = launches === 1 ? { ...options.env, AGENTS_PROCESS_CHILD_EXIT_ON_TERM: "1" } : launches === 3 ? { ...options.env, AGENTS_PROCESS_CHILD_EXIT_AFTER_READY: "1" } : options.env;
37
+ const child = nodeSpawn(process.execPath, [fixture], { cwd: options.cwd, env, detached: options.detached, stdio: ["pipe", "pipe", "pipe"] });
38
+ const launchIndex = launches - 1;
39
+ ownedPids.push(child.pid!);
40
+ if (launches === 1) {
41
+ firstPid = child.pid;
42
+ firstDetached = options.detached === true;
43
+ }
44
+ let output = "";
45
+ child.stdout.on("data", (chunk: Buffer) => {
46
+ output += chunk.toString();
47
+ const match = output.match(/DESCENDANT:(\d+)/);
48
+ if (match) {
49
+ descendantPids[launchIndex] = Number(match[1]);
50
+ armRuntimeTimeout(launchIndex);
51
+ }
52
+ });
53
+ return child;
54
+ },
55
+ now: Date.now,
56
+ schedule: (fn, ms) => {
57
+ if (ms === 500) {
58
+ const timer = { fn, timer: undefined, cancelled: false };
59
+ const index = launches - 1;
60
+ runtimeTimers[index] = timer;
61
+ if (descendantPids[index] !== undefined) armRuntimeTimeout(index);
62
+ return () => {
63
+ timer.cancelled = true;
64
+ if (timer.timer) clearTimeout(timer.timer);
65
+ };
66
+ }
67
+ const timer = setTimeout(fn, ms);
68
+ return () => clearTimeout(timer);
69
+ },
70
+ pi: { command: process.execPath, args: [fixture] },
71
+ };
72
+ const runner = new AgentRunner(store, { maxConcurrency: 1, stallTimeoutMs: 500 }, deps, { askUser: async () => ({ cancelled: true }) });
73
+ const first = runner.run(request("first"));
74
+ const second = runner.run(request("second"));
75
+ const third = runner.run(request("third"));
76
+ const fourth = runner.run(request("fourth"));
77
+ try {
78
+ await waitFor(() => launches === 1 && descendantPids[0] !== undefined);
79
+ assert.equal(firstDetached, true, "the first child owns a POSIX process group");
80
+ assert.equal(runner.cancel(first.id), true);
81
+ assert.equal(store.get(first.id)?.status, TASK_STATUS.RUNNING, "leader exit does not release its live descendant group");
82
+ await new Promise((resolve) => setTimeout(resolve, 40));
83
+ assert.doesNotThrow(() => process.kill(descendantPids[0]!, 0), "the exact TERM-resisting descendant remains alive");
84
+ assert.equal(launches, 1, "the queued task cannot use the slot during SIGTERM grace");
85
+ await waitFor(() => store.get(first.id)?.status === TASK_STATUS.CANCELLED);
86
+ assert.equal(launches, 2, "the slot opens only after the owned group exits");
87
+ assert.throws(() => process.kill(-firstPid!, 0), { code: "ESRCH" }, "SIGKILL cleaned the owned child group, including its descendant");
88
+ await waitFor(() => store.get(second.id)?.status === TASK_STATUS.TIMED_OUT);
89
+ assert.throws(() => process.kill(-ownedPids[1], 0), { code: "ESRCH" }, "timeout also bounds cleanup of its owned group");
90
+ await waitFor(() => launches === 3 && descendantPids[2] !== undefined);
91
+ await new Promise((resolve) => setTimeout(resolve, 40));
92
+ assert.doesNotThrow(() => process.kill(descendantPids[2]!, 0), "the natural-exit descendant remains alive");
93
+ assert.equal(launches, 3, "natural leader exit does not release the queue slot");
94
+ await waitFor(() => store.get(third.id)?.status === TASK_STATUS.FAILED);
95
+ assert.equal(launches, 4, "the queue resumes after natural-exit group cleanup");
96
+ assert.throws(() => process.kill(-ownedPids[2], 0), { code: "ESRCH" }, "natural exit also cleans its owned group");
97
+ } finally {
98
+ if (firstDetached) {
99
+ for (const pid of ownedPids) {
100
+ try { process.kill(-pid, "SIGKILL"); } catch {}
101
+ }
102
+ } else {
103
+ for (const pid of [firstPid, ...descendantPids, ...ownedPids]) {
104
+ if (pid) try { process.kill(pid, "SIGKILL"); } catch {}
105
+ }
106
+ }
107
+ runner.cancel(second.id);
108
+ runner.cancel(third.id);
109
+ runner.cancel(fourth.id);
110
+ }
111
+ });