gentle-pi 3.4.0 → 3.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -13,6 +13,7 @@ const sources = [
13
13
  "review-risk-assessment",
14
14
  "native-review-cli",
15
15
  "telemetry-trigger",
16
+ "gentle-shell-launcher",
16
17
  ];
17
18
  const header = "// Generated by scripts/build-runtime-modules.mjs. Do not edit.\n";
18
19
 
@@ -3,8 +3,10 @@
3
3
  // hardcoded copies of it survived a pin bump once and reported installing
4
4
  // v2.1.11 while writing v2.2.0 to disk, which is the one moment an operator
5
5
  // most needs the number to be true.
6
+ import { dirname } from "node:path";
7
+ import { fileURLToPath } from "node:url";
6
8
  import { INSTALLER_VERSION, installGentleAi } from "./gentle-ai-installer.mjs";
7
- import { installTuiModeSetting } from "./install-tui-mode-setting.mjs";
9
+ import { installTuiModeSetting, isPiManagedInstall } from "./install-tui-mode-setting.mjs";
8
10
 
9
11
  if (process.env.GENTLE_PI_SKIP_GENTLE_AI_INSTALL === "1") {
10
12
  console.warn("GENTLE_PI_SKIP_GENTLE_AI_INSTALL=1: skipped package-local Gentle AI installation; native review operations will fail with package-local-binary-missing until gentle-pi is reinstalled.");
@@ -20,11 +22,16 @@ if (process.env.GENTLE_PI_SKIP_GENTLE_AI_INSTALL === "1") {
20
22
 
21
23
  // A native failure never changes settings; the explicit native-only skip does.
22
24
  if (!process.exitCode) {
23
- try {
24
- const result = await installTuiModeSetting();
25
- if (result.changed) console.log("gentle-pi enabled fullscreen in global Pi settings; /settings can switch back to regular.");
26
- } catch (error) {
27
- console.error(`gentle-pi could not enable fullscreen: ${error instanceof Error ? error.message : String(error)}`);
28
- process.exitCode = 1;
25
+ const packageDir = dirname(dirname(fileURLToPath(import.meta.url)));
26
+ if (!isPiManagedInstall(packageDir)) {
27
+ console.log(`gentle-pi skipped enabling fullscreen in global Pi settings: ${packageDir} is not a pi-managed install (npm install -g, a git checkout, and npx all land here).`);
28
+ } else {
29
+ try {
30
+ const result = await installTuiModeSetting();
31
+ if (result.changed) console.log("gentle-pi enabled fullscreen in global Pi settings; /settings can switch back to regular.");
32
+ } catch (error) {
33
+ console.error(`gentle-pi could not enable fullscreen: ${error instanceof Error ? error.message : String(error)}`);
34
+ process.exitCode = 1;
35
+ }
29
36
  }
30
37
  }
@@ -1,10 +1,44 @@
1
1
  import { closeSync, constants, fchmodSync, fsyncSync, fstatSync, lstatSync, mkdirSync, openSync, readFileSync, realpathSync, renameSync, rmdirSync, unlinkSync, writeFileSync } from "node:fs";
2
2
  import { homedir } from "node:os";
3
- import { dirname, join, resolve } from "node:path";
3
+ import { dirname, join, resolve, sep } from "node:path";
4
4
  import { fileURLToPath } from "node:url";
5
5
  import { randomUUID } from "node:crypto";
6
6
  import { setTimeout as delay } from "node:timers/promises";
7
7
 
8
+ // The two directory layouts Pi's own package manager creates when it installs
9
+ // a package into an agent home: the npm-backed `npm/node_modules/<package>`
10
+ // (user scope `<agent dir>/npm/node_modules/gentle-pi`, project scope
11
+ // `.pi/npm/node_modules/gentle-pi`) and the git-backed
12
+ // `git/github.com/Gentleman-Programming/<package>` layout. A path is checked
13
+ // for these sequences anywhere in its segments, not anchored to a specific
14
+ // resolved agent home — unlike installTuiModeSetting's own ownership check.
15
+ const PI_MANAGED_SEGMENT_SEQUENCES = [
16
+ ["npm", "node_modules"],
17
+ ["git", "github.com", "Gentleman-Programming"],
18
+ ];
19
+
20
+ function containsSequence(segments, sequence) {
21
+ for (let start = 0; start + sequence.length <= segments.length; start += 1) {
22
+ if (sequence.every((part, offset) => segments[start + offset] === part)) return true;
23
+ }
24
+ return false;
25
+ }
26
+
27
+ /** Gates the POSTINSTALL entry point only (scripts/install-gentle-ai.mjs), not
28
+ * installTuiModeSetting or installIsolatedTuiModeSetting: true when
29
+ * `packageDir` (the directory of the gentle-pi package actually running,
30
+ * typically derived from that script's own import.meta.url) sits under one of
31
+ * the directory layouts above. A plain `npm install -g gentle-pi`, a
32
+ * development git checkout, an `npx` cache directory, or a pnpm store never
33
+ * match, so the postinstall entry skips writing the user's global Pi settings
34
+ * for those instead of relying solely on installTuiModeSetting's own,
35
+ * differently-scoped ownership check.
36
+ */
37
+ export function isPiManagedInstall(packageDir) {
38
+ const segments = resolve(packageDir).split(sep);
39
+ return PI_MANAGED_SEGMENT_SEQUENCES.some((sequence) => containsSequence(segments, sequence));
40
+ }
41
+
8
42
  function inspect(path) {
9
43
  try { return lstatSync(path); }
10
44
  catch (error) { if (error.code === "ENOENT") return undefined; throw error; }
@@ -112,3 +146,46 @@ export async function installTuiModeSetting(options = {}) {
112
146
  }
113
147
  }
114
148
  }
149
+
150
+ /** Writes tuiMode: "fullscreen" into a directory gentle-shell's own isolated-home
151
+ * bootstrap (T2) just created and owns. Unlike installTuiModeSetting, there is no
152
+ * "physically installed under this home's npm/node_modules" ownership check to
153
+ * satisfy: the caller already knows it created `dir` moments ago as gentle-shell's
154
+ * dedicated agent home, so the only safety property that still matters is the one
155
+ * every writer here needs — atomic, non-symlink, cooperative-lock-respecting.
156
+ */
157
+ export async function installIsolatedTuiModeSetting(dir) {
158
+ const home = realpathSync(resolve(dir));
159
+ assertDirectories([home]);
160
+ const settingsPath = join(home, "settings.json");
161
+ const lockPath = `${settingsPath}.lock`;
162
+ const lock = await acquireLock(lockPath);
163
+ const started = Date.now();
164
+ let staging;
165
+ try {
166
+ assertDirectories([home]);
167
+ const original = readSettings(settingsPath);
168
+ if (original.value.tuiMode === "fullscreen") return { changed: false, recognized: true };
169
+ staging = join(home, `.settings-fullscreen-${randomUUID()}.tmp`);
170
+ const fd = openSync(staging, "wx", original.stat ? original.stat.mode & 0o777 : 0o600);
171
+ try {
172
+ if (original.stat) fchmodSync(fd, original.stat.mode & 0o777);
173
+ writeFileSync(fd, `${JSON.stringify({ ...original.value, tuiMode: "fullscreen" }, null, 2)}\n`, "utf8");
174
+ fsyncSync(fd);
175
+ } finally { closeSync(fd); }
176
+ assertDirectories([home]);
177
+ const latest = readSettings(settingsPath);
178
+ if (latest.text !== original.text || (original.stat ? !sameFile(original.stat, latest.stat) : latest.stat !== undefined)) {
179
+ throw new Error("settings.json changed concurrently; retry installation");
180
+ }
181
+ if (!sameFile(lock, inspect(lockPath)) || Date.now() - started >= 5000) throw new Error("Fullscreen settings lock ownership expired; retry installation");
182
+ renameSync(staging, settingsPath);
183
+ staging = undefined;
184
+ return { changed: true, recognized: true };
185
+ } finally {
186
+ if (realpathSync(home) === home) {
187
+ if (staging) unlinkSync(staging);
188
+ if (sameFile(lock, inspect(lockPath))) rmdirSync(lockPath);
189
+ }
190
+ }
191
+ }
@@ -8,6 +8,7 @@ import { fileURLToPath, pathToFileURL } from "node:url";
8
8
  const root = join(fileURLToPath(new URL("..", import.meta.url)));
9
9
 
10
10
  const requiredPaths = [
11
+ "bin/gentle-shell.mjs",
11
12
  "assets/orchestrator.md",
12
13
  "assets/orchestrator-delegation.md",
13
14
  "assets/orchestrator-memory.md",
@@ -54,6 +55,7 @@ const requiredPaths = [
54
55
  "extensions/sdd-init.ts",
55
56
  "extensions/skill-registry.ts",
56
57
  "lib/gentle-ai-binary.ts",
58
+ "lib/gentle-shell-launcher.ts",
57
59
  "lib/native-review-cli.ts",
58
60
  "lib/provider-contract-bundle.ts",
59
61
  "lib/review-host-relay.ts",
@@ -62,6 +64,7 @@ const requiredPaths = [
62
64
  "lib/sdd-preflight.ts",
63
65
  "lib/telemetry-trigger.ts",
64
66
  "runtime/gentle-ai-binary.mjs",
67
+ "runtime/gentle-shell-launcher.mjs",
65
68
  "runtime/native-review-cli.mjs",
66
69
  "runtime/review-integration-v2.mjs",
67
70
  "runtime/review-risk-assessment.mjs",
@@ -70,6 +73,7 @@ const requiredPaths = [
70
73
  "scripts/check-provider-contract.mjs",
71
74
  "scripts/gentle-ai-installer.mjs",
72
75
  "scripts/install-gentle-ai.mjs",
76
+ "scripts/install-tui-mode-setting.mjs",
73
77
  "scripts/mirror-provider-contract.mjs",
74
78
  "tests/fixtures/native-review-cli/v2.1.3/start.json",
75
79
  "tests/fixtures/provider-contract-bundle/v1.1.0/README.md",
@@ -0,0 +1,473 @@
1
+ import assert from "node:assert/strict";
2
+ import test from "node:test";
3
+ import { TASK_EVENT, TASK_STATUS, TaskStore, type TaskRecord, type TaskStatus } from "../lib/agents-protocol.ts";
4
+ import {
5
+ ACTIVITY_SCHEMA,
6
+ ACTIVITY_WIDGET_KEY,
7
+ createRpcActivityPublisher,
8
+ encodeActivityLines,
9
+ projectRpcActivity,
10
+ type RpcActivity,
11
+ type RpcTask,
12
+ } from "../lib/agents-rpc-publisher.ts";
13
+
14
+ // gentle-agents RPC publisher: a pure projection of TaskStore state into the
15
+ // bounded JSON payload an interactive RPC host receives through setWidget,
16
+ // plus the store-driven, coalesced publisher that pushes it.
17
+
18
+ function task(id: string, parentSessionId: string, overrides: Partial<TaskRecord> = {}): TaskRecord {
19
+ return {
20
+ id,
21
+ agent: "worker",
22
+ mode: "task",
23
+ prompt: "p",
24
+ label: "p",
25
+ cwd: "/r",
26
+ parentSessionId,
27
+ status: TASK_STATUS.RUNNING,
28
+ createdAt: 1000,
29
+ startedAt: 1000,
30
+ endedAt: null,
31
+ model: "gpt",
32
+ thinking: undefined,
33
+ sessionPath: "/sessions/child.jsonl",
34
+ error: null,
35
+ result: null,
36
+ lastStep: "working",
37
+ lastActivityAt: 1000,
38
+ turns: 0,
39
+ toolCalls: 0,
40
+ tokens: 0,
41
+ cost: 0,
42
+ ...overrides,
43
+ };
44
+ }
45
+
46
+ /** Build one already-projected `RpcTask` fixture directly, bypassing `projectRpcActivity`, so `encodeActivityLines` shrink-stage tests control exact sizes. */
47
+ function rpcTask(id: string, status: TaskStatus, endedAt: number | null, opts: { items?: number; itemTextLen?: number; labelLen?: number } = {}): RpcTask {
48
+ const { items = 1, itemTextLen = 20, labelLen = 4 } = opts;
49
+ return {
50
+ summary: {
51
+ id,
52
+ agent: "a",
53
+ label: "l".repeat(labelLen),
54
+ prompt: "p",
55
+ status,
56
+ createdAt: 1,
57
+ startedAt: 1,
58
+ endedAt,
59
+ lastStep: "s",
60
+ lastActivityAt: 1,
61
+ turns: 0,
62
+ toolCalls: 0,
63
+ error: null,
64
+ },
65
+ thread: {
66
+ version: 1,
67
+ dropped: 0,
68
+ items: Array.from({ length: items }, (_v, index) => ({ kind: "note" as const, text: `${"n".repeat(itemTextLen)}${index}` })),
69
+ },
70
+ };
71
+ }
72
+
73
+ function byteLength(value: string): number {
74
+ return Buffer.byteLength(value, "utf8");
75
+ }
76
+
77
+ test("projectRpcActivity whitelists task fields and orders running, waiting, queued, then finished by endedAt desc", () => {
78
+ const store = new TaskStore();
79
+ store.add(task("running-1", "s1", { status: TASK_STATUS.RUNNING, agent: "explore", label: "Explore X", prompt: "short prompt", turns: 2, toolCalls: 1, createdAt: 1, lastActivityAt: 5 }));
80
+ store.add(task("waiting-1", "s1", { status: TASK_STATUS.WAITING, createdAt: 2, lastActivityAt: 6 }));
81
+ store.add(task("queued-1", "s1", { status: TASK_STATUS.QUEUED, createdAt: 3, startedAt: null, lastActivityAt: 3 }));
82
+ store.add(task("finished-old", "s1", { status: TASK_STATUS.COMPLETED, createdAt: 4, endedAt: 100, lastActivityAt: 100 }));
83
+ store.add(task("finished-new", "s1", { status: TASK_STATUS.FAILED, createdAt: 5, endedAt: 200, lastActivityAt: 200, error: "boom" }));
84
+
85
+ const activity = projectRpcActivity(store);
86
+
87
+ assert.equal(activity.schema, ACTIVITY_SCHEMA);
88
+ assert.deepEqual(activity.summary, store.summary());
89
+ assert.deepEqual(activity.tasks.map((entry) => entry.summary.id), ["running-1", "waiting-1", "queued-1", "finished-new", "finished-old"]);
90
+
91
+ const runningSummary = activity.tasks[0]!.summary;
92
+ assert.deepEqual(
93
+ Object.keys(runningSummary).sort(),
94
+ ["id", "agent", "label", "prompt", "status", "createdAt", "startedAt", "endedAt", "lastStep", "lastActivityAt", "turns", "toolCalls", "error"].sort(),
95
+ "only the whitelisted fields are projected: cwd, parentSessionId, mode, model, thinking, sessionPath, result, tokens, and cost never leak",
96
+ );
97
+ assert.equal(runningSummary.agent, "explore");
98
+ assert.equal(runningSummary.label, "Explore X");
99
+ assert.equal(runningSummary.prompt, "short prompt");
100
+ assert.equal(activity.tasks[4]!.summary.error, null);
101
+ assert.equal(activity.tasks[3]!.summary.error, "boom");
102
+ });
103
+
104
+ test("projectRpcActivity truncates the prompt, tool args, and tool output to their bounds", () => {
105
+ const store = new TaskStore();
106
+ store.add(task("t1", "s1", { prompt: "x".repeat(250) }));
107
+ store.apply("t1", { type: TASK_EVENT.TOOL_START, callId: "c1", name: "bash", args: { note: "y".repeat(600) } }, 1);
108
+ store.apply("t1", { type: TASK_EVENT.TOOL_END, callId: "c1", output: "z".repeat(600), isError: false }, 2);
109
+
110
+ const activity = projectRpcActivity(store);
111
+ const summary = activity.tasks[0]!.summary;
112
+ assert.equal(summary.prompt.length, 200);
113
+ assert.ok(summary.prompt.endsWith("…"));
114
+
115
+ const toolItem = activity.tasks[0]!.thread.items.at(-1) as { kind: string; name: string; args: string; running: boolean; isError: boolean; output: string };
116
+ assert.equal(toolItem.kind, "tool");
117
+ assert.equal(toolItem.name, "bash");
118
+ assert.equal(toolItem.running, false);
119
+ assert.equal(toolItem.isError, false);
120
+ assert.ok(toolItem.args.length <= 500 && toolItem.args.endsWith("…"));
121
+ assert.ok(toolItem.output.length <= 500 && toolItem.output.endsWith("…"));
122
+ assert.match(toolItem.args, /^\{"note":"y+…$/);
123
+ });
124
+
125
+ test("projectRpcActivity keeps only the last N thread items per task (default 40)", () => {
126
+ const store = new TaskStore();
127
+ store.add(task("t1", "s1"));
128
+ for (let index = 0; index < 50; index += 1) store.apply("t1", { type: TASK_EVENT.NOTE, text: `note-${index}` }, index);
129
+
130
+ const activity = projectRpcActivity(store);
131
+ const items = activity.tasks[0]!.thread.items as { text: string }[];
132
+ assert.equal(items.length, 40);
133
+ assert.equal(items[0]!.text, "note-10");
134
+ assert.equal(items.at(-1)!.text, "note-49");
135
+ });
136
+
137
+ test("projectRpcActivity truncates text, thinking, and note item text to 2000 characters", () => {
138
+ const store = new TaskStore();
139
+ store.add(task("t1", "s1"));
140
+ store.apply("t1", { type: TASK_EVENT.TEXT, text: "a".repeat(3000) }, 1);
141
+ store.apply("t1", { type: TASK_EVENT.THINKING, text: "b".repeat(3000) }, 2);
142
+ store.apply("t1", { type: TASK_EVENT.NOTE, text: "c".repeat(3000) }, 3);
143
+
144
+ const activity = projectRpcActivity(store);
145
+ const items = activity.tasks[0]!.thread.items as { kind: string; text: string }[];
146
+
147
+ assert.equal(items.length, 3);
148
+ for (const item of items) {
149
+ assert.ok(item.text.length <= 2000, `${item.kind} item must be bounded to 2000 characters`);
150
+ assert.ok(item.text.endsWith("…"), `${item.kind} item must carry the truncation marker`);
151
+ }
152
+ });
153
+
154
+ test("projectRpcActivity truncates a synthetic 1 MiB text item to 2000 characters", () => {
155
+ const store = new TaskStore();
156
+ store.add(task("t1", "s1"));
157
+ store.apply("t1", { type: TASK_EVENT.TEXT, text: "x".repeat(1024 * 1024) }, 1);
158
+
159
+ const activity = projectRpcActivity(store);
160
+ const item = activity.tasks[0]!.thread.items[0] as { text: string };
161
+
162
+ assert.equal(item.text.length, 2000);
163
+ assert.ok(item.text.endsWith("…"));
164
+ });
165
+
166
+ test("projectRpcActivity truncates the summary error, label, and lastStep fields to 500 characters", () => {
167
+ const store = new TaskStore();
168
+ store.add(task("t1", "s1", { label: "l".repeat(600) }));
169
+ store.apply("t1", { type: TASK_EVENT.AGENT_END, text: "", outcome: "error", diagnostic: "e".repeat(600) }, 1);
170
+
171
+ const activity = projectRpcActivity(store);
172
+ const summary = activity.tasks[0]!.summary;
173
+
174
+ assert.ok(summary.label.length <= 500 && summary.label.endsWith("…"));
175
+ assert.ok(summary.lastStep.length <= 500 && summary.lastStep.endsWith("…"));
176
+ assert.ok(summary.error !== null && summary.error.length <= 500 && summary.error.endsWith("…"));
177
+ });
178
+
179
+ test("projectRpcActivity honors a custom maxThreadItems override", () => {
180
+ const store = new TaskStore();
181
+ store.add(task("t1", "s1"));
182
+ for (let index = 0; index < 5; index += 1) store.apply("t1", { type: TASK_EVENT.NOTE, text: `n${index}` }, index);
183
+
184
+ const activity = projectRpcActivity(store, { maxThreadItems: 2 });
185
+ assert.deepEqual((activity.tasks[0]!.thread.items as { text: string }[]).map((item) => item.text), ["n3", "n4"]);
186
+ });
187
+
188
+ // Regression for the desktop app's Helpers tab showing helpers from every
189
+ // session: `TaskStore` restores finished tasks of every session from disk at
190
+ // startup, so an RPC push scoped to `opts.parentSessionId` must project only
191
+ // the current session's tasks, and its `summary` counts must match.
192
+ test("projectRpcActivity with parentSessionId keeps only that session's tasks and computes the summary from them", () => {
193
+ const store = new TaskStore();
194
+ store.add(task("mine-running", "s1", { status: TASK_STATUS.RUNNING }));
195
+ store.add(task("mine-finished", "s1", { status: TASK_STATUS.COMPLETED, endedAt: 100 }));
196
+ store.add(task("other-running", "s2", { status: TASK_STATUS.RUNNING }));
197
+ store.add(task("other-finished", "s2", { status: TASK_STATUS.COMPLETED, endedAt: 200 }));
198
+
199
+ const activity = projectRpcActivity(store, { parentSessionId: "s1" });
200
+
201
+ assert.deepEqual(
202
+ activity.tasks.map((entry) => entry.summary.id).sort(),
203
+ ["mine-finished", "mine-running"],
204
+ "only the requested session's tasks are projected",
205
+ );
206
+ assert.deepEqual(activity.summary, store.summary("s1"), "summary counts are scoped to the same session filter, not the whole store");
207
+ assert.notDeepEqual(activity.summary, store.summary(), "an unfiltered summary would have counted the other session's tasks too");
208
+ });
209
+
210
+ // Tasks restored on demand (e.g. opening an older task from another session
211
+ // in the overlay) can land in the shared store without ever being live in
212
+ // the current session; the RPC projection must still exclude them by id.
213
+ test("projectRpcActivity with parentSessionId excludes a task from another session even when it is the only one in the store", () => {
214
+ const store = new TaskStore();
215
+ store.add(task("not-mine", "other-session"));
216
+
217
+ const activity = projectRpcActivity(store, { parentSessionId: "current-session" });
218
+
219
+ assert.deepEqual(activity.tasks, []);
220
+ assert.deepEqual(activity.summary, { running: 0, queued: 0, waiting: 0, finished: 0 });
221
+ });
222
+
223
+ test("encodeActivityLines returns the activity untouched as a single line when it already fits", () => {
224
+ const activity: RpcActivity = { schema: ACTIVITY_SCHEMA, summary: { running: 0, queued: 0, waiting: 0, finished: 0 }, tasks: [] };
225
+
226
+ const lines = encodeActivityLines(activity);
227
+
228
+ assert.equal(lines.length, 1);
229
+ assert.deepEqual(JSON.parse(lines[0]!), activity);
230
+ });
231
+
232
+ test("encodeActivityLines halves thread items per task when only halving can help", () => {
233
+ const activity: RpcActivity = {
234
+ schema: ACTIVITY_SCHEMA,
235
+ summary: { running: 1, queued: 0, waiting: 0, finished: 0 },
236
+ tasks: [rpcTask("r1", TASK_STATUS.RUNNING, null, { items: 16, itemTextLen: 50 })],
237
+ };
238
+ const maxBytes = Math.floor(byteLength(JSON.stringify(activity)) / 2);
239
+
240
+ const [line] = encodeActivityLines(activity, maxBytes);
241
+ const shrunk = JSON.parse(line!) as RpcActivity;
242
+
243
+ assert.equal(shrunk.tasks.length, 1);
244
+ assert.ok(shrunk.tasks[0]!.thread.items.length < 16, "items must have been halved down");
245
+ assert.ok(shrunk.tasks[0]!.thread.items.length >= 1, "halving never removes the last item on a running task");
246
+ });
247
+
248
+ test("encodeActivityLines drops a finished task's thread items before dropping the task itself", () => {
249
+ const activity: RpcActivity = {
250
+ schema: ACTIVITY_SCHEMA,
251
+ summary: { running: 1, queued: 0, waiting: 0, finished: 1 },
252
+ tasks: [rpcTask("r1", TASK_STATUS.RUNNING, null, { items: 1, itemTextLen: 10 }), rpcTask("f1", TASK_STATUS.COMPLETED, 100, { items: 1, itemTextLen: 5000 })],
253
+ };
254
+ const full = byteLength(JSON.stringify(activity));
255
+ const maxBytes = full - 4000; // enough slack that only emptying f1's thread is needed
256
+
257
+ const [line] = encodeActivityLines(activity, maxBytes);
258
+ const shrunk = JSON.parse(line!) as RpcActivity;
259
+
260
+ const finishedTask = shrunk.tasks.find((entry) => entry.summary.id === "f1")!;
261
+ assert.deepEqual(finishedTask.thread.items, [], "the finished task's thread is emptied, not the task itself");
262
+ assert.ok(shrunk.tasks.some((entry) => entry.summary.id === "r1"), "the running task is untouched");
263
+ });
264
+
265
+ test("encodeActivityLines drops whole finished tasks, oldest first, once emptying threads is not enough", () => {
266
+ const activity: RpcActivity = {
267
+ schema: ACTIVITY_SCHEMA,
268
+ summary: { running: 1, queued: 0, waiting: 0, finished: 2 },
269
+ tasks: [
270
+ rpcTask("r1", TASK_STATUS.RUNNING, null, { items: 1, itemTextLen: 10 }),
271
+ rpcTask("f-new", TASK_STATUS.COMPLETED, 200, { items: 1, itemTextLen: 10, labelLen: 3000 }),
272
+ rpcTask("f-old", TASK_STATUS.COMPLETED, 100, { items: 1, itemTextLen: 10, labelLen: 3000 }),
273
+ ],
274
+ };
275
+ const emptiedThreads: RpcActivity = { ...activity, tasks: activity.tasks.map((entry) => (entry.summary.status === TASK_STATUS.COMPLETED ? { ...entry, thread: { ...entry.thread, items: [] } } : entry)) };
276
+ const maxBytes = byteLength(JSON.stringify(emptiedThreads)) - 1000; // still too big with both finished threads emptied
277
+
278
+ const [line] = encodeActivityLines(activity, maxBytes);
279
+ const shrunk = JSON.parse(line!) as RpcActivity;
280
+
281
+ assert.ok(!shrunk.tasks.some((entry) => entry.summary.id === "f-old"), "the oldest finished task must be dropped first");
282
+ assert.ok(shrunk.tasks.some((entry) => entry.summary.id === "f-new"), "the more recently finished task survives longer");
283
+ assert.ok(shrunk.tasks.some((entry) => entry.summary.id === "r1"), "the running task must never be dropped");
284
+ });
285
+
286
+ test("encodeActivityLines empties every remaining active task's thread as a last resort, emitting a summary-only payload", () => {
287
+ const activity: RpcActivity = {
288
+ schema: ACTIVITY_SCHEMA,
289
+ summary: { running: 2, queued: 0, waiting: 0, finished: 0 },
290
+ tasks: [
291
+ rpcTask("r1", TASK_STATUS.RUNNING, null, { items: 1, itemTextLen: 2000 }),
292
+ rpcTask("r2", TASK_STATUS.RUNNING, null, { items: 1, itemTextLen: 2000 }),
293
+ ],
294
+ };
295
+ // No finished tasks to drop, and both running tasks are already down to
296
+ // one item each; only emptying every thread can still shrink this.
297
+ const maxBytes = byteLength(JSON.stringify(activity)) - 1000;
298
+
299
+ const [line] = encodeActivityLines(activity, maxBytes);
300
+ const shrunk = JSON.parse(line!) as RpcActivity;
301
+
302
+ assert.equal(shrunk.tasks.length, 2, "no active task's summary is ever dropped");
303
+ for (const entry of shrunk.tasks) assert.deepEqual(entry.thread.items, [], "every task's thread is emptied, summary-only");
304
+ });
305
+
306
+ test("encodeActivityLines never throws even when nothing left can be dropped to fit", () => {
307
+ const activity: RpcActivity = {
308
+ schema: ACTIVITY_SCHEMA,
309
+ summary: { running: 1, queued: 0, waiting: 0, finished: 0 },
310
+ tasks: [rpcTask("r1", TASK_STATUS.RUNNING, null, { items: 3, itemTextLen: 50 })],
311
+ };
312
+
313
+ const lines = encodeActivityLines(activity, 1);
314
+
315
+ assert.equal(lines.length, 1);
316
+ assert.doesNotThrow(() => JSON.parse(lines[0]!));
317
+ });
318
+
319
+ function fakeScheduler() {
320
+ const pending: Array<{ id: number; fn: () => void }> = [];
321
+ let nextId = 0;
322
+ return {
323
+ schedule: (fn: () => void, _ms: number) => {
324
+ const id = nextId++;
325
+ pending.push({ id, fn });
326
+ return () => {
327
+ const index = pending.findIndex((entry) => entry.id === id);
328
+ if (index !== -1) pending.splice(index, 1);
329
+ };
330
+ },
331
+ pendingCount: () => pending.length,
332
+ flushAll: () => {
333
+ const due = pending.splice(0, pending.length);
334
+ for (const entry of due) entry.fn();
335
+ },
336
+ };
337
+ }
338
+
339
+ test("createRpcActivityPublisher coalesces multiple task changes into one flush per window", () => {
340
+ const store = new TaskStore();
341
+ store.add(task("t1", "s1"));
342
+ const scheduler = fakeScheduler();
343
+ const calls: string[][] = [];
344
+ const publisher = createRpcActivityPublisher({ store, ui: { setWidget: (key, lines) => { assert.equal(key, ACTIVITY_WIDGET_KEY); calls.push(lines); } }, schedule: scheduler.schedule });
345
+
346
+ publisher.start();
347
+ assert.equal(scheduler.pendingCount(), 1, "start schedules exactly one coalescing flush");
348
+ store.apply("t1", { type: TASK_EVENT.TEXT, text: "hello" }, 1);
349
+ store.apply("t1", { type: TASK_EVENT.TEXT, text: " world" }, 2);
350
+ assert.equal(scheduler.pendingCount(), 1, "further changes inside the coalescing window do not add timers");
351
+ assert.equal(calls.length, 0, "nothing is published before the coalescing timer fires");
352
+
353
+ scheduler.flushAll();
354
+
355
+ assert.equal(calls.length, 1);
356
+ const activity = JSON.parse(calls[0]![0]!) as RpcActivity;
357
+ assert.equal(activity.schema, ACTIVITY_SCHEMA);
358
+ assert.equal(activity.tasks[0]!.summary.id, "t1");
359
+ });
360
+
361
+ test("createRpcActivityPublisher tracks a task created after start through the summary subscription", () => {
362
+ const store = new TaskStore();
363
+ const scheduler = fakeScheduler();
364
+ const calls: string[][] = [];
365
+ const publisher = createRpcActivityPublisher({ store, ui: { setWidget: (_key, lines) => calls.push(lines) }, schedule: scheduler.schedule });
366
+
367
+ publisher.start();
368
+ scheduler.flushAll();
369
+ calls.length = 0;
370
+
371
+ store.add(task("late-task", "s1"));
372
+ assert.equal(scheduler.pendingCount(), 1, "adding a task schedules a coalesced flush");
373
+ scheduler.flushAll();
374
+ calls.length = 0;
375
+
376
+ store.apply("late-task", { type: TASK_EVENT.TEXT, text: "streamed" }, 1);
377
+ assert.equal(scheduler.pendingCount(), 1, "the new task's own thread changes must also be tracked");
378
+ scheduler.flushAll();
379
+
380
+ const activity = JSON.parse(calls[0]![0]!) as RpcActivity;
381
+ assert.deepEqual((activity.tasks[0]!.thread.items as { text: string }[]).map((item) => item.text), ["streamed"]);
382
+ });
383
+
384
+ test("createRpcActivityPublisher stop unsubscribes everything and publishes exactly one final frame", () => {
385
+ const store = new TaskStore();
386
+ store.add(task("t1", "s1"));
387
+ const scheduler = fakeScheduler();
388
+ const calls: string[][] = [];
389
+ const publisher = createRpcActivityPublisher({ store, ui: { setWidget: (_key, lines) => calls.push(lines) }, schedule: scheduler.schedule });
390
+ publisher.start();
391
+ scheduler.flushAll();
392
+ calls.length = 0;
393
+
394
+ publisher.stop();
395
+
396
+ assert.equal(calls.length, 1, "stop publishes exactly one final frame");
397
+ assert.equal(scheduler.pendingCount(), 0, "stop cancels any pending coalescing timer");
398
+
399
+ store.apply("t1", { type: TASK_EVENT.TEXT, text: "after stop" }, 10);
400
+ assert.equal(scheduler.pendingCount(), 0, "task subscriptions are torn down by stop");
401
+ assert.equal(calls.length, 1, "no further frame is published after stop");
402
+ });
403
+
404
+ // `parentSessionId` is a value captured at construction, not a live getter.
405
+ // Honoring a mid-process session switch (a resumed/new/forked session) is
406
+ // the caller's job: stop the old publisher and construct a new one scoped
407
+ // to the new session id, exactly as `extensions/gentle-agents.ts` does on
408
+ // every `session_start`. This locks in that only the newly scoped publisher
409
+ // ever sees the other session's tasks.
410
+ test("createRpcActivityPublisher recreated with a new parentSessionId after a session switch publishes only the new session's tasks", () => {
411
+ const store = new TaskStore();
412
+ store.add(task("session-a-task", "session-a"));
413
+ const scheduler = fakeScheduler();
414
+ const calls: string[][] = [];
415
+ const ui = { setWidget: (_key: string, lines: string[]) => calls.push(lines) };
416
+
417
+ const first = createRpcActivityPublisher({ store, ui, schedule: scheduler.schedule, parentSessionId: "session-a" });
418
+ first.start();
419
+ scheduler.flushAll();
420
+ const firstActivity = JSON.parse(calls.at(-1)![0]!) as RpcActivity;
421
+ assert.deepEqual(firstActivity.tasks.map((entry) => entry.summary.id), ["session-a-task"]);
422
+
423
+ first.stop();
424
+ calls.length = 0;
425
+ store.add(task("session-b-task", "session-b"));
426
+
427
+ const second = createRpcActivityPublisher({ store, ui, schedule: scheduler.schedule, parentSessionId: "session-b" });
428
+ second.start();
429
+ scheduler.flushAll();
430
+
431
+ const secondActivity = JSON.parse(calls.at(-1)![0]!) as RpcActivity;
432
+ assert.deepEqual(secondActivity.tasks.map((entry) => entry.summary.id), ["session-b-task"], "the publisher scoped to the new session must never carry the old session's task");
433
+ });
434
+
435
+ test("createRpcActivityPublisher flush publishes immediately and cancels a pending coalescing timer", () => {
436
+ const store = new TaskStore();
437
+ store.add(task("t1", "s1"));
438
+ const scheduler = fakeScheduler();
439
+ const calls: string[][] = [];
440
+ const publisher = createRpcActivityPublisher({ store, ui: { setWidget: (_key, lines) => calls.push(lines) }, schedule: scheduler.schedule });
441
+ publisher.start();
442
+ scheduler.flushAll();
443
+ calls.length = 0;
444
+
445
+ store.apply("t1", { type: TASK_EVENT.TEXT, text: "pending" }, 1);
446
+ assert.equal(scheduler.pendingCount(), 1);
447
+
448
+ publisher.flush();
449
+
450
+ assert.equal(calls.length, 1);
451
+ assert.equal(scheduler.pendingCount(), 0, "flush cancels the timer it preempted");
452
+ });
453
+
454
+ test("createRpcActivityPublisher swallows setWidget errors through an injectable onError", () => {
455
+ const store = new TaskStore();
456
+ store.add(task("t1", "s1"));
457
+ const scheduler = fakeScheduler();
458
+ const errors: unknown[] = [];
459
+ const publisher = createRpcActivityPublisher({
460
+ store,
461
+ ui: { setWidget: () => { throw new Error("boom"); } },
462
+ schedule: scheduler.schedule,
463
+ onError: (error) => errors.push(error),
464
+ });
465
+
466
+ assert.doesNotThrow(() => {
467
+ publisher.start();
468
+ scheduler.flushAll();
469
+ });
470
+ assert.equal(errors.length, 1);
471
+ assert.match(String((errors[0] as Error).message), /boom/);
472
+ assert.doesNotThrow(() => publisher.stop());
473
+ });
@@ -5,6 +5,7 @@ import { AGENT_MODE, parseAgentsConfig, resolveAgentProfile, type AgentDefinitio
5
5
  import { TASK_STATUS, TaskStore, type TaskRecord } from "../lib/agents-protocol.ts";
6
6
  import { AgentRunner, childArguments, JsonLines, piCommand, abortReasonText, type ChildLike, type RunnerDeps, type RunnerHooks, type TaskRequest } from "../lib/agents-runner.ts";
7
7
  import { fakeChild, type FakeChild } from "./agents-fake-child.ts";
8
+ import { INTERACTIVE_HOST_ENV } from "../lib/rpc-host.ts";
8
9
 
9
10
  // Gentle Agents runner: every subagent is a child `pi --mode rpc` process.
10
11
  // The host only parses JSON lines, applies deltas to the store, answers
@@ -806,6 +807,15 @@ for (const [platform, detached] of [["win32", false], ["linux", true]] as const)
806
807
  assert.equal((await runner.waitFor(task.id)).status, TASK_STATUS.COMPLETED);
807
808
  });
808
809
 
810
+ test("AgentRunner strips the interactive-host signal from every spawned child env", async () => {
811
+ const { runner, spawnOptions } = harness();
812
+ runner.run(request({ env: { PATH: "/fixture", [INTERACTIVE_HOST_ENV]: "1" } }));
813
+ await tick();
814
+
815
+ assert.equal(spawnOptions[0]?.env[INTERACTIVE_HOST_ENV], undefined, "subagent children never see the interactive-host signal");
816
+ assert.equal(spawnOptions[0]?.env.PATH, "/fixture", "unrelated inherited env is preserved");
817
+ });
818
+
809
819
  function ipcCleanupHarness(connected: boolean | undefined) {
810
820
  const child = fakeChild({ exitOnKill: false });
811
821
  const disconnectListeners: Array<(...args: unknown[]) => void> = [];