pi-better-harness 0.3.5 → 0.3.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-better-background-tasks",
3
- "version": "0.2.11",
3
+ "version": "0.2.12",
4
4
  "description": "Pi extension for durable background shell tasks, watchers, logs, and status inspection.",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -103,6 +103,12 @@ const StatusActionParams = Type.Object({
103
103
  verbose: Type.Optional(Type.Boolean()),
104
104
  });
105
105
 
106
+ const BACKGROUND_ORCHESTRATION_GUIDELINES = [
107
+ "Use background tasks for genuinely long-running processes or repeated checks. Run short commands in the foreground.",
108
+ "When a structured plan is active, keep it as the coordinator ledger: launch relevant background work early, continue unblocked foreground work without polling, and update the plan after inspecting each terminal result or failure.",
109
+ "Do not treat launch as completion of the parent milestone; relevant background work must be terminal, inspected, and integrated before verification or completion.",
110
+ ];
111
+
106
112
  export function registerTools(pi: ExtensionAPI): void {
107
113
  let activeSession: BackgroundTaskCallbackOrigin | undefined;
108
114
  const getActiveSession = () => activeSession;
@@ -125,6 +131,7 @@ export function registerTools(pi: ExtensionAPI): void {
125
131
  name: "bg_task_spawn",
126
132
  label: "BG Spawn",
127
133
  description: "Start a long-running background process and return immediately with its task id. For remote work, prefer structured ssh: pass ssh:{host,user} and put the remote command in command; spawn defaults to a remote tmux session with durable local logs and real remote stop. For short synchronous remote commands that should return output now, use remote_bash from pi-better-ssh. If tmux is missing, the preset attempts to install tmux non-interactively and fails closed with operator guidance when setup cannot proceed. Explicit remote.session=direct skips tmux, but direct mode has weaker stop semantics and may leave the remote process running. Never wait or poll in the foreground.",
134
+ promptGuidelines: BACKGROUND_ORCHESTRATION_GUIDELINES,
128
135
  parameters: SpawnParams,
129
136
  async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
130
137
  activeSession = getCallbackOrigin(ctx);
@@ -138,6 +145,7 @@ export function registerTools(pi: ExtensionAPI): void {
138
145
  name: "bg_task_watch",
139
146
  label: "BG Watch",
140
147
  description: "Poll a command in the background until success_when, failure_when, or timeout matches. For remote work, prefer structured ssh: pass ssh:{host,user} and provide the remote command in command; each interval opens a direct one-shot SSH poll without tmux installation. For short synchronous remote commands that should return output now, use remote_bash from pi-better-ssh. Returns immediately with its task id. Default timeout 900 seconds; pass timeout_seconds:0 to disable.",
148
+ promptGuidelines: BACKGROUND_ORCHESTRATION_GUIDELINES,
141
149
  parameters: WatchParams,
142
150
  async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
143
151
  activeSession = getCallbackOrigin(ctx);
@@ -198,6 +206,7 @@ export function registerTools(pi: ExtensionAPI): void {
198
206
  name: "bg_task",
199
207
  label: "BG Task",
200
208
  description: "Action wrapper for background tasks: spawn, watch, list, status, log, stop, or clear. For remote work, prefer structured ssh: pass ssh:{host,user} and provide the remote command in command. For short synchronous remote commands that should return output now, use remote_bash from pi-better-ssh. SSH spawn defaults to durable tmux; SSH watches use direct one-shot polls without tmux installation; remote.session=direct is a weaker-stop spawn escape hatch. Spawn/watch return immediately; do not poll in foreground. For action:status, default compact output and use verbose:true only for full metadata. For action:log, default compact tail and use tail_lines:0 only for explicit full logs.",
209
+ promptGuidelines: BACKGROUND_ORCHESTRATION_GUIDELINES,
201
210
  parameters: ActionParams,
202
211
  renderResult(result: unknown, options: unknown, theme: unknown) {
203
212
  return renderBackgroundTaskLogDisplay(result, options, theme);
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-better-goal",
3
- "version": "0.2.1",
3
+ "version": "0.2.2",
4
4
  "description": "Pi extension for goal tracking with background-aware continuation.",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -713,7 +713,7 @@ export default function (pi: ExtensionAPI): void {
713
713
  }
714
714
 
715
715
  const backgroundInstruction = snapshot.backgroundRunning
716
- ? ` The goal still has delegated background work running (${summarizeActiveBackground(snapshot)}). Foreground idleness alone is not goal completion; do not mark the goal complete until delegated background work reaches a terminal state and its result or failure has been inspected.`
716
+ ? ` The goal still has delegated background work running (${summarizeActiveBackground(snapshot)}). Foreground idleness alone is not goal completion; keep any structured plan current and do not mark verification, the plan, or the goal complete until every relevant delegated task reaches a terminal state and its result or failure has been inspected and integrated.`
717
717
  : "";
718
718
  return {
719
719
  systemPrompt:
@@ -7,10 +7,17 @@
7
7
  - Gives models `update_plan` and `get_plan` tools for atomic, explicit progress updates.
8
8
  - Shows the complete checklist of completed, active, pending, and blocked steps above the editor.
9
9
  - Persists plan state and display preferences on the active Pi session branch.
10
+ - Keeps a completed plan visible for 30 seconds, then clears it automatically.
10
11
  - Opens the complete plan with `/plan`.
11
12
 
12
13
  Plan progress is checklist progress, not an estimate of effort. The extension never infers completion from prose or successful tool calls.
13
14
 
15
+ ## Coordinating Delegated Work
16
+
17
+ Use the plan as the foreground coordinator's milestone ledger. Delegate independent, sufficiently substantial work early with subagents, and use background tasks for long-running processes or repeated checks. Keep doing unblocked foreground work after launch; do not poll workers.
18
+
19
+ Concurrent workers belong under one `in_progress` coordinator step rather than one active plan step per worker. Worker tools and the background-work navigator own individual run status. Complete verification and the plan only after every relevant delegated task is terminal and its result or failure has been inspected and integrated.
20
+
14
21
  ## Install
15
22
 
16
23
  ```sh
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-better-plan",
3
- "version": "0.1.3",
3
+ "version": "0.1.5",
4
4
  "description": "Structured execution plans with persistent progress for Pi.",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -70,4 +70,4 @@
70
70
  "engines": {
71
71
  "node": ">=22.0.0"
72
72
  }
73
- }
73
+ }
@@ -4,6 +4,7 @@ import { Text } from "@earendil-works/pi-tui";
4
4
  import { Type } from "typebox";
5
5
 
6
6
  import {
7
+ completedPlanClearDelay,
7
8
  planClearEntry,
8
9
  planDisplayEntry,
9
10
  planProgress,
@@ -42,21 +43,53 @@ export default function planExtension(pi: ExtensionAPI): void {
42
43
  let currentPlan: PlanSnapshot | null = null;
43
44
  let displayMode: PlanDisplayMode = "auto";
44
45
  let refreshWidget: ((force?: boolean) => void) | undefined;
46
+ let completedPlanClearTimer: ReturnType<typeof setTimeout> | undefined;
45
47
 
46
48
  const refresh = (force = false): void => refreshWidget?.(force);
47
49
 
50
+ const cancelCompletedPlanClear = (): void => {
51
+ if (completedPlanClearTimer) clearTimeout(completedPlanClearTimer);
52
+ completedPlanClearTimer = undefined;
53
+ };
54
+
55
+ const clearPlan = (): void => {
56
+ cancelCompletedPlanClear();
57
+ currentPlan = null;
58
+ pi.appendEntry(EXTENSION_NAME, planClearEntry());
59
+ refresh(true);
60
+ };
61
+
62
+ const scheduleCompletedPlanClear = (): void => {
63
+ cancelCompletedPlanClear();
64
+ if (!currentPlan) return;
65
+ const delay = completedPlanClearDelay(currentPlan);
66
+ if (delay === null) return;
67
+ const completedPlan = currentPlan;
68
+ const timer = setTimeout(() => {
69
+ if (completedPlanClearTimer !== timer) return;
70
+ completedPlanClearTimer = undefined;
71
+ if (currentPlan?.planId !== completedPlan.planId || currentPlan.revision !== completedPlan.revision) return;
72
+ clearPlan();
73
+ }, delay);
74
+ completedPlanClearTimer = timer;
75
+ timer.unref?.();
76
+ };
77
+
48
78
  const restore = (ctx: ExtensionContext): void => {
79
+ cancelCompletedPlanClear();
49
80
  const state = reconstructPlanState(ctx.sessionManager.getBranch());
50
81
  currentPlan = state.plan;
51
82
  displayMode = state.displayMode;
52
83
  if (ctx.hasUI) ctx.ui.setStatus(LEGACY_PLAN_NAV_STATUS_KEY, undefined);
53
84
  refresh(true);
85
+ scheduleCompletedPlanClear();
54
86
  };
55
87
 
56
88
  const persistPlan = (plan: PlanSnapshot): void => {
57
89
  currentPlan = plan;
58
90
  pi.appendEntry(EXTENSION_NAME, planSetEntry(plan));
59
91
  refresh(true);
92
+ scheduleCompletedPlanClear();
60
93
  };
61
94
 
62
95
  const setDisplayMode = (mode: PlanDisplayMode): void => {
@@ -119,6 +152,8 @@ export default function planExtension(pi: ExtensionAPI): void {
119
152
  promptGuidelines: [
120
153
  "Use update_plan for work with three or more meaningful steps, and update it immediately when a step completes, becomes blocked, or scope changes.",
121
154
  "Keep at most one update_plan step in_progress and do not mark a step completed until its required verification succeeds.",
155
+ "Use the plan as the foreground coordinator's milestone ledger. Delegate independent, sufficiently substantial work early when subagent or background-task tools are available, and continue unblocked foreground work instead of waiting or polling.",
156
+ "Represent concurrent delegated work under one in_progress coordinator step. Before completing verification or the plan, ensure every relevant delegated task is terminal, then inspect and integrate its result or failure.",
122
157
  ],
123
158
  parameters: UpdatePlanSchema,
124
159
  async execute(_toolCallId, params) {
@@ -171,9 +206,7 @@ export default function planExtension(pi: ExtensionAPI): void {
171
206
  const input = args.trim().toLowerCase();
172
207
  if (!input) return showFullPlan(ctx);
173
208
  if (input === "clear") {
174
- currentPlan = null;
175
- pi.appendEntry(EXTENSION_NAME, planClearEntry());
176
- refresh(true);
209
+ clearPlan();
177
210
  ctx.ui.notify("Plan cleared.", "info");
178
211
  return;
179
212
  }
@@ -209,6 +242,7 @@ export default function planExtension(pi: ExtensionAPI): void {
209
242
  };
210
243
  });
211
244
  pi.on("session_shutdown", async (_event, ctx) => {
245
+ cancelCompletedPlanClear();
212
246
  refreshWidget = undefined;
213
247
  try {
214
248
  ctx.ui.setStatus(LEGACY_PLAN_NAV_STATUS_KEY, undefined);
@@ -238,6 +272,8 @@ function planPrompt(plan: PlanSnapshot): string {
238
272
  return [
239
273
  "Current structured execution plan:",
240
274
  ...plan.steps.map((item, index) => `${index + 1}. [${item.status}] ${item.step}`),
241
- "Use update_plan immediately when a step completes, becomes blocked, or the scope changes. Completion must be explicit and evidence-backed.",
275
+ "Use update_plan immediately when a step completes, becomes blocked, or the scope changes.",
276
+ "Treat the plan as the foreground coordinator's milestone ledger: delegate independent, sufficiently substantial work early, keep one coordinator step in_progress across concurrent workers, and continue unblocked foreground work instead of waiting or polling.",
277
+ "Do not complete verification or the plan until relevant delegated work is terminal, its results or failures have been inspected and integrated, and the outcome is evidence-backed.",
242
278
  ].join("\n");
243
279
  }
@@ -4,6 +4,7 @@ import { EXTENSION_NAME } from "./types.js";
4
4
  const MAX_STEPS = 50;
5
5
  const MAX_STEP_CHARS = 500;
6
6
  const MAX_EXPLANATION_CHARS = 2_000;
7
+ export const COMPLETED_PLAN_RETENTION_MS = 30_000;
7
8
 
8
9
  interface SessionEntryLike {
9
10
  type: string;
@@ -54,6 +55,7 @@ export function replacePlan(
54
55
  steps: readonly PlanStepInput[],
55
56
  explanation?: string,
56
57
  now = nowSeconds(),
58
+ completedAtMs = Date.now(),
57
59
  ): PlanSnapshot {
58
60
  const error = validatePlanInput(steps, explanation);
59
61
  if (error) throw new Error(error);
@@ -68,6 +70,8 @@ export function replacePlan(
68
70
  status: item.status,
69
71
  };
70
72
  });
73
+ const isComplete = normalizedSteps.every((item) => item.status === "completed");
74
+ const currentIsComplete = current?.steps.every((item) => item.status === "completed") === true;
71
75
 
72
76
  return {
73
77
  version: 1,
@@ -77,9 +81,18 @@ export function replacePlan(
77
81
  steps: normalizedSteps,
78
82
  createdAt: current?.createdAt ?? now,
79
83
  updatedAt: now,
84
+ ...(isComplete
85
+ ? { completedAtMs: currentIsComplete ? current.completedAtMs ?? current.updatedAt * 1_000 : completedAtMs }
86
+ : {}),
80
87
  };
81
88
  }
82
89
 
90
+ export function completedPlanClearDelay(plan: PlanSnapshot, now = Date.now()): number | null {
91
+ if (planProgress(plan).state !== "complete") return null;
92
+ const completedAtMs = plan.completedAtMs ?? plan.updatedAt * 1_000;
93
+ return Math.max(0, completedAtMs + COMPLETED_PLAN_RETENTION_MS - now);
94
+ }
95
+
83
96
  export function planProgress(plan: PlanSnapshot): PlanProgress {
84
97
  const completed = plan.steps.filter((item) => item.status === "completed").length;
85
98
  const pending = plan.steps.filter((item) => item.status === "pending").length;
@@ -149,6 +162,7 @@ function isPlanSnapshot(value: unknown): value is PlanSnapshot {
149
162
  typeof candidate.revision === "number" &&
150
163
  Array.isArray(candidate.steps) &&
151
164
  typeof candidate.createdAt === "number" &&
152
- typeof candidate.updatedAt === "number"
165
+ typeof candidate.updatedAt === "number" &&
166
+ (candidate.completedAtMs === undefined || typeof candidate.completedAtMs === "number")
153
167
  );
154
168
  }
@@ -20,6 +20,7 @@ export interface PlanSnapshot {
20
20
  steps: PlanStep[];
21
21
  createdAt: number;
22
22
  updatedAt: number;
23
+ completedAtMs?: number;
23
24
  }
24
25
 
25
26
  export interface PlanProgress {
@@ -127,6 +127,11 @@ const SUBAGENT_TOOLS = [
127
127
  "subagent_result",
128
128
  ];
129
129
 
130
+ const SUBAGENT_ORCHESTRATION_GUIDELINES = [
131
+ "When a structured plan is active, keep it as the parent-owned coordinator ledger: delegate bounded independent work, continue unblocked foreground work, and update the plan after integrating each result or failure.",
132
+ "Do not treat launching a subagent as completion of the parent milestone; relevant terminal results must be inspected and integrated before verification or completion.",
133
+ ];
134
+
130
135
  const GOAL_READY_EVENT = "pi-better-goal:ready";
131
136
  const GOAL_REGISTER_PROVIDER_EVENT = "pi-better-goal:register-provider";
132
137
  const goalReadySubscriptions = new WeakSet<ExtensionAPI>();
@@ -1267,7 +1272,8 @@ export default function (pi: ExtensionAPI) {
1267
1272
  promptGuidelines: [
1268
1273
  "Use subagent_spawn for independent work the user should not have to wait on. It returns at once with a run id; that return IS the deliverable — report the id to the user and continue.",
1269
1274
  "After subagent_spawn, do NOT call subagent_output or subagent_result in a loop to wait for the result, and do NOT sleep. The run completes on its own and reports back on the next turn.",
1270
- "Only call subagent_result / subagent_output when the user explicitly asks how a run is going or for its result.",
1275
+ "Call subagent_result after a completion or attention callback, or when the user explicitly asks for the result. Use subagent_output only when the user explicitly asks how a run is progressing; never use either tool to poll.",
1276
+ ...SUBAGENT_ORCHESTRATION_GUIDELINES,
1271
1277
  "The tools param is both the tool allowlist AND what determines which extensions load in the child (e.g. tools='read,bash,web_fetch' loads only the web-tools package). Ask for the tools the task needs and nothing more; clean:true gives a built-ins-only child. Pick a model with the model param (e.g. 'xai/grok-4.5@high'); providerless model patterns are resolved by Pi, while provider/model is deterministic and loads mapped provider extensions.",
1272
1278
  "By default the subagent is sandboxed (writes confined to its working dir, reads and network open) and triggers completion here on finish. Set callback:false to finish quietly — then read the result on demand via subagent_result.",
1273
1279
  "Use git_clone_workspace:true when the subagent will mutate Git in a sandbox. The parent prepares a disposable, self-contained clone with a real .git/ directory inside the sandbox root, so linked-worktree metadata outside the sandbox cannot stall the child.",
@@ -1336,6 +1342,7 @@ export default function (pi: ExtensionAPI) {
1336
1342
  "Use subagent_spawn_batch when you have several independent tasks to delegate. It returns immediately with a batch id and one run id per launched job.",
1337
1343
  "Each job is a normal subagent run; use subagent_result / subagent_output / subagent_stop with the individual run ids just like subagent_spawn.",
1338
1344
  "Do NOT poll for results. Each job reports back on its own when it finishes.",
1345
+ ...SUBAGENT_ORCHESTRATION_GUIDELINES,
1339
1346
  "By default the whole batch is rejected if there is not enough capacity. Set onCapacity to 'launch-available' to launch as many as fit and report the rest as skipped.",
1340
1347
  ],
1341
1348
  parameters: Type.Object({
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-better-subagents",
3
- "version": "0.1.26",
3
+ "version": "0.1.27",
4
4
  "description": "Pi extension for detached, sandboxed subagent runs that keep the foreground session free.",
5
5
  "license": "MIT",
6
6
  "type": "module",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-better-harness",
3
- "version": "0.3.5",
3
+ "version": "0.3.7",
4
4
  "description": "Pi extension bundle for a write sandbox, subagents, background tasks, SSH, goals, and structured plans.",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -50,12 +50,12 @@
50
50
  "test": "node --test test/*.test.mjs"
51
51
  },
52
52
  "dependencies": {
53
- "pi-better-background-tasks": "0.2.11",
54
- "pi-better-goal": "0.2.1",
55
- "pi-better-plan": "0.1.3",
53
+ "pi-better-background-tasks": "0.2.12",
54
+ "pi-better-goal": "0.2.2",
55
+ "pi-better-plan": "0.1.5",
56
56
  "pi-better-sandbox": "0.3.0",
57
57
  "pi-better-ssh": "0.1.1",
58
- "pi-better-subagents": "0.1.26"
58
+ "pi-better-subagents": "0.1.27"
59
59
  },
60
60
  "bundledDependencies": [
61
61
  "pi-better-background-tasks",