pi-better-harness 0.3.5 → 0.3.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/node_modules/pi-better-background-tasks/package.json +1 -1
- package/node_modules/pi-better-background-tasks/src/tools.ts +9 -0
- package/node_modules/pi-better-goal/package.json +1 -1
- package/node_modules/pi-better-goal/src/index.ts +1 -1
- package/node_modules/pi-better-plan/README.md +7 -0
- package/node_modules/pi-better-plan/package.json +2 -2
- package/node_modules/pi-better-plan/src/index.ts +40 -4
- package/node_modules/pi-better-plan/src/plan-state.ts +15 -1
- package/node_modules/pi-better-plan/src/types.ts +1 -0
- package/node_modules/pi-better-subagents/index.ts +8 -1
- package/node_modules/pi-better-subagents/package.json +1 -1
- package/package.json +5 -5
|
@@ -103,6 +103,12 @@ const StatusActionParams = Type.Object({
|
|
|
103
103
|
verbose: Type.Optional(Type.Boolean()),
|
|
104
104
|
});
|
|
105
105
|
|
|
106
|
+
const BACKGROUND_ORCHESTRATION_GUIDELINES = [
|
|
107
|
+
"Use background tasks for genuinely long-running processes or repeated checks. Run short commands in the foreground.",
|
|
108
|
+
"When a structured plan is active, keep it as the coordinator ledger: launch relevant background work early, continue unblocked foreground work without polling, and update the plan after inspecting each terminal result or failure.",
|
|
109
|
+
"Do not treat launch as completion of the parent milestone; relevant background work must be terminal, inspected, and integrated before verification or completion.",
|
|
110
|
+
];
|
|
111
|
+
|
|
106
112
|
export function registerTools(pi: ExtensionAPI): void {
|
|
107
113
|
let activeSession: BackgroundTaskCallbackOrigin | undefined;
|
|
108
114
|
const getActiveSession = () => activeSession;
|
|
@@ -125,6 +131,7 @@ export function registerTools(pi: ExtensionAPI): void {
|
|
|
125
131
|
name: "bg_task_spawn",
|
|
126
132
|
label: "BG Spawn",
|
|
127
133
|
description: "Start a long-running background process and return immediately with its task id. For remote work, prefer structured ssh: pass ssh:{host,user} and put the remote command in command; spawn defaults to a remote tmux session with durable local logs and real remote stop. For short synchronous remote commands that should return output now, use remote_bash from pi-better-ssh. If tmux is missing, the preset attempts to install tmux non-interactively and fails closed with operator guidance when setup cannot proceed. Explicit remote.session=direct skips tmux, but direct mode has weaker stop semantics and may leave the remote process running. Never wait or poll in the foreground.",
|
|
134
|
+
promptGuidelines: BACKGROUND_ORCHESTRATION_GUIDELINES,
|
|
128
135
|
parameters: SpawnParams,
|
|
129
136
|
async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
|
|
130
137
|
activeSession = getCallbackOrigin(ctx);
|
|
@@ -138,6 +145,7 @@ export function registerTools(pi: ExtensionAPI): void {
|
|
|
138
145
|
name: "bg_task_watch",
|
|
139
146
|
label: "BG Watch",
|
|
140
147
|
description: "Poll a command in the background until success_when, failure_when, or timeout matches. For remote work, prefer structured ssh: pass ssh:{host,user} and provide the remote command in command; each interval opens a direct one-shot SSH poll without tmux installation. For short synchronous remote commands that should return output now, use remote_bash from pi-better-ssh. Returns immediately with its task id. Default timeout 900 seconds; pass timeout_seconds:0 to disable.",
|
|
148
|
+
promptGuidelines: BACKGROUND_ORCHESTRATION_GUIDELINES,
|
|
141
149
|
parameters: WatchParams,
|
|
142
150
|
async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
|
|
143
151
|
activeSession = getCallbackOrigin(ctx);
|
|
@@ -198,6 +206,7 @@ export function registerTools(pi: ExtensionAPI): void {
|
|
|
198
206
|
name: "bg_task",
|
|
199
207
|
label: "BG Task",
|
|
200
208
|
description: "Action wrapper for background tasks: spawn, watch, list, status, log, stop, or clear. For remote work, prefer structured ssh: pass ssh:{host,user} and provide the remote command in command. For short synchronous remote commands that should return output now, use remote_bash from pi-better-ssh. SSH spawn defaults to durable tmux; SSH watches use direct one-shot polls without tmux installation; remote.session=direct is a weaker-stop spawn escape hatch. Spawn/watch return immediately; do not poll in foreground. For action:status, default compact output and use verbose:true only for full metadata. For action:log, default compact tail and use tail_lines:0 only for explicit full logs.",
|
|
209
|
+
promptGuidelines: BACKGROUND_ORCHESTRATION_GUIDELINES,
|
|
201
210
|
parameters: ActionParams,
|
|
202
211
|
renderResult(result: unknown, options: unknown, theme: unknown) {
|
|
203
212
|
return renderBackgroundTaskLogDisplay(result, options, theme);
|
|
@@ -713,7 +713,7 @@ export default function (pi: ExtensionAPI): void {
|
|
|
713
713
|
}
|
|
714
714
|
|
|
715
715
|
const backgroundInstruction = snapshot.backgroundRunning
|
|
716
|
-
? ` The goal still has delegated background work running (${summarizeActiveBackground(snapshot)}). Foreground idleness alone is not goal completion; do not mark the goal complete until delegated
|
|
716
|
+
? ` The goal still has delegated background work running (${summarizeActiveBackground(snapshot)}). Foreground idleness alone is not goal completion; keep any structured plan current and do not mark verification, the plan, or the goal complete until every relevant delegated task reaches a terminal state and its result or failure has been inspected and integrated.`
|
|
717
717
|
: "";
|
|
718
718
|
return {
|
|
719
719
|
systemPrompt:
|
|
@@ -7,10 +7,17 @@
|
|
|
7
7
|
- Gives models `update_plan` and `get_plan` tools for atomic, explicit progress updates.
|
|
8
8
|
- Shows the complete checklist of completed, active, pending, and blocked steps above the editor.
|
|
9
9
|
- Persists plan state and display preferences on the active Pi session branch.
|
|
10
|
+
- Keeps a completed plan visible for 30 seconds, then clears it automatically.
|
|
10
11
|
- Opens the complete plan with `/plan`.
|
|
11
12
|
|
|
12
13
|
Plan progress is checklist progress, not an estimate of effort. The extension never infers completion from prose or successful tool calls.
|
|
13
14
|
|
|
15
|
+
## Coordinating Delegated Work
|
|
16
|
+
|
|
17
|
+
Use the plan as the foreground coordinator's milestone ledger. Delegate independent, sufficiently substantial work early with subagents, and use background tasks for long-running processes or repeated checks. Keep doing unblocked foreground work after launch; do not poll workers.
|
|
18
|
+
|
|
19
|
+
Concurrent workers belong under one `in_progress` coordinator step rather than one active plan step per worker. Worker tools and the background-work navigator own individual run status. Complete verification and the plan only after every relevant delegated task is terminal and its result or failure has been inspected and integrated.
|
|
20
|
+
|
|
14
21
|
## Install
|
|
15
22
|
|
|
16
23
|
```sh
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-better-plan",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.5",
|
|
4
4
|
"description": "Structured execution plans with persistent progress for Pi.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"type": "module",
|
|
@@ -70,4 +70,4 @@
|
|
|
70
70
|
"engines": {
|
|
71
71
|
"node": ">=22.0.0"
|
|
72
72
|
}
|
|
73
|
-
}
|
|
73
|
+
}
|
|
@@ -4,6 +4,7 @@ import { Text } from "@earendil-works/pi-tui";
|
|
|
4
4
|
import { Type } from "typebox";
|
|
5
5
|
|
|
6
6
|
import {
|
|
7
|
+
completedPlanClearDelay,
|
|
7
8
|
planClearEntry,
|
|
8
9
|
planDisplayEntry,
|
|
9
10
|
planProgress,
|
|
@@ -42,21 +43,53 @@ export default function planExtension(pi: ExtensionAPI): void {
|
|
|
42
43
|
let currentPlan: PlanSnapshot | null = null;
|
|
43
44
|
let displayMode: PlanDisplayMode = "auto";
|
|
44
45
|
let refreshWidget: ((force?: boolean) => void) | undefined;
|
|
46
|
+
let completedPlanClearTimer: ReturnType<typeof setTimeout> | undefined;
|
|
45
47
|
|
|
46
48
|
const refresh = (force = false): void => refreshWidget?.(force);
|
|
47
49
|
|
|
50
|
+
const cancelCompletedPlanClear = (): void => {
|
|
51
|
+
if (completedPlanClearTimer) clearTimeout(completedPlanClearTimer);
|
|
52
|
+
completedPlanClearTimer = undefined;
|
|
53
|
+
};
|
|
54
|
+
|
|
55
|
+
const clearPlan = (): void => {
|
|
56
|
+
cancelCompletedPlanClear();
|
|
57
|
+
currentPlan = null;
|
|
58
|
+
pi.appendEntry(EXTENSION_NAME, planClearEntry());
|
|
59
|
+
refresh(true);
|
|
60
|
+
};
|
|
61
|
+
|
|
62
|
+
const scheduleCompletedPlanClear = (): void => {
|
|
63
|
+
cancelCompletedPlanClear();
|
|
64
|
+
if (!currentPlan) return;
|
|
65
|
+
const delay = completedPlanClearDelay(currentPlan);
|
|
66
|
+
if (delay === null) return;
|
|
67
|
+
const completedPlan = currentPlan;
|
|
68
|
+
const timer = setTimeout(() => {
|
|
69
|
+
if (completedPlanClearTimer !== timer) return;
|
|
70
|
+
completedPlanClearTimer = undefined;
|
|
71
|
+
if (currentPlan?.planId !== completedPlan.planId || currentPlan.revision !== completedPlan.revision) return;
|
|
72
|
+
clearPlan();
|
|
73
|
+
}, delay);
|
|
74
|
+
completedPlanClearTimer = timer;
|
|
75
|
+
timer.unref?.();
|
|
76
|
+
};
|
|
77
|
+
|
|
48
78
|
const restore = (ctx: ExtensionContext): void => {
|
|
79
|
+
cancelCompletedPlanClear();
|
|
49
80
|
const state = reconstructPlanState(ctx.sessionManager.getBranch());
|
|
50
81
|
currentPlan = state.plan;
|
|
51
82
|
displayMode = state.displayMode;
|
|
52
83
|
if (ctx.hasUI) ctx.ui.setStatus(LEGACY_PLAN_NAV_STATUS_KEY, undefined);
|
|
53
84
|
refresh(true);
|
|
85
|
+
scheduleCompletedPlanClear();
|
|
54
86
|
};
|
|
55
87
|
|
|
56
88
|
const persistPlan = (plan: PlanSnapshot): void => {
|
|
57
89
|
currentPlan = plan;
|
|
58
90
|
pi.appendEntry(EXTENSION_NAME, planSetEntry(plan));
|
|
59
91
|
refresh(true);
|
|
92
|
+
scheduleCompletedPlanClear();
|
|
60
93
|
};
|
|
61
94
|
|
|
62
95
|
const setDisplayMode = (mode: PlanDisplayMode): void => {
|
|
@@ -119,6 +152,8 @@ export default function planExtension(pi: ExtensionAPI): void {
|
|
|
119
152
|
promptGuidelines: [
|
|
120
153
|
"Use update_plan for work with three or more meaningful steps, and update it immediately when a step completes, becomes blocked, or scope changes.",
|
|
121
154
|
"Keep at most one update_plan step in_progress and do not mark a step completed until its required verification succeeds.",
|
|
155
|
+
"Use the plan as the foreground coordinator's milestone ledger. Delegate independent, sufficiently substantial work early when subagent or background-task tools are available, and continue unblocked foreground work instead of waiting or polling.",
|
|
156
|
+
"Represent concurrent delegated work under one in_progress coordinator step. Before completing verification or the plan, ensure every relevant delegated task is terminal, then inspect and integrate its result or failure.",
|
|
122
157
|
],
|
|
123
158
|
parameters: UpdatePlanSchema,
|
|
124
159
|
async execute(_toolCallId, params) {
|
|
@@ -171,9 +206,7 @@ export default function planExtension(pi: ExtensionAPI): void {
|
|
|
171
206
|
const input = args.trim().toLowerCase();
|
|
172
207
|
if (!input) return showFullPlan(ctx);
|
|
173
208
|
if (input === "clear") {
|
|
174
|
-
|
|
175
|
-
pi.appendEntry(EXTENSION_NAME, planClearEntry());
|
|
176
|
-
refresh(true);
|
|
209
|
+
clearPlan();
|
|
177
210
|
ctx.ui.notify("Plan cleared.", "info");
|
|
178
211
|
return;
|
|
179
212
|
}
|
|
@@ -209,6 +242,7 @@ export default function planExtension(pi: ExtensionAPI): void {
|
|
|
209
242
|
};
|
|
210
243
|
});
|
|
211
244
|
pi.on("session_shutdown", async (_event, ctx) => {
|
|
245
|
+
cancelCompletedPlanClear();
|
|
212
246
|
refreshWidget = undefined;
|
|
213
247
|
try {
|
|
214
248
|
ctx.ui.setStatus(LEGACY_PLAN_NAV_STATUS_KEY, undefined);
|
|
@@ -238,6 +272,8 @@ function planPrompt(plan: PlanSnapshot): string {
|
|
|
238
272
|
return [
|
|
239
273
|
"Current structured execution plan:",
|
|
240
274
|
...plan.steps.map((item, index) => `${index + 1}. [${item.status}] ${item.step}`),
|
|
241
|
-
"Use update_plan immediately when a step completes, becomes blocked, or the scope changes.
|
|
275
|
+
"Use update_plan immediately when a step completes, becomes blocked, or the scope changes.",
|
|
276
|
+
"Treat the plan as the foreground coordinator's milestone ledger: delegate independent, sufficiently substantial work early, keep one coordinator step in_progress across concurrent workers, and continue unblocked foreground work instead of waiting or polling.",
|
|
277
|
+
"Do not complete verification or the plan until relevant delegated work is terminal, its results or failures have been inspected and integrated, and the outcome is evidence-backed.",
|
|
242
278
|
].join("\n");
|
|
243
279
|
}
|
|
@@ -4,6 +4,7 @@ import { EXTENSION_NAME } from "./types.js";
|
|
|
4
4
|
const MAX_STEPS = 50;
|
|
5
5
|
const MAX_STEP_CHARS = 500;
|
|
6
6
|
const MAX_EXPLANATION_CHARS = 2_000;
|
|
7
|
+
export const COMPLETED_PLAN_RETENTION_MS = 30_000;
|
|
7
8
|
|
|
8
9
|
interface SessionEntryLike {
|
|
9
10
|
type: string;
|
|
@@ -54,6 +55,7 @@ export function replacePlan(
|
|
|
54
55
|
steps: readonly PlanStepInput[],
|
|
55
56
|
explanation?: string,
|
|
56
57
|
now = nowSeconds(),
|
|
58
|
+
completedAtMs = Date.now(),
|
|
57
59
|
): PlanSnapshot {
|
|
58
60
|
const error = validatePlanInput(steps, explanation);
|
|
59
61
|
if (error) throw new Error(error);
|
|
@@ -68,6 +70,8 @@ export function replacePlan(
|
|
|
68
70
|
status: item.status,
|
|
69
71
|
};
|
|
70
72
|
});
|
|
73
|
+
const isComplete = normalizedSteps.every((item) => item.status === "completed");
|
|
74
|
+
const currentIsComplete = current?.steps.every((item) => item.status === "completed") === true;
|
|
71
75
|
|
|
72
76
|
return {
|
|
73
77
|
version: 1,
|
|
@@ -77,9 +81,18 @@ export function replacePlan(
|
|
|
77
81
|
steps: normalizedSteps,
|
|
78
82
|
createdAt: current?.createdAt ?? now,
|
|
79
83
|
updatedAt: now,
|
|
84
|
+
...(isComplete
|
|
85
|
+
? { completedAtMs: currentIsComplete ? current.completedAtMs ?? current.updatedAt * 1_000 : completedAtMs }
|
|
86
|
+
: {}),
|
|
80
87
|
};
|
|
81
88
|
}
|
|
82
89
|
|
|
90
|
+
export function completedPlanClearDelay(plan: PlanSnapshot, now = Date.now()): number | null {
|
|
91
|
+
if (planProgress(plan).state !== "complete") return null;
|
|
92
|
+
const completedAtMs = plan.completedAtMs ?? plan.updatedAt * 1_000;
|
|
93
|
+
return Math.max(0, completedAtMs + COMPLETED_PLAN_RETENTION_MS - now);
|
|
94
|
+
}
|
|
95
|
+
|
|
83
96
|
export function planProgress(plan: PlanSnapshot): PlanProgress {
|
|
84
97
|
const completed = plan.steps.filter((item) => item.status === "completed").length;
|
|
85
98
|
const pending = plan.steps.filter((item) => item.status === "pending").length;
|
|
@@ -149,6 +162,7 @@ function isPlanSnapshot(value: unknown): value is PlanSnapshot {
|
|
|
149
162
|
typeof candidate.revision === "number" &&
|
|
150
163
|
Array.isArray(candidate.steps) &&
|
|
151
164
|
typeof candidate.createdAt === "number" &&
|
|
152
|
-
typeof candidate.updatedAt === "number"
|
|
165
|
+
typeof candidate.updatedAt === "number" &&
|
|
166
|
+
(candidate.completedAtMs === undefined || typeof candidate.completedAtMs === "number")
|
|
153
167
|
);
|
|
154
168
|
}
|
|
@@ -127,6 +127,11 @@ const SUBAGENT_TOOLS = [
|
|
|
127
127
|
"subagent_result",
|
|
128
128
|
];
|
|
129
129
|
|
|
130
|
+
const SUBAGENT_ORCHESTRATION_GUIDELINES = [
|
|
131
|
+
"When a structured plan is active, keep it as the parent-owned coordinator ledger: delegate bounded independent work, continue unblocked foreground work, and update the plan after integrating each result or failure.",
|
|
132
|
+
"Do not treat launching a subagent as completion of the parent milestone; relevant terminal results must be inspected and integrated before verification or completion.",
|
|
133
|
+
];
|
|
134
|
+
|
|
130
135
|
const GOAL_READY_EVENT = "pi-better-goal:ready";
|
|
131
136
|
const GOAL_REGISTER_PROVIDER_EVENT = "pi-better-goal:register-provider";
|
|
132
137
|
const goalReadySubscriptions = new WeakSet<ExtensionAPI>();
|
|
@@ -1267,7 +1272,8 @@ export default function (pi: ExtensionAPI) {
|
|
|
1267
1272
|
promptGuidelines: [
|
|
1268
1273
|
"Use subagent_spawn for independent work the user should not have to wait on. It returns at once with a run id; that return IS the deliverable — report the id to the user and continue.",
|
|
1269
1274
|
"After subagent_spawn, do NOT call subagent_output or subagent_result in a loop to wait for the result, and do NOT sleep. The run completes on its own and reports back on the next turn.",
|
|
1270
|
-
"
|
|
1275
|
+
"Call subagent_result after a completion or attention callback, or when the user explicitly asks for the result. Use subagent_output only when the user explicitly asks how a run is progressing; never use either tool to poll.",
|
|
1276
|
+
...SUBAGENT_ORCHESTRATION_GUIDELINES,
|
|
1271
1277
|
"The tools param is both the tool allowlist AND what determines which extensions load in the child (e.g. tools='read,bash,web_fetch' loads only the web-tools package). Ask for the tools the task needs and nothing more; clean:true gives a built-ins-only child. Pick a model with the model param (e.g. 'xai/grok-4.5@high'); providerless model patterns are resolved by Pi, while provider/model is deterministic and loads mapped provider extensions.",
|
|
1272
1278
|
"By default the subagent is sandboxed (writes confined to its working dir, reads and network open) and triggers completion here on finish. Set callback:false to finish quietly — then read the result on demand via subagent_result.",
|
|
1273
1279
|
"Use git_clone_workspace:true when the subagent will mutate Git in a sandbox. The parent prepares a disposable, self-contained clone with a real .git/ directory inside the sandbox root, so linked-worktree metadata outside the sandbox cannot stall the child.",
|
|
@@ -1336,6 +1342,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
1336
1342
|
"Use subagent_spawn_batch when you have several independent tasks to delegate. It returns immediately with a batch id and one run id per launched job.",
|
|
1337
1343
|
"Each job is a normal subagent run; use subagent_result / subagent_output / subagent_stop with the individual run ids just like subagent_spawn.",
|
|
1338
1344
|
"Do NOT poll for results. Each job reports back on its own when it finishes.",
|
|
1345
|
+
...SUBAGENT_ORCHESTRATION_GUIDELINES,
|
|
1339
1346
|
"By default the whole batch is rejected if there is not enough capacity. Set onCapacity to 'launch-available' to launch as many as fit and report the rest as skipped.",
|
|
1340
1347
|
],
|
|
1341
1348
|
parameters: Type.Object({
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-better-harness",
|
|
3
|
-
"version": "0.3.
|
|
3
|
+
"version": "0.3.7",
|
|
4
4
|
"description": "Pi extension bundle for a write sandbox, subagents, background tasks, SSH, goals, and structured plans.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"type": "module",
|
|
@@ -50,12 +50,12 @@
|
|
|
50
50
|
"test": "node --test test/*.test.mjs"
|
|
51
51
|
},
|
|
52
52
|
"dependencies": {
|
|
53
|
-
"pi-better-background-tasks": "0.2.
|
|
54
|
-
"pi-better-goal": "0.2.
|
|
55
|
-
"pi-better-plan": "0.1.
|
|
53
|
+
"pi-better-background-tasks": "0.2.12",
|
|
54
|
+
"pi-better-goal": "0.2.2",
|
|
55
|
+
"pi-better-plan": "0.1.5",
|
|
56
56
|
"pi-better-sandbox": "0.3.0",
|
|
57
57
|
"pi-better-ssh": "0.1.1",
|
|
58
|
-
"pi-better-subagents": "0.1.
|
|
58
|
+
"pi-better-subagents": "0.1.27"
|
|
59
59
|
},
|
|
60
60
|
"bundledDependencies": [
|
|
61
61
|
"pi-better-background-tasks",
|