pi-plans 0.5.7 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +126 -0
- package/README.md +17 -9
- package/agents/executor.md +26 -0
- package/agents/ref-analyst.md +7 -4
- package/index.ts +38 -11
- package/package.json +2 -1
- package/references/pi-planning-workflow.md +7 -4
- package/references/state-and-config.md +7 -7
- package/scripts/validate.ts +3 -2
- package/skills/debug-and-plan/SKILL.md +1 -1
- package/skills/plan-big/SKILL.md +2 -2
- package/skills/plan-normal/SKILL.md +2 -2
- package/skills/plan-small/SKILL.md +1 -1
- package/skills/plan-with-refs/SKILL.md +3 -3
- package/skills/planning/SKILL.md +1 -1
- package/src/config-command.ts +8 -3
- package/src/exec.ts +234 -3
- package/src/guard.ts +19 -5
- package/src/refine-prompts.ts +2 -2
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +1 -1
- package/src/resume-command.ts +6 -2
- package/src/resume.ts +15 -17
- package/src/run-context.ts +12 -4
- package/src/run-picker.ts +98 -0
- package/src/state.ts +110 -7
- package/src/subagent.ts +42 -1
- package/src/workflow-state.ts +18 -2
- package/tests/ask-choice-pros-cons.test.ts +147 -0
- package/tests/multi-run.test.ts +284 -0
- package/tests/resume.test.ts +10 -7
- package/tools/ask-choice.ts +20 -4
- package/tools/execute-plan.ts +93 -12
- package/tools/graph-aware-file-tools.ts +8 -0
- package/tools/plans.ts +1 -1
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared run picker (v0.6.0 multi-run support): a descriptive selector form
|
|
3
|
+
* used by /plans-abandon, /plans-execute, and /resume-plans whenever more than
|
|
4
|
+
* one candidate run exists in the workdir. Labels show topic · status · skill ·
|
|
5
|
+
* updated_at, width-fitted; the recommended run (session binding or shared
|
|
6
|
+
* active) is listed first.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import * as path from "node:path";
|
|
10
|
+
import { readdirSync } from "node:fs";
|
|
11
|
+
import { boundRunId } from "./run-context.ts";
|
|
12
|
+
import { listRuns, TERMINAL_RUN_STATUSES, type RunSummary } from "./state.ts";
|
|
13
|
+
import { truncateToWidth, visibleWidth } from "./refine-ui-helpers.ts";
|
|
14
|
+
|
|
15
|
+
export interface RunPickerContext {
|
|
16
|
+
cwd: string;
|
|
17
|
+
sessionManager: unknown;
|
|
18
|
+
ui: {
|
|
19
|
+
select: (title: string, options: string[]) => Promise<string | undefined>;
|
|
20
|
+
};
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
/** Pick-window budget: keep labels to one selector row in normal terminals. */
|
|
24
|
+
const LABEL_WIDTH_BUDGET = 88;
|
|
25
|
+
|
|
26
|
+
/** True when the run's artifact dir contains at least one PLAN_vN.md. */
|
|
27
|
+
function hasPlanFile(run: RunSummary): boolean {
|
|
28
|
+
try {
|
|
29
|
+
return readdirSync(run.artifact_dir).some((name) => /^PLAN_v\d+\.md$/i.test(name));
|
|
30
|
+
} catch {
|
|
31
|
+
return false;
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/** Non-terminal runs (planning/accepted/executing/stopped) with ≥1 plan file. */
|
|
36
|
+
export function executionCandidates(workdir: string): RunSummary[] {
|
|
37
|
+
return listRuns(workdir).filter((run) => !TERMINAL_RUN_STATUSES.has(run.status) && hasPlanFile(run));
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
/** Non-terminal runs (anything that can still be abandoned). */
|
|
41
|
+
export function abandonCandidates(workdir: string): RunSummary[] {
|
|
42
|
+
return listRuns(workdir).filter((run) => !TERMINAL_RUN_STATUSES.has(run.status));
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/** One-line descriptive label: topic · status · skill · updated_at. */
|
|
46
|
+
export function runPickerLabel(run: RunSummary, recommended: boolean): string {
|
|
47
|
+
const label = `${recommended ? "★ " : ""}${run.topic} · ${run.status} · ${run.skill} · ${run.updated_at} · ${run.run_id}`;
|
|
48
|
+
if (visibleWidth(label) <= LABEL_WIDTH_BUDGET) return label;
|
|
49
|
+
return truncateToWidth(label, LABEL_WIDTH_BUDGET, "");
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Show the descriptive run picker. Returns the chosen RunSummary, or null when
|
|
54
|
+
* cancelled / no UI. `candidates` must be pre-sorted newest-first (listRuns
|
|
55
|
+
* order); the recommended run is moved to the front.
|
|
56
|
+
*/
|
|
57
|
+
export async function pickRun(
|
|
58
|
+
ctx: RunPickerContext,
|
|
59
|
+
options: { candidates: RunSummary[]; recommendedId?: string | null; title: string },
|
|
60
|
+
): Promise<RunSummary | null> {
|
|
61
|
+
const candidates = [...options.candidates];
|
|
62
|
+
if (options.recommendedId) {
|
|
63
|
+
const index = candidates.findIndex((run) => run.run_id === options.recommendedId);
|
|
64
|
+
if (index > 0) {
|
|
65
|
+
const [recommended] = candidates.splice(index, 1);
|
|
66
|
+
candidates.unshift(recommended);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
const labels = candidates.map((run) => runPickerLabel(run, run.run_id === options.recommendedId));
|
|
70
|
+
const selected = await ctx.ui.select(options.title, labels);
|
|
71
|
+
if (selected === undefined) return null;
|
|
72
|
+
const index = labels.indexOf(selected);
|
|
73
|
+
return candidates[index] ?? null;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* Resolve the run a command should operate on, binding-first:
|
|
78
|
+
* 1. the session-bound run when it is among the candidates;
|
|
79
|
+
* 2. exactly one candidate → direct (no form — 0.5.7 parity);
|
|
80
|
+
* 3. zero candidates → null;
|
|
81
|
+
* 4. multiple candidates → the descriptive picker (recommended = bound run).
|
|
82
|
+
*/
|
|
83
|
+
export async function resolveCommandRun(
|
|
84
|
+
ctx: RunPickerContext,
|
|
85
|
+
options: { candidates: RunSummary[]; title: string },
|
|
86
|
+
): Promise<RunSummary | null> {
|
|
87
|
+
const bound = boundRunId(ctx.sessionManager, ctx.cwd);
|
|
88
|
+
const boundCandidate = bound ? options.candidates.find((run) => run.run_id === bound) ?? null : null;
|
|
89
|
+
if (boundCandidate !== null) return boundCandidate;
|
|
90
|
+
if (options.candidates.length === 0) return null;
|
|
91
|
+
if (options.candidates.length === 1) return options.candidates[0];
|
|
92
|
+
return pickRun(ctx, { ...options, recommendedId: bound });
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/** Convert a RunSummary into the ActiveInfo shape used downstream. */
|
|
96
|
+
export function activeInfoOf(run: RunSummary, stateRoot: string): { run_id: string; run_dir: string; artifact_dir: string } {
|
|
97
|
+
return { run_id: run.run_id, run_dir: path.join(stateRoot, "runs", run.run_id), artifact_dir: run.artifact_dir };
|
|
98
|
+
}
|
package/src/state.ts
CHANGED
|
@@ -51,6 +51,8 @@ export interface PlansConfig {
|
|
|
51
51
|
/** null = never asked; the plans tool surfaces a hint so the agent asks once. */
|
|
52
52
|
graph_enabled: boolean | null;
|
|
53
53
|
graph_enabled_updated_at: string | null;
|
|
54
|
+
/** Delegated executor child timeout in minutes (v0.6.0). 0/absent = default (60). */
|
|
55
|
+
executor_timeout_minutes?: number;
|
|
54
56
|
}
|
|
55
57
|
|
|
56
58
|
const DEFAULT_ARTIFACT_ROOT = "./docs/pi-plans";
|
|
@@ -137,6 +139,93 @@ export interface ActiveInfo {
|
|
|
137
139
|
artifact_dir: string;
|
|
138
140
|
}
|
|
139
141
|
|
|
142
|
+
/** Lightweight registry view of one run (RunInfo minus the heavy fields). */
|
|
143
|
+
export interface RunSummary {
|
|
144
|
+
run_id: string;
|
|
145
|
+
topic: string;
|
|
146
|
+
skill: string;
|
|
147
|
+
status: string;
|
|
148
|
+
created_at: string;
|
|
149
|
+
updated_at: string;
|
|
150
|
+
artifact_dir: string;
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
/** Terminal statuses: a run that can no longer be resumed or guarded. */
|
|
154
|
+
export const TERMINAL_RUN_STATUSES = new Set(["abandoned", "done"]);
|
|
155
|
+
|
|
156
|
+
function summarize(run: RunInfo): RunSummary {
|
|
157
|
+
return {
|
|
158
|
+
run_id: run.run_id,
|
|
159
|
+
topic: run.topic,
|
|
160
|
+
skill: run.skill,
|
|
161
|
+
status: run.status,
|
|
162
|
+
created_at: run.created_at,
|
|
163
|
+
updated_at: run.updated_at,
|
|
164
|
+
artifact_dir: run.artifact_dir,
|
|
165
|
+
};
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/**
|
|
169
|
+
* Filesystem-derived run registry (v0.6.0): scans `<stateRoot>/runs/<runId>/run.json`
|
|
170
|
+
* and returns summaries sorted by `updated_at` desc (ties go to run_id desc so the
|
|
171
|
+
* newest-created run wins within one timestamp tick). Corrupt or partial run
|
|
172
|
+
* dirs are skipped, never thrown. This replaces the racy shared `active.json`
|
|
173
|
+
* pointer: there is no new shared mutable file, and per-run files are written
|
|
174
|
+
* only by the flow that owns the run.
|
|
175
|
+
*/
|
|
176
|
+
export function listRuns(workdir: string): RunSummary[] {
|
|
177
|
+
const stateRoot = resolveStateRootOrNull(workdir);
|
|
178
|
+
if (stateRoot === null) return [];
|
|
179
|
+
const runsRoot = path.join(stateRoot, "runs");
|
|
180
|
+
let entries: string[] = [];
|
|
181
|
+
try {
|
|
182
|
+
entries = fs.readdirSync(runsRoot);
|
|
183
|
+
} catch {
|
|
184
|
+
return [];
|
|
185
|
+
}
|
|
186
|
+
const summaries: Array<RunSummary & { mtimeMs: number }> = [];
|
|
187
|
+
for (const entry of entries) {
|
|
188
|
+
const runPath = path.join(runsRoot, entry, "run.json");
|
|
189
|
+
if (!existsSync(runPath)) continue;
|
|
190
|
+
try {
|
|
191
|
+
const run = JSON.parse(readFileSync(runPath, "utf8")) as RunInfo;
|
|
192
|
+
if (typeof run?.run_id !== "string" || typeof run?.status !== "string") continue;
|
|
193
|
+
// utcNow() has second precision: same-second runs tie on updated_at, so
|
|
194
|
+
// the per-run run.json mtime (written only by the owning flow) is the
|
|
195
|
+
// race-free recency tie-break.
|
|
196
|
+
summaries.push({ ...summarize(run), mtimeMs: Number(fs.statSync(runPath, { bigint: true }).mtimeNs) / 1e6 });
|
|
197
|
+
} catch {
|
|
198
|
+
/* corrupt run.json: skip, never fail the registry scan */
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
summaries.sort((a, b) =>
|
|
202
|
+
a.updated_at === b.updated_at
|
|
203
|
+
? a.mtimeMs === b.mtimeMs
|
|
204
|
+
? (a.run_id < b.run_id ? 1 : -1)
|
|
205
|
+
: b.mtimeMs - a.mtimeMs
|
|
206
|
+
: a.updated_at < b.updated_at ? 1 : -1,
|
|
207
|
+
);
|
|
208
|
+
return summaries.map(({ mtimeMs: _mtimeMs, ...summary }) => summary);
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
/** The newest non-terminal run (planning/accepted/executing/stopped), or null. */
|
|
212
|
+
export function newestNonTerminalRun(workdir: string): RunSummary | null {
|
|
213
|
+
return listRuns(workdir).find((run) => !TERMINAL_RUN_STATUSES.has(run.status)) ?? null;
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
/** The newest run of ANY status — display-only (status widget), never attribution. */
|
|
217
|
+
export function latestRun(workdir: string): RunSummary | null {
|
|
218
|
+
return listRuns(workdir)[0] ?? null;
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
function activeInfoFromSummary(stateRoot: string, run: RunSummary): ActiveInfo {
|
|
222
|
+
return {
|
|
223
|
+
run_id: run.run_id,
|
|
224
|
+
run_dir: path.join(stateRoot, "runs", run.run_id),
|
|
225
|
+
artifact_dir: run.artifact_dir,
|
|
226
|
+
};
|
|
227
|
+
}
|
|
228
|
+
|
|
140
229
|
export interface DecisionEntry {
|
|
141
230
|
question: string;
|
|
142
231
|
options: string[];
|
|
@@ -442,7 +531,15 @@ export function startRun(workdir: string, options: StartRunOptions): StartRunRes
|
|
|
442
531
|
let artifactRoot = config.artifact_root ?? DEFAULT_ARTIFACT_ROOT;
|
|
443
532
|
if (!path.isAbsolute(artifactRoot)) artifactRoot = path.resolve(workdir, artifactRoot);
|
|
444
533
|
const dateSlug = now.slice(0, 10);
|
|
445
|
-
const
|
|
534
|
+
const baseArtifactDir = path.join(artifactRoot, `${dateSlug}-${topicSlug}`);
|
|
535
|
+
// v0.6.0: same-topic runs on the same day (concurrent sessions) must not
|
|
536
|
+
// share an artifact directory — suffix until unused, mirroring the run-id loop.
|
|
537
|
+
let artifactDir = baseArtifactDir;
|
|
538
|
+
let artifactSuffix = 2;
|
|
539
|
+
while (existsSync(artifactDir)) {
|
|
540
|
+
artifactDir = `${baseArtifactDir}-${artifactSuffix}`;
|
|
541
|
+
artifactSuffix += 1;
|
|
542
|
+
}
|
|
446
543
|
const runDir = path.join(stateRoot, "runs", runId);
|
|
447
544
|
mkdirSync(runDir, { recursive: true });
|
|
448
545
|
mkdirSync(artifactDir, { recursive: true });
|
|
@@ -464,11 +561,8 @@ export function startRun(workdir: string, options: StartRunOptions): StartRunRes
|
|
|
464
561
|
const ledger = path.join(runDir, name);
|
|
465
562
|
if (!existsSync(ledger)) writeFileSync(ledger, "", "utf8");
|
|
466
563
|
}
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
run_dir: runDir,
|
|
470
|
-
artifact_dir: artifactDir,
|
|
471
|
-
} satisfies ActiveInfo);
|
|
564
|
+
// v0.6.0: the shared active.json pointer is no longer written — the run
|
|
565
|
+
// registry is derived from runs/*/run.json (race-free across sessions).
|
|
472
566
|
if (options.onStart) {
|
|
473
567
|
try {
|
|
474
568
|
options.onStart(run);
|
|
@@ -479,10 +573,19 @@ export function startRun(workdir: string, options: StartRunOptions): StartRunRes
|
|
|
479
573
|
return { run, notices };
|
|
480
574
|
}
|
|
481
575
|
|
|
482
|
-
/**
|
|
576
|
+
/**
|
|
577
|
+
* v0.6.0 registry-backed resolution (replaces the shared `active.json`
|
|
578
|
+
* pointer): the newest NON-TERMINAL run, or null when every run is terminal.
|
|
579
|
+
* Legacy fallback: when the scan finds zero runs but a pre-0.6.0 `active.json`
|
|
580
|
+
* exists, honor it once (one-release migration shim; deprecation is surfaced
|
|
581
|
+
* on the `/plans` and `/resume-plans` command surfaces, not here — this is a
|
|
582
|
+
* read-only hot path with no notices channel).
|
|
583
|
+
*/
|
|
483
584
|
export function readActive(workdir: string): ActiveInfo | null {
|
|
484
585
|
const stateRoot = resolveStateRootOrNull(workdir);
|
|
485
586
|
if (stateRoot === null) return null;
|
|
587
|
+
const newest = newestNonTerminalRun(workdir);
|
|
588
|
+
if (newest !== null) return activeInfoFromSummary(stateRoot, newest);
|
|
486
589
|
const activePath = path.join(stateRoot, "active.json");
|
|
487
590
|
if (!existsSync(activePath)) return null;
|
|
488
591
|
try {
|
package/src/subagent.ts
CHANGED
|
@@ -40,6 +40,16 @@ export interface SubagentOptions {
|
|
|
40
40
|
timeoutMs?: number;
|
|
41
41
|
/** Optional normalized progress sink. Exceptions from the sink are ignored. */
|
|
42
42
|
onProgress?: (event: SubagentProgressEvent) => void;
|
|
43
|
+
/**
|
|
44
|
+
* Child env marker (v0.6.0): "refiner" (default — read-only reviewer/
|
|
45
|
+
* criticizer/ref-analyst children; sets PI_PLANS_REFINER=1, which the code
|
|
46
|
+
* graph gates treat as read-only), "executor" (delegated plan executor;
|
|
47
|
+
* sets PI_PLANS_EXECUTOR=1 so the write guard and the graph-aware file
|
|
48
|
+
* tools bypass staging and run natively), or "none".
|
|
49
|
+
*/
|
|
50
|
+
envMarker?: "refiner" | "executor" | "none";
|
|
51
|
+
/** Pin the run id a delegated executor child operates on (PI_PLANS_RUN_ID). */
|
|
52
|
+
runId?: string;
|
|
43
53
|
}
|
|
44
54
|
|
|
45
55
|
export interface SubagentResult {
|
|
@@ -283,6 +293,37 @@ function emitProgress(options: SubagentOptions, event: SubagentProgressEvent): v
|
|
|
283
293
|
|
|
284
294
|
const DEFAULT_TIMEOUT_MS = 60 * 60 * 1000;
|
|
285
295
|
|
|
296
|
+
/**
|
|
297
|
+
* Build the child process env for a subagent run (v0.6.0): refiner children
|
|
298
|
+
* carry PI_PLANS_REFINER=1, executor children PI_PLANS_EXECUTOR=1 plus an
|
|
299
|
+
* optional PI_PLANS_RUN_ID pin, and marker keys never leak across kinds.
|
|
300
|
+
*/
|
|
301
|
+
export function subagentChildEnv(
|
|
302
|
+
options: Pick<SubagentOptions, "envMarker" | "runId">,
|
|
303
|
+
parentEnv: NodeJS.ProcessEnv = process.env,
|
|
304
|
+
): NodeJS.ProcessEnv {
|
|
305
|
+
const childEnv: NodeJS.ProcessEnv = { ...parentEnv };
|
|
306
|
+
switch (options.envMarker ?? "refiner") {
|
|
307
|
+
case "refiner":
|
|
308
|
+
childEnv.PI_PLANS_REFINER = "1";
|
|
309
|
+
delete childEnv.PI_PLANS_EXECUTOR;
|
|
310
|
+
delete childEnv.PI_PLANS_RUN_ID;
|
|
311
|
+
break;
|
|
312
|
+
case "executor":
|
|
313
|
+
childEnv.PI_PLANS_EXECUTOR = "1";
|
|
314
|
+
delete childEnv.PI_PLANS_REFINER;
|
|
315
|
+
if (options.runId) childEnv.PI_PLANS_RUN_ID = options.runId;
|
|
316
|
+
else delete childEnv.PI_PLANS_RUN_ID;
|
|
317
|
+
break;
|
|
318
|
+
case "none":
|
|
319
|
+
delete childEnv.PI_PLANS_REFINER;
|
|
320
|
+
delete childEnv.PI_PLANS_EXECUTOR;
|
|
321
|
+
delete childEnv.PI_PLANS_RUN_ID;
|
|
322
|
+
break;
|
|
323
|
+
}
|
|
324
|
+
return childEnv;
|
|
325
|
+
}
|
|
326
|
+
|
|
286
327
|
export async function runPiSubagent(options: SubagentOptions): Promise<SubagentResult> {
|
|
287
328
|
const tools = options.tools ?? ["read", "grep", "find", "ls"];
|
|
288
329
|
let tmpDir = "";
|
|
@@ -322,7 +363,7 @@ export async function runPiSubagent(options: SubagentOptions): Promise<SubagentR
|
|
|
322
363
|
cwd: options.cwd,
|
|
323
364
|
shell: false,
|
|
324
365
|
stdio: ["ignore", "pipe", "pipe"],
|
|
325
|
-
env:
|
|
366
|
+
env: subagentChildEnv(options),
|
|
326
367
|
});
|
|
327
368
|
let buffer = "";
|
|
328
369
|
let closed = false;
|
package/src/workflow-state.ts
CHANGED
|
@@ -144,6 +144,9 @@ export interface ExecutionCheckpoint {
|
|
|
144
144
|
reverifyAll?: boolean;
|
|
145
145
|
/** True when this approval/progress was produced in a different (origin) worktree. */
|
|
146
146
|
originWorktree?: string;
|
|
147
|
+
/** v0.6.0: set while a delegated executor child owns the implementation;
|
|
148
|
+
* stale after a restart (orphaned delegate — the child died with the parent). */
|
|
149
|
+
delegate?: { modelSelector: string; startedAt: string };
|
|
147
150
|
}
|
|
148
151
|
|
|
149
152
|
export interface OwnerInfo {
|
|
@@ -451,7 +454,7 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
|
|
|
451
454
|
const record = asRecord(value, label);
|
|
452
455
|
rejectExtraKeys(
|
|
453
456
|
record,
|
|
454
|
-
new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree"]),
|
|
457
|
+
new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree", "delegate"]),
|
|
455
458
|
label,
|
|
456
459
|
);
|
|
457
460
|
const execution: ExecutionCheckpoint = {
|
|
@@ -473,6 +476,14 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
|
|
|
473
476
|
if (record.pausedReason !== undefined) execution.pausedReason = asString(record.pausedReason, `${label}.pausedReason`);
|
|
474
477
|
if (record.reverifyAll !== undefined) execution.reverifyAll = asBool(record.reverifyAll, `${label}.reverifyAll`);
|
|
475
478
|
if (record.originWorktree !== undefined) execution.originWorktree = asString(record.originWorktree, `${label}.originWorktree`);
|
|
479
|
+
if (record.delegate !== undefined && record.delegate !== null) {
|
|
480
|
+
const delegate = asRecord(record.delegate, `${label}.delegate`);
|
|
481
|
+
rejectExtraKeys(delegate, new Set(["modelSelector", "startedAt"]), `${label}.delegate`);
|
|
482
|
+
execution.delegate = {
|
|
483
|
+
modelSelector: asString(delegate.modelSelector, `${label}.delegate.modelSelector`),
|
|
484
|
+
startedAt: asString(delegate.startedAt, `${label}.delegate.startedAt`),
|
|
485
|
+
};
|
|
486
|
+
}
|
|
476
487
|
return execution;
|
|
477
488
|
}
|
|
478
489
|
|
|
@@ -1035,6 +1046,8 @@ export interface ExecutionProgressInput {
|
|
|
1035
1046
|
currentI?: string;
|
|
1036
1047
|
usage?: { inToks: number; outToks: number };
|
|
1037
1048
|
pausedReason?: string | null;
|
|
1049
|
+
/** v0.6.0: set/clear the delegated-executor record; null clears it. */
|
|
1050
|
+
delegate?: { modelSelector: string; startedAt: string } | null;
|
|
1038
1051
|
}
|
|
1039
1052
|
|
|
1040
1053
|
export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: ExecutionProgressInput): WorkflowCheckpoint {
|
|
@@ -1051,6 +1064,8 @@ export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: Executi
|
|
|
1051
1064
|
}
|
|
1052
1065
|
if (progress.pausedReason === null) delete execution.pausedReason;
|
|
1053
1066
|
else if (progress.pausedReason !== undefined) execution.pausedReason = progress.pausedReason;
|
|
1067
|
+
if (progress.delegate === null) delete execution.delegate;
|
|
1068
|
+
else if (progress.delegate !== undefined) execution.delegate = progress.delegate;
|
|
1054
1069
|
return { ...cp, execution };
|
|
1055
1070
|
}
|
|
1056
1071
|
|
|
@@ -1148,7 +1163,8 @@ function migrationNextAction(cp: WorkflowCheckpoint): NextAction {
|
|
|
1148
1163
|
/** Mark a paused stop without erasing the last phase (D-008). */
|
|
1149
1164
|
export function applyExecutionStopped(cp: WorkflowCheckpoint, reason: string): WorkflowCheckpoint {
|
|
1150
1165
|
if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
|
|
1151
|
-
|
|
1166
|
+
const { delegate: _delegate, ...execution } = cp.execution;
|
|
1167
|
+
return { ...cp, execution: { ...execution, pausedReason: reason } };
|
|
1152
1168
|
}
|
|
1153
1169
|
|
|
1154
1170
|
// ---------------------------------------------------------------------------
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Feature 5 (v0.6.0): every AI-written ask_choice option must state its
|
|
3
|
+
* advantage AND its drawback as '✓ <advantage> / ✗ <drawback>' in the
|
|
4
|
+
* configured language.
|
|
5
|
+
*
|
|
6
|
+
* The rule is SOFT GUIDANCE (deliberately — no runtime validation, no new
|
|
7
|
+
* schema fields), so these tests pin the *contract text* the model actually
|
|
8
|
+
* reads: the Option schema, the tool description, the promptGuidelines, the
|
|
9
|
+
* batch/single `options` arrays, the six skills, and the normative reference
|
|
10
|
+
* doc. They also prove the convention is actually RENDERABLE by the existing
|
|
11
|
+
* form renderer, so guidance can never drift away from what users see.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
import * as assert from "node:assert/strict";
|
|
15
|
+
import * as fs from "node:fs";
|
|
16
|
+
import * as path from "node:path";
|
|
17
|
+
import { fileURLToPath } from "node:url";
|
|
18
|
+
import { describe, it } from "node:test";
|
|
19
|
+
import { Value } from "typebox/value";
|
|
20
|
+
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
21
|
+
import { Option, AskChoiceParams, BatchQuestionParams, registerAskChoiceTool } from "../tools/ask-choice.ts";
|
|
22
|
+
import { createFormState, formRender, type FormQuestion } from "../src/ask-form.ts";
|
|
23
|
+
|
|
24
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..");
|
|
25
|
+
|
|
26
|
+
const PRO = "✓";
|
|
27
|
+
const CON = "✗";
|
|
28
|
+
|
|
29
|
+
interface ToolDef {
|
|
30
|
+
name: string;
|
|
31
|
+
description: string;
|
|
32
|
+
promptSnippet: string;
|
|
33
|
+
promptGuidelines: string[];
|
|
34
|
+
parameters: unknown;
|
|
35
|
+
execute: (id: string, params: unknown, signal: undefined, update: undefined, ctx: unknown) => Promise<unknown>;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
function loadTool(): ToolDef {
|
|
39
|
+
let tool: ToolDef | undefined;
|
|
40
|
+
const pi = { registerTool: (definition: ToolDef) => { tool = definition; } } as unknown as ExtensionAPI;
|
|
41
|
+
registerAskChoiceTool(pi);
|
|
42
|
+
if (!tool) throw new Error("ask_choice tool not registered");
|
|
43
|
+
return tool;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/** Read a TypeBox property description out of a schema object. */
|
|
47
|
+
function propertyDescription(schema: unknown, property: string): string {
|
|
48
|
+
const properties = (schema as { properties?: Record<string, { description?: string }> }).properties;
|
|
49
|
+
const found = properties?.[property]?.description;
|
|
50
|
+
assert.equal(typeof found, "string", `expected a description on property ${property}`);
|
|
51
|
+
return found!;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
const SKILL_DIRS = ["planning", "debug-and-plan", "plan-small", "plan-normal", "plan-big", "plan-with-refs"];
|
|
55
|
+
|
|
56
|
+
describe("ask_choice pros/cons contract (v0.6.0 feature 5)", () => {
|
|
57
|
+
it("Option.description requires both halves via the ✓ / ✗ markers", () => {
|
|
58
|
+
const text = propertyDescription(Option, "description");
|
|
59
|
+
assert.ok(text.includes(PRO), "description must show the advantage marker");
|
|
60
|
+
assert.ok(text.includes(CON), "description must show the drawback marker");
|
|
61
|
+
assert.ok(/drawback/i.test(text), "description must name the drawback half");
|
|
62
|
+
assert.ok(/advantage/i.test(text), "description must name the advantage half");
|
|
63
|
+
assert.ok(/configured language/i.test(text), "description must pin the configured language");
|
|
64
|
+
});
|
|
65
|
+
|
|
66
|
+
it("the registered tool description states the rule and exempts the tool-appended tails", () => {
|
|
67
|
+
const tool = loadTool();
|
|
68
|
+
assert.ok(tool.description.includes(PRO) && tool.description.includes(CON));
|
|
69
|
+
assert.ok(/advantage/i.test(tool.description) && /drawback/i.test(tool.description));
|
|
70
|
+
assert.ok(/Other/.test(tool.description) && /Auto-complete/.test(tool.description));
|
|
71
|
+
});
|
|
72
|
+
|
|
73
|
+
it("promptGuidelines carry the rule as an imperative, not as new schema fields", () => {
|
|
74
|
+
const tool = loadTool();
|
|
75
|
+
assert.ok(Array.isArray(tool.promptGuidelines));
|
|
76
|
+
const joined = tool.promptGuidelines.join("\n");
|
|
77
|
+
assert.ok(joined.includes(PRO) && joined.includes(CON), "a guideline must carry the markers");
|
|
78
|
+
assert.ok(/drawback/i.test(joined), "a guideline must name the drawback half");
|
|
79
|
+
// The failure mode this guards: a model inventing `pros`/`cons` keys,
|
|
80
|
+
// which Option rejects (additionalProperties: false).
|
|
81
|
+
assert.ok(
|
|
82
|
+
/no separate pros\/cons fields|there are no separate pros\/cons fields/i.test(joined),
|
|
83
|
+
"guidelines must forbid inventing pros/cons fields",
|
|
84
|
+
);
|
|
85
|
+
});
|
|
86
|
+
|
|
87
|
+
it("both the batch and single-question options arrays propagate the rule", () => {
|
|
88
|
+
const batch = propertyDescription(BatchQuestionParams, "options");
|
|
89
|
+
assert.ok(batch.includes(PRO) && batch.includes(CON));
|
|
90
|
+
assert.ok(/configured language/i.test(batch));
|
|
91
|
+
|
|
92
|
+
const single = propertyDescription(AskChoiceParams, "options");
|
|
93
|
+
assert.ok(single.includes(PRO) && single.includes(CON));
|
|
94
|
+
assert.ok(/configured language/i.test(single));
|
|
95
|
+
});
|
|
96
|
+
|
|
97
|
+
it("every skill states the per-option pros/drawbacks rule", () => {
|
|
98
|
+
for (const dir of SKILL_DIRS) {
|
|
99
|
+
const file = path.join(ROOT, "skills", dir, "SKILL.md");
|
|
100
|
+
const text = fs.readFileSync(file, "utf8");
|
|
101
|
+
assert.ok(text.includes(PRO), `${dir}: must show the advantage marker`);
|
|
102
|
+
assert.ok(text.includes(CON), `${dir}: must show the drawback marker`);
|
|
103
|
+
}
|
|
104
|
+
});
|
|
105
|
+
|
|
106
|
+
it("the normative workflow doc states the rule at the options clause", () => {
|
|
107
|
+
const text = fs.readFileSync(path.join(ROOT, "references", "pi-planning-workflow.md"), "utf8");
|
|
108
|
+
const optionsBullet = text.split("\n").find((line) => line.startsWith("- `options`:"));
|
|
109
|
+
assert.ok(optionsBullet, "the options clause must exist");
|
|
110
|
+
assert.ok(optionsBullet!.includes(PRO) && optionsBullet!.includes(CON));
|
|
111
|
+
assert.ok(/configured language/i.test(optionsBullet!));
|
|
112
|
+
assert.ok(/accept\/execute/.test(optionsBullet!), "the handoff must be in scope");
|
|
113
|
+
});
|
|
114
|
+
|
|
115
|
+
it("the convention is renderable by the existing form renderer", () => {
|
|
116
|
+
const question: FormQuestion = {
|
|
117
|
+
question: "Which storage approach?",
|
|
118
|
+
options: [
|
|
119
|
+
{ label: "Reuse the existing table", description: `${PRO} no migration / ${CON} needs a backfill`, recommended: true },
|
|
120
|
+
{ label: "Add a new table", description: `${PRO} clean isolation / ${CON} doubles write cost` },
|
|
121
|
+
],
|
|
122
|
+
allowOther: true,
|
|
123
|
+
questionId: "q1",
|
|
124
|
+
autoComplete: true,
|
|
125
|
+
};
|
|
126
|
+
const rows = formRender(createFormState([question]), 100);
|
|
127
|
+
const rendered = rows.join("\n");
|
|
128
|
+
assert.ok(rendered.includes(PRO), "the advantage half must reach the user");
|
|
129
|
+
assert.ok(rendered.includes(CON), "the drawback half must reach the user");
|
|
130
|
+
assert.ok(rendered.includes("no migration"), "the advantage text must be shown");
|
|
131
|
+
assert.ok(rendered.includes("backfill"), "the drawback text must be shown");
|
|
132
|
+
});
|
|
133
|
+
|
|
134
|
+
it("a convention-following option still validates against the Option schema", () => {
|
|
135
|
+
// No new keys: the whole point of riding `description` is that the
|
|
136
|
+
// existing schema accepts it unchanged.
|
|
137
|
+
const ok = Value.Check(Option, {
|
|
138
|
+
label: "Reuse the existing table",
|
|
139
|
+
description: `${PRO} no migration / ${CON} needs a backfill`,
|
|
140
|
+
recommended: true,
|
|
141
|
+
});
|
|
142
|
+
assert.equal(ok, true);
|
|
143
|
+
// And stray pros/cons keys are still rejected loudly.
|
|
144
|
+
const stray = Value.Check(Option, { label: "x", pros: "a", cons: "b" });
|
|
145
|
+
assert.equal(stray, false);
|
|
146
|
+
});
|
|
147
|
+
});
|