pi-plans 0.5.7 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +126 -0
- package/README.md +17 -9
- package/agents/executor.md +26 -0
- package/agents/ref-analyst.md +7 -4
- package/index.ts +38 -11
- package/package.json +2 -1
- package/references/pi-planning-workflow.md +7 -4
- package/references/state-and-config.md +7 -7
- package/scripts/validate.ts +3 -2
- package/skills/debug-and-plan/SKILL.md +1 -1
- package/skills/plan-big/SKILL.md +2 -2
- package/skills/plan-normal/SKILL.md +2 -2
- package/skills/plan-small/SKILL.md +1 -1
- package/skills/plan-with-refs/SKILL.md +3 -3
- package/skills/planning/SKILL.md +1 -1
- package/src/config-command.ts +8 -3
- package/src/exec.ts +234 -3
- package/src/guard.ts +19 -5
- package/src/refine-prompts.ts +2 -2
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +1 -1
- package/src/resume-command.ts +6 -2
- package/src/resume.ts +15 -17
- package/src/run-context.ts +12 -4
- package/src/run-picker.ts +98 -0
- package/src/state.ts +110 -7
- package/src/subagent.ts +42 -1
- package/src/workflow-state.ts +18 -2
- package/tests/ask-choice-pros-cons.test.ts +147 -0
- package/tests/multi-run.test.ts +284 -0
- package/tests/resume.test.ts +10 -7
- package/tools/ask-choice.ts +20 -4
- package/tools/execute-plan.ts +93 -12
- package/tools/graph-aware-file-tools.ts +8 -0
- package/tools/plans.ts +1 -1
|
@@ -16,10 +16,10 @@ Read `../../references/pi-planning-workflow.md` and `../../references/state-and-
|
|
|
16
16
|
1. Inspect the target Git repo read-only before external research so search terms match the actual codebase and constraints.
|
|
17
17
|
2. Create the `.git/pi_plans` run state and planning artifact directory once the topic is clear (`plans` action `start-run`).
|
|
18
18
|
3. Search proactively for related projects, articles, papers, docs, and prior art. Prefer a websearch skill when installed; otherwise use bash tools such as `curl` or `gh` when already available.
|
|
19
|
-
4. Before the first download, check `refs_root` in `.git/pi_plans/config.json` (`plans` action `show`). If it is unset, ask exactly one `ask_choice` question — recommended `.git/pi-plans/refs/` (inside the git dir, never tracked), second `./refs/`, third `~/.cache/pi-plans/refs/` — and persist with `plans` (`set-refs-root`); this question does not count against the planning-question limit. Then download
|
|
20
|
-
5. For every reference, record source metadata and local path in `REF_ANALYSIS.md` and in the run's `refs.jsonl` (via `plans` action `record-ref`): title, URL, kind, retrieval method, date accessed, local path, coverage, and evidence gaps.
|
|
19
|
+
4. Before the first download, check `refs_root` in `.git/pi_plans/config.json` (`plans` action `show`). If it is unset, ask exactly one `ask_choice` question — recommended `.git/pi-plans/refs/` (inside the git dir, never tracked), second `./refs/`, third `~/.cache/pi-plans/refs/` — with each option's `description` set to `✓ <advantage> / ✗ <drawback>` in the configured language, kept terse, and persist with `plans` (`set-refs-root`); this question does not count against the planning-question limit. Then download at least 3 credible references across at least 2 distinct origins before writing `PLAN_v1.md`, under the configured refs root (one subdirectory per reference — `analyze_refs` requires directories). References are NOT limited to GitHub repositories: papers (e.g. arXiv), engineering blog posts, and documentation sites are first-class, and theoretical references count exactly as much as implementation references. A download is qualified per medium: repo = clone (full working tree); paper = the full text (arXiv HTML preferred, else the PDF with extracted text into `paper.txt`/`paper.md`; an abstract alone never qualifies); blog/docs site = the full-article readable markdown saved locally (single-page posts are fine — they ARE the source; only fragments or teasers fail). Landing pages, README-only snapshots, abstracts, package metadata, or curl-only fragments do not count when deeper source material is available.
|
|
20
|
+
5. For every reference, record source metadata and local path in `REF_ANALYSIS.md` and in the run's `refs.jsonl` (via `plans` action `record-ref`): title, URL, kind (`project` for repos, `paper` for papers, `article` for blog posts, `docs` for documentation sites), retrieval method, date accessed, local path, coverage, and evidence gaps.
|
|
21
21
|
6. For every reference, run the `analyze_refs` tool (required path — it replaces manual structured reads): one independent read-only subagent per reference deep-reads it and returns structured sections (Overview / Key Mechanisms And Design Tradeoffs / Adoptable Ideas For The Target Repo / Pitfalls And Anti-Patterns / Evidence Citations / Coverage / Evidence Gaps). Paste each analysis into `REF_ANALYSIS.md` and fill `coverage` and `gaps` in `refs.jsonl` via `plans` (`record-ref`) before asking adoption questions.
|
|
22
|
-
7. For every reference after analysis, ask at least 3 ref-specific adoption questions via `ask_choice` before using its ideas in `PLAN_v1.md`; each based on downloaded content, recommended option first, `Other` second-last, `Auto-complete` last (the tool appends both). Batch one reference's adoption questions into a single `questions: [...]` form call.
|
|
22
|
+
7. For every reference after analysis, ask at least 3 ref-specific adoption questions via `ask_choice` before using its ideas in `PLAN_v1.md`; each based on downloaded content, recommended option first, `Other` second-last, `Auto-complete` last (the tool appends both). Every option you write carries a `description` of `✓ <advantage> / ✗ <drawback>` in the configured language, kept terse — adoption choices trade a real gain against a real cost, so state both. Batch one reference's adoption questions into a single `questions: [...]` form call.
|
|
23
23
|
8. Block rather than pad if fewer than 3 credible references exist, unless the user explicitly narrows the topic or waives the minimum. `Auto-complete` cannot grant this waiver.
|
|
24
24
|
9. Continue with big-plan depth: at least 10 planning questions, required web research during brainstorming and refinement (`refine` `reviewers: 3` reviewer round, then a criticizer round), no refinement limit, at most five high-priority comments or questions per refinement round. Then the merged accept/execute question (ask_choice with `autoComplete: false`: ✓ Accept & execute now / Accept, don't execute yet / another round) and the `execute_plan` tool.
|
|
25
25
|
|
package/skills/planning/SKILL.md
CHANGED
|
@@ -19,4 +19,4 @@ Use this skill when a task is planning-related but the right specialist is not o
|
|
|
19
19
|
|
|
20
20
|
## Pi Setup
|
|
21
21
|
|
|
22
|
-
Use the same language, `ask_choice`, `refine`, reviewer, criticizer, Auto-complete, and `.git/pi_plans` rules as the selected specialist skill and the shared workflow. Questions prefer the 0.4.0 batch form: one `ask_choice` call with `questions: [...]` (2-8) opens a tabbed multiple-choice form; scope confirmation and execution handoff stay single-question with `autoComplete: false`.
|
|
22
|
+
Use the same language, `ask_choice`, `refine`, reviewer, criticizer, Auto-complete, and `.git/pi_plans` rules as the selected specialist skill and the shared workflow — including the per-option `✓ <advantage> / ✗ <drawback>` descriptions, which every option you write carries in the configured language, kept terse. Questions prefer the 0.4.0 batch form: one `ask_choice` call with `questions: [...]` (2-8) opens a tabbed multiple-choice form; scope confirmation and execution handoff stay single-question with `autoComplete: false`.
|
package/src/config-command.ts
CHANGED
|
@@ -7,7 +7,7 @@ interface ModelLike {
|
|
|
7
7
|
id?: unknown;
|
|
8
8
|
}
|
|
9
9
|
|
|
10
|
-
interface ConfigCommandContext {
|
|
10
|
+
export interface ConfigCommandContext {
|
|
11
11
|
cwd: string;
|
|
12
12
|
hasUI: boolean;
|
|
13
13
|
model?: unknown;
|
|
@@ -34,7 +34,7 @@ function cancelled<T>(reason: "user" | "invalid" = "user"): ChoiceResult<T> {
|
|
|
34
34
|
return { cancelled: true, reason };
|
|
35
35
|
}
|
|
36
36
|
|
|
37
|
-
function modelSelectorOf(value: unknown): string | null {
|
|
37
|
+
export function modelSelectorOf(value: unknown): string | null {
|
|
38
38
|
if (!value || typeof value !== "object") return null;
|
|
39
39
|
const model = value as ModelLike;
|
|
40
40
|
if (typeof model.provider !== "string" || typeof model.id !== "string") return null;
|
|
@@ -42,7 +42,12 @@ function modelSelectorOf(value: unknown): string | null {
|
|
|
42
42
|
return `${model.provider}/${model.id}`;
|
|
43
43
|
}
|
|
44
44
|
|
|
45
|
-
|
|
45
|
+
/**
|
|
46
|
+
* Merge the session-visible model selectors (ctx.model + ctx.scopedModels +
|
|
47
|
+
* the model registry), deduped, EXCLUDING `currentSelector` — the result is a
|
|
48
|
+
* switch-target list (v0.6.0 delegated-execution model picker reuses this).
|
|
49
|
+
*/
|
|
50
|
+
export function collectModelSelectors(ctx: ConfigCommandContext, currentSelector: string | null): string[] {
|
|
46
51
|
const selectors: string[] = [];
|
|
47
52
|
const push = (selector: string | null): void => {
|
|
48
53
|
if (!selector) return;
|
package/src/exec.ts
CHANGED
|
@@ -36,9 +36,11 @@ import {
|
|
|
36
36
|
type VccCompactionBuildResult,
|
|
37
37
|
type VccCompactionStats,
|
|
38
38
|
} from "./compaction.ts";
|
|
39
|
-
import { getRun, lintPlanIntoNotices, readActive, resolveStateRootOrNull, setRunStatus, StateError, utcNow } from "./state.ts";
|
|
39
|
+
import { TERMINAL_RUN_STATUSES, getRun, latestRun, lintPlanIntoNotices, loadConfig, readActive, resolveStateRootOrNull, setRunStatus, StateError, utcNow } from "./state.ts";
|
|
40
40
|
import { execChrome, resolveUiLanguage, type UiLanguage } from "./ui-language.ts";
|
|
41
41
|
import { bindRun, resolveActiveRun } from "./run-context.ts";
|
|
42
|
+
import { runPiSubagent, type SubagentProgressEvent } from "./subagent.ts";
|
|
43
|
+
import { RefineOverlayController, refineOverlayContext } from "./refine-ui.ts";
|
|
42
44
|
import { OwnershipError } from "./run-ownership.ts";
|
|
43
45
|
import {
|
|
44
46
|
applyExecutionApproved,
|
|
@@ -110,6 +112,10 @@ export interface ExecState {
|
|
|
110
112
|
uiLanguage?: UiLanguage;
|
|
111
113
|
currentI?: string;
|
|
112
114
|
goalWait?: GoalWaitState;
|
|
115
|
+
/** v0.6.0: set while a delegated executor child owns the implementation.
|
|
116
|
+
* Persisted subset only (modelSelector + startedAt); the AbortController is
|
|
117
|
+
* runtime state kept in delegatedRuntime, never persisted. */
|
|
118
|
+
delegate?: { modelSelector: string; startedAt: string };
|
|
113
119
|
}
|
|
114
120
|
|
|
115
121
|
/**
|
|
@@ -283,6 +289,19 @@ export function loadExecutionFromCheckpoint(
|
|
|
283
289
|
resetGoalWaitRuntime(ctx);
|
|
284
290
|
pendingExecutionFlush = false; // restored state: no inherited flush debt
|
|
285
291
|
resetExecutionCompactionState(ctx);
|
|
292
|
+
// v0.6.0 orphaned-delegate detection: a restart never carries a live child.
|
|
293
|
+
// If the checkpoint/session carries delegate state, surface it so the user
|
|
294
|
+
// knows the previous executor died mid-run (VC state is intact, resumable).
|
|
295
|
+
if (cp.execution.delegate) {
|
|
296
|
+
try {
|
|
297
|
+
ctx.ui.notify?.(
|
|
298
|
+
`pi-plans: the previous delegated executor (${cp.execution.delegate.modelSelector}) did not finish before this session ended. Verified VC state is preserved; resume with /plans-execute.`,
|
|
299
|
+
"warning",
|
|
300
|
+
);
|
|
301
|
+
} catch {
|
|
302
|
+
/* notification is best-effort */
|
|
303
|
+
}
|
|
304
|
+
}
|
|
286
305
|
if (headChanged) {
|
|
287
306
|
withExecutionCheckpoint(ctx, (current) => applyExecutionHeadChanged(current));
|
|
288
307
|
}
|
|
@@ -463,7 +482,13 @@ let loopPanelRegistered = false;
|
|
|
463
482
|
function implReviewLoopState(
|
|
464
483
|
ctx: ExtensionContext,
|
|
465
484
|
): { topic: string; review: { terminationCondition?: string; reviewerCount?: number; completedRounds: number } } | null {
|
|
466
|
-
|
|
485
|
+
// v0.6.0: resolveActiveRun only returns NON-TERMINAL runs, but this state
|
|
486
|
+
// is by definition attached to a DONE run — fall back to the newest run of
|
|
487
|
+
// any status (display-only) when the active resolution is null.
|
|
488
|
+
const active = resolveActiveRun(ctx.sessionManager, ctx.cwd) ?? (() => {
|
|
489
|
+
const latest = latestRun(ctx.cwd);
|
|
490
|
+
return latest === null ? null : { run_id: latest.run_id, artifact_dir: latest.artifact_dir };
|
|
491
|
+
})();
|
|
467
492
|
if (!active) return null;
|
|
468
493
|
if (getRun(ctx.cwd, active.run_id)?.status !== "done") return null;
|
|
469
494
|
const load = loadCheckpoint(ctx.cwd, active.run_id);
|
|
@@ -564,7 +589,10 @@ export function updateStatusWidget(ctx: ExtensionContext): void {
|
|
|
564
589
|
if (active) {
|
|
565
590
|
// Idle indicator depends on the run's lifecycle, not just its existence:
|
|
566
591
|
// done reads as finished, abandoned as closed, stopped/accepted as paused.
|
|
567
|
-
|
|
592
|
+
// v0.6.0: resolveActiveRun only returns NON-TERMINAL runs; for the pure
|
|
593
|
+
// display line below, fall back to the newest run of any status so a
|
|
594
|
+
// finished workdir still shows its last run's outcome.
|
|
595
|
+
const status = getRun(ctx.cwd, active.run_id)?.status ?? latestRun(ctx.cwd)?.status;
|
|
568
596
|
if (status === "done") {
|
|
569
597
|
// D-3/D-015: while the implementation-review loop is live (checkpoint
|
|
570
598
|
// still in the implementation-review phase), the status line mirrors
|
|
@@ -599,6 +627,24 @@ export function updateStatusWidget(ctx: ExtensionContext): void {
|
|
|
599
627
|
}
|
|
600
628
|
// unknown status: no indicator.
|
|
601
629
|
}
|
|
630
|
+
// Terminal-only workdir: resolveActiveRun is null, but the display line
|
|
631
|
+
// still reports the newest run's outcome (done/abandoned, plus its live
|
|
632
|
+
// implementation-review loop).
|
|
633
|
+
const terminalLatest = latestRun(ctx.cwd);
|
|
634
|
+
if (terminalLatest && TERMINAL_RUN_STATUSES.has(terminalLatest.status)) {
|
|
635
|
+
if (terminalLatest.status === "done") {
|
|
636
|
+
const loop = implReviewLoopState(ctx);
|
|
637
|
+
if (loop) {
|
|
638
|
+
const line = formatImplReviewLoopSummaryLine(deriveImplReviewLoopModel(loop.topic, loop.review));
|
|
639
|
+
ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", line));
|
|
640
|
+
return;
|
|
641
|
+
}
|
|
642
|
+
ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("success", `🎯 plans: ${terminalLatest.run_id} (done)`));
|
|
643
|
+
} else {
|
|
644
|
+
ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("error", `🚫 plans: ${terminalLatest.run_id}`));
|
|
645
|
+
}
|
|
646
|
+
return;
|
|
647
|
+
}
|
|
602
648
|
ctx.ui.setStatus("pi-plans", undefined);
|
|
603
649
|
}
|
|
604
650
|
|
|
@@ -614,6 +660,7 @@ function persist(pi: ExtensionAPI): void {
|
|
|
614
660
|
implWarning: execution.implWarning ?? null,
|
|
615
661
|
currentI: execution.currentI,
|
|
616
662
|
goalWait: execution.goalWait,
|
|
663
|
+
delegate: execution.delegate ?? null,
|
|
617
664
|
});
|
|
618
665
|
}
|
|
619
666
|
|
|
@@ -641,6 +688,7 @@ export async function startExecution(
|
|
|
641
688
|
planPath: string,
|
|
642
689
|
items: CheckItem[],
|
|
643
690
|
implItems?: ImplItem[],
|
|
691
|
+
opts?: StartExecutionOptions,
|
|
644
692
|
): Promise<void> {
|
|
645
693
|
execution = {
|
|
646
694
|
planPath,
|
|
@@ -721,6 +769,189 @@ export async function startExecution(
|
|
|
721
769
|
},
|
|
722
770
|
{ triggerTurn: false },
|
|
723
771
|
);
|
|
772
|
+
// Delegated runtime (v0.6.0): one executor child implements the whole plan
|
|
773
|
+
// while this tool call blocks; the parent mirrors VC progress from the
|
|
774
|
+
// child's streamed assistant messages (R-9..R-12).
|
|
775
|
+
updateStatusWidget(ctx);
|
|
776
|
+
if (opts?.runtime && opts.runtime !== "current-session") {
|
|
777
|
+
await runDelegatedExecution(pi, ctx, opts.runtime.modelSelector, opts.signal);
|
|
778
|
+
}
|
|
779
|
+
}
|
|
780
|
+
|
|
781
|
+
/** Where a chosen execution runs: this session, or a delegated executor child. */
|
|
782
|
+
export type ExecutionRuntime = "current-session" | { modelSelector: string };
|
|
783
|
+
|
|
784
|
+
export interface StartExecutionOptions {
|
|
785
|
+
/** "current-session" (default) or a delegated executor model selector. */
|
|
786
|
+
runtime?: ExecutionRuntime;
|
|
787
|
+
/** Tool-call abort signal, threaded into the delegated child. */
|
|
788
|
+
signal?: AbortSignal;
|
|
789
|
+
}
|
|
790
|
+
|
|
791
|
+
/** Default delegated executor timeout when config omits executor_timeout_minutes. */
|
|
792
|
+
const DELEGATE_DEFAULT_TIMEOUT_MINUTES = 60;
|
|
793
|
+
void DELEGATE_DEFAULT_TIMEOUT_MINUTES;
|
|
794
|
+
|
|
795
|
+
/** Live AbortController for the delegated executor child (runtime-only state). */
|
|
796
|
+
let delegatedRuntime: AbortController | null = null;
|
|
797
|
+
|
|
798
|
+
/** Abort the delegated executor child, if one is running (used by /plans-stop). */
|
|
799
|
+
export function abortDelegatedExecutor(): boolean {
|
|
800
|
+
if (delegatedRuntime === null) return false;
|
|
801
|
+
delegatedRuntime.abort();
|
|
802
|
+
return true;
|
|
803
|
+
}
|
|
804
|
+
|
|
805
|
+
function executorAgentPrompt(): string {
|
|
806
|
+
try {
|
|
807
|
+
const agentPath = new URL("../agents/executor.md", import.meta.url);
|
|
808
|
+
return fs.readFileSync(agentPath, "utf8");
|
|
809
|
+
} catch {
|
|
810
|
+
return "You are a delegated plan executor in the pi-plans workflow. Implement the accepted plan autonomously and emit [DONE:VC-xxx] markers in your replies as verifier items pass.";
|
|
811
|
+
}
|
|
812
|
+
}
|
|
813
|
+
|
|
814
|
+
function delegatedExecutorTimeoutMs(): number | undefined {
|
|
815
|
+
try {
|
|
816
|
+
const stateRoot = resolveStateRootOrNull(process.cwd());
|
|
817
|
+
if (stateRoot === null) return undefined;
|
|
818
|
+
const minutes = loadConfig(stateRoot).executor_timeout_minutes;
|
|
819
|
+
if (typeof minutes !== "number" || minutes <= 0) return undefined;
|
|
820
|
+
return minutes * 60 * 1000;
|
|
821
|
+
} catch {
|
|
822
|
+
return undefined;
|
|
823
|
+
}
|
|
824
|
+
}
|
|
825
|
+
|
|
826
|
+
/**
|
|
827
|
+
* Delegated execution (R-9..R-12): spawn ONE executor child for the whole
|
|
828
|
+
* plan (write-capable tools, chosen model, PI_PLANS_EXECUTOR=1 + pinned run
|
|
829
|
+
* id), stream its progress into the overlay, parse full-text assistant
|
|
830
|
+
* messages for [DONE:VC-xxx]/[I-xxx] markers, and on exit verify the
|
|
831
|
+
* remaining items. Abort/timeout → stopExecution (stopped, resumable);
|
|
832
|
+
* clean exit with items left → stay executing (resumable under either
|
|
833
|
+
* runtime); clean exit complete → normal completion flow.
|
|
834
|
+
*/
|
|
835
|
+
async function runDelegatedExecution(
|
|
836
|
+
pi: ExtensionAPI,
|
|
837
|
+
ctx: ExtensionContext,
|
|
838
|
+
modelSelector: string,
|
|
839
|
+
parentSignal: AbortSignal | undefined,
|
|
840
|
+
): Promise<void> {
|
|
841
|
+
if (!execution) return;
|
|
842
|
+
const controller = new AbortController();
|
|
843
|
+
delegatedRuntime = controller;
|
|
844
|
+
const relayAbort = () => controller.abort();
|
|
845
|
+
if (parentSignal?.aborted) controller.abort();
|
|
846
|
+
else parentSignal?.addEventListener("abort", relayAbort, { once: true });
|
|
847
|
+
const lang = resolveUiLanguage(ctx.cwd);
|
|
848
|
+
const overlay = ctx.mode === "tui" ? new RefineOverlayController("executor", [{ id: "executor" }], relayAbort, lang) : undefined;
|
|
849
|
+
if (overlay) {
|
|
850
|
+
try {
|
|
851
|
+
overlay.open(refineOverlayContext(ctx), modelSelector);
|
|
852
|
+
} catch {
|
|
853
|
+
/* overlay is best-effort; the blocking call itself must not fail */
|
|
854
|
+
}
|
|
855
|
+
}
|
|
856
|
+
execution.delegate = { modelSelector, startedAt: utcNow() };
|
|
857
|
+
withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { delegate: execution?.delegate ?? null }));
|
|
858
|
+
persist(pi);
|
|
859
|
+
const runId = executionRunId;
|
|
860
|
+
const remainingAtStart = execution.items.filter((item) => !item.done).map((item) => item.id);
|
|
861
|
+
const task = [
|
|
862
|
+
`Implement the accepted plan at ${execution.planPath} (workdir: ${ctx.cwd}).`,
|
|
863
|
+
"Read the plan file first; it is the source of truth for scope, sequencing, and verification steps.",
|
|
864
|
+
remainingAtStart.length > 0
|
|
865
|
+
? `Verifier items still open: ${remainingAtStart.join(", ")}. Emit [DONE:VC-xxx] markers in your replies as each item's stated evidence passes.`
|
|
866
|
+
: "All verifier items already passed; verify the plan end-to-end and report.",
|
|
867
|
+
execution.implItems?.length
|
|
868
|
+
? `Implementation items: ${execution.implItems.map((item) => item.id).join(", ")} — emit [I-###:implemented]/[I-###:validating] markers as you progress.`
|
|
869
|
+
: "",
|
|
870
|
+
"Finish with the structured summary your system prompt specifies.",
|
|
871
|
+
]
|
|
872
|
+
.filter((line) => line !== "")
|
|
873
|
+
.join("\n");
|
|
874
|
+
let result: Awaited<ReturnType<typeof runPiSubagent>>;
|
|
875
|
+
try {
|
|
876
|
+
result = await runPiSubagent({
|
|
877
|
+
systemPrompt: executorAgentPrompt(),
|
|
878
|
+
task,
|
|
879
|
+
cwd: ctx.cwd,
|
|
880
|
+
model: modelSelector,
|
|
881
|
+
tools: ["read", "write", "edit", "bash", "grep", "find", "ls"],
|
|
882
|
+
envMarker: "executor",
|
|
883
|
+
runId: runId ?? undefined,
|
|
884
|
+
signal: controller.signal,
|
|
885
|
+
timeoutMs: delegatedExecutorTimeoutMs(),
|
|
886
|
+
onProgress: (event: SubagentProgressEvent) => {
|
|
887
|
+
try {
|
|
888
|
+
overlay?.update("executor", event);
|
|
889
|
+
} catch {
|
|
890
|
+
/* display must not fail the child runner */
|
|
891
|
+
}
|
|
892
|
+
if (
|
|
893
|
+
event.type === "transcript"
|
|
894
|
+
&& event.phase === "end"
|
|
895
|
+
&& event.entryType === "assistant-text"
|
|
896
|
+
&& typeof event.text === "string"
|
|
897
|
+
) {
|
|
898
|
+
mirrorDelegateMarkers(pi, ctx, event.text);
|
|
899
|
+
}
|
|
900
|
+
},
|
|
901
|
+
});
|
|
902
|
+
} finally {
|
|
903
|
+
delegatedRuntime = null;
|
|
904
|
+
try {
|
|
905
|
+
await overlay?.close();
|
|
906
|
+
} catch {
|
|
907
|
+
/* best-effort */
|
|
908
|
+
}
|
|
909
|
+
parentSignal?.removeEventListener("abort", relayAbort);
|
|
910
|
+
}
|
|
911
|
+
if (!result.ok) {
|
|
912
|
+
const reason = result.timedOut
|
|
913
|
+
? `delegated executor timed out (${result.errorMessage ?? "no output"})`
|
|
914
|
+
: result.cancelled || controller.signal.aborted
|
|
915
|
+
? "delegated executor aborted by user"
|
|
916
|
+
: `delegated executor failed: ${result.errorMessage ?? "unknown error"}${result.stderr ? `; stderr: ${result.stderr.slice(0, 500)}` : ""}`;
|
|
917
|
+
await stopExecution(pi, ctx, reason);
|
|
918
|
+
return;
|
|
919
|
+
}
|
|
920
|
+
// Clean exit: land any markers from the final output text, then verify.
|
|
921
|
+
mirrorDelegateMarkers(pi, ctx, result.output);
|
|
922
|
+
if (isExecutionComplete()) {
|
|
923
|
+
await completeExecution(pi, ctx);
|
|
924
|
+
return;
|
|
925
|
+
}
|
|
926
|
+
// Items remain: keep the run executing (resumable via /plans-execute under
|
|
927
|
+
// either runtime); the delegate bookkeeping is cleared so restarts do not
|
|
928
|
+
// treat this as an orphaned child.
|
|
929
|
+
if (execution) {
|
|
930
|
+
execution.delegate = undefined;
|
|
931
|
+
withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { delegate: null }));
|
|
932
|
+
persist(pi);
|
|
933
|
+
}
|
|
934
|
+
const remaining = execution?.items.filter((item) => !item.done).map((item) => item.id) ?? [];
|
|
935
|
+
pi.sendMessage(
|
|
936
|
+
{
|
|
937
|
+
customType: "pi-plans-exec-delegate-exit",
|
|
938
|
+
content: `**pi-plans: delegated executor exited with items remaining** — ${remaining.join(", ") || "(none)"}. Run stays executing; resume with /plans-execute (either runtime). Executor summary:\n${result.output.slice(0, 2000)}`,
|
|
939
|
+
display: true,
|
|
940
|
+
},
|
|
941
|
+
{ triggerTurn: false },
|
|
942
|
+
);
|
|
943
|
+
ctx.ui.notify?.(`Delegated executor exited; ${remaining.length} verifier item(s) remain. Resume with /plans-execute.`, "warning");
|
|
944
|
+
updateStatusWidget(ctx);
|
|
945
|
+
}
|
|
946
|
+
|
|
947
|
+
/** Apply VC/I markers parsed from a delegated child's full-text message. Exported for tests. */
|
|
948
|
+
export function mirrorDelegateMarkers(pi: ExtensionAPI, ctx: ExtensionContext, text: string): void {
|
|
949
|
+
const changedVc = applyDoneMarkers(text);
|
|
950
|
+
const changedImpl = applyImplMarkers(text);
|
|
951
|
+
applyCurrentIMarker(text);
|
|
952
|
+
if (changedVc.length === 0 && changedImpl.length === 0) return;
|
|
953
|
+
withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp));
|
|
954
|
+
persist(pi);
|
|
724
955
|
updateStatusWidget(ctx);
|
|
725
956
|
}
|
|
726
957
|
|
package/src/guard.ts
CHANGED
|
@@ -7,12 +7,17 @@
|
|
|
7
7
|
|
|
8
8
|
import * as os from "node:os";
|
|
9
9
|
import * as path from "node:path";
|
|
10
|
-
import { getRun, loadConfig, readActive, resolveStateRootOrNull } from "./state.ts";
|
|
11
10
|
import { activeInfoById } from "./run-context.ts";
|
|
11
|
+
import { getRun, loadConfig, readActive, resolveStateRootOrNull } from "./state.ts";
|
|
12
12
|
|
|
13
13
|
const GUARDED_TOOLS = new Set(["write", "edit"]);
|
|
14
14
|
const GUARDED_STATUSES = new Set(["planning", "accepted"]);
|
|
15
15
|
|
|
16
|
+
/** True when running inside a delegated executor child (PI_PLANS_EXECUTOR=1). */
|
|
17
|
+
export function isExecutorChild(): boolean {
|
|
18
|
+
return process.env.PI_PLANS_EXECUTOR === "1";
|
|
19
|
+
}
|
|
20
|
+
|
|
16
21
|
export interface GuardInput {
|
|
17
22
|
workdir: string;
|
|
18
23
|
toolName: string;
|
|
@@ -24,10 +29,19 @@ export interface GuardInput {
|
|
|
24
29
|
/** Returns a block reason when the write must be blocked, or null when allowed. */
|
|
25
30
|
export function planningWriteBlockReason(input: GuardInput): string | null {
|
|
26
31
|
if (!GUARDED_TOOLS.has(input.toolName)) return null;
|
|
32
|
+
// Delegated executor children write natively with the parent's approval
|
|
33
|
+
// already recorded — the guard must never block them, even when another
|
|
34
|
+
// session's planning run is the newest non-terminal run in the workdir.
|
|
35
|
+
if (isExecutorChild()) return null;
|
|
36
|
+
// Executor children also pin their run via PI_PLANS_RUN_ID; honor it for
|
|
37
|
+
// any child that is not marked executor (defense in depth).
|
|
38
|
+
const envRunId = typeof process.env.PI_PLANS_RUN_ID === "string" ? process.env.PI_PLANS_RUN_ID.trim() : "";
|
|
27
39
|
const active =
|
|
28
|
-
|
|
29
|
-
? activeInfoById(input.workdir,
|
|
30
|
-
:
|
|
40
|
+
envRunId !== ""
|
|
41
|
+
? activeInfoById(input.workdir, envRunId)
|
|
42
|
+
: input.activeRunId !== undefined && input.activeRunId !== null
|
|
43
|
+
? activeInfoById(input.workdir, input.activeRunId)
|
|
44
|
+
: readActive(input.workdir);
|
|
31
45
|
if (!active) return null;
|
|
32
46
|
const run = getRun(input.workdir, active.run_id);
|
|
33
47
|
if (!run || !GUARDED_STATUSES.has(run.status)) return null;
|
|
@@ -53,5 +67,5 @@ export function planningWriteBlockReason(input: GuardInput): string | null {
|
|
|
53
67
|
const allowed = allowedRoots.some((root) => target === root || target.startsWith(`${root}${path.sep}`));
|
|
54
68
|
if (allowed) return null;
|
|
55
69
|
|
|
56
|
-
return `pi-plans: active planning run "${active.run_id}" is read-only outside planning artifacts. Allowed write roots: ${allowedRoots.join(", ")}. Finish planning and get execution approval (execute_plan tool or /plans-execute), or abandon the run (/plans-abandon).`;
|
|
70
|
+
return `pi-plans: active planning run "${active.run_id}" is read-only outside planning artifacts (this session is not bound to it; if you are operating on a different run, bind it via /resume-plans or pick the run explicitly). Allowed write roots: ${allowedRoots.join(", ")}. Finish planning and get execution approval (execute_plan tool or /plans-execute), or abandon the run (/plans-abandon).`;
|
|
57
71
|
}
|
package/src/refine-prompts.ts
CHANGED
|
@@ -144,7 +144,7 @@ Local path (your working directory): ${opts.localPath}
|
|
|
144
144
|
|
|
145
145
|
Authority boundary: read-only analysis only. Do not edit, write, delete, commit, push, or spawn subagents. Stay inside the reference directory.
|
|
146
146
|
|
|
147
|
-
Evidence: inspect the reference with read, grep, find, and ls before judging it. Cite
|
|
147
|
+
Evidence: inspect the reference with read, grep, find, and ls before judging it. Cite evidence for every claim in the medium's format — code: <relative-path>:<line>; papers: section/theorem/table numbers with a short quote; blogs/docs: the heading or quoted passage. Quote only what you verified. Medium-aware deep-read: repos go through entry points, core modules, tests, and configuration; papers through claims, method, limitations, and experiments; blogs/docs through technique, measurements, and caveats. Theoretical grounding counts — an algorithm, a formal property, or a measured tradeoff is as adoptable as an implementation pattern.${contextLine}${languageLine}
|
|
148
148
|
|
|
149
149
|
Success criteria: a structured analysis the main agent can paste into REF_ANALYSIS.md and turn into adoption questions.
|
|
150
150
|
|
|
@@ -157,7 +157,7 @@ Section contracts:
|
|
|
157
157
|
- Key Mechanisms And Design Tradeoffs: the mechanisms that make it work and the tradeoffs they embody.
|
|
158
158
|
- Adoptable Ideas For The Target Repo: concrete, portable ideas ranked by expected value; name the target-repo surface each would touch.
|
|
159
159
|
- Pitfalls And Anti-Patterns: what to avoid when borrowing; failure modes the reference itself documents or exhibits.
|
|
160
|
-
- Evidence Citations: the
|
|
160
|
+
- Evidence Citations: the evidence references backing the claims above (file:line for code; section/theorem/table + quote for papers; heading/quote for blogs and docs).
|
|
161
161
|
- Coverage: which parts of the reference you actually read versus skipped.
|
|
162
162
|
- Evidence Gaps: what you could not determine from the reference alone.`;
|
|
163
163
|
}
|
package/src/refine-ui-state.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import type { SubagentProgressEvent, SubagentResult } from "./subagent.ts";
|
|
2
2
|
|
|
3
|
-
export type RefineOverlayRole = "reviewer" | "criticizer" | "refs";
|
|
3
|
+
export type RefineOverlayRole = "reviewer" | "criticizer" | "refs" | "executor";
|
|
4
4
|
export type RefineLaneStatus = "queued" | "running" | "complete" | "failed" | "cancelled";
|
|
5
5
|
export type RefineTranscriptEntryType = "assistant-text" | "thinking" | "tool-call" | "tool-result" | "diagnostic";
|
|
6
6
|
|
package/src/refine-ui.ts
CHANGED
|
@@ -166,7 +166,7 @@ function summaryFor(role: RefineOverlayRole, lanes: RefineLaneState[], modelLabe
|
|
|
166
166
|
const complete = lanes.filter((lane) => lane.status === "complete").length;
|
|
167
167
|
const terminal = lanes.filter((lane) => ["complete", "failed", "cancelled"].includes(lane.status)).length;
|
|
168
168
|
const running = lanes.filter((lane) => lane.status === "running").length;
|
|
169
|
-
const title = role === "reviewer" ? "Reviewer" : role === "refs" ? "Refs" : "Criticizer";
|
|
169
|
+
const title = role === "reviewer" ? "Reviewer" : role === "refs" ? "Refs" : role === "executor" ? "Executor" : "Criticizer";
|
|
170
170
|
const visibleTitle = modelLabel ? `${title} (${modelLabel})` : title;
|
|
171
171
|
const state = terminal === lanes.length ? "done" : running > 0 ? `${running} running` : "queued";
|
|
172
172
|
return `${visibleTitle} · ${complete}/${lanes.length} done · ${state}`;
|
package/src/resume-command.ts
CHANGED
|
@@ -17,7 +17,7 @@ import * as fs from "node:fs";
|
|
|
17
17
|
import { existsSync } from "node:fs";
|
|
18
18
|
import * as path from "node:path";
|
|
19
19
|
import { loadExecutionFromCheckpoint } from "./exec.ts";
|
|
20
|
-
import { bindRun } from "./run-context.ts";
|
|
20
|
+
import { bindRun, boundRunId } from "./run-context.ts";
|
|
21
21
|
import { acquireOwnership, OwnershipError, releaseOwnership } from "./run-ownership.ts";
|
|
22
22
|
import { loadConfig, resolveStateRootOrNull, setRunStatus, updateRunWorkdir } from "./state.ts";
|
|
23
23
|
import {
|
|
@@ -73,7 +73,11 @@ async function run(pi: ExtensionAPI, ctx: ExtensionContext, baseDir: string): Pr
|
|
|
73
73
|
return;
|
|
74
74
|
}
|
|
75
75
|
|
|
76
|
-
|
|
76
|
+
// v0.6.0 (D-1): a session-bound resumable run resumes directly; the
|
|
77
|
+
// active-pointer auto-win is gone — ambiguity opens the descriptive form.
|
|
78
|
+
const bound = boundRunId(ctx.sessionManager, ctx.cwd);
|
|
79
|
+
const boundCandidate = bound ? candidates.find((candidate) => candidate.runId === bound) ?? null : null;
|
|
80
|
+
let candidate = boundCandidate ?? pickDefaultCandidate(ctx.cwd, candidates);
|
|
77
81
|
if (candidate === null) {
|
|
78
82
|
const labels = candidates.map((entry, index) => {
|
|
79
83
|
const cross = entry.crossWorktree ? " · cross-worktree" : "";
|
package/src/resume.ts
CHANGED
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
import * as fs from "node:fs";
|
|
10
10
|
import { existsSync, readFileSync } from "node:fs";
|
|
11
11
|
import * as path from "node:path";
|
|
12
|
-
import { getRun, readActive, resolveStateRootOrNull, runDirPath, type RunInfo } from "./state.ts";
|
|
12
|
+
import { getRun, listRuns, readActive, resolveStateRootOrNull, runDirPath, type RunInfo } from "./state.ts";
|
|
13
13
|
import { loadCheckpoint, mutateCheckpoint, type WorkflowCheckpoint } from "./workflow-state.ts";
|
|
14
14
|
|
|
15
15
|
export interface ResumeCandidate {
|
|
@@ -81,22 +81,17 @@ function planVersionOf(checkpoint: WorkflowCheckpoint | null, run: RunInfo): num
|
|
|
81
81
|
/**
|
|
82
82
|
* Enumerate resumable runs for the repo containing `workdir`. Corrupt
|
|
83
83
|
* checkpoints are surfaced (not hidden) so the command can report them;
|
|
84
|
-
* read errors never abort discovery of other runs.
|
|
84
|
+
* read errors never abort discovery of other runs. v0.6.0: enumeration is
|
|
85
|
+
* driven by the filesystem-derived registry (`listRuns`) so ordering and
|
|
86
|
+
* corrupt-run tolerance match every other multi-run surface.
|
|
85
87
|
*/
|
|
86
88
|
export function listResumeCandidates(workdir: string): ResumeCandidate[] {
|
|
87
89
|
const stateRoot = resolveStateRootOrNull(workdir);
|
|
88
90
|
if (stateRoot === null) return [];
|
|
89
|
-
const runsDir = path.join(stateRoot, "runs");
|
|
90
|
-
if (!existsSync(runsDir)) return [];
|
|
91
91
|
const active = readActive(workdir);
|
|
92
92
|
const candidates: ResumeCandidate[] = [];
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
entries = fs.readdirSync(runsDir);
|
|
96
|
-
} catch {
|
|
97
|
-
return [];
|
|
98
|
-
}
|
|
99
|
-
for (const runId of entries) {
|
|
93
|
+
for (const summary of listRuns(workdir)) {
|
|
94
|
+
const runId = summary.run_id;
|
|
100
95
|
const run = getRun(workdir, runId);
|
|
101
96
|
if (!run) continue;
|
|
102
97
|
const runDir = runDirPath(workdir, runId);
|
|
@@ -130,7 +125,7 @@ export function listResumeCandidates(workdir: string): ResumeCandidate[] {
|
|
|
130
125
|
updatedAt: checkpoint?.updatedAt ?? run.updated_at,
|
|
131
126
|
});
|
|
132
127
|
}
|
|
133
|
-
//
|
|
128
|
+
// Newest updated first; the active run (registry hint) gets priority.
|
|
134
129
|
candidates.sort((a, b) => {
|
|
135
130
|
const aActive = active?.run_id === a.runId ? 1 : 0;
|
|
136
131
|
const bActive = active?.run_id === b.runId ? 1 : 0;
|
|
@@ -140,14 +135,17 @@ export function listResumeCandidates(workdir: string): ResumeCandidate[] {
|
|
|
140
135
|
return candidates;
|
|
141
136
|
}
|
|
142
137
|
|
|
143
|
-
/**
|
|
138
|
+
/**
|
|
139
|
+
* Pick the default candidate (v0.6.0 D-1): the SESSION BINDING is resolved by
|
|
140
|
+
* the command (it owns the SessionManager); here a unique candidate goes
|
|
141
|
+
* direct and everything else is ambiguous — the active-pointer auto-win is
|
|
142
|
+
* gone (multi-run workdirs must not silently pick the registry hint).
|
|
143
|
+
*/
|
|
144
144
|
export function pickDefaultCandidate(workdir: string, candidates: ResumeCandidate[]): ResumeCandidate | null {
|
|
145
|
+
void workdir;
|
|
145
146
|
if (candidates.length === 0) return null;
|
|
146
|
-
const active = readActive(workdir);
|
|
147
|
-
const activeCandidate = active ? candidates.find((candidate) => candidate.runId === active.run_id) ?? null : null;
|
|
148
|
-
if (activeCandidate !== null) return activeCandidate;
|
|
149
147
|
if (candidates.length === 1) return candidates[0]!;
|
|
150
|
-
return null; // ambiguous: the command must ask
|
|
148
|
+
return null; // ambiguous: the command must ask (binding first, then form)
|
|
151
149
|
}
|
|
152
150
|
|
|
153
151
|
export interface DecisionLedgerEntry {
|
package/src/run-context.ts
CHANGED
|
@@ -10,7 +10,7 @@
|
|
|
10
10
|
*/
|
|
11
11
|
|
|
12
12
|
import * as path from "node:path";
|
|
13
|
-
import { getRun, readActive, runDirPath, type ActiveInfo } from "./state.ts";
|
|
13
|
+
import { getRun, newestNonTerminalRun, readActive, runDirPath, type ActiveInfo } from "./state.ts";
|
|
14
14
|
|
|
15
15
|
interface RunBinding {
|
|
16
16
|
runId: string;
|
|
@@ -54,11 +54,19 @@ export function activeInfoById(workdir: string, runId: string): ActiveInfo | nul
|
|
|
54
54
|
}
|
|
55
55
|
|
|
56
56
|
/**
|
|
57
|
-
* Resolve the run this session should attribute work to.
|
|
58
|
-
*
|
|
59
|
-
*
|
|
57
|
+
* Resolve the run this session should attribute work to. Resolution order:
|
|
58
|
+
* 1. `PI_PLANS_RUN_ID` env (delegated executor children — deterministic even
|
|
59
|
+
* when several runs are active concurrently);
|
|
60
|
+
* 2. the session binding (a session-bound run always wins);
|
|
61
|
+
* 3. registry fallback: the newest non-terminal run (`readActive` shim), or
|
|
62
|
+
* null when every run is terminal.
|
|
60
63
|
*/
|
|
61
64
|
export function resolveActiveRun(session: unknown, workdir: string): ActiveInfo | null {
|
|
65
|
+
const envRunId = process.env.PI_PLANS_RUN_ID;
|
|
66
|
+
if (typeof envRunId === "string" && envRunId.trim() !== "") {
|
|
67
|
+
const pinned = activeInfoById(workdir, envRunId.trim());
|
|
68
|
+
if (pinned !== null) return pinned;
|
|
69
|
+
}
|
|
62
70
|
if (session !== undefined && session !== null) {
|
|
63
71
|
const runId = boundRunId(session, workdir);
|
|
64
72
|
if (runId !== null) {
|