pi-plans 0.6.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +3 -3
- package/README.md +39 -37
- package/agents/reviewer.md +12 -3
- package/index.ts +42 -35
- package/package.json +1 -1
- package/references/pi-planning-workflow.md +44 -60
- package/references/plan-artifact-template.md +71 -60
- package/references/state-and-config.md +59 -43
- package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
- package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
- package/scripts/run-tests.ts +12 -1
- package/scripts/validate.ts +20 -9
- package/skills/debug-and-plan/SKILL.md +3 -3
- package/skills/plan-big/SKILL.md +3 -3
- package/skills/plan-normal/SKILL.md +3 -3
- package/skills/plan-small/SKILL.md +4 -4
- package/skills/plan-with-refs/SKILL.md +6 -6
- package/skills/planning/SKILL.md +1 -1
- package/src/ask-form.ts +4 -4
- package/src/auditor.ts +126 -0
- package/src/auto-approve.ts +1 -1
- package/src/autocomplete.ts +19 -17
- package/src/code-graph/commands.ts +2 -2
- package/src/code-graph/community.ts +1 -1
- package/src/code-graph/paths.ts +1 -1
- package/src/code-graph/watch.ts +2 -2
- package/src/compaction.ts +3 -3
- package/src/config-command.ts +146 -73
- package/src/dashboard.ts +257 -0
- package/src/exec.ts +692 -919
- package/src/global-state.ts +304 -0
- package/src/guard.ts +18 -19
- package/src/messaging.ts +44 -0
- package/src/plan.ts +421 -112
- package/src/query-hook.ts +4 -4
- package/src/refine-prompts.ts +12 -70
- package/src/refine-ui-helpers.ts +24 -5
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +1 -1
- package/src/resume-command.ts +34 -128
- package/src/role-panels.ts +542 -0
- package/src/run-context.ts +3 -10
- package/src/state.ts +272 -72
- package/src/subagent.ts +19 -29
- package/src/task-tool.ts +100 -0
- package/src/tasks.ts +189 -0
- package/src/thinking-levels.ts +67 -0
- package/src/ui-language.ts +3 -54
- package/src/workflow-state.ts +63 -58
- package/tests/analyze-refs.test.ts +35 -18
- package/tests/ask-choice-schema.test.ts +0 -12
- package/tests/ask-choice.test.ts +2 -49
- package/tests/ask-form-tool.test.ts +4 -5
- package/tests/ask-form.test.ts +2 -2
- package/tests/auditor.test.ts +111 -0
- package/tests/auto-approve.test.ts +7 -10
- package/tests/autocomplete.test.ts +8 -11
- package/tests/code-graph-apply-action.test.ts +2 -2
- package/tests/code-graph-commands.test.ts +2 -2
- package/tests/code-graph-index.test.ts +2 -2
- package/tests/code-graph-loop.e2e.test.ts +1 -1
- package/tests/code-graph-mutations.test.ts +1 -1
- package/tests/code-graph-rollback.test.ts +1 -1
- package/tests/code-graph-v05.test.ts +2 -2
- package/tests/compaction.test.ts +1 -1
- package/tests/config-command.test.ts +103 -100
- package/tests/dashboard.test.ts +268 -0
- package/tests/exec-lifecycle.test.ts +181 -115
- package/tests/exec-panel-lifecycle.test.ts +106 -251
- package/tests/exec.test.ts +617 -1706
- package/tests/execute-plan.test.ts +44 -19
- package/tests/extension-load.test.ts +48 -0
- package/tests/global-state.test.ts +371 -0
- package/tests/graph-aware-file-tools.test.ts +5 -5
- package/tests/guard.test.ts +1 -1
- package/tests/multi-run.test.ts +3 -103
- package/tests/plan.test.ts +139 -62
- package/tests/plans.test.ts +7 -79
- package/tests/refine-prompts.test.ts +20 -71
- package/tests/refine-resume.test.ts +27 -22
- package/tests/refine-ui.test.ts +6 -15
- package/tests/resume-lifecycle.test.ts +37 -22
- package/tests/resume.test.ts +33 -81
- package/tests/role-panels.test.ts +391 -0
- package/tests/run-context.test.ts +1 -1
- package/tests/run-ownership.test.ts +1 -1
- package/tests/stale-ctx.test.ts +218 -0
- package/tests/state.test.ts +151 -32
- package/tests/subagent-thinking.test.ts +65 -0
- package/tests/subagent-usage.test.ts +1 -1
- package/tests/task-tool.test.ts +61 -0
- package/tests/thinking-levels.test.ts +77 -0
- package/tests/ui-language.test.ts +2 -17
- package/tests/workflow-state.test.ts +17 -99
- package/tools/analyze-refs.ts +67 -32
- package/tools/ask-choice.ts +7 -53
- package/tools/code-graph.ts +2 -2
- package/tools/execute-plan.ts +48 -99
- package/tools/graph-aware-file-tools.ts +4 -10
- package/tools/plans.ts +40 -66
- package/tools/refine.ts +101 -164
- package/agents/criticizer.md +0 -18
- package/agents/executor.md +0 -26
- package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
- package/src/panel.ts +0 -473
- package/src/termination-prompt.ts +0 -73
- package/tests/goal-wait.test.ts +0 -269
- package/tests/panel-i-zero.test.ts +0 -420
- package/tests/panel.test.ts +0 -355
|
@@ -7,14 +7,11 @@ import * as os from "node:os";
|
|
|
7
7
|
import * as path from "node:path";
|
|
8
8
|
import { after, before, describe, it } from "node:test";
|
|
9
9
|
import {
|
|
10
|
-
applyCompleted,
|
|
11
10
|
applyExecutionApproved,
|
|
12
11
|
applyExecutionCompleted,
|
|
13
12
|
applyExecutionProgress,
|
|
14
13
|
applyExecutionHeadChanged,
|
|
15
14
|
applyExecutionStopped,
|
|
16
|
-
applyImplementationReviewConfigured,
|
|
17
|
-
applyImplementationRoundFinished,
|
|
18
15
|
applyLaneResult,
|
|
19
16
|
applyMigration,
|
|
20
17
|
applyPlanWritten,
|
|
@@ -331,33 +328,21 @@ describe("state machine reducers", () => {
|
|
|
331
328
|
assert.throws(() => applyLaneResult(staged, "r2", "l1", { ok: true }), StateError);
|
|
332
329
|
});
|
|
333
330
|
|
|
334
|
-
it("implementation
|
|
335
|
-
const { workdir, runId } = setupRun("sm-impl");
|
|
331
|
+
it("implementation-review write side is gone; legacy checkpoints stay readable (D-018)", () => {
|
|
332
|
+
const { workdir, runId } = setupRun("sm-impl-legacy");
|
|
336
333
|
createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
|
|
337
334
|
let cp = baseCheckpoint(workdir, runId);
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
assert.
|
|
348
|
-
cp = applyReviewRoundStarted(cp, { roundId: "i1", role: "reviewer", target: "implementation", reviewers: 1, lanes: [{ laneId: "l1" }] });
|
|
349
|
-
const file = writeReviewOutput(workdir, runId, "i1", "l1", "out");
|
|
350
|
-
cp = applyLaneResult(cp, "i1", "l1", { ok: true, resultFile: file });
|
|
351
|
-
assert.throws(() => applyImplementationRoundFinished(cp), StateError);
|
|
352
|
-
cp = applyReviewConsolidated(cp, "i1");
|
|
353
|
-
cp = applyImplementationRoundFinished(cp);
|
|
335
|
+
// Simulate a persisted 0.6.0 checkpoint in the legacy phase with its
|
|
336
|
+
// review bookkeeping — the schema still parses it read-only.
|
|
337
|
+
cp = {
|
|
338
|
+
...cp,
|
|
339
|
+
phase: "implementation-review",
|
|
340
|
+
nextAction: "run-review",
|
|
341
|
+
implementationReview: { terminationCondition: "until-no-high", reviewerCount: 2, completedRounds: 1 },
|
|
342
|
+
};
|
|
343
|
+
const round = applyReviewRoundStarted(cp, { roundId: "i1", role: "reviewer", target: "implementation", reviewers: 2, lanes: [{ laneId: "l1" }] });
|
|
344
|
+
assert.equal(round.reviewRounds.length, 1);
|
|
354
345
|
assert.equal(cp.implementationReview?.completedRounds, 1);
|
|
355
|
-
// Completion requires evidence AND a termination condition.
|
|
356
|
-
assert.throws(() => applyCompleted({ ...cp, implementationReview: undefined }, "evidence"), StateError);
|
|
357
|
-
assert.throws(() => applyCompleted(cp, " "), StateError);
|
|
358
|
-
const done = applyCompleted(cp, "no findings in final round");
|
|
359
|
-
assert.equal(done.phase, "completed");
|
|
360
|
-
assert.equal(done.nextAction, "none");
|
|
361
346
|
});
|
|
362
347
|
|
|
363
348
|
it("execution approval and progress (D-003/D-011)", () => {
|
|
@@ -394,10 +379,12 @@ describe("state machine reducers", () => {
|
|
|
394
379
|
assert.equal(cp.execution?.pausedReason, "stopped by user");
|
|
395
380
|
const resumed = applyExecutionProgress(cp, { pausedReason: null });
|
|
396
381
|
assert.equal(resumed.execution?.pausedReason, undefined);
|
|
397
|
-
// Completion
|
|
382
|
+
// Completion: v0.6.1 passes a completed audit straight to the terminal
|
|
383
|
+
// phase (the implementation-review loop is gone, D-018).
|
|
398
384
|
const finished = applyExecutionCompleted(cp);
|
|
399
|
-
assert.equal(finished.phase, "
|
|
400
|
-
assert.equal(finished.nextAction, "
|
|
385
|
+
assert.equal(finished.phase, "completed");
|
|
386
|
+
assert.equal(finished.nextAction, "none");
|
|
387
|
+
assert.equal(finished.execution?.audit?.passed, true);
|
|
401
388
|
});
|
|
402
389
|
|
|
403
390
|
it("migration resets rounds and approval, keeps termination (F-003)", () => {
|
|
@@ -431,72 +418,3 @@ describe("state machine reducers", () => {
|
|
|
431
418
|
});
|
|
432
419
|
});
|
|
433
420
|
|
|
434
|
-
describe("implementationReview.reviewerCount (0.5.4)", () => {
|
|
435
|
-
it("configured persists reviewerCount and survives checkpoint roundtrip", () => {
|
|
436
|
-
const { workdir, runId } = setupRun("rc-persist");
|
|
437
|
-
createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
|
|
438
|
-
let cp = baseCheckpoint(workdir, runId);
|
|
439
|
-
cp = { ...cp, phase: "implementation-review", nextAction: "ask-question" };
|
|
440
|
-
cp = applyImplementationReviewConfigured(cp, "until-no-high", 2);
|
|
441
|
-
assert.equal(cp.implementationReview?.reviewerCount, 2);
|
|
442
|
-
assert.equal(cp.nextAction, "run-review");
|
|
443
|
-
mutateCheckpoint(workdir, runId, () => cp);
|
|
444
|
-
const reloaded = loadCheckpoint(workdir, runId);
|
|
445
|
-
assert.ok(reloaded.status === "ok");
|
|
446
|
-
assert.equal(reloaded.checkpoint.implementationReview?.reviewerCount, 2);
|
|
447
|
-
});
|
|
448
|
-
|
|
449
|
-
it("omitted reviewerCount stays undefined (legacy checkpoints unchanged)", () => {
|
|
450
|
-
const { workdir, runId } = setupRun("rc-legacy");
|
|
451
|
-
createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
|
|
452
|
-
let cp = baseCheckpoint(workdir, runId);
|
|
453
|
-
cp = { ...cp, phase: "implementation-review", nextAction: "ask-question" };
|
|
454
|
-
cp = applyImplementationReviewConfigured(cp, "1 round");
|
|
455
|
-
assert.equal(cp.implementationReview?.reviewerCount, undefined);
|
|
456
|
-
mutateCheckpoint(workdir, runId, () => cp);
|
|
457
|
-
assert.ok(loadCheckpoint(workdir, runId).status === "ok");
|
|
458
|
-
});
|
|
459
|
-
|
|
460
|
-
it("rejects out-of-range and non-integer reviewerCount on load", () => {
|
|
461
|
-
const { workdir, runId } = setupRun("rc-invalid");
|
|
462
|
-
createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
|
|
463
|
-
const file = checkpointFilePath(workdir, runId)!;
|
|
464
|
-
const base = JSON.parse(fs.readFileSync(file, "utf8")) as { implementationReview?: unknown };
|
|
465
|
-
for (const bad of [0, 4, "3", 1.5]) {
|
|
466
|
-
const doc = {
|
|
467
|
-
...base,
|
|
468
|
-
implementationReview: {
|
|
469
|
-
terminationCondition: "1 round",
|
|
470
|
-
reviewerCount: bad,
|
|
471
|
-
completedRounds: 0,
|
|
472
|
-
},
|
|
473
|
-
};
|
|
474
|
-
fs.writeFileSync(file, JSON.stringify(doc), "utf8");
|
|
475
|
-
const loaded = loadCheckpoint(workdir, runId);
|
|
476
|
-
assert.ok(loaded.status === "corrupt", `reviewerCount ${JSON.stringify(bad)} rejected`);
|
|
477
|
-
}
|
|
478
|
-
// Corrupt bytes refuse overwrite (mutateCheckpoint guard), which is the
|
|
479
|
-
// intended fail-loud behavior — no restore attempted here.
|
|
480
|
-
});
|
|
481
|
-
|
|
482
|
-
it("applyMigration preserves an explicit reviewerCount (CQ1/D-4)", () => {
|
|
483
|
-
const { workdir, runId } = setupRun("rc-migrate");
|
|
484
|
-
createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
|
|
485
|
-
let cp = baseCheckpoint(workdir, runId);
|
|
486
|
-
cp = { ...cp, phase: "implementation-review", nextAction: "run-review" };
|
|
487
|
-
cp = applyImplementationReviewConfigured(cp, "until-no-high", 3);
|
|
488
|
-
cp = applyReviewRoundStarted(cp, { roundId: "i1", role: "reviewer", target: "implementation", reviewers: 3, lanes: [{ laneId: "l1" }] });
|
|
489
|
-
const migrated = applyMigration(cp, { workdir: "/target/wt", worktreeRoot: "/target/wt", commonDir: "/target/.git" });
|
|
490
|
-
assert.equal(migrated.implementationReview?.reviewerCount, 3);
|
|
491
|
-
assert.equal(migrated.implementationReview?.completedRounds, 0);
|
|
492
|
-
});
|
|
493
|
-
|
|
494
|
-
it("second configuration write is rejected even with identical values (replay guard)", () => {
|
|
495
|
-
const { workdir, runId } = setupRun("rc-replay");
|
|
496
|
-
createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
|
|
497
|
-
let cp = baseCheckpoint(workdir, runId);
|
|
498
|
-
cp = { ...cp, phase: "implementation-review", nextAction: "ask-question" };
|
|
499
|
-
cp = applyImplementationReviewConfigured(cp, "until-no-high", 3);
|
|
500
|
-
assert.throws(() => applyImplementationReviewConfigured(cp, "until-no-high", 3), StateError);
|
|
501
|
-
});
|
|
502
|
-
});
|
package/tools/analyze-refs.ts
CHANGED
|
@@ -1,9 +1,12 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* `analyze_refs` tool — plan-with-refs per-reference analysis via read-only Pi
|
|
3
3
|
* subagents with isolated context. One lane per reference (cwd = the ref's own
|
|
4
|
-
* directory), reusing the reviewer
|
|
5
|
-
* and the concurrent
|
|
6
|
-
*
|
|
4
|
+
* directory), reusing the reviewer model confirmation from the GLOBAL config
|
|
5
|
+
* (`~/.pi/pi-plans/config.json`) and the concurrent overlay (title "Refs").
|
|
6
|
+
* analyze_refs is spawn-only by nature, so the reviewer MODE is deliberately
|
|
7
|
+
* not consulted here (Q-4=B): a current-session reviewer still gets spawned
|
|
8
|
+
* ref-analyst lanes, with a one-time notice in the result. Batches are capped
|
|
9
|
+
* at three concurrent lanes; larger ref sets run as sequential batches.
|
|
7
10
|
*
|
|
8
11
|
* Recording is best-effort: spawns land in `subagents.jsonl` (role
|
|
9
12
|
* `ref-analyst`) only when an active planning run exists. Analysis output is
|
|
@@ -17,7 +20,18 @@ import { Text } from "@earendil-works/pi-tui";
|
|
|
17
20
|
import { Type } from "typebox";
|
|
18
21
|
import * as fs from "node:fs";
|
|
19
22
|
import * as path from "node:path";
|
|
20
|
-
import {
|
|
23
|
+
import {
|
|
24
|
+
loadConfig,
|
|
25
|
+
normalizeWorkdir,
|
|
26
|
+
readActive,
|
|
27
|
+
recordSubagent,
|
|
28
|
+
resolveEffectiveReviewer,
|
|
29
|
+
resolveGlobalConfigPath,
|
|
30
|
+
resolveStateRootOrNull,
|
|
31
|
+
StateError,
|
|
32
|
+
} from "../src/state.ts";
|
|
33
|
+
import { runFirstUseFlow, firstUseCancelledError, firstUseTextGuidance, availableModels, findModel, type RolePanelHost } from "../src/role-panels.ts";
|
|
34
|
+
import { roleModelLabel } from "../src/thinking-levels.ts";
|
|
21
35
|
import type { SubagentUsage } from "../src/subagent.ts";
|
|
22
36
|
import { resolveActiveRun } from "../src/run-context.ts";
|
|
23
37
|
import { buildRefAnalystTask, type RefAnalystTaskInput } from "../src/refine-prompts.ts";
|
|
@@ -45,25 +59,39 @@ const AnalyzeRefsParams = Type.Object({
|
|
|
45
59
|
workdir: Type.Optional(Type.String({ description: "Target workspace; default current working directory" })),
|
|
46
60
|
});
|
|
47
61
|
|
|
48
|
-
function gateError(problem: "state" | "
|
|
62
|
+
function gateError(problem: "state" | "confirm", guidance?: string): StateError {
|
|
49
63
|
if (problem === "state") {
|
|
50
64
|
return new StateError("no pi-plans state found; run the plans tool (action: init) first");
|
|
51
65
|
}
|
|
52
|
-
if (problem === "mode") {
|
|
53
|
-
return new StateError(
|
|
54
|
-
"The reviewer role mode is missing or invalid in .git/pi_plans/config.json (analyze_refs reuses the reviewer gates). Ask the role-setting question with ask_choice first: 1. Delegated subagent (recommended; read-only pi subprocess with isolated context) 2. Current session 3. Other 4. Auto-complete — then persist with the plans tool (set-role, role=reviewer).",
|
|
55
|
-
);
|
|
56
|
-
}
|
|
57
|
-
if (problem === "current-session") {
|
|
58
|
-
return new StateError(
|
|
59
|
-
"The reviewer role mode is current-session, but analyze_refs only spawns delegated read-only subagents (one per reference). Ask the user to switch the reviewer mode to delegated-subagent via ask_choice, persist with the plans tool (set-role, role=reviewer, mode=delegated-subagent), then retry analyze_refs.",
|
|
60
|
-
);
|
|
61
|
-
}
|
|
62
66
|
return new StateError(
|
|
63
|
-
|
|
67
|
+
guidance ??
|
|
68
|
+
firstUseTextGuidance([], resolveGlobalConfigPath()),
|
|
64
69
|
);
|
|
65
70
|
}
|
|
66
71
|
|
|
72
|
+
/** First-use model confirmation for the spawn-only ref-analyst path: native
|
|
73
|
+
* panels in TUI, menus for hasUI non-TUI, embedded text guidance otherwise.
|
|
74
|
+
* The reviewer MODE is not consulted (Q-4=B), but because analysis always
|
|
75
|
+
* spawns, a confirmed CONCRETE model is required even when the stored mode
|
|
76
|
+
* is current-session (model confirmation “as usual”). */
|
|
77
|
+
async function ensureRefAnalystModelReady(
|
|
78
|
+
host: RolePanelHost,
|
|
79
|
+
role: { mode: string; model_selector: string | null; thinking_level: string | null; confirmed_at: string | null },
|
|
80
|
+
): Promise<{ mode: string; model_selector: string | null; thinking_level: string | null; confirmed_at: string | null; name_prefix: string }> {
|
|
81
|
+
if (role.confirmed_at !== null && role.model_selector !== null) return role as never;
|
|
82
|
+
let outcome = await runFirstUseFlow(host, role.thinking_level);
|
|
83
|
+
if (outcome.status === "confirmed" && outcome.model_selector !== null && availableModels(host).length > 0 && findModel(host, outcome.model_selector) === null) {
|
|
84
|
+
// F-008: a manually entered selector that the registry does not know —
|
|
85
|
+
// one re-pick, then let spawn-side errors surface precisely.
|
|
86
|
+
outcome = await runFirstUseFlow(host, outcome.role.thinking_level);
|
|
87
|
+
}
|
|
88
|
+
if (outcome.status === "cancelled") throw firstUseCancelledError("analyze_refs");
|
|
89
|
+
const guidance = firstUseTextGuidance(availableModels(host), resolveGlobalConfigPath());
|
|
90
|
+
if (outcome.status === "unavailable") throw gateError("confirm", guidance);
|
|
91
|
+
if (outcome.role.model_selector === null) throw gateError("confirm", guidance);
|
|
92
|
+
return outcome.role;
|
|
93
|
+
}
|
|
94
|
+
|
|
67
95
|
interface AnalysisJob {
|
|
68
96
|
input: RefAnalystTaskInput;
|
|
69
97
|
name: string;
|
|
@@ -72,14 +100,14 @@ interface AnalysisJob {
|
|
|
72
100
|
missing: string | null;
|
|
73
101
|
}
|
|
74
102
|
|
|
75
|
-
export function registerAnalyzeRefsTool(
|
|
103
|
+
export function registerAnalyzeRefsTool(ext: ExtensionAPI, baseDir: string): void {
|
|
76
104
|
const agentPrompt = stripFrontmatter(fs.readFileSync(path.join(baseDir, "agents", "ref-analyst.md"), "utf8"));
|
|
77
105
|
|
|
78
|
-
|
|
106
|
+
ext.registerTool({
|
|
79
107
|
name: "analyze_refs",
|
|
80
108
|
label: "Analyze Refs",
|
|
81
109
|
description:
|
|
82
|
-
"plan-with-refs: analyze downloaded references via independent read-only Pi subagents — one lane per reference (cwd = the ref directory), reusing the reviewer
|
|
110
|
+
"plan-with-refs: analyze downloaded references via independent read-only Pi subagents — one lane per reference (cwd = the ref directory), reusing the reviewer model confirmation from the global config and the concurrent overlay. Batches of at most 3 lanes run sequentially; results are structured per-reference sections for REF_ANALYSIS.md. Recording into subagents.jsonl is best-effort (active run only); refs.jsonl stays owned by the main agent via the plans record-ref action. The reviewer mode is not consulted (spawn-only); first use pops native model/effort panels in TUI.",
|
|
83
111
|
promptSnippet: "Analyze plan-with-refs references with per-ref read-only subagents",
|
|
84
112
|
parameters: AnalyzeRefsParams,
|
|
85
113
|
|
|
@@ -92,17 +120,22 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
|
|
|
92
120
|
throw gateError("state");
|
|
93
121
|
}
|
|
94
122
|
const config = loadConfig(root);
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
if (reviewer.mode === "current-session") {
|
|
100
|
-
throw gateError("current-session");
|
|
101
|
-
}
|
|
102
|
-
if (reviewer.confirmed_at === null) {
|
|
103
|
-
throw gateError("confirm");
|
|
123
|
+
|
|
124
|
+
// F-005: cheap validations BEFORE any first-use panel.
|
|
125
|
+
if (params.refs.length === 0) {
|
|
126
|
+
throw new StateError("analyze_refs requires at least one reference");
|
|
104
127
|
}
|
|
105
128
|
|
|
129
|
+
// Effective reviewer from the global config (mode NOT consulted —
|
|
130
|
+
// analyze_refs is spawn-only, Q-4=B; a notice surfaces when the stored
|
|
131
|
+
// mode is current-session so the switch is never silent).
|
|
132
|
+
const { reviewer: initialReviewer } = resolveEffectiveReviewer(root);
|
|
133
|
+
const modeIgnoredNotice =
|
|
134
|
+
initialReviewer.mode === "current-session"
|
|
135
|
+
? `note: the reviewer mode is ${initialReviewer.mode}, but analyze_refs always spawns read-only subagents; the mode is ignored here and unchanged.`
|
|
136
|
+
: null;
|
|
137
|
+
const reviewer = await ensureRefAnalystModelReady(ctx as unknown as RolePanelHost, initialReviewer);
|
|
138
|
+
|
|
106
139
|
// Resolve refs and validate directories up front; missing ones become
|
|
107
140
|
// FAILED sections instead of aborting the whole batch.
|
|
108
141
|
const active = resolveActiveRun(ctx.sessionManager, workdir);
|
|
@@ -122,7 +155,7 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
|
|
|
122
155
|
}
|
|
123
156
|
|
|
124
157
|
const model = reviewer.model_selector ?? (ctx.model ? `${ctx.model.provider}/${ctx.model.id}` : undefined);
|
|
125
|
-
const modelLabel = model ?? "inherit";
|
|
158
|
+
const modelLabel = roleModelLabel(model ?? "inherit", reviewer.thinking_level);
|
|
126
159
|
let languageTag: string | null = null;
|
|
127
160
|
if (active) {
|
|
128
161
|
try {
|
|
@@ -140,6 +173,7 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
|
|
|
140
173
|
role: "ref-analyst",
|
|
141
174
|
name,
|
|
142
175
|
model: okModel ?? model ?? null,
|
|
176
|
+
thinking_level: reviewer.thinking_level,
|
|
143
177
|
// I-010: meter subagent token/cost for benchmark accounting.
|
|
144
178
|
usage: usage
|
|
145
179
|
? { input: usage.input, output: usage.output, cache_read: usage.cacheRead, cache_write: usage.cacheWrite, cost: usage.cost }
|
|
@@ -160,6 +194,7 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
|
|
|
160
194
|
task: buildRefAnalystTask({ ...job.input, languageTag }),
|
|
161
195
|
cwd: job.dir,
|
|
162
196
|
model,
|
|
197
|
+
thinkingLevel: reviewer.thinking_level ?? undefined,
|
|
163
198
|
tools: READ_ONLY_TOOLS,
|
|
164
199
|
signal: relay.signal,
|
|
165
200
|
onProgress: (event) => overlay?.update(job.laneId, event),
|
|
@@ -224,7 +259,7 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
|
|
|
224
259
|
|
|
225
260
|
if (failures === jobs.length) {
|
|
226
261
|
throw new Error(
|
|
227
|
-
`all reference analysis subagents failed (${failures}/${jobs.length})${model ? `\nIf the model selector "${model}" is unavailable, reset the reviewer confirmation (plans set-role, role=reviewer, resetConfirmation: true)
|
|
262
|
+
`all reference analysis subagents failed (${failures}/${jobs.length})${model ? `\nIf the model selector "${model}" is unavailable, reset the reviewer confirmation (plans set-role, role=reviewer, resetConfirmation: true) — the next analyze_refs opens the native model panel to re-confirm.` : ""}`,
|
|
228
263
|
);
|
|
229
264
|
}
|
|
230
265
|
|
|
@@ -237,13 +272,13 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
|
|
|
237
272
|
content: [
|
|
238
273
|
{
|
|
239
274
|
type: "text",
|
|
240
|
-
text: `${text}\n\n---\nPersist: paste each reference's analysis into REF_ANALYSIS.md, call the plans tool (record-ref) per reference with coverage and gaps filled from the analysis, then ask at least three ref-specific adoption questions per reference with ask_choice before using its ideas in PLAN_v1.md.`,
|
|
275
|
+
text: `${modeIgnoredNotice ? `${modeIgnoredNotice}\n\n` : ""}${text}\n\n---\nPersist: paste each reference's analysis into REF_ANALYSIS.md, call the plans tool (record-ref) per reference with coverage and gaps filled from the analysis, then ask at least three ref-specific adoption questions per reference with ask_choice before using its ideas in PLAN_v1.md.`,
|
|
241
276
|
},
|
|
242
277
|
],
|
|
243
278
|
details: {
|
|
244
279
|
mode: "delegated-subagent",
|
|
245
280
|
role: "ref-analyst",
|
|
246
|
-
reviewerGates: { mode:
|
|
281
|
+
reviewerGates: { mode: initialReviewer.mode, modeIgnored: modeIgnoredNotice !== null, model, thinkingLevel: reviewer.thinking_level },
|
|
247
282
|
batches: Math.ceil(jobs.length / BATCH_SIZE),
|
|
248
283
|
model,
|
|
249
284
|
outputs,
|
package/tools/ask-choice.ts
CHANGED
|
@@ -16,13 +16,6 @@ import { Text } from "@earendil-works/pi-tui";
|
|
|
16
16
|
import { Type } from "typebox";
|
|
17
17
|
import { disableAutoComplete, enableAutoComplete, isAutoCompleteEnabled, recordAskChoice } from "../src/autocomplete.ts";
|
|
18
18
|
import { assertAutoApprovable, isAutoApproveEnabled } from "../src/auto-approve.ts";
|
|
19
|
-
import {
|
|
20
|
-
TERMINATION_QUESTION,
|
|
21
|
-
TERMINATION_OPTIONS,
|
|
22
|
-
TERMINATION_RECORDING_INSTRUCTIONS,
|
|
23
|
-
implReviewerCountPromptLine,
|
|
24
|
-
renderTerminationOptions,
|
|
25
|
-
} from "../src/termination-prompt.ts";
|
|
26
19
|
import { truncateToWidth, visibleWidth } from "../src/refine-ui-helpers.ts";
|
|
27
20
|
import { stripRecommendedMarker } from "../src/ask-form.ts";
|
|
28
21
|
import {
|
|
@@ -63,8 +56,8 @@ export const FALLBACK_ROWS = 30;
|
|
|
63
56
|
/** Minimal-form floor for tiny terminals (stage-3 width). */
|
|
64
57
|
const MINIMAL_LINE_WIDTH = 20;
|
|
65
58
|
/**
|
|
66
|
-
* Truncation floor for fixed tail labels (Other…/Auto-complete
|
|
67
|
-
*
|
|
59
|
+
* Truncation floor for fixed tail labels (Other…/Auto-complete): the
|
|
60
|
+
* longest magic prefix ("Auto-complete", 14 cols) plus slack.
|
|
68
61
|
* These labels drive startsWith() answer routing and must never lose it.
|
|
69
62
|
*/
|
|
70
63
|
const FIXED_LABEL_FLOOR = 18;
|
|
@@ -74,7 +67,7 @@ export interface PanelItem {
|
|
|
74
67
|
core: string;
|
|
75
68
|
/** Full display label: core + description (degradation stage 0). */
|
|
76
69
|
display: string;
|
|
77
|
-
/** Fixed tail labels (Other…/Auto-complete
|
|
70
|
+
/** Fixed tail labels (Other…/Auto-complete): truncation keeps at least the magic prefix. */
|
|
78
71
|
fixed?: boolean;
|
|
79
72
|
}
|
|
80
73
|
|
|
@@ -213,12 +206,6 @@ export const AskChoiceParams = Type.Object(
|
|
|
213
206
|
purpose: Type.Optional(
|
|
214
207
|
Type.String({ description: "Short machine-readable purpose (e.g. 'scope', 'termination-condition')." }),
|
|
215
208
|
),
|
|
216
|
-
trailing: Type.Optional(
|
|
217
|
-
StringEnum(["auto-refine-loop"] as const, {
|
|
218
|
-
description:
|
|
219
|
-
'Replace the trailing Auto-complete option with "Auto-refine loop" (post-execution amelioration prompt). Selecting it returns instructions to ask the rounds/termination follow-up; Auto-complete is suppressed entirely for this question.',
|
|
220
|
-
}),
|
|
221
|
-
),
|
|
222
209
|
workdir: Type.Optional(Type.String({ description: "Target workspace; default current working directory" })),
|
|
223
210
|
},
|
|
224
211
|
{ additionalProperties: false },
|
|
@@ -589,12 +576,12 @@ function formatBatchAnswers(batch: NonNullable<AskChoiceDetails["batch"]>): stri
|
|
|
589
576
|
}
|
|
590
577
|
const NL = "\n";
|
|
591
578
|
|
|
592
|
-
export function registerAskChoiceTool(
|
|
593
|
-
|
|
579
|
+
export function registerAskChoiceTool(ext: ExtensionAPI): void {
|
|
580
|
+
ext.registerTool({
|
|
594
581
|
name: "ask_choice",
|
|
595
582
|
label: "Ask Choice",
|
|
596
583
|
description:
|
|
597
|
-
"Ask the user planning or refinement questions as numbered choice prompts: recommended option first, alternatives next, then Other and Auto-complete. Two shapes: questions: [...] (2-8 questions) opens ONE tabbed multiple-choice form with a submit page — use it to batch a round of questions (≤8), then think about the answers and follow up in later calls (phased questioning stays agent-driven); question + options asks one question at a time (classic flow). Use ask_choice for every user-facing planning question, the final scope confirmation, refinement-mode questions, language/role/model settings, and the execution handoff. Scope confirmation and the execution handoff MUST stay single-question calls (autoComplete: false); batches reject autoComplete: false items and the
|
|
584
|
+
"Ask the user planning or refinement questions as numbered choice prompts: recommended option first, alternatives next, then Other and Auto-complete. Two shapes: questions: [...] (2-8 questions) opens ONE tabbed multiple-choice form with a submit page — use it to batch a round of questions (≤8), then think about the answers and follow up in later calls (phased questioning stays agent-driven); question + options asks one question at a time (classic flow). Use ask_choice for every user-facing planning question, the final scope confirmation, refinement-mode questions, language/role/model settings, and the execution handoff. Scope confirmation and the execution handoff MUST stay single-question calls (autoComplete: false); batches reject autoComplete: false items and the questionIds reserved for handoff. EVERY option you author — including the accept/execute handoff — must set description to '✓ <advantage> / ✗ <drawback>' in the configured language, so the user can see what each option gains and what it costs. Other and Auto-complete are appended by this tool and need no description.",
|
|
598
585
|
promptSnippet: "Ask structured planning questions with recommended/Other/Auto-complete ordering; batch ≤8 questions per form",
|
|
599
586
|
promptGuidelines: [
|
|
600
587
|
"Use ask_choice for every pi-plans question to the user instead of plain-text questions; it enforces option ordering and records decisions.",
|
|
@@ -605,13 +592,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
|
|
|
605
592
|
executionMode: "sequential",
|
|
606
593
|
|
|
607
594
|
async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
|
|
608
|
-
// R-13 (defense-in-depth): a delegated executor child has no user to
|
|
609
|
-
// answer — refuse instead of blocking a headless run on a UI prompt.
|
|
610
|
-
if (process.env.PI_PLANS_EXECUTOR === "1") {
|
|
611
|
-
throw new Error(
|
|
612
|
-
"ask_choice is unavailable in a delegated executor session: no interactive user. Decide autonomously, proceed, and record the deviation in your final summary.",
|
|
613
|
-
);
|
|
614
|
-
}
|
|
615
595
|
// 0.4.0 batch mode: one tabbed form for a whole round of questions
|
|
616
596
|
// (2-8). The single-question path below is untouched (C-004).
|
|
617
597
|
// F-005 (impl review r1): ambiguous shapes fail loudly instead of
|
|
@@ -620,9 +600,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
|
|
|
620
600
|
if (params.question !== undefined || params.options !== undefined) {
|
|
621
601
|
throw new Error("ask_choice accepts either question+options or questions, not both");
|
|
622
602
|
}
|
|
623
|
-
if (params.trailing !== undefined) {
|
|
624
|
-
throw new Error("ask_choice batch mode does not support trailing (single-question only)");
|
|
625
|
-
}
|
|
626
603
|
return executeAskChoiceBatch({ questions: params.questions, workdir: params.workdir }, ctx);
|
|
627
604
|
}
|
|
628
605
|
const workdir = normalizeWorkdir(params.workdir ?? ctx.cwd);
|
|
@@ -661,10 +638,7 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
|
|
|
661
638
|
/* the decisions ledger already holds the answer; F-005 reconcile covers the gap */
|
|
662
639
|
}
|
|
663
640
|
};
|
|
664
|
-
|
|
665
|
-
// so an erroneously passed autoComplete flag is suppressed here.
|
|
666
|
-
const trailing = params.trailing;
|
|
667
|
-
const autoComplete = (params.autoComplete ?? true) && trailing === undefined;
|
|
641
|
+
const autoComplete = params.autoComplete ?? true;
|
|
668
642
|
const options = params.options;
|
|
669
643
|
if (options.length === 0) throw new Error("ask_choice requires at least one option");
|
|
670
644
|
const recommended = options.find((option) => option.recommended) ?? options[0];
|
|
@@ -764,8 +738,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
|
|
|
764
738
|
};
|
|
765
739
|
}
|
|
766
740
|
|
|
767
|
-
const AUTO_REFINE_LOOP_LABEL =
|
|
768
|
-
"Auto-refine loop (run refinement rounds until no high-severity finding or the 5-round cap)";
|
|
769
741
|
const panelItems: PanelItem[] = options.map((option, index) => {
|
|
770
742
|
const label = stripRecommendedMarker(option.label);
|
|
771
743
|
const isRec = option === recommended;
|
|
@@ -777,7 +749,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
|
|
|
777
749
|
});
|
|
778
750
|
if (allowOther) panelItems.push({ core: "Other… (type your own answer)", display: "Other… (type your own answer)", fixed: true });
|
|
779
751
|
if (autoComplete) panelItems.push({ core: "Auto-complete (take the recommended option)", display: "Auto-complete (take the recommended option)", fixed: true });
|
|
780
|
-
else if (trailing) panelItems.push({ core: AUTO_REFINE_LOOP_LABEL, display: AUTO_REFINE_LOOP_LABEL, fixed: true });
|
|
781
752
|
|
|
782
753
|
const panel = fitAskChoicePanel(
|
|
783
754
|
params.question,
|
|
@@ -820,23 +791,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
|
|
|
820
791
|
};
|
|
821
792
|
}
|
|
822
793
|
|
|
823
|
-
if (trailing && selected.startsWith("Auto-refine loop")) {
|
|
824
|
-
recordAskChoice(ctx, false);
|
|
825
|
-
record("Auto-refine loop", "user");
|
|
826
|
-
// Skill-aware reviewer-count default (D-1/D-4): same mapping the
|
|
827
|
-
// goal-running continuation in src/exec.ts renders.
|
|
828
|
-
const activeSkill = resolveActiveRun(ctx.sessionManager, workdir)?.skill;
|
|
829
|
-
return {
|
|
830
|
-
content: [
|
|
831
|
-
{
|
|
832
|
-
type: "text",
|
|
833
|
-
text: `User selected Auto-refine loop. Immediately ask the follow-up with ask_choice (autoComplete: false, in the session language): "${TERMINATION_QUESTION}" Options (recommended first): ${renderTerminationOptions()}. ${TERMINATION_RECORDING_INSTRUCTIONS} ${implReviewerCountPromptLine(activeSkill)} Then run the loop per the completion instructions: each round calls refine (role: "reviewer", target: "implementation", reviewers: <configured reviewerCount>), accepts findings on evidence, applies fixes, re-runs relevant tests, and continues until the chosen termination condition — the goal-wait option keeps the loop running until no unpassed VCs remain.`,
|
|
834
|
-
},
|
|
835
|
-
],
|
|
836
|
-
details: details("Auto-refine loop", "user"),
|
|
837
|
-
};
|
|
838
|
-
}
|
|
839
|
-
|
|
840
794
|
if (allowOther && selected.startsWith("Other…")) {
|
|
841
795
|
const typed = await ctx.ui.input(`${params.question} — your answer:`);
|
|
842
796
|
if (typed === undefined || !typed.trim()) {
|
package/tools/code-graph.ts
CHANGED
|
@@ -128,8 +128,8 @@ export async function ensureRuntime(workdir: string, ctx: CodeGraphContext): Pro
|
|
|
128
128
|
return { entry: runtimeCache, status };
|
|
129
129
|
}
|
|
130
130
|
|
|
131
|
-
export function registerCodeGraphTool(
|
|
132
|
-
|
|
131
|
+
export function registerCodeGraphTool(ext: ExtensionAPI): void {
|
|
132
|
+
ext.registerTool({
|
|
133
133
|
name: "code_graph",
|
|
134
134
|
label: "Code Graph",
|
|
135
135
|
description:
|