pi-plans 0.5.7 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +126 -0
- package/README.md +49 -39
- package/agents/ref-analyst.md +7 -4
- package/agents/reviewer.md +12 -3
- package/index.ts +74 -40
- package/package.json +2 -1
- package/references/pi-planning-workflow.md +45 -58
- package/references/plan-artifact-template.md +71 -60
- package/references/state-and-config.md +63 -47
- package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
- package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
- package/scripts/run-tests.ts +12 -1
- package/scripts/validate.ts +22 -10
- package/skills/debug-and-plan/SKILL.md +4 -4
- package/skills/plan-big/SKILL.md +5 -5
- package/skills/plan-normal/SKILL.md +5 -5
- package/skills/plan-small/SKILL.md +5 -5
- package/skills/plan-with-refs/SKILL.md +8 -8
- package/skills/planning/SKILL.md +1 -1
- package/src/ask-form.ts +4 -4
- package/src/auditor.ts +126 -0
- package/src/auto-approve.ts +1 -1
- package/src/autocomplete.ts +19 -17
- package/src/code-graph/commands.ts +2 -2
- package/src/code-graph/community.ts +1 -1
- package/src/code-graph/paths.ts +1 -1
- package/src/code-graph/watch.ts +2 -2
- package/src/compaction.ts +3 -3
- package/src/config-command.ts +154 -76
- package/src/dashboard.ts +257 -0
- package/src/exec.ts +709 -705
- package/src/global-state.ts +304 -0
- package/src/guard.ts +16 -3
- package/src/messaging.ts +44 -0
- package/src/plan.ts +421 -112
- package/src/query-hook.ts +4 -4
- package/src/refine-prompts.ts +14 -72
- package/src/refine-ui-helpers.ts +24 -5
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +1 -1
- package/src/resume-command.ts +40 -130
- package/src/resume.ts +15 -17
- package/src/role-panels.ts +542 -0
- package/src/run-context.ts +5 -4
- package/src/run-picker.ts +98 -0
- package/src/state.ts +380 -77
- package/src/subagent.ts +32 -1
- package/src/task-tool.ts +100 -0
- package/src/tasks.ts +189 -0
- package/src/thinking-levels.ts +67 -0
- package/src/ui-language.ts +3 -54
- package/src/workflow-state.ts +78 -57
- package/tests/analyze-refs.test.ts +35 -18
- package/tests/ask-choice-pros-cons.test.ts +147 -0
- package/tests/ask-choice-schema.test.ts +0 -12
- package/tests/ask-choice.test.ts +2 -49
- package/tests/ask-form-tool.test.ts +4 -5
- package/tests/ask-form.test.ts +2 -2
- package/tests/auditor.test.ts +111 -0
- package/tests/auto-approve.test.ts +7 -10
- package/tests/autocomplete.test.ts +8 -11
- package/tests/code-graph-apply-action.test.ts +2 -2
- package/tests/code-graph-commands.test.ts +2 -2
- package/tests/code-graph-index.test.ts +2 -2
- package/tests/code-graph-loop.e2e.test.ts +1 -1
- package/tests/code-graph-mutations.test.ts +1 -1
- package/tests/code-graph-rollback.test.ts +1 -1
- package/tests/code-graph-v05.test.ts +2 -2
- package/tests/compaction.test.ts +1 -1
- package/tests/config-command.test.ts +103 -100
- package/tests/dashboard.test.ts +268 -0
- package/tests/exec-lifecycle.test.ts +181 -115
- package/tests/exec-panel-lifecycle.test.ts +106 -251
- package/tests/exec.test.ts +617 -1706
- package/tests/execute-plan.test.ts +44 -19
- package/tests/extension-load.test.ts +48 -0
- package/tests/global-state.test.ts +371 -0
- package/tests/graph-aware-file-tools.test.ts +5 -5
- package/tests/guard.test.ts +1 -1
- package/tests/multi-run.test.ts +184 -0
- package/tests/plan.test.ts +139 -62
- package/tests/plans.test.ts +7 -79
- package/tests/refine-prompts.test.ts +20 -71
- package/tests/refine-resume.test.ts +27 -22
- package/tests/refine-ui.test.ts +6 -15
- package/tests/resume-lifecycle.test.ts +37 -22
- package/tests/resume.test.ts +43 -88
- package/tests/role-panels.test.ts +391 -0
- package/tests/run-context.test.ts +1 -1
- package/tests/run-ownership.test.ts +1 -1
- package/tests/stale-ctx.test.ts +218 -0
- package/tests/state.test.ts +151 -32
- package/tests/subagent-thinking.test.ts +65 -0
- package/tests/subagent-usage.test.ts +1 -1
- package/tests/task-tool.test.ts +61 -0
- package/tests/thinking-levels.test.ts +77 -0
- package/tests/ui-language.test.ts +2 -17
- package/tests/workflow-state.test.ts +17 -99
- package/tools/analyze-refs.ts +67 -32
- package/tools/ask-choice.ts +19 -49
- package/tools/code-graph.ts +2 -2
- package/tools/execute-plan.ts +63 -33
- package/tools/graph-aware-file-tools.ts +6 -4
- package/tools/plans.ts +40 -66
- package/tools/refine.ts +101 -164
- package/agents/criticizer.md +0 -18
- package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
- package/src/panel.ts +0 -473
- package/src/termination-prompt.ts +0 -73
- package/tests/goal-wait.test.ts +0 -269
- package/tests/panel-i-zero.test.ts +0 -420
- package/tests/panel.test.ts +0 -355
|
@@ -7,14 +7,11 @@ import * as os from "node:os";
|
|
|
7
7
|
import * as path from "node:path";
|
|
8
8
|
import { after, before, describe, it } from "node:test";
|
|
9
9
|
import {
|
|
10
|
-
applyCompleted,
|
|
11
10
|
applyExecutionApproved,
|
|
12
11
|
applyExecutionCompleted,
|
|
13
12
|
applyExecutionProgress,
|
|
14
13
|
applyExecutionHeadChanged,
|
|
15
14
|
applyExecutionStopped,
|
|
16
|
-
applyImplementationReviewConfigured,
|
|
17
|
-
applyImplementationRoundFinished,
|
|
18
15
|
applyLaneResult,
|
|
19
16
|
applyMigration,
|
|
20
17
|
applyPlanWritten,
|
|
@@ -331,33 +328,21 @@ describe("state machine reducers", () => {
|
|
|
331
328
|
assert.throws(() => applyLaneResult(staged, "r2", "l1", { ok: true }), StateError);
|
|
332
329
|
});
|
|
333
330
|
|
|
334
|
-
it("implementation
|
|
335
|
-
const { workdir, runId } = setupRun("sm-impl");
|
|
331
|
+
it("implementation-review write side is gone; legacy checkpoints stay readable (D-018)", () => {
|
|
332
|
+
const { workdir, runId } = setupRun("sm-impl-legacy");
|
|
336
333
|
createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
|
|
337
334
|
let cp = baseCheckpoint(workdir, runId);
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
assert.
|
|
348
|
-
cp = applyReviewRoundStarted(cp, { roundId: "i1", role: "reviewer", target: "implementation", reviewers: 1, lanes: [{ laneId: "l1" }] });
|
|
349
|
-
const file = writeReviewOutput(workdir, runId, "i1", "l1", "out");
|
|
350
|
-
cp = applyLaneResult(cp, "i1", "l1", { ok: true, resultFile: file });
|
|
351
|
-
assert.throws(() => applyImplementationRoundFinished(cp), StateError);
|
|
352
|
-
cp = applyReviewConsolidated(cp, "i1");
|
|
353
|
-
cp = applyImplementationRoundFinished(cp);
|
|
335
|
+
// Simulate a persisted 0.6.0 checkpoint in the legacy phase with its
|
|
336
|
+
// review bookkeeping — the schema still parses it read-only.
|
|
337
|
+
cp = {
|
|
338
|
+
...cp,
|
|
339
|
+
phase: "implementation-review",
|
|
340
|
+
nextAction: "run-review",
|
|
341
|
+
implementationReview: { terminationCondition: "until-no-high", reviewerCount: 2, completedRounds: 1 },
|
|
342
|
+
};
|
|
343
|
+
const round = applyReviewRoundStarted(cp, { roundId: "i1", role: "reviewer", target: "implementation", reviewers: 2, lanes: [{ laneId: "l1" }] });
|
|
344
|
+
assert.equal(round.reviewRounds.length, 1);
|
|
354
345
|
assert.equal(cp.implementationReview?.completedRounds, 1);
|
|
355
|
-
// Completion requires evidence AND a termination condition.
|
|
356
|
-
assert.throws(() => applyCompleted({ ...cp, implementationReview: undefined }, "evidence"), StateError);
|
|
357
|
-
assert.throws(() => applyCompleted(cp, " "), StateError);
|
|
358
|
-
const done = applyCompleted(cp, "no findings in final round");
|
|
359
|
-
assert.equal(done.phase, "completed");
|
|
360
|
-
assert.equal(done.nextAction, "none");
|
|
361
346
|
});
|
|
362
347
|
|
|
363
348
|
it("execution approval and progress (D-003/D-011)", () => {
|
|
@@ -394,10 +379,12 @@ describe("state machine reducers", () => {
|
|
|
394
379
|
assert.equal(cp.execution?.pausedReason, "stopped by user");
|
|
395
380
|
const resumed = applyExecutionProgress(cp, { pausedReason: null });
|
|
396
381
|
assert.equal(resumed.execution?.pausedReason, undefined);
|
|
397
|
-
// Completion
|
|
382
|
+
// Completion: v0.6.1 passes a completed audit straight to the terminal
|
|
383
|
+
// phase (the implementation-review loop is gone, D-018).
|
|
398
384
|
const finished = applyExecutionCompleted(cp);
|
|
399
|
-
assert.equal(finished.phase, "
|
|
400
|
-
assert.equal(finished.nextAction, "
|
|
385
|
+
assert.equal(finished.phase, "completed");
|
|
386
|
+
assert.equal(finished.nextAction, "none");
|
|
387
|
+
assert.equal(finished.execution?.audit?.passed, true);
|
|
401
388
|
});
|
|
402
389
|
|
|
403
390
|
it("migration resets rounds and approval, keeps termination (F-003)", () => {
|
|
@@ -431,72 +418,3 @@ describe("state machine reducers", () => {
|
|
|
431
418
|
});
|
|
432
419
|
});
|
|
433
420
|
|
|
434
|
-
describe("implementationReview.reviewerCount (0.5.4)", () => {
|
|
435
|
-
it("configured persists reviewerCount and survives checkpoint roundtrip", () => {
|
|
436
|
-
const { workdir, runId } = setupRun("rc-persist");
|
|
437
|
-
createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
|
|
438
|
-
let cp = baseCheckpoint(workdir, runId);
|
|
439
|
-
cp = { ...cp, phase: "implementation-review", nextAction: "ask-question" };
|
|
440
|
-
cp = applyImplementationReviewConfigured(cp, "until-no-high", 2);
|
|
441
|
-
assert.equal(cp.implementationReview?.reviewerCount, 2);
|
|
442
|
-
assert.equal(cp.nextAction, "run-review");
|
|
443
|
-
mutateCheckpoint(workdir, runId, () => cp);
|
|
444
|
-
const reloaded = loadCheckpoint(workdir, runId);
|
|
445
|
-
assert.ok(reloaded.status === "ok");
|
|
446
|
-
assert.equal(reloaded.checkpoint.implementationReview?.reviewerCount, 2);
|
|
447
|
-
});
|
|
448
|
-
|
|
449
|
-
it("omitted reviewerCount stays undefined (legacy checkpoints unchanged)", () => {
|
|
450
|
-
const { workdir, runId } = setupRun("rc-legacy");
|
|
451
|
-
createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
|
|
452
|
-
let cp = baseCheckpoint(workdir, runId);
|
|
453
|
-
cp = { ...cp, phase: "implementation-review", nextAction: "ask-question" };
|
|
454
|
-
cp = applyImplementationReviewConfigured(cp, "1 round");
|
|
455
|
-
assert.equal(cp.implementationReview?.reviewerCount, undefined);
|
|
456
|
-
mutateCheckpoint(workdir, runId, () => cp);
|
|
457
|
-
assert.ok(loadCheckpoint(workdir, runId).status === "ok");
|
|
458
|
-
});
|
|
459
|
-
|
|
460
|
-
it("rejects out-of-range and non-integer reviewerCount on load", () => {
|
|
461
|
-
const { workdir, runId } = setupRun("rc-invalid");
|
|
462
|
-
createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
|
|
463
|
-
const file = checkpointFilePath(workdir, runId)!;
|
|
464
|
-
const base = JSON.parse(fs.readFileSync(file, "utf8")) as { implementationReview?: unknown };
|
|
465
|
-
for (const bad of [0, 4, "3", 1.5]) {
|
|
466
|
-
const doc = {
|
|
467
|
-
...base,
|
|
468
|
-
implementationReview: {
|
|
469
|
-
terminationCondition: "1 round",
|
|
470
|
-
reviewerCount: bad,
|
|
471
|
-
completedRounds: 0,
|
|
472
|
-
},
|
|
473
|
-
};
|
|
474
|
-
fs.writeFileSync(file, JSON.stringify(doc), "utf8");
|
|
475
|
-
const loaded = loadCheckpoint(workdir, runId);
|
|
476
|
-
assert.ok(loaded.status === "corrupt", `reviewerCount ${JSON.stringify(bad)} rejected`);
|
|
477
|
-
}
|
|
478
|
-
// Corrupt bytes refuse overwrite (mutateCheckpoint guard), which is the
|
|
479
|
-
// intended fail-loud behavior — no restore attempted here.
|
|
480
|
-
});
|
|
481
|
-
|
|
482
|
-
it("applyMigration preserves an explicit reviewerCount (CQ1/D-4)", () => {
|
|
483
|
-
const { workdir, runId } = setupRun("rc-migrate");
|
|
484
|
-
createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
|
|
485
|
-
let cp = baseCheckpoint(workdir, runId);
|
|
486
|
-
cp = { ...cp, phase: "implementation-review", nextAction: "run-review" };
|
|
487
|
-
cp = applyImplementationReviewConfigured(cp, "until-no-high", 3);
|
|
488
|
-
cp = applyReviewRoundStarted(cp, { roundId: "i1", role: "reviewer", target: "implementation", reviewers: 3, lanes: [{ laneId: "l1" }] });
|
|
489
|
-
const migrated = applyMigration(cp, { workdir: "/target/wt", worktreeRoot: "/target/wt", commonDir: "/target/.git" });
|
|
490
|
-
assert.equal(migrated.implementationReview?.reviewerCount, 3);
|
|
491
|
-
assert.equal(migrated.implementationReview?.completedRounds, 0);
|
|
492
|
-
});
|
|
493
|
-
|
|
494
|
-
it("second configuration write is rejected even with identical values (replay guard)", () => {
|
|
495
|
-
const { workdir, runId } = setupRun("rc-replay");
|
|
496
|
-
createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
|
|
497
|
-
let cp = baseCheckpoint(workdir, runId);
|
|
498
|
-
cp = { ...cp, phase: "implementation-review", nextAction: "ask-question" };
|
|
499
|
-
cp = applyImplementationReviewConfigured(cp, "until-no-high", 3);
|
|
500
|
-
assert.throws(() => applyImplementationReviewConfigured(cp, "until-no-high", 3), StateError);
|
|
501
|
-
});
|
|
502
|
-
});
|
package/tools/analyze-refs.ts
CHANGED
|
@@ -1,9 +1,12 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* `analyze_refs` tool — plan-with-refs per-reference analysis via read-only Pi
|
|
3
3
|
* subagents with isolated context. One lane per reference (cwd = the ref's own
|
|
4
|
-
* directory), reusing the reviewer
|
|
5
|
-
* and the concurrent
|
|
6
|
-
*
|
|
4
|
+
* directory), reusing the reviewer model confirmation from the GLOBAL config
|
|
5
|
+
* (`~/.pi/pi-plans/config.json`) and the concurrent overlay (title "Refs").
|
|
6
|
+
* analyze_refs is spawn-only by nature, so the reviewer MODE is deliberately
|
|
7
|
+
* not consulted here (Q-4=B): a current-session reviewer still gets spawned
|
|
8
|
+
* ref-analyst lanes, with a one-time notice in the result. Batches are capped
|
|
9
|
+
* at three concurrent lanes; larger ref sets run as sequential batches.
|
|
7
10
|
*
|
|
8
11
|
* Recording is best-effort: spawns land in `subagents.jsonl` (role
|
|
9
12
|
* `ref-analyst`) only when an active planning run exists. Analysis output is
|
|
@@ -17,7 +20,18 @@ import { Text } from "@earendil-works/pi-tui";
|
|
|
17
20
|
import { Type } from "typebox";
|
|
18
21
|
import * as fs from "node:fs";
|
|
19
22
|
import * as path from "node:path";
|
|
20
|
-
import {
|
|
23
|
+
import {
|
|
24
|
+
loadConfig,
|
|
25
|
+
normalizeWorkdir,
|
|
26
|
+
readActive,
|
|
27
|
+
recordSubagent,
|
|
28
|
+
resolveEffectiveReviewer,
|
|
29
|
+
resolveGlobalConfigPath,
|
|
30
|
+
resolveStateRootOrNull,
|
|
31
|
+
StateError,
|
|
32
|
+
} from "../src/state.ts";
|
|
33
|
+
import { runFirstUseFlow, firstUseCancelledError, firstUseTextGuidance, availableModels, findModel, type RolePanelHost } from "../src/role-panels.ts";
|
|
34
|
+
import { roleModelLabel } from "../src/thinking-levels.ts";
|
|
21
35
|
import type { SubagentUsage } from "../src/subagent.ts";
|
|
22
36
|
import { resolveActiveRun } from "../src/run-context.ts";
|
|
23
37
|
import { buildRefAnalystTask, type RefAnalystTaskInput } from "../src/refine-prompts.ts";
|
|
@@ -45,25 +59,39 @@ const AnalyzeRefsParams = Type.Object({
|
|
|
45
59
|
workdir: Type.Optional(Type.String({ description: "Target workspace; default current working directory" })),
|
|
46
60
|
});
|
|
47
61
|
|
|
48
|
-
function gateError(problem: "state" | "
|
|
62
|
+
function gateError(problem: "state" | "confirm", guidance?: string): StateError {
|
|
49
63
|
if (problem === "state") {
|
|
50
64
|
return new StateError("no pi-plans state found; run the plans tool (action: init) first");
|
|
51
65
|
}
|
|
52
|
-
if (problem === "mode") {
|
|
53
|
-
return new StateError(
|
|
54
|
-
"The reviewer role mode is missing or invalid in .git/pi_plans/config.json (analyze_refs reuses the reviewer gates). Ask the role-setting question with ask_choice first: 1. Delegated subagent (recommended; read-only pi subprocess with isolated context) 2. Current session 3. Other 4. Auto-complete — then persist with the plans tool (set-role, role=reviewer).",
|
|
55
|
-
);
|
|
56
|
-
}
|
|
57
|
-
if (problem === "current-session") {
|
|
58
|
-
return new StateError(
|
|
59
|
-
"The reviewer role mode is current-session, but analyze_refs only spawns delegated read-only subagents (one per reference). Ask the user to switch the reviewer mode to delegated-subagent via ask_choice, persist with the plans tool (set-role, role=reviewer, mode=delegated-subagent), then retry analyze_refs.",
|
|
60
|
-
);
|
|
61
|
-
}
|
|
62
66
|
return new StateError(
|
|
63
|
-
|
|
67
|
+
guidance ??
|
|
68
|
+
firstUseTextGuidance([], resolveGlobalConfigPath()),
|
|
64
69
|
);
|
|
65
70
|
}
|
|
66
71
|
|
|
72
|
+
/** First-use model confirmation for the spawn-only ref-analyst path: native
|
|
73
|
+
* panels in TUI, menus for hasUI non-TUI, embedded text guidance otherwise.
|
|
74
|
+
* The reviewer MODE is not consulted (Q-4=B), but because analysis always
|
|
75
|
+
* spawns, a confirmed CONCRETE model is required even when the stored mode
|
|
76
|
+
* is current-session (model confirmation “as usual”). */
|
|
77
|
+
async function ensureRefAnalystModelReady(
|
|
78
|
+
host: RolePanelHost,
|
|
79
|
+
role: { mode: string; model_selector: string | null; thinking_level: string | null; confirmed_at: string | null },
|
|
80
|
+
): Promise<{ mode: string; model_selector: string | null; thinking_level: string | null; confirmed_at: string | null; name_prefix: string }> {
|
|
81
|
+
if (role.confirmed_at !== null && role.model_selector !== null) return role as never;
|
|
82
|
+
let outcome = await runFirstUseFlow(host, role.thinking_level);
|
|
83
|
+
if (outcome.status === "confirmed" && outcome.model_selector !== null && availableModels(host).length > 0 && findModel(host, outcome.model_selector) === null) {
|
|
84
|
+
// F-008: a manually entered selector that the registry does not know —
|
|
85
|
+
// one re-pick, then let spawn-side errors surface precisely.
|
|
86
|
+
outcome = await runFirstUseFlow(host, outcome.role.thinking_level);
|
|
87
|
+
}
|
|
88
|
+
if (outcome.status === "cancelled") throw firstUseCancelledError("analyze_refs");
|
|
89
|
+
const guidance = firstUseTextGuidance(availableModels(host), resolveGlobalConfigPath());
|
|
90
|
+
if (outcome.status === "unavailable") throw gateError("confirm", guidance);
|
|
91
|
+
if (outcome.role.model_selector === null) throw gateError("confirm", guidance);
|
|
92
|
+
return outcome.role;
|
|
93
|
+
}
|
|
94
|
+
|
|
67
95
|
interface AnalysisJob {
|
|
68
96
|
input: RefAnalystTaskInput;
|
|
69
97
|
name: string;
|
|
@@ -72,14 +100,14 @@ interface AnalysisJob {
|
|
|
72
100
|
missing: string | null;
|
|
73
101
|
}
|
|
74
102
|
|
|
75
|
-
export function registerAnalyzeRefsTool(
|
|
103
|
+
export function registerAnalyzeRefsTool(ext: ExtensionAPI, baseDir: string): void {
|
|
76
104
|
const agentPrompt = stripFrontmatter(fs.readFileSync(path.join(baseDir, "agents", "ref-analyst.md"), "utf8"));
|
|
77
105
|
|
|
78
|
-
|
|
106
|
+
ext.registerTool({
|
|
79
107
|
name: "analyze_refs",
|
|
80
108
|
label: "Analyze Refs",
|
|
81
109
|
description:
|
|
82
|
-
"plan-with-refs: analyze downloaded references via independent read-only Pi subagents — one lane per reference (cwd = the ref directory), reusing the reviewer
|
|
110
|
+
"plan-with-refs: analyze downloaded references via independent read-only Pi subagents — one lane per reference (cwd = the ref directory), reusing the reviewer model confirmation from the global config and the concurrent overlay. Batches of at most 3 lanes run sequentially; results are structured per-reference sections for REF_ANALYSIS.md. Recording into subagents.jsonl is best-effort (active run only); refs.jsonl stays owned by the main agent via the plans record-ref action. The reviewer mode is not consulted (spawn-only); first use pops native model/effort panels in TUI.",
|
|
83
111
|
promptSnippet: "Analyze plan-with-refs references with per-ref read-only subagents",
|
|
84
112
|
parameters: AnalyzeRefsParams,
|
|
85
113
|
|
|
@@ -92,17 +120,22 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
|
|
|
92
120
|
throw gateError("state");
|
|
93
121
|
}
|
|
94
122
|
const config = loadConfig(root);
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
if (reviewer.mode === "current-session") {
|
|
100
|
-
throw gateError("current-session");
|
|
101
|
-
}
|
|
102
|
-
if (reviewer.confirmed_at === null) {
|
|
103
|
-
throw gateError("confirm");
|
|
123
|
+
|
|
124
|
+
// F-005: cheap validations BEFORE any first-use panel.
|
|
125
|
+
if (params.refs.length === 0) {
|
|
126
|
+
throw new StateError("analyze_refs requires at least one reference");
|
|
104
127
|
}
|
|
105
128
|
|
|
129
|
+
// Effective reviewer from the global config (mode NOT consulted —
|
|
130
|
+
// analyze_refs is spawn-only, Q-4=B; a notice surfaces when the stored
|
|
131
|
+
// mode is current-session so the switch is never silent).
|
|
132
|
+
const { reviewer: initialReviewer } = resolveEffectiveReviewer(root);
|
|
133
|
+
const modeIgnoredNotice =
|
|
134
|
+
initialReviewer.mode === "current-session"
|
|
135
|
+
? `note: the reviewer mode is ${initialReviewer.mode}, but analyze_refs always spawns read-only subagents; the mode is ignored here and unchanged.`
|
|
136
|
+
: null;
|
|
137
|
+
const reviewer = await ensureRefAnalystModelReady(ctx as unknown as RolePanelHost, initialReviewer);
|
|
138
|
+
|
|
106
139
|
// Resolve refs and validate directories up front; missing ones become
|
|
107
140
|
// FAILED sections instead of aborting the whole batch.
|
|
108
141
|
const active = resolveActiveRun(ctx.sessionManager, workdir);
|
|
@@ -122,7 +155,7 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
|
|
|
122
155
|
}
|
|
123
156
|
|
|
124
157
|
const model = reviewer.model_selector ?? (ctx.model ? `${ctx.model.provider}/${ctx.model.id}` : undefined);
|
|
125
|
-
const modelLabel = model ?? "inherit";
|
|
158
|
+
const modelLabel = roleModelLabel(model ?? "inherit", reviewer.thinking_level);
|
|
126
159
|
let languageTag: string | null = null;
|
|
127
160
|
if (active) {
|
|
128
161
|
try {
|
|
@@ -140,6 +173,7 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
|
|
|
140
173
|
role: "ref-analyst",
|
|
141
174
|
name,
|
|
142
175
|
model: okModel ?? model ?? null,
|
|
176
|
+
thinking_level: reviewer.thinking_level,
|
|
143
177
|
// I-010: meter subagent token/cost for benchmark accounting.
|
|
144
178
|
usage: usage
|
|
145
179
|
? { input: usage.input, output: usage.output, cache_read: usage.cacheRead, cache_write: usage.cacheWrite, cost: usage.cost }
|
|
@@ -160,6 +194,7 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
|
|
|
160
194
|
task: buildRefAnalystTask({ ...job.input, languageTag }),
|
|
161
195
|
cwd: job.dir,
|
|
162
196
|
model,
|
|
197
|
+
thinkingLevel: reviewer.thinking_level ?? undefined,
|
|
163
198
|
tools: READ_ONLY_TOOLS,
|
|
164
199
|
signal: relay.signal,
|
|
165
200
|
onProgress: (event) => overlay?.update(job.laneId, event),
|
|
@@ -224,7 +259,7 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
|
|
|
224
259
|
|
|
225
260
|
if (failures === jobs.length) {
|
|
226
261
|
throw new Error(
|
|
227
|
-
`all reference analysis subagents failed (${failures}/${jobs.length})${model ? `\nIf the model selector "${model}" is unavailable, reset the reviewer confirmation (plans set-role, role=reviewer, resetConfirmation: true)
|
|
262
|
+
`all reference analysis subagents failed (${failures}/${jobs.length})${model ? `\nIf the model selector "${model}" is unavailable, reset the reviewer confirmation (plans set-role, role=reviewer, resetConfirmation: true) — the next analyze_refs opens the native model panel to re-confirm.` : ""}`,
|
|
228
263
|
);
|
|
229
264
|
}
|
|
230
265
|
|
|
@@ -237,13 +272,13 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
|
|
|
237
272
|
content: [
|
|
238
273
|
{
|
|
239
274
|
type: "text",
|
|
240
|
-
text: `${text}\n\n---\nPersist: paste each reference's analysis into REF_ANALYSIS.md, call the plans tool (record-ref) per reference with coverage and gaps filled from the analysis, then ask at least three ref-specific adoption questions per reference with ask_choice before using its ideas in PLAN_v1.md.`,
|
|
275
|
+
text: `${modeIgnoredNotice ? `${modeIgnoredNotice}\n\n` : ""}${text}\n\n---\nPersist: paste each reference's analysis into REF_ANALYSIS.md, call the plans tool (record-ref) per reference with coverage and gaps filled from the analysis, then ask at least three ref-specific adoption questions per reference with ask_choice before using its ideas in PLAN_v1.md.`,
|
|
241
276
|
},
|
|
242
277
|
],
|
|
243
278
|
details: {
|
|
244
279
|
mode: "delegated-subagent",
|
|
245
280
|
role: "ref-analyst",
|
|
246
|
-
reviewerGates: { mode:
|
|
281
|
+
reviewerGates: { mode: initialReviewer.mode, modeIgnored: modeIgnoredNotice !== null, model, thinkingLevel: reviewer.thinking_level },
|
|
247
282
|
batches: Math.ceil(jobs.length / BATCH_SIZE),
|
|
248
283
|
model,
|
|
249
284
|
outputs,
|
package/tools/ask-choice.ts
CHANGED
|
@@ -16,13 +16,6 @@ import { Text } from "@earendil-works/pi-tui";
|
|
|
16
16
|
import { Type } from "typebox";
|
|
17
17
|
import { disableAutoComplete, enableAutoComplete, isAutoCompleteEnabled, recordAskChoice } from "../src/autocomplete.ts";
|
|
18
18
|
import { assertAutoApprovable, isAutoApproveEnabled } from "../src/auto-approve.ts";
|
|
19
|
-
import {
|
|
20
|
-
TERMINATION_QUESTION,
|
|
21
|
-
TERMINATION_OPTIONS,
|
|
22
|
-
TERMINATION_RECORDING_INSTRUCTIONS,
|
|
23
|
-
implReviewerCountPromptLine,
|
|
24
|
-
renderTerminationOptions,
|
|
25
|
-
} from "../src/termination-prompt.ts";
|
|
26
19
|
import { truncateToWidth, visibleWidth } from "../src/refine-ui-helpers.ts";
|
|
27
20
|
import { stripRecommendedMarker } from "../src/ask-form.ts";
|
|
28
21
|
import {
|
|
@@ -63,8 +56,8 @@ export const FALLBACK_ROWS = 30;
|
|
|
63
56
|
/** Minimal-form floor for tiny terminals (stage-3 width). */
|
|
64
57
|
const MINIMAL_LINE_WIDTH = 20;
|
|
65
58
|
/**
|
|
66
|
-
* Truncation floor for fixed tail labels (Other…/Auto-complete
|
|
67
|
-
*
|
|
59
|
+
* Truncation floor for fixed tail labels (Other…/Auto-complete): the
|
|
60
|
+
* longest magic prefix ("Auto-complete", 14 cols) plus slack.
|
|
68
61
|
* These labels drive startsWith() answer routing and must never lose it.
|
|
69
62
|
*/
|
|
70
63
|
const FIXED_LABEL_FLOOR = 18;
|
|
@@ -74,7 +67,7 @@ export interface PanelItem {
|
|
|
74
67
|
core: string;
|
|
75
68
|
/** Full display label: core + description (degradation stage 0). */
|
|
76
69
|
display: string;
|
|
77
|
-
/** Fixed tail labels (Other…/Auto-complete
|
|
70
|
+
/** Fixed tail labels (Other…/Auto-complete): truncation keeps at least the magic prefix. */
|
|
78
71
|
fixed?: boolean;
|
|
79
72
|
}
|
|
80
73
|
|
|
@@ -156,7 +149,12 @@ export function fitAskChoicePanel(question: string, items: PanelItem[], columns:
|
|
|
156
149
|
export const Option = Type.Object(
|
|
157
150
|
{
|
|
158
151
|
label: Type.String({ description: "Option label" }),
|
|
159
|
-
description: Type.Optional(
|
|
152
|
+
description: Type.Optional(
|
|
153
|
+
Type.String({
|
|
154
|
+
description:
|
|
155
|
+
"REQUIRED on every option you author: '✓ <advantage> / ✗ <drawback>' — the user compares options side by side, so each one must state what it gains AND what it costs. Write BOTH halves in this single description string, in the configured language, tersely (≈8 words per half). If a side is genuinely absent write '—' rather than dropping it. Do NOT invent separate pros/cons fields: Option accepts no other keys.",
|
|
156
|
+
}),
|
|
157
|
+
),
|
|
160
158
|
recommended: Type.Optional(Type.Boolean({ description: "Mark exactly one recommended option; put it first. Never embed (推荐)/(recommended) text in labels — the UI renders the ★ marker automatically" })),
|
|
161
159
|
},
|
|
162
160
|
{ additionalProperties: false },
|
|
@@ -165,7 +163,10 @@ export const Option = Type.Object(
|
|
|
165
163
|
export const BatchQuestionParams = Type.Object(
|
|
166
164
|
{
|
|
167
165
|
question: Type.String({ description: "The question to ask, in the configured language" }),
|
|
168
|
-
options: Type.Array(Option, {
|
|
166
|
+
options: Type.Array(Option, {
|
|
167
|
+
description:
|
|
168
|
+
"Ordered options: recommended first, alternatives next. Every option's description states its advantage AND its drawback as '✓ <advantage> / ✗ <drawback>' in the configured language. Do not include Other or Auto-complete yourself.",
|
|
169
|
+
}),
|
|
169
170
|
allowOther: Type.Optional(Type.Boolean({ description: "Offer free-form input for this question (default true)" })),
|
|
170
171
|
autoComplete: Type.Optional(
|
|
171
172
|
Type.Boolean({
|
|
@@ -182,7 +183,7 @@ export const BatchQuestionParams = Type.Object(
|
|
|
182
183
|
export const AskChoiceParams = Type.Object(
|
|
183
184
|
{
|
|
184
185
|
question: Type.Optional(Type.String({ description: "The single question to ask, in the configured language (mutually exclusive with questions)." })),
|
|
185
|
-
options: Type.Optional(Type.Array(Option, { description: "Ordered options (single-question form): recommended first, alternatives next. Do not include Other or Auto-complete yourself." })),
|
|
186
|
+
options: Type.Optional(Type.Array(Option, { description: "Ordered options (single-question form): recommended first, alternatives next. Every option's description states its advantage AND its drawback as '✓ <advantage> / ✗ <drawback>' in the configured language. Do not include Other or Auto-complete yourself." })),
|
|
186
187
|
questions: Type.Optional(
|
|
187
188
|
Type.Array(BatchQuestionParams, {
|
|
188
189
|
description:
|
|
@@ -205,12 +206,6 @@ export const AskChoiceParams = Type.Object(
|
|
|
205
206
|
purpose: Type.Optional(
|
|
206
207
|
Type.String({ description: "Short machine-readable purpose (e.g. 'scope', 'termination-condition')." }),
|
|
207
208
|
),
|
|
208
|
-
trailing: Type.Optional(
|
|
209
|
-
StringEnum(["auto-refine-loop"] as const, {
|
|
210
|
-
description:
|
|
211
|
-
'Replace the trailing Auto-complete option with "Auto-refine loop" (post-execution amelioration prompt). Selecting it returns instructions to ask the rounds/termination follow-up; Auto-complete is suppressed entirely for this question.',
|
|
212
|
-
}),
|
|
213
|
-
),
|
|
214
209
|
workdir: Type.Optional(Type.String({ description: "Target workspace; default current working directory" })),
|
|
215
210
|
},
|
|
216
211
|
{ additionalProperties: false },
|
|
@@ -581,16 +576,17 @@ function formatBatchAnswers(batch: NonNullable<AskChoiceDetails["batch"]>): stri
|
|
|
581
576
|
}
|
|
582
577
|
const NL = "\n";
|
|
583
578
|
|
|
584
|
-
export function registerAskChoiceTool(
|
|
585
|
-
|
|
579
|
+
export function registerAskChoiceTool(ext: ExtensionAPI): void {
|
|
580
|
+
ext.registerTool({
|
|
586
581
|
name: "ask_choice",
|
|
587
582
|
label: "Ask Choice",
|
|
588
583
|
description:
|
|
589
|
-
"Ask the user planning or refinement questions as numbered choice prompts: recommended option first, alternatives next, then Other and Auto-complete. Two shapes: questions: [...] (2-8 questions) opens ONE tabbed multiple-choice form with a submit page — use it to batch a round of questions (≤8), then think about the answers and follow up in later calls (phased questioning stays agent-driven); question + options asks one question at a time (classic flow). Use ask_choice for every user-facing planning question, the final scope confirmation, refinement-mode questions, language/role/model settings, and the execution handoff. Scope confirmation and the execution handoff MUST stay single-question calls (autoComplete: false); batches reject autoComplete: false items and the
|
|
584
|
+
"Ask the user planning or refinement questions as numbered choice prompts: recommended option first, alternatives next, then Other and Auto-complete. Two shapes: questions: [...] (2-8 questions) opens ONE tabbed multiple-choice form with a submit page — use it to batch a round of questions (≤8), then think about the answers and follow up in later calls (phased questioning stays agent-driven); question + options asks one question at a time (classic flow). Use ask_choice for every user-facing planning question, the final scope confirmation, refinement-mode questions, language/role/model settings, and the execution handoff. Scope confirmation and the execution handoff MUST stay single-question calls (autoComplete: false); batches reject autoComplete: false items and the questionIds reserved for handoff. EVERY option you author — including the accept/execute handoff — must set description to '✓ <advantage> / ✗ <drawback>' in the configured language, so the user can see what each option gains and what it costs. Other and Auto-complete are appended by this tool and need no description.",
|
|
590
585
|
promptSnippet: "Ask structured planning questions with recommended/Other/Auto-complete ordering; batch ≤8 questions per form",
|
|
591
586
|
promptGuidelines: [
|
|
592
587
|
"Use ask_choice for every pi-plans question to the user instead of plain-text questions; it enforces option ordering and records decisions.",
|
|
593
588
|
"Batch a round's questions into one ask_choice call (questions: [...], 2-8 items) instead of asking one at a time, then think after the answers and follow up with later calls. Scope confirmation and execution handoff are always separate single-question calls (autoComplete: false).",
|
|
589
|
+
"Give every option you author a description of the form '✓ <advantage> / ✗ <drawback>' — the user's whole point is seeing what each option wins and what it costs, in the configured language. Keep each half terse (~8 words). Put both halves in the description string; there are no separate pros/cons fields, and Other/Auto-complete are added by the tool.",
|
|
594
590
|
],
|
|
595
591
|
parameters: AskChoiceParams,
|
|
596
592
|
executionMode: "sequential",
|
|
@@ -604,9 +600,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
|
|
|
604
600
|
if (params.question !== undefined || params.options !== undefined) {
|
|
605
601
|
throw new Error("ask_choice accepts either question+options or questions, not both");
|
|
606
602
|
}
|
|
607
|
-
if (params.trailing !== undefined) {
|
|
608
|
-
throw new Error("ask_choice batch mode does not support trailing (single-question only)");
|
|
609
|
-
}
|
|
610
603
|
return executeAskChoiceBatch({ questions: params.questions, workdir: params.workdir }, ctx);
|
|
611
604
|
}
|
|
612
605
|
const workdir = normalizeWorkdir(params.workdir ?? ctx.cwd);
|
|
@@ -645,10 +638,7 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
|
|
|
645
638
|
/* the decisions ledger already holds the answer; F-005 reconcile covers the gap */
|
|
646
639
|
}
|
|
647
640
|
};
|
|
648
|
-
|
|
649
|
-
// so an erroneously passed autoComplete flag is suppressed here.
|
|
650
|
-
const trailing = params.trailing;
|
|
651
|
-
const autoComplete = (params.autoComplete ?? true) && trailing === undefined;
|
|
641
|
+
const autoComplete = params.autoComplete ?? true;
|
|
652
642
|
const options = params.options;
|
|
653
643
|
if (options.length === 0) throw new Error("ask_choice requires at least one option");
|
|
654
644
|
const recommended = options.find((option) => option.recommended) ?? options[0];
|
|
@@ -748,8 +738,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
|
|
|
748
738
|
};
|
|
749
739
|
}
|
|
750
740
|
|
|
751
|
-
const AUTO_REFINE_LOOP_LABEL =
|
|
752
|
-
"Auto-refine loop (run refinement rounds until no high-severity finding or the 5-round cap)";
|
|
753
741
|
const panelItems: PanelItem[] = options.map((option, index) => {
|
|
754
742
|
const label = stripRecommendedMarker(option.label);
|
|
755
743
|
const isRec = option === recommended;
|
|
@@ -761,7 +749,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
|
|
|
761
749
|
});
|
|
762
750
|
if (allowOther) panelItems.push({ core: "Other… (type your own answer)", display: "Other… (type your own answer)", fixed: true });
|
|
763
751
|
if (autoComplete) panelItems.push({ core: "Auto-complete (take the recommended option)", display: "Auto-complete (take the recommended option)", fixed: true });
|
|
764
|
-
else if (trailing) panelItems.push({ core: AUTO_REFINE_LOOP_LABEL, display: AUTO_REFINE_LOOP_LABEL, fixed: true });
|
|
765
752
|
|
|
766
753
|
const panel = fitAskChoicePanel(
|
|
767
754
|
params.question,
|
|
@@ -804,23 +791,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
|
|
|
804
791
|
};
|
|
805
792
|
}
|
|
806
793
|
|
|
807
|
-
if (trailing && selected.startsWith("Auto-refine loop")) {
|
|
808
|
-
recordAskChoice(ctx, false);
|
|
809
|
-
record("Auto-refine loop", "user");
|
|
810
|
-
// Skill-aware reviewer-count default (D-1/D-4): same mapping the
|
|
811
|
-
// goal-running continuation in src/exec.ts renders.
|
|
812
|
-
const activeSkill = resolveActiveRun(ctx.sessionManager, workdir)?.skill;
|
|
813
|
-
return {
|
|
814
|
-
content: [
|
|
815
|
-
{
|
|
816
|
-
type: "text",
|
|
817
|
-
text: `User selected Auto-refine loop. Immediately ask the follow-up with ask_choice (autoComplete: false, in the session language): "${TERMINATION_QUESTION}" Options (recommended first): ${renderTerminationOptions()}. ${TERMINATION_RECORDING_INSTRUCTIONS} ${implReviewerCountPromptLine(activeSkill)} Then run the loop per the completion instructions: each round calls refine (role: "reviewer", target: "implementation", reviewers: <configured reviewerCount>), accepts findings on evidence, applies fixes, re-runs relevant tests, and continues until the chosen termination condition — the goal-wait option keeps the loop running until no unpassed VCs remain.`,
|
|
818
|
-
},
|
|
819
|
-
],
|
|
820
|
-
details: details("Auto-refine loop", "user"),
|
|
821
|
-
};
|
|
822
|
-
}
|
|
823
|
-
|
|
824
794
|
if (allowOther && selected.startsWith("Other…")) {
|
|
825
795
|
const typed = await ctx.ui.input(`${params.question} — your answer:`);
|
|
826
796
|
if (typed === undefined || !typed.trim()) {
|
package/tools/code-graph.ts
CHANGED
|
@@ -128,8 +128,8 @@ export async function ensureRuntime(workdir: string, ctx: CodeGraphContext): Pro
|
|
|
128
128
|
return { entry: runtimeCache, status };
|
|
129
129
|
}
|
|
130
130
|
|
|
131
|
-
export function registerCodeGraphTool(
|
|
132
|
-
|
|
131
|
+
export function registerCodeGraphTool(ext: ExtensionAPI): void {
|
|
132
|
+
ext.registerTool({
|
|
133
133
|
name: "code_graph",
|
|
134
134
|
label: "Code Graph",
|
|
135
135
|
description:
|