pi-plans 0.5.7 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +126 -0
- package/README.md +49 -39
- package/agents/ref-analyst.md +7 -4
- package/agents/reviewer.md +12 -3
- package/index.ts +74 -40
- package/package.json +2 -1
- package/references/pi-planning-workflow.md +45 -58
- package/references/plan-artifact-template.md +71 -60
- package/references/state-and-config.md +63 -47
- package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
- package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
- package/scripts/run-tests.ts +12 -1
- package/scripts/validate.ts +22 -10
- package/skills/debug-and-plan/SKILL.md +4 -4
- package/skills/plan-big/SKILL.md +5 -5
- package/skills/plan-normal/SKILL.md +5 -5
- package/skills/plan-small/SKILL.md +5 -5
- package/skills/plan-with-refs/SKILL.md +8 -8
- package/skills/planning/SKILL.md +1 -1
- package/src/ask-form.ts +4 -4
- package/src/auditor.ts +126 -0
- package/src/auto-approve.ts +1 -1
- package/src/autocomplete.ts +19 -17
- package/src/code-graph/commands.ts +2 -2
- package/src/code-graph/community.ts +1 -1
- package/src/code-graph/paths.ts +1 -1
- package/src/code-graph/watch.ts +2 -2
- package/src/compaction.ts +3 -3
- package/src/config-command.ts +154 -76
- package/src/dashboard.ts +257 -0
- package/src/exec.ts +709 -705
- package/src/global-state.ts +304 -0
- package/src/guard.ts +16 -3
- package/src/messaging.ts +44 -0
- package/src/plan.ts +421 -112
- package/src/query-hook.ts +4 -4
- package/src/refine-prompts.ts +14 -72
- package/src/refine-ui-helpers.ts +24 -5
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +1 -1
- package/src/resume-command.ts +40 -130
- package/src/resume.ts +15 -17
- package/src/role-panels.ts +542 -0
- package/src/run-context.ts +5 -4
- package/src/run-picker.ts +98 -0
- package/src/state.ts +380 -77
- package/src/subagent.ts +32 -1
- package/src/task-tool.ts +100 -0
- package/src/tasks.ts +189 -0
- package/src/thinking-levels.ts +67 -0
- package/src/ui-language.ts +3 -54
- package/src/workflow-state.ts +78 -57
- package/tests/analyze-refs.test.ts +35 -18
- package/tests/ask-choice-pros-cons.test.ts +147 -0
- package/tests/ask-choice-schema.test.ts +0 -12
- package/tests/ask-choice.test.ts +2 -49
- package/tests/ask-form-tool.test.ts +4 -5
- package/tests/ask-form.test.ts +2 -2
- package/tests/auditor.test.ts +111 -0
- package/tests/auto-approve.test.ts +7 -10
- package/tests/autocomplete.test.ts +8 -11
- package/tests/code-graph-apply-action.test.ts +2 -2
- package/tests/code-graph-commands.test.ts +2 -2
- package/tests/code-graph-index.test.ts +2 -2
- package/tests/code-graph-loop.e2e.test.ts +1 -1
- package/tests/code-graph-mutations.test.ts +1 -1
- package/tests/code-graph-rollback.test.ts +1 -1
- package/tests/code-graph-v05.test.ts +2 -2
- package/tests/compaction.test.ts +1 -1
- package/tests/config-command.test.ts +103 -100
- package/tests/dashboard.test.ts +268 -0
- package/tests/exec-lifecycle.test.ts +181 -115
- package/tests/exec-panel-lifecycle.test.ts +106 -251
- package/tests/exec.test.ts +617 -1706
- package/tests/execute-plan.test.ts +44 -19
- package/tests/extension-load.test.ts +48 -0
- package/tests/global-state.test.ts +371 -0
- package/tests/graph-aware-file-tools.test.ts +5 -5
- package/tests/guard.test.ts +1 -1
- package/tests/multi-run.test.ts +184 -0
- package/tests/plan.test.ts +139 -62
- package/tests/plans.test.ts +7 -79
- package/tests/refine-prompts.test.ts +20 -71
- package/tests/refine-resume.test.ts +27 -22
- package/tests/refine-ui.test.ts +6 -15
- package/tests/resume-lifecycle.test.ts +37 -22
- package/tests/resume.test.ts +43 -88
- package/tests/role-panels.test.ts +391 -0
- package/tests/run-context.test.ts +1 -1
- package/tests/run-ownership.test.ts +1 -1
- package/tests/stale-ctx.test.ts +218 -0
- package/tests/state.test.ts +151 -32
- package/tests/subagent-thinking.test.ts +65 -0
- package/tests/subagent-usage.test.ts +1 -1
- package/tests/task-tool.test.ts +61 -0
- package/tests/thinking-levels.test.ts +77 -0
- package/tests/ui-language.test.ts +2 -17
- package/tests/workflow-state.test.ts +17 -99
- package/tools/analyze-refs.ts +67 -32
- package/tools/ask-choice.ts +19 -49
- package/tools/code-graph.ts +2 -2
- package/tools/execute-plan.ts +63 -33
- package/tools/graph-aware-file-tools.ts +6 -4
- package/tools/plans.ts +40 -66
- package/tools/refine.ts +101 -164
- package/agents/criticizer.md +0 -18
- package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
- package/src/panel.ts +0 -473
- package/src/termination-prompt.ts +0 -73
- package/tests/goal-wait.test.ts +0 -269
- package/tests/panel-i-zero.test.ts +0 -420
- package/tests/panel.test.ts +0 -355
|
@@ -3,79 +3,90 @@
|
|
|
3
3
|
Status: draft | reviewed | accepted
|
|
4
4
|
Plan version: N
|
|
5
5
|
Artifact directory: `<artifact_root>/YYYY-MM-DD-topic/`
|
|
6
|
-
State directory: `.git/
|
|
6
|
+
State directory: `.git/pi-plans/runs/<run-id>/` (resolved git common dir)
|
|
7
7
|
Language: `<BCP47 tag>`
|
|
8
8
|
|
|
9
9
|
## Original Request
|
|
10
10
|
|
|
11
|
-
|
|
11
|
+
One paragraph summarizing the user's request.
|
|
12
12
|
|
|
13
|
-
##
|
|
13
|
+
## Tasks
|
|
14
14
|
|
|
15
|
-
- `
|
|
15
|
+
- `Task-1`: <title> — deps: <Task-ids, optional>; files: <paths, optional>; wave: <number, optional>
|
|
16
|
+
- `Task-2`: <title> — deps: Task-1; files: src/a.ts, src/b.ts; wave: 2
|
|
17
|
+
- `Task-2.1`: <subtask title> — deps: Task-1; files: src/a.ts
|
|
18
|
+
- `Task-3`: <title> — deps: Task-1, Task-2; files: src/c.ts; wave: 3
|
|
16
19
|
|
|
17
|
-
|
|
20
|
+
### Execution Waves
|
|
18
21
|
|
|
19
|
-
-
|
|
22
|
+
- wave 1: Task-1 — <why these can run first / in parallel>
|
|
23
|
+
- wave 2: Task-2 — <files disjoint within the wave; deps satisfied by earlier waves>
|
|
24
|
+
- wave 3: Task-3 — <serial finish>
|
|
20
25
|
|
|
21
|
-
##
|
|
26
|
+
## Verification Checks
|
|
22
27
|
|
|
23
|
-
- `
|
|
24
|
-
- `
|
|
25
|
-
|
|
26
|
-
## Repo Evidence
|
|
27
|
-
|
|
28
|
-
- `E-REPO-001`: Path or command inspected, what it proves, and any uncertainty.
|
|
29
|
-
|
|
30
|
-
## External Evidence
|
|
31
|
-
|
|
32
|
-
- `E-EXT-001`: URL or local ref path, what it supports, and date accessed.
|
|
33
|
-
|
|
34
|
-
## Resolved Decisions
|
|
35
|
-
|
|
36
|
-
- `D-001`: Question, chosen answer, answer source (user | Auto-complete), and rationale.
|
|
37
|
-
|
|
38
|
-
## Requirements
|
|
39
|
-
|
|
40
|
-
- `R-001`: Requirement tied to goals and decisions.
|
|
41
|
-
|
|
42
|
-
## Constraints
|
|
43
|
-
|
|
44
|
-
- `C-001`: Compatibility, style, interface, performance, safety, or ownership constraint.
|
|
45
|
-
|
|
46
|
-
## Implementation Items
|
|
47
|
-
|
|
48
|
-
- `I-001`: Work item with affected paths, dependencies, and expected code or doc changes.
|
|
49
|
-
|
|
50
|
-
## Acceptance Criteria
|
|
51
|
-
|
|
52
|
-
- `AC-001`: Observable result tied to one or more requirements.
|
|
53
|
-
|
|
54
|
-
## Verification Plan
|
|
55
|
-
|
|
56
|
-
- `V-001`: Command, manual check, screenshot, log review, or static inspection required after implementation.
|
|
57
|
-
|
|
58
|
-
## Verifier Checklist
|
|
59
|
-
|
|
60
|
-
- [ ] `VC-001` covers `I-001`; pass condition: describe pass condition; evidence: describe expected evidence; metric: threshold or reason not quantified.
|
|
61
|
-
|
|
62
|
-
## Risks And Mitigations
|
|
63
|
-
|
|
64
|
-
- `Risk-001`: Risk and mitigation.
|
|
65
|
-
|
|
66
|
-
## Refinement Settings
|
|
67
|
-
|
|
68
|
-
- Reviewer mode: `delegated-subagent | current-session`; model selector: `inherit | <selector>`.
|
|
69
|
-
- Criticizer mode: `delegated-subagent | current-session`; model selector: `inherit | <selector>`.
|
|
28
|
+
- [ ] `VC-001` covers `Task-1`; pass condition: <observable condition>; evidence: <expected evidence>; metric: <threshold or "not quantified">.
|
|
29
|
+
- [ ] `VC-002` covers `Task-2` and `Task-2.1`; pass condition: …; evidence: …; metric: ….
|
|
70
30
|
|
|
71
31
|
## Execution Handoff Notes
|
|
72
32
|
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
## Termination Recording (implementation review)
|
|
76
|
-
|
|
77
|
-
When the post-execution implementation-review loop starts, the termination question is asked with `ask_choice` using `questionId: "termination-condition"` and persisted via `plans record-checkpoint` (`transition: "implementation-review-configured"`). Each disposed round records `implementation-round-finished`; the loop closes with `completed` plus evidence. These records make `/resume-plans` continue the loop with its original condition and round count.
|
|
33
|
+
Ordering, files to avoid, verification commands, and anything the executor must know. The handoff still requires explicit user approval (`ask_choice` with `autoComplete: false`, then the `execute_plan` tool) and is never auto-completed. Once approved, execution mode tracks every task through the `plans_update_task` tool (status + evidence); when all tasks are terminal, the independent completion auditor verifies each check above before the run completes.
|
|
78
34
|
|
|
79
35
|
## Revision Ledger
|
|
80
36
|
|
|
81
|
-
- `PLAN_v1`:
|
|
37
|
+
- `PLAN_v1`: <one line per revision: what changed and why>.
|
|
38
|
+
|
|
39
|
+
---
|
|
40
|
+
|
|
41
|
+
## Format specification (normative)
|
|
42
|
+
|
|
43
|
+
The plan body is exactly two sections plus the metadata header shown above:
|
|
44
|
+
`## Original Request` (one paragraph), `## Tasks`, and `## Verification Checks`.
|
|
45
|
+
Everything else the workflow needs lives in the run state (decisions ledger,
|
|
46
|
+
review rounds, refs), not in the plan file. `## Execution Handoff Notes` and
|
|
47
|
+
`## Revision Ledger` are the two permitted auxiliary sections.
|
|
48
|
+
|
|
49
|
+
### Tasks microsyntax
|
|
50
|
+
|
|
51
|
+
- Top-level task line: ``- `Task-N`: <title> — deps: <ids>; files: <paths>; wave: <n>``.
|
|
52
|
+
The separator between title and metadata is an em dash `—` (tolerated: `——`, `--`, `–`).
|
|
53
|
+
With no separator the whole body is the title (no fields).
|
|
54
|
+
- Fields are separated by `;` (tolerated `;`); multi-values by `,`
|
|
55
|
+
(tolerated `,` `、`). All three fields are optional; missing `wave` values
|
|
56
|
+
derive from deps (1 + max(dep wave)), and the `### Execution Waves`
|
|
57
|
+
subsection wins over inline `wave:` on conflict (linted).
|
|
58
|
+
- Subtasks: one indented bullet level, ``- `Task-N.M`: <title> — …``. Only one
|
|
59
|
+
nesting level is valid; deeper ids lint as drift. Subtasks may carry
|
|
60
|
+
`deps:`/`files:`; the wave is inherited from the parent (inline `wave:` on a
|
|
61
|
+
subtask is ignored with a lint notice).
|
|
62
|
+
- `### Execution Waves` rows: ``- wave <n>: Task-1, Task-2 — <rationale>``.
|
|
63
|
+
The wave table is the authoritative parallel-execution order: within one
|
|
64
|
+
wave, top-level tasks' file sets must be disjoint (children roll up to their
|
|
65
|
+
parent), and every `deps:` target must sit in an earlier wave.
|
|
66
|
+
|
|
67
|
+
### Verification Checks microsyntax
|
|
68
|
+
|
|
69
|
+
- Row: ``- [ ] `VC-###` covers `Task-2` and `Task-3`; pass condition: …; evidence: …; metric: …``.
|
|
70
|
+
- `covers` accepts multiple targets and `Task-N.M` subtask ids; the clause ends
|
|
71
|
+
at the first `;`. Checks covering zero tasks never enter the completion
|
|
72
|
+
audit; a check whose covered tasks are ALL skipped passes as skipped-pass.
|
|
73
|
+
|
|
74
|
+
### Lint and compatibility
|
|
75
|
+
|
|
76
|
+
- Planning-time lint (`lintPlanIntoNotices`): zero parsed tasks under an
|
|
77
|
+
existing `## Tasks` header, over-deep subtasks, non-consecutive top-level
|
|
78
|
+
numbering, unknown dep/coverage targets, same-wave file overlaps,
|
|
79
|
+
deps inside the same-or-later wave, inline-vs-subsection wave conflicts.
|
|
80
|
+
Lint notices are advisory while planning and hard-rejected at the execution gate.
|
|
81
|
+
- Legacy compatibility: plans without `## Tasks` fall back to parsing
|
|
82
|
+
`## Implementation Items` (`I-001` → `Task-1`, `covers \`I-001\`` normalizes
|
|
83
|
+
the same way); checklist-only pre-0.5 artifacts synthesize one serial task
|
|
84
|
+
per check. The execution gate surfaces an upgrade notice on fallback.
|
|
85
|
+
|
|
86
|
+
### Refinement
|
|
87
|
+
|
|
88
|
+
One Reviewer role: each round returns findings (`F-###`) and up to five
|
|
89
|
+
questions (`Q-1..Q-5`). Default sequences: plan-big → one round of three
|
|
90
|
+
concurrent reviewers (questions included); plan-normal / plan-small → one
|
|
91
|
+
reviewer round. The main agent asks every question with `ask_choice` and
|
|
92
|
+
records the answers before revising the plan.
|
|
@@ -1,18 +1,20 @@
|
|
|
1
1
|
# State And Config
|
|
2
2
|
|
|
3
|
-
pi-plans stores
|
|
3
|
+
pi-plans stores planning preferences and run state in the target workspace's git directory as `<git-common-dir>/pi-plans/` — in an ordinary repository this is simply `.git/pi-plans/` — resolving the git common dir with `git rev-parse --git-common-dir` from the workspace. Because the state lives inside the git dir, git never tracks it and no `.gitignore` entries are needed. The target workspace is the current working directory unless the user explicitly names another repository.
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
The one workspace-independent piece is the **reviewer role** (v0.7.0): it lives in the global config `~/.pi/pi-plans/config.json` (override the directory with `PI_PLANS_GLOBAL_DIR`), confirmed once and shared across every workspace. It is a standalone pi-plans file, never Pi's own settings.
|
|
6
|
+
|
|
7
|
+
Do not store pi-plans preferences in Pi's own settings (`~/.pi/agent/settings.json`); pi-plans uses `.git/pi-plans/config.json` for workspace state and `~/.pi/pi-plans/config.json` for the reviewer role.
|
|
6
8
|
|
|
7
9
|
## State Root Resolution
|
|
8
10
|
|
|
9
11
|
- Git runs with `GIT_DIR`, `GIT_COMMON_DIR`, and `GIT_WORK_TREE` scrubbed from the environment, so leaked env vars cannot misdirect state into an unrelated repository. Relative results (`.git`, `../.git`) resolve against the workdir.
|
|
10
|
-
- Granularity is **per enclosing repository**: running from a subdirectory uses the enclosing repo's git dir (a one-line notice names that repo). Linked worktrees share one common dir; run directories are unique,
|
|
11
|
-
- State does not travel with clones: a fresh clone starts with empty state
|
|
12
|
+
- Granularity is **per enclosing repository**: running from a subdirectory uses the enclosing repo's git dir (a one-line notice names that repo). Linked worktrees share one common dir; run directories are unique, and since v0.6.0 the run registry is derived from `runs/` (no shared pointer to race). The legacy `active.json` is deprecated: reads fall back to it only when no `runs/` entries exist (pre-0.6.0 migration).
|
|
13
|
+
- State does not travel with clones: a fresh clone starts with empty state, and the default plan artifacts live in the git dir with it. Point `artifact_root` at `./docs/pi-plans` when you want the plans committed and public instead.
|
|
12
14
|
|
|
13
15
|
## Auto Git Init
|
|
14
16
|
|
|
15
|
-
When a mutating state action (`init`, `set-language`, `set-
|
|
17
|
+
When a mutating state action (`init`, `set-language`, `set-artifact-root`, `set-refs-root`, `set-graph-enabled`, `start-run`, record-*) runs in a workdir that is not a git repository, the helper auto-runs `git init` there (with a one-line notice) and then creates the state dir. It never creates commits. `set-role` is deliberately NOT in this list: it writes only the global config and never triggers auto-init or any workspace write. Auto-init runs only when ALL of the following hold:
|
|
16
18
|
|
|
17
19
|
- the workdir has no `.git` entry (a pre-existing `.git` file or directory that git cannot resolve is a fatal error, never a silent reinit);
|
|
18
20
|
- the workdir is not inside any git work tree (a subdirectory of a repo uses the enclosing repo instead);
|
|
@@ -23,12 +25,12 @@ Bare repositories are refused with a clear error. A missing `git` executable is
|
|
|
23
25
|
## Directory Layout
|
|
24
26
|
|
|
25
27
|
```text
|
|
26
|
-
<git-common-dir>/
|
|
28
|
+
<git-common-dir>/pi-plans/
|
|
27
29
|
config.json
|
|
28
30
|
pi-vcc-config.json
|
|
29
|
-
active.json
|
|
30
|
-
runs/ #
|
|
31
|
-
<run-id>/
|
|
31
|
+
active.json # deprecated (v0.6.0): legacy pointer, read only when runs/ is empty
|
|
32
|
+
runs/ # the run registry derives from runs/<run-id>/run.json
|
|
33
|
+
<run-id>/ # note: the refs root is a sibling — .git/pi-plans/refs (hyphenated), not under pi-plans/
|
|
32
34
|
run.json
|
|
33
35
|
decisions.jsonl
|
|
34
36
|
subagents.jsonl
|
|
@@ -37,29 +39,17 @@ Bare repositories are refused with a clear error. A missing `git` executable is
|
|
|
37
39
|
cache/
|
|
38
40
|
```
|
|
39
41
|
|
|
40
|
-
`config.json` is stable workspace preference state. `pi-vcc-config.json` is the repo-private compaction config used only by pi-plans' VCC-style compact hook. `
|
|
42
|
+
`config.json` is stable workspace preference state. `pi-vcc-config.json` is the repo-private compaction config used only by pi-plans' VCC-style compact hook. `runs/` is the run state and the source of the run registry: `listRuns` scans `runs/<run-id>/run.json` (sorted by `updated_at` desc, corrupt entries skipped) and the un-bound "active" resolution is the newest NON-TERMINAL run (null when every run is done/abandoned). Multiple concurrent planning runs in one workdir are supported: each session binds to the run it started (binding first, registry fallback after), and `/plans-abandon`, `/plans-execute`, and `/resume-plans` open a descriptive run-picker form when more than one candidate exists. Reference downloads go to the configured `refs_root` (asked once per workspace when unset; the recommended `.git/pi-plans/refs/` sits inside the git dir so git never tracks it), with metadata recorded in the run state and public artifacts.
|
|
41
43
|
|
|
42
44
|
## Config Schema
|
|
43
45
|
|
|
44
|
-
The default config is:
|
|
46
|
+
The default workspace config is (no `reviewer` key — the role lives globally since v0.7.0):
|
|
45
47
|
|
|
46
48
|
```json
|
|
47
49
|
{
|
|
48
50
|
"schema": 1,
|
|
49
51
|
"language": { "tag": null, "source": "unset", "updated_at": null },
|
|
50
|
-
"
|
|
51
|
-
"mode": "delegated-subagent",
|
|
52
|
-
"model_selector": null,
|
|
53
|
-
"name_prefix": "pi-plans-reviewer",
|
|
54
|
-
"confirmed_at": null
|
|
55
|
-
},
|
|
56
|
-
"criticizer": {
|
|
57
|
-
"mode": "delegated-subagent",
|
|
58
|
-
"model_selector": null,
|
|
59
|
-
"name_prefix": "pi-plans-criticizer",
|
|
60
|
-
"confirmed_at": null
|
|
61
|
-
},
|
|
62
|
-
"artifact_root": "./docs/pi-plans",
|
|
52
|
+
"artifact_root": "./.git/pi-plans/plans",
|
|
63
53
|
"artifact_root_source": "unset",
|
|
64
54
|
"artifact_root_updated_at": null,
|
|
65
55
|
"refs_root": null,
|
|
@@ -72,18 +62,42 @@ Rules:
|
|
|
72
62
|
|
|
73
63
|
- `schema` must be `1`.
|
|
74
64
|
- `language.tag` is a BCP47-style tag such as `zh-Hans`, `en`, or `zh-Hant`, or `null` before selection; `language.source` is `user`, `auto`, or `unset`.
|
|
75
|
-
- `reviewer
|
|
76
|
-
- `model_selector` is `null` to inherit the dispatching session's model, or an exact `provider/model` selector matching Pi's model registry.
|
|
77
|
-
- `confirmed_at` is `null` until the user has confirmed the role's model at first use; see below.
|
|
65
|
+
- A legacy workspace `reviewer` block (pre-0.7.0, and the removed v0.6.0 `criticizer` key) is read-tolerated: the first mutating state call seeds the global config from intent blocks and the next workspace write strips the key. Read-only paths resolve the effective reviewer in memory (global first, legacy block second) and never write.
|
|
78
66
|
- `artifact_root` is relative to the target workspace unless absolute.
|
|
79
67
|
- `artifact_root_source` is `user`, `auto`, or `unset`.
|
|
80
68
|
- `artifact_root_updated_at` is the selection timestamp or `null` before confirmation.
|
|
81
69
|
- `refs_root` is where plan-with-refs downloads references, relative to the target workspace unless absolute, or `null` before selection; `refs_root_source` is `user`, `auto`, or `unset`; `refs_root_updated_at` is the selection timestamp or `null`.
|
|
82
|
-
|
|
70
|
+
|
|
71
|
+
## Global Reviewer Config (v0.7.0)
|
|
72
|
+
|
|
73
|
+
The reviewer role is stored in `~/.pi/pi-plans/config.json` (`PI_PLANS_GLOBAL_DIR` overrides the directory; tests, CI, and bench containers rely on it):
|
|
74
|
+
|
|
75
|
+
```json
|
|
76
|
+
{
|
|
77
|
+
"schema": 1,
|
|
78
|
+
"reviewer": {
|
|
79
|
+
"mode": "delegated-subagent",
|
|
80
|
+
"model_selector": "devin/claude-sonnet-5.5",
|
|
81
|
+
"thinking_level": null,
|
|
82
|
+
"name_prefix": "pi-plans-reviewer",
|
|
83
|
+
"confirmed_at": "2026-09-30T07:54:12Z"
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Rules:
|
|
89
|
+
|
|
90
|
+
- `mode` is `delegated-subagent` or `current-session`.
|
|
91
|
+
- `model_selector` is an exact `provider/model` selector. There is no inherit entry point anymore: after first-use confirmation the delegated reviewer always runs a concrete model (`modelSelector: "inherit"` in `set-role` is a full reset — it clears the selector AND `confirmed_at`).
|
|
92
|
+
- `thinking_level` is `null` (default — spawn WITHOUT `--thinking`, letting the child pi resolve its own chain: per-model settings → `defaultThinkingLevel` → `medium`, then model clamping) or an explicit level (`off | minimal | low | medium | high | xhigh | max`; the domain comes from the chosen model's `thinkingLevelMap` via pi-ai's `getSupportedThinkingLevels`). `"off"` is a real level and deliberately distinct from `null`. Changing `modelSelector` without passing `thinkingLevel` resets the level.
|
|
93
|
+
- `confirmed_at` is stamped only by a real confirmation. `reviewerReady(role)` = current-session, or delegated with `confirmed_at` set AND a concrete `model_selector` — a confirmed null selector can never pass (the old confirmed-inherit state is unreachable).
|
|
94
|
+
- A corrupt or wrong-schema global file yields defaults plus a notice and is NEVER clobbered by reads.
|
|
95
|
+
- Migration (Q-1=A): the first mutating pi-plans call in a workspace with a legacy intent block (`confirmed_at` set, an explicit selector, or a non-default mode) seeds the global file once — first touched workspace wins; other workspaces get a one-time "ignored" notice. A confirmed-inherit block seeds with the selector null and NO confirmation, so the next `refine` re-asks once via the native panel. Scaffold-only blocks are dropped silently.
|
|
96
|
+
- The completion auditor's spawns are NOT governed by this role: it runs its own model; `subagents.jsonl` records the reviewer/ref-analyst thinking level actually passed (`thinking_level` field, older entries have none).
|
|
83
97
|
|
|
84
98
|
## VCC Compact Config
|
|
85
99
|
|
|
86
|
-
`pi-vcc-config.json` is scaffolded under the resolved `<git-common-dir>/
|
|
100
|
+
`pi-vcc-config.json` is scaffolded under the resolved `<git-common-dir>/pi-plans/` state root when an active planning or execution compaction hook first needs it. It is independent from `config.json` so planning preferences, run state, and compact policy can evolve separately.
|
|
87
101
|
|
|
88
102
|
Default values:
|
|
89
103
|
|
|
@@ -117,7 +131,7 @@ Before the first product planning question, check the persisted config (`plans`
|
|
|
117
131
|
4. `Other` — user provides a BCP47 tag.
|
|
118
132
|
5. `Auto-complete` — select the recommended language.
|
|
119
133
|
|
|
120
|
-
Persist with `plans` (`set-language`, `languageSource: "user"`). Use the selected language for visible questions, choices, review summaries,
|
|
134
|
+
Persist with `plans` (`set-language`, `languageSource: "user"`). Use the selected language for visible questions, choices, review summaries, reviewer questions, and Markdown artifacts. Keep IDs, file paths, command names, JSON keys, and protocol labels stable in English.
|
|
121
135
|
|
|
122
136
|
## Code Graph Enabled
|
|
123
137
|
|
|
@@ -125,13 +139,13 @@ Persist with `plans` (`set-language`, `languageSource: "user"`). Use the selecte
|
|
|
125
139
|
|
|
126
140
|
## `/config-pi-plans`
|
|
127
141
|
|
|
128
|
-
`/config-pi-plans` is an interactive workspace configuration wizard. It re-asks the workspace language, planning docs root, refs root, code graph toggle, reviewer mode
|
|
142
|
+
`/config-pi-plans` is an interactive workspace configuration wizard. It re-asks the workspace language, planning docs root, refs root, and code graph toggle (written to `.git/pi-plans/config.json`), plus the reviewer mode and model. The reviewer steps live in the GLOBAL config: the mode switch persists immediately, and the model step shows a keep/change entry menu — `Keep current (provider/model · level)` or `Choose model & thinking level…`. When the mode is `current-session` the model step is skipped entirely (an existing selector is never cleared). Choosing opens the native model panel + effort panel in TUI, or model/effort menus otherwise; Esc keeps the current role and the wizard CONTINUES instead of discarding earlier answers. A failed global write is reported explicitly while workspace settings still save. When code graph is enabled, the extension also overrides built-in `read`/`write`/`edit` for indexed source files so graph-backed source reads and DB-first edits happen automatically. If a run is already active, only the workspace defaults change; the active run's `artifact_dir` and `language_tag` stay unchanged.
|
|
129
143
|
|
|
130
144
|
|
|
131
145
|
Before the first product planning question, check the persisted config again. If `artifact_root_source` is missing or `unset`, ask exactly one `ask_choice` question:
|
|
132
146
|
|
|
133
|
-
1.
|
|
134
|
-
2.
|
|
147
|
+
1. `./.git/pi-plans/plans` — recommended; the default. Planning docs stay private to the repository and are never tracked.
|
|
148
|
+
2. `./docs/pi-plans` — planning docs live in the working tree, are public, and can be committed with the repository.
|
|
135
149
|
3. `Other` — user provides a custom path.
|
|
136
150
|
4. `Auto-complete` — select the recommended path.
|
|
137
151
|
|
|
@@ -147,27 +161,29 @@ Before downloading any reference in a plan-with-refs flow, check the persisted c
|
|
|
147
161
|
Persist with `plans` (`set-refs-root`, `refsRoot: <selected path>`, `refsRootSource: "user"` or `"auto"`). Download references under this root. This question does not count against the planning-question limit.
|
|
148
162
|
|
|
149
163
|
|
|
150
|
-
Before running a `refine` round, read the role setting from the persisted config.
|
|
164
|
+
Before running a `refine` round, read the role setting from the persisted global config.
|
|
151
165
|
|
|
152
|
-
If the role's `mode` is missing or invalid, ask exactly one `ask_choice` question and persist:
|
|
166
|
+
If the role's `mode` is missing or invalid, ask exactly one `ask_choice` question and persist (the mode question stays agent-mediated):
|
|
153
167
|
|
|
154
168
|
1. `Delegated subagent` — recommended; read-only `pi` subprocess with isolated context.
|
|
155
169
|
2. `Current session` — run the read-only pass in the current foreground session.
|
|
156
170
|
3. `Other` / 4. `Auto-complete` — select the recommended delegated subagent.
|
|
157
171
|
|
|
158
|
-
|
|
172
|
+
The **model + thinking level** are confirmed once, at first actual use, through NATIVE panels — not ask_choice:
|
|
173
|
+
|
|
174
|
+
- **TUI**: the gate itself pops a `/model`-style searchable model panel (pi's `ModelSelectorComponent`; the runtime adapter maps the public model registry facade — `getAvailable`/`find`/`getError`/`refresh` — onto the four runtime methods, with the private `.runtime` field as a secondary attempt and menus as the construction fallback), then a `/thinking`-style effort panel whose first row is `Default (no --thinking flag: the child pi resolves per-model settings → defaultThinkingLevel → medium)` followed by the chosen model's `thinkingLevelMap` levels. Both panels must complete; the result persists to the global config (`modelSelector` + `thinkingLevel`, `confirmed: true`) only then, and the same invocation continues with the returned values.
|
|
175
|
+
- **hasUI non-TUI (RPC/ACP)**: native `ctx.ui.select` menus over the same data (model list, then effort list).
|
|
176
|
+
- **UI-less (print/json/bench)**: the gate returns text guidance embedding the available selectors and the exact `set-role` call; persist that way. Automation may also pre-write `~/.pi/pi-plans/config.json` directly.
|
|
159
177
|
|
|
160
|
-
|
|
161
|
-
2. `Choose a model` — pick from the models available in this Pi install (check `/model` or `ctx.scopedModels`); persist the exact `provider/model` selector; do not invent model names.
|
|
162
|
-
3. `Other` / 4. `Auto-complete` — select inherit.
|
|
178
|
+
Pressing Esc on either panel CANCELS the whole gate: nothing is persisted and the tool returns a dedicated error (`details.cancelled`) instructing the agent NOT to re-ask via ask_choice and NOT to retry unless the user asks — suggest `/config-pi-plans` instead. Cheap validations (plan path, refs) run BEFORE any panel so a typo never walks the user through panels.
|
|
163
179
|
|
|
164
|
-
|
|
180
|
+
The confirmation applies only to `delegated-subagent` mode (current-session runs in the main session and needs no model). If a stored selector is missing from the registry at spawn time, TUI re-opens the panel; other modes return an error naming the selector — reset with `set-role` (`resetConfirmation: true`) and re-confirm.
|
|
165
181
|
|
|
166
|
-
|
|
182
|
+
`set-role` invariants: `confirmed: true` requires a concrete `provider/model` selector in delegated mode; `modelSelector: "inherit"` resets BOTH the selector and `confirmed_at`; `thinkingLevel: "default"` stores `null`; changing `modelSelector` without `thinkingLevel` resets the level.
|
|
167
183
|
|
|
168
184
|
## Subagent Spawning
|
|
169
185
|
|
|
170
|
-
When `mode` is `delegated-subagent`, the `refine` tool spawns a read-only `pi` subprocess (`--mode json -p --no-session --tools read,grep,find,ls`) whose system prompt comes from `agents/reviewer.md
|
|
186
|
+
When `mode` is `delegated-subagent`, the `refine` tool spawns a read-only `pi` subprocess (`--mode json -p --no-session --tools read,grep,find,ls`, plus `--model <provider/model>` and — only when `thinking_level` is set — `--thinking <level>`) whose system prompt comes from `agents/reviewer.md`. In TUI mode, delegated runs also show a standalone `Reviewer` overlay with live lane/tool status (header label `provider/model:level`); the child is awaited and the overlay is closed before the tool result returns. The subagent:
|
|
171
187
|
|
|
172
188
|
- performs read-only analysis and never edits files;
|
|
173
189
|
- receives the full plan text and a review/criticism brief;
|
|
@@ -175,19 +191,19 @@ When `mode` is `delegated-subagent`, the `refine` tool spawns a read-only `pi` s
|
|
|
175
191
|
|
|
176
192
|
The main agent consolidates the results, records dispositions, revises the plan, and asks the next merged accept/execute question — all in the same turn.
|
|
177
193
|
|
|
178
|
-
The `analyze_refs` tool (plan-with-refs) uses the same spawning machinery with the **reviewer** role's
|
|
194
|
+
The `analyze_refs` tool (plan-with-refs) uses the same spawning machinery with the **reviewer** role's model confirmation from the global config — but NOT its mode (v0.7.0, Q-4): analysis is spawn-only by nature, so a `current-session` reviewer still gets spawned ref-analyst lanes (a one-time notice says the mode is ignored and unchanged) while the confirmed concrete model remains required. Each downloaded reference gets one independent read-only subagent whose system prompt comes from `agents/ref-analyst.md` and whose working directory is that reference's own directory; lanes never get `code_graph`. Lanes run in sequential batches of at most 3 under a standalone overlay titled `Refs`; each batch's controller opens and closes exactly like a single refine round. Successful spawns are recorded best-effort in `subagents.jsonl` with role `ref-analyst` and the thinking level actually passed (skipped when no active run exists, e.g. adhoc calls). The structured per-reference sections come back as the tool result; the main agent owns `REF_ANALYSIS.md` and fills `coverage`/`gaps` in `refs.jsonl` via `plans` (`record-ref`).
|
|
179
195
|
|
|
180
196
|
## Run State
|
|
181
197
|
|
|
182
|
-
One run directory per planning request: `<git-common-dir>/
|
|
198
|
+
One run directory per planning request: `<git-common-dir>/pi-plans/runs/<YYYYMMDDTHHMMSSZ-topic>/` (second-precision; `-2`, `-3` suffixes on collision).
|
|
183
199
|
|
|
184
200
|
`run.json` includes: run ID; skill name; original request; target workspace; artifact directory; language tag; status (`planning` → `accepted` → `executing` → `done`, with `stopped`/`abandoned` as exits); timestamps.
|
|
185
201
|
|
|
186
|
-
`decisions.jsonl` is appended automatically by `ask_choice` (question, options, answer, answer source). `subagents.jsonl` records reviewer/
|
|
202
|
+
`decisions.jsonl` is appended automatically by `ask_choice` (question, options, answer, answer source). `subagents.jsonl` records reviewer/ref-analyst spawns (legacy 0.6.0 entries with `criticizer` remain readable). `refs.jsonl` records reference metadata via `plans` (`record-ref`); its `kind` field is `project` (repos), `paper` (arXiv etc.), `article` (blog posts), or `docs` (documentation sites).
|
|
187
203
|
|
|
188
204
|
## Workflow Checkpoints (`/resume-plans`)
|
|
189
205
|
|
|
190
|
-
Each run may carry a `checkpoint.json` — the durable, cross-session workflow state that `/resume-plans` restores in the current session. It records: logical `phase` (`planning | reviewing | executing | implementation-review
|
|
206
|
+
Each run may carry a `checkpoint.json` — the durable, cross-session workflow state that `/resume-plans` restores in the current session. It records: logical `phase` (`planning | reviewing | executing | completed`; the legacy 0.6.0 `implementation-review` phase is read-tolerated and maps to done), `nextAction`, the exact plan identity (path + version + SHA-256), pending/answered questions (stable `questionId`), review rounds with per-lane status and result-file references, execution approval evidence (plan digest, worktree, `git rev-parse HEAD` at approval, task progress map, audit rounds, verified VC set, usage), and ownership metadata. Full review outputs live in separate `reviews/` files; the checkpoint keeps only validated references.
|
|
191
207
|
|
|
192
208
|
Rules:
|
|
193
209
|
|
|
@@ -200,4 +216,4 @@ Rules:
|
|
|
200
216
|
|
|
201
217
|
## Run Ownership
|
|
202
218
|
|
|
203
|
-
A run may be held by at most one live owner (`owner.json`: host, pid, process start time via `ps -o lstart=`, session id, random process token, generation). Acquisition is an atomic exclusive create; takeovers require proof the previous owner is dead (process gone, or pid alive with a different start time — PID reuse). Foreign hosts, corrupt records, and unverifiable liveness are conservatively refused; `/resume-plans` never queues or interrupts. Sessions bind to the run they start/execute/resume (restored from `pi-plans-run-start` entries on the current branch), and attribution (tools, write guard, autocomplete, execution bookkeeping, code-graph apply gate) prefers the binding
|
|
219
|
+
A run may be held by at most one live owner (`owner.json`: host, pid, process start time via `ps -o lstart=`, session id, random process token, generation). Acquisition is an atomic exclusive create; takeovers require proof the previous owner is dead (process gone, or pid alive with a different start time — PID reuse). Foreign hosts, corrupt records, and unverifiable liveness are conservatively refused; `/resume-plans` never queues or interrupts. Sessions bind to the run they start/execute/resume (restored from `pi-plans-run-start` entries on the current branch), and attribution (tools, write guard, autocomplete, execution bookkeeping, code-graph apply gate) prefers the binding, falling back to the registry's newest non-terminal run (v0.6.0 — the shared `active.json` pointer is deprecated). The v0.6.0 delegated-executor env pins (`PI_PLANS_RUN_ID` / `PI_PLANS_EXECUTOR`) are gone; read-only subagent children carry `PI_PLANS_REFINER=1`.
|
|
@@ -25,7 +25,7 @@ Usage with harbor::
|
|
|
25
25
|
|
|
26
26
|
Fairness (D-010/D-020): both arms share one container image; the only
|
|
27
27
|
difference is the agent-level configuration above. Seeded evaluation state
|
|
28
|
-
lives under ``.git/
|
|
28
|
+
lives under ``.git/pi-plans/`` with the artifact root pointed OUTSIDE the
|
|
29
29
|
graded tree (``/tmp/pi-plans-bench``); the pre-registered pre-oracle
|
|
30
30
|
snapshot-diff restore was NOT implemented in this run — recorded as a
|
|
31
31
|
limitation (TB oracles read /app artifacts only, so scoring impact is
|
|
@@ -63,27 +63,33 @@ TREATMENT_SYSTEM_PROMPT = (
|
|
|
63
63
|
"it step by step with the verifier checklist."
|
|
64
64
|
)
|
|
65
65
|
|
|
66
|
-
# D-015 seeded
|
|
67
|
-
#
|
|
66
|
+
# D-015 seeded configs: deterministic, no first-use Q&A rounds, graph off,
|
|
67
|
+
# artifact root outside the graded workspace. Since v0.7.0 the reviewer role
|
|
68
|
+
# lives in the GLOBAL config (~/.pi/pi-plans/config.json inside the
|
|
69
|
+
# container) with a CONCRETE provider/model selector (inherit was removed);
|
|
70
|
+
# the workspace config carries no reviewer/criticizer keys at all.
|
|
68
71
|
SEEDED_CONFIG = {
|
|
69
72
|
"schema": 1,
|
|
70
73
|
"artifact_root": "/tmp/pi-plans-bench/docs",
|
|
71
74
|
"artifact_root_source": "user",
|
|
72
75
|
"language": {"tag": "en", "source": "user"},
|
|
73
|
-
"reviewer": {
|
|
74
|
-
"mode": "delegated-subagent",
|
|
75
|
-
"model_selector": None, # None => inherit the main agent's model (flash)
|
|
76
|
-
"confirmed_at": "1970-01-01T00:00:00Z",
|
|
77
|
-
},
|
|
78
|
-
"criticizer": {
|
|
79
|
-
"mode": "delegated-subagent",
|
|
80
|
-
"model_selector": None,
|
|
81
|
-
"confirmed_at": "1970-01-01T00:00:00Z",
|
|
82
|
-
},
|
|
83
76
|
"graph_enabled": False,
|
|
84
77
|
}
|
|
85
78
|
|
|
86
79
|
|
|
80
|
+
def _seeded_global_config(provider: str, model_id: str) -> dict:
|
|
81
|
+
return {
|
|
82
|
+
"schema": 1,
|
|
83
|
+
"reviewer": {
|
|
84
|
+
"mode": "delegated-subagent",
|
|
85
|
+
"model_selector": f"{provider}/{model_id}",
|
|
86
|
+
"thinking_level": None, # default: child pi resolves its own chain
|
|
87
|
+
"name_prefix": "pi-plans-reviewer",
|
|
88
|
+
"confirmed_at": "1970-01-01T00:00:00Z",
|
|
89
|
+
},
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
|
|
87
93
|
def _read_text(name: str) -> str:
|
|
88
94
|
return (ADAPTER_DIR / name).read_text(encoding="utf-8")
|
|
89
95
|
|
|
@@ -199,20 +205,25 @@ class PiPlansBench(Pi):
|
|
|
199
205
|
filename=".bench-system-prompt.md",
|
|
200
206
|
)
|
|
201
207
|
|
|
202
|
-
async def _seed_pi_plans_config(self, environment: BaseEnvironment) -> None:
|
|
203
|
-
"""Write the deterministic pi-plans
|
|
208
|
+
async def _seed_pi_plans_config(self, environment: BaseEnvironment, provider: str, model_id: str) -> None:
|
|
209
|
+
"""Write the deterministic pi-plans configs into the trial container (D-015).
|
|
204
210
|
|
|
205
|
-
The config lives under ``.git/
|
|
206
|
-
before oracle scoring) with the artifact root pointed at
|
|
207
|
-
``/tmp/pi-plans-bench`` so plan artifacts never land in the graded
|
|
211
|
+
The workspace config lives under ``.git/pi-plans/`` (diff-allowlist
|
|
212
|
+
path, removed before oracle scoring) with the artifact root pointed at
|
|
213
|
+
``/tmp/pi-plans-bench`` so plan artifacts never land in the graded
|
|
214
|
+
tree. The reviewer role is seeded into the GLOBAL config under the
|
|
215
|
+
container's ``~/.pi/pi-plans/`` with the exact driver model (v0.7.0:
|
|
216
|
+
no inherit, no criticizer), so refine gates pass without any panel.
|
|
208
217
|
"""
|
|
209
218
|
config = json.dumps(SEEDED_CONFIG, indent=2)
|
|
219
|
+
global_config = json.dumps(_seeded_global_config(provider, model_id), indent=2)
|
|
210
220
|
await self.exec_as_agent(
|
|
211
221
|
environment,
|
|
212
222
|
command=(
|
|
213
223
|
"set -euo pipefail; "
|
|
214
|
-
"mkdir -p .git/
|
|
215
|
-
f"printf {shlex.quote(config)} > .git/
|
|
224
|
+
"mkdir -p .git/pi-plans ~/.pi/pi-plans " f"{BENCH_CONFIG_DIR} && "
|
|
225
|
+
f"printf {shlex.quote(config)} > .git/pi-plans/config.json && "
|
|
226
|
+
f"printf {shlex.quote(global_config)} > ~/.pi/pi-plans/config.json"
|
|
216
227
|
),
|
|
217
228
|
)
|
|
218
229
|
|
|
@@ -224,7 +235,7 @@ class PiPlansBench(Pi):
|
|
|
224
235
|
provider, model_id = self.model_name.split("/", 1)
|
|
225
236
|
|
|
226
237
|
if self.arm == "treatment":
|
|
227
|
-
await self._seed_pi_plans_config(environment)
|
|
238
|
+
await self._seed_pi_plans_config(environment, provider, model_id)
|
|
228
239
|
|
|
229
240
|
# printf interprets backslash escapes; ship sources base64-encoded so
|
|
230
241
|
# the driver/task bytes survive verbatim.
|
|
@@ -291,7 +302,7 @@ class PiPlansBench(Pi):
|
|
|
291
302
|
if subagent_usage is not None:
|
|
292
303
|
metadata["pi_plans_subagent_usage"] = subagent_usage
|
|
293
304
|
# Fold child tokens/cost into the top-level accounting so treatment
|
|
294
|
-
# totals are comparable (the extension's reviewer/
|
|
305
|
+
# totals are comparable (the extension's reviewer/ref-analyst calls).
|
|
295
306
|
totals = subagent_usage.get("totals") or {}
|
|
296
307
|
context.n_input_tokens = (context.n_input_tokens or 0) + int(totals.get("input", 0))
|
|
297
308
|
context.n_output_tokens = (context.n_output_tokens or 0) + int(totals.get("output", 0))
|
|
@@ -184,11 +184,11 @@ function snapshotPlanArtifacts(dest) {
|
|
|
184
184
|
if (!dest) return;
|
|
185
185
|
// Plan artifact locations: the seeded artifact root (/tmp/pi-plans-bench/docs)
|
|
186
186
|
// and pi-plans' own fallbacks — the repo-local docs/pi-plans and the state
|
|
187
|
-
// root .git/
|
|
187
|
+
// root .git/pi-plans/docs (where run plans actually land).
|
|
188
188
|
const roots = [
|
|
189
189
|
"/tmp/pi-plans-bench/docs",
|
|
190
190
|
path.join(process.cwd(), "docs", "pi-plans"),
|
|
191
|
-
path.join(process.cwd(), ".git", "
|
|
191
|
+
path.join(process.cwd(), ".git", "pi-plans", "docs"),
|
|
192
192
|
];
|
|
193
193
|
try {
|
|
194
194
|
mkdirSync(dest, { recursive: true });
|
|
@@ -209,8 +209,8 @@ function snapshotPlanArtifacts(dest) {
|
|
|
209
209
|
|
|
210
210
|
function snapshotSubagentUsage(dest) {
|
|
211
211
|
if (!dest) return;
|
|
212
|
-
// pi-plans state root convention: <workdir>/.git/
|
|
213
|
-
const stateRoots = [path.join(process.cwd(), ".git", "
|
|
212
|
+
// pi-plans state root convention: <workdir>/.git/pi-plans/runs/<id>/subagents.jsonl
|
|
213
|
+
const stateRoots = [path.join(process.cwd(), ".git", "pi-plans", "runs")];
|
|
214
214
|
const totals = { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, cost: 0, children: 0 };
|
|
215
215
|
try {
|
|
216
216
|
for (const root of stateRoots) {
|
package/scripts/run-tests.ts
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
|
|
3
3
|
import { spawnSync } from "node:child_process";
|
|
4
|
-
import { existsSync, readdirSync } from "node:fs";
|
|
4
|
+
import { existsSync, mkdtempSync, readdirSync } from "node:fs";
|
|
5
|
+
import { tmpdir } from "node:os";
|
|
5
6
|
import { join, resolve } from "node:path";
|
|
6
7
|
|
|
7
8
|
const testDir = resolve("tests");
|
|
@@ -10,6 +11,15 @@ if (!existsSync(testDir)) {
|
|
|
10
11
|
process.exit(1);
|
|
11
12
|
}
|
|
12
13
|
|
|
14
|
+
// Safety net (F-002): unless the caller pinned one, point the pi-plans
|
|
15
|
+
// GLOBAL config at a throwaway directory so concurrent test files can never
|
|
16
|
+
// clobber the developer's real ~/.pi/pi-plans/config.json. Individual suites
|
|
17
|
+
// may still override per-test with their own PI_PLANS_GLOBAL_DIR.
|
|
18
|
+
const env = { ...process.env };
|
|
19
|
+
if (!env.PI_PLANS_GLOBAL_DIR) {
|
|
20
|
+
env.PI_PLANS_GLOBAL_DIR = mkdtempSync(join(tmpdir(), "pi-plans-global-"));
|
|
21
|
+
}
|
|
22
|
+
|
|
13
23
|
const tests = readdirSync(testDir)
|
|
14
24
|
.filter((entry) => entry.endsWith(".test.ts"))
|
|
15
25
|
.sort()
|
|
@@ -22,6 +32,7 @@ if (tests.length === 0) {
|
|
|
22
32
|
|
|
23
33
|
const result = spawnSync(process.execPath, ["--experimental-strip-types", "--test", ...tests], {
|
|
24
34
|
stdio: "inherit",
|
|
35
|
+
env,
|
|
25
36
|
});
|
|
26
37
|
|
|
27
38
|
if (result.error) {
|