@jwilger/pi-development-system 0.83.0 → 0.85.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -2
- package/extensions/development-system.ts +9 -0
- package/package.json +1 -1
- package/prompts/devsys-adr.md +12 -0
- package/prompts/devsys-lens-review.md +12 -0
- package/prompts/devsys-plan.md +1 -1
- package/skills/product-planning/SKILL.md +86 -0
- package/skills/product-planning/references/brief.md +48 -0
- package/skills/product-planning/references/decisions.md +33 -0
- package/skills/product-planning/references/followups.md +16 -0
- package/skills/product-planning/references/journeys.md +14 -0
- package/skills/product-planning/references/terminology.md +10 -0
- package/src/context/phase-guide.ts +3 -3
- package/src/context/turn-verifier.ts +7 -1
- package/src/core/lifecycle.ts +51 -0
- package/src/gates/commit-guard.ts +98 -20
- package/src/gates/red-first-guard.ts +20 -4
- package/src/jev/questions/architecture.ts +43 -0
- package/src/jev/questions/product-lenses.ts +64 -0
- package/src/jev/questions/solution-detail.ts +43 -0
- package/src/planning/adr-tool.ts +70 -0
- package/src/planning/adr.ts +46 -0
- package/src/planning/brief-lint.ts +71 -0
- package/src/planning/intake-tool.ts +3 -4
- package/src/planning/slice-close.ts +195 -0
- package/src/review/lens-review-tool.ts +205 -0
- package/src/review/lens-review.ts +168 -0
- package/src/review/review-tools.ts +3 -0
package/README.md
CHANGED
|
@@ -19,8 +19,20 @@ Everything the system offers is reachable without remembering a command. Describ
|
|
|
19
19
|
plain words: when a prompt asks for new work, a fix or a review, the system adds a guideline
|
|
20
20
|
naming the tool to use (`devsys_intake`, `devsys_review_start`). A `devsys` tool always shows
|
|
21
21
|
what the current phase expects. When you say a slice is finished with no review round recorded,
|
|
22
|
-
it tells you to start one. The slash commands (`/devsys-start`, `/devsys-plan`, `/devsys-review
|
|
23
|
-
are shortcuts to the same tools.
|
|
22
|
+
it tells you to start one. The slash commands (`/devsys-start`, `/devsys-plan`, `/devsys-review`,
|
|
23
|
+
`/devsys-lens-review`, `/devsys-adr`) are shortcuts to the same tools.
|
|
24
|
+
|
|
25
|
+
Product planning has a skill (`product-planning`: brief, decision register, follow-ups,
|
|
26
|
+
terminology, journeys). `devsys_lens_review` plans a review of the brief by five product lenses and
|
|
27
|
+
writes the packets to `docs/product/reviews/`; `devsys_adr_new` creates the next numbered ADR, and
|
|
28
|
+
a commit that shapes the architecture without one is stopped by the soft gate `adr.missing`.
|
|
29
|
+
|
|
30
|
+
A slice has a life cycle: `implementing` → `reviewing` (when a review round starts) → `delivering`
|
|
31
|
+
(review satisfied) → `idle`. Editing production source while delivering reopens `implementing`,
|
|
32
|
+
and red-first stays on throughout. A push of a clean tree closes a delivered slice by itself (a commit
|
|
33
|
+
in `local-only` mode; CI is not awaited), and `devsys_finish_slice` closes it explicitly or, with a
|
|
34
|
+
reason, abandons it. A plan's increments each end in a push, so each increment is its own slice: the
|
|
35
|
+
next one starts with `devsys_intake`.
|
|
24
36
|
|
|
25
37
|
With `codemode` enabled (`"defaultTools": ["+codemode"]` in pi settings), rarely used tools are
|
|
26
38
|
reached through scripts and the `judge_*` Jev wrappers exist for scripts only; without codemode
|
|
@@ -27,9 +27,12 @@ import { createRequestApprovalTool } from "../src/gates/request-approval-tool.ts
|
|
|
27
27
|
import { registerTestGuard } from "../src/gates/test-guard.ts";
|
|
28
28
|
import { createJevHolder } from "../src/jev/holder.ts";
|
|
29
29
|
import { createJudgeTools } from "../src/jev/judge-tools.ts";
|
|
30
|
+
import { createAdrNewTool } from "../src/planning/adr-tool.ts";
|
|
30
31
|
import { createBeginWorkTool } from "../src/planning/begin-tool.ts";
|
|
31
32
|
import { createIntakeTool } from "../src/planning/intake-tool.ts";
|
|
33
|
+
import { createFinishSliceTool, registerSliceClose } from "../src/planning/slice-close.ts";
|
|
32
34
|
import { createTaskCheckTool } from "../src/planning/task-check-tool.ts";
|
|
35
|
+
import { createLensReviewTool } from "../src/review/lens-review-tool.ts";
|
|
33
36
|
import { createReviewRecordTool, createReviewStartTool } from "../src/review/review-tools.ts";
|
|
34
37
|
import { registerCiCommand } from "../src/state/ci-command.ts";
|
|
35
38
|
import { loadConfig } from "../src/state/config.ts";
|
|
@@ -156,6 +159,7 @@ export function createDevelopmentSystem(pi: ExtensionAPI) {
|
|
|
156
159
|
},
|
|
157
160
|
});
|
|
158
161
|
registerRedFirstGuard({ pi, state });
|
|
162
|
+
registerSliceClose({ pi, state, exec });
|
|
159
163
|
registerLintSuppressionGuard({ pi, state });
|
|
160
164
|
pi.registerTool(createRecordDepartureTool({ pi, state }));
|
|
161
165
|
pi.registerTool(createRequestApprovalTool({ pi, approvals }));
|
|
@@ -174,6 +178,11 @@ export function createDevelopmentSystem(pi: ExtensionAPI) {
|
|
|
174
178
|
}
|
|
175
179
|
pi.registerTool(createPhaseTool({ state }));
|
|
176
180
|
pi.registerTool(createBeginWorkTool({ state }));
|
|
181
|
+
pi.registerTool(createAdrNewTool({ now: () => new Date() }));
|
|
182
|
+
pi.registerTool(createFinishSliceTool({ state, exec }));
|
|
183
|
+
pi.registerTool(
|
|
184
|
+
createLensReviewTool({ state, jev: (ctx) => jevHolder.forContext(ctx), now: () => new Date() }),
|
|
185
|
+
);
|
|
177
186
|
pi.registerTool(createIntakeTool({ pi, state, jev: (ctx) => jevHolder.forContext(ctx) }));
|
|
178
187
|
const reviewDeps = {
|
|
179
188
|
state,
|
package/package.json
CHANGED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: Record a hard-to-reverse technical decision as the next numbered ADR
|
|
3
|
+
argument-hint: "<decision title>"
|
|
4
|
+
---
|
|
5
|
+
Record an architecture decision with the product-planning skill.
|
|
6
|
+
|
|
7
|
+
1. Call `devsys_adr_new` with the title: $ARGUMENTS. It creates `docs/adr/NNNN-<slug>.md` from the template and returns the path.
|
|
8
|
+
2. Fill in Context (the forces and the problem), Decision (stated plainly), Consequences (positive and negative), Alternatives (each with why it was rejected) and Revisit when (a concrete condition), then link related ADRs and plan sections.
|
|
9
|
+
3. Set the status to `accepted` only when the user has agreed; leave it `proposed` otherwise.
|
|
10
|
+
4. Report the path and a one-line summary of the decision.
|
|
11
|
+
|
|
12
|
+
Use an ADR for a boundary, dependency, data format or protocol that is hard to reverse. Product decisions belong in the decision register instead.
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: Run the five product lenses over the brief and record the packets
|
|
3
|
+
argument-hint: "[brief path] [round 1|2]"
|
|
4
|
+
---
|
|
5
|
+
Review the product brief with the product-planning skill.
|
|
6
|
+
|
|
7
|
+
1. Call `devsys_lens_review` (brief path and round: $ARGUMENTS; omit either for `docs/product/brief.md` and round 1).
|
|
8
|
+
2. Run the codemode script it returns, unchanged. It spawns the lens agents, waits, writes the packets to `docs/product/reviews/<date>-round<n>.md` and returns only a verdict per lens and the path. If codemode is unavailable, use the `agent_spawn` payloads it lists.
|
|
9
|
+
3. Read the review file, not the agents' full output. Report each lens verdict and the findings with their `path:line`.
|
|
10
|
+
4. After round 2, write the synthesis (R-table and a one-question agenda) from the template in the reply, then ask the user that one question.
|
|
11
|
+
|
|
12
|
+
Agreement among agents is useful critique, not customer evidence.
|
package/prompts/devsys-plan.md
CHANGED
|
@@ -9,4 +9,4 @@ Write the plan for the current work with the work-intake-and-slicing skill.
|
|
|
9
9
|
3. Write each task so that someone who sees only that task could finish it. Never write `TBD`.
|
|
10
10
|
4. Run `devsys_task_check` on every task record you wrote and fix each one until it says ready. Split any that come back too-big.
|
|
11
11
|
5. Stop and ask the user to review the plan before any implementation starts. Do not start a goal yet: a goal tool may continue on its own and begin implementing before the review.
|
|
12
|
-
6. Only after the user approves the plan: call `devsys_begin_work` (so the review and red-first gates are on while the
|
|
12
|
+
6. Only after the user approves the plan: call `devsys_begin_work` (so the review and red-first gates are on while the increment is built; it does nothing if the work is already implementing). A push of a clean, reviewed tree closes the slice, so each later increment starts with `devsys_intake`. Then, if a `create_goal` tool exists in this session, call it with the objective "complete docs/plan/<slug>.md honouring its Constraints and Progress checklist, one increment at a time, stopping after each increment's Release step". Otherwise tell the user the plan path so they can start a goal on it. Do not depend on any goal tool's file format.
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: product-planning
|
|
3
|
+
description: Plan a product or capability before building it - discovery interview, brief, decision register, follow-ups, terminology, journey inventory, lens review and ADRs. Use when sizing says capability or product, when asked for a brief or a discovery interview, when an answer must be recorded as a decision, or when a plan needs a product review.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Product planning
|
|
7
|
+
|
|
8
|
+
Planning here finds out what is worth building. It is not a spec for code. The artifacts are a small set of
|
|
9
|
+
linked documents; each has one job and a template under `references/`.
|
|
10
|
+
|
|
11
|
+
| Artifact | Job | Template |
|
|
12
|
+
|---|---|---|
|
|
13
|
+
| Brief | The one narrative: outcome, customers, four risks, assumptions, scope, non-goals | [brief](references/brief.md) |
|
|
14
|
+
| Decision register | The spine: D (decisions), Q (questions), F (scheduling and exclusions) | [decisions](references/decisions.md) |
|
|
15
|
+
| Follow-ups | P items and R review findings: not lost, not now | [followups](references/followups.md) |
|
|
16
|
+
| Terminology | One meaning per word | [terminology](references/terminology.md) |
|
|
17
|
+
| Journeys | J items: a user, actions, an outcome | [journeys](references/journeys.md) |
|
|
18
|
+
|
|
19
|
+
Put them under `docs/product/` (`brief.md`, `decisions.md`, `followups.md`, `terminology.md`, `journeys.md`).
|
|
20
|
+
Copy a template only when the work needs that artifact: `capability` work gets a short brief and journeys,
|
|
21
|
+
`product` work gets the lot. Skipping one is a recorded departure (`artifact.skipped`).
|
|
22
|
+
|
|
23
|
+
## The interview loop
|
|
24
|
+
|
|
25
|
+
Quote this rule and follow it exactly:
|
|
26
|
+
|
|
27
|
+
> Update docs after each answer, ask one next question, yield.
|
|
28
|
+
|
|
29
|
+
1. Ask one question. Make it the most decision-relevant open Q item.
|
|
30
|
+
2. Stop. Wait for the answer. Do not ask a second question or guess ahead.
|
|
31
|
+
3. Record the answer as a D-item in the register, in the owner's words, and add a dated line to the interview log.
|
|
32
|
+
4. Edit the brief **only where the answered question touches it**. Bump its version. Nothing else changes.
|
|
33
|
+
5. Ask the next question and yield again.
|
|
34
|
+
|
|
35
|
+
Every answer becomes a D-item, including "no" and "not now" (a Deferred item with a revisit point).
|
|
36
|
+
|
|
37
|
+
## Do not over-edit the brief
|
|
38
|
+
|
|
39
|
+
The commonest failure is an agent that rewrites the brief between answers. It misframed the team, moved the
|
|
40
|
+
timing boundary, let later-phase ideas leak in and relocated the business case, and the owner had to find and
|
|
41
|
+
undo each one. So:
|
|
42
|
+
|
|
43
|
+
- An edit is limited to the answered question. If you notice something else that is wrong, add a Q item, do not fix it silently.
|
|
44
|
+
- Ideas for later go to follow-ups. Nothing speculative goes in the brief.
|
|
45
|
+
- Show the brief diff after each answer so the owner can see exactly what moved.
|
|
46
|
+
- The owner owns judgements about the domain; you own process and where things are filed.
|
|
47
|
+
|
|
48
|
+
## Outcome, risks, assumptions
|
|
49
|
+
|
|
50
|
+
- Open with an **outcome** (direction plus target), not a feature. State the problem to solve, not the solution to build.
|
|
51
|
+
- Assess the four risks (value, usability, feasibility, viability). "Unassessed" is a legal, recorded state.
|
|
52
|
+
- Write assumptions specifically; the smaller the assumption, the smaller the test. Test the riskiest first
|
|
53
|
+
(critical to success, least evidence). Compare at least two solutions, and record why when you only had one.
|
|
54
|
+
- Say whether the work builds to learn (disposable, fewer gates) or builds to earn (full gates). That is a decision, never an inference.
|
|
55
|
+
|
|
56
|
+
## Deferral is not exclusion
|
|
57
|
+
|
|
58
|
+
Postponing the details of a capability does not remove it from scope. Only an explicit scope decision does.
|
|
59
|
+
A deferred item records who deferred it, why, the interim constraint and when to revisit. A non-goal is
|
|
60
|
+
something we decided not to do.
|
|
61
|
+
|
|
62
|
+
## Journeys
|
|
63
|
+
|
|
64
|
+
Cut the backlog from journeys, not features. A journey passes if it tells the story of a user performing a
|
|
65
|
+
set of actions to achieve an outcome. Decide journeys before generating tasks; feature-shaped backlogs had to
|
|
66
|
+
be thrown away and redone.
|
|
67
|
+
|
|
68
|
+
## Lens review
|
|
69
|
+
|
|
70
|
+
Run `devsys_lens_review` (or `/devsys-lens-review`) on the brief when the outcome, scope or risks are settled
|
|
71
|
+
enough to be wrong in an interesting way. Five fresh agents read it, each through one lens (Cagan, Torres,
|
|
72
|
+
Pichler, Perri, Rumelt). Round one is independent; round two is a peer exchange; you then synthesise an R table
|
|
73
|
+
and a one-question agenda, and the interview loop resumes. **Agreement among agents is useful critique, not
|
|
74
|
+
customer evidence.** Record rejected recommendations in the register instead of dropping them.
|
|
75
|
+
|
|
76
|
+
## Decisions that need an ADR
|
|
77
|
+
|
|
78
|
+
A decision about how the code is built that is hard to reverse (a boundary, a dependency, a data format, a
|
|
79
|
+
protocol) gets an ADR through `devsys_adr_new`. Product decisions stay in the register. Do not write an ADR
|
|
80
|
+
for something the register already holds, or the other way round.
|
|
81
|
+
|
|
82
|
+
## Status words
|
|
83
|
+
|
|
84
|
+
Everything produced here is "proposed" or "advisory" until a person approves a specific revision. Say
|
|
85
|
+
"readiness" for declared checks, "approval" for a person's decision on a named revision, and never let one
|
|
86
|
+
stand in for the other.
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
# Brief: <product or change name>
|
|
2
|
+
|
|
3
|
+
- **Context:** <which product or process this brief is about>
|
|
4
|
+
- **Status:** v0.1 · <date> · links: [decisions](decisions.md), [follow-ups](followups.md)
|
|
5
|
+
- **Not an authorisation:** this brief records what is understood so far. It does not approve building anything.
|
|
6
|
+
|
|
7
|
+
## Outcome
|
|
8
|
+
|
|
9
|
+
One sentence: the change in customer behaviour or business result we want, with a direction and a target
|
|
10
|
+
(for example "first-session success from 22% to 25%"). An output ("ship X") is not an outcome.
|
|
11
|
+
|
|
12
|
+
## Customers and the problem
|
|
13
|
+
|
|
14
|
+
Who has the problem, what they do today, and the evidence for it (link the source, date it). Say plainly what
|
|
15
|
+
is evidence and what is belief.
|
|
16
|
+
|
|
17
|
+
## Four risks
|
|
18
|
+
|
|
19
|
+
| Risk | What could be wrong | Evidence so far | Cheapest test |
|
|
20
|
+
|---|---|---|---|
|
|
21
|
+
| Value | Nobody wants or uses it | unassessed | |
|
|
22
|
+
| Usability | People cannot work it out | unassessed | |
|
|
23
|
+
| Feasibility | We cannot build it | unassessed | |
|
|
24
|
+
| Viability | It does not work for the business | unassessed | |
|
|
25
|
+
|
|
26
|
+
"Unassessed" is a legal state. Writing it is better than guessing.
|
|
27
|
+
|
|
28
|
+
## Assumptions
|
|
29
|
+
|
|
30
|
+
Each assumption is specific, tagged with a risk category and a test status. Riskiest first (critical to
|
|
31
|
+
success, little evidence). Every assumption gets a D-item in the register once answered.
|
|
32
|
+
|
|
33
|
+
| ID | Assumption | Category | Test status |
|
|
34
|
+
|---|---|---|---|
|
|
35
|
+
| A01 | | desirability / viability / feasibility / usability / ethical | untested |
|
|
36
|
+
|
|
37
|
+
## Scope
|
|
38
|
+
|
|
39
|
+
What the first release does, in the customer's words.
|
|
40
|
+
|
|
41
|
+
## Non-goals
|
|
42
|
+
|
|
43
|
+
Things we have decided not to do. Only an explicit scope decision belongs here.
|
|
44
|
+
**Deferral is not exclusion:** an item that is merely later goes to [follow-ups](followups.md), not here.
|
|
45
|
+
|
|
46
|
+
## Open questions
|
|
47
|
+
|
|
48
|
+
Pointers only (Q-ids in [decisions](decisions.md)). The question text lives in the register, not the brief.
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
# Decision register
|
|
2
|
+
|
|
3
|
+
The referential spine: every brief, review and journey cites these ids.
|
|
4
|
+
|
|
5
|
+
## Closure rules
|
|
6
|
+
|
|
7
|
+
- **Resolved:** the owner answered or accepted it.
|
|
8
|
+
- **Deferred:** postponed to a later phase. State who decided, why, the interim constraint and the revisit
|
|
9
|
+
point. Deferring the specification does not defer the capability; only an explicit scope decision does.
|
|
10
|
+
- **Open:** not yet answered.
|
|
11
|
+
- Superseding keeps the old row and points at the new one (`D06 → D56`). Rows are never deleted.
|
|
12
|
+
|
|
13
|
+
## Decisions (D)
|
|
14
|
+
|
|
15
|
+
| ID | Decision | Status | Source |
|
|
16
|
+
|---|---|---|---|
|
|
17
|
+
| D01 | | Resolved | <answer, date> |
|
|
18
|
+
|
|
19
|
+
## Questions (Q)
|
|
20
|
+
|
|
21
|
+
| ID | Question | Status | Needed for closure |
|
|
22
|
+
|---|---|---|---|
|
|
23
|
+
| Q01 | | Open | |
|
|
24
|
+
|
|
25
|
+
## Scheduling, conditional scope and exclusions (F)
|
|
26
|
+
|
|
27
|
+
| ID | Item | Disposition | Rationale | Revisit when |
|
|
28
|
+
|---|---|---|---|---|
|
|
29
|
+
| F01 | | true exclusion / conditional / scheduled | | |
|
|
30
|
+
|
|
31
|
+
## Interview log
|
|
32
|
+
|
|
33
|
+
One dated bullet per answer: the question, the answer in the owner's words, and "Recorded as D<nn>".
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
# Planning follow-ups
|
|
2
|
+
|
|
3
|
+
Where "do not lose it, do not do it now" goes. **Scheduled planning work is not an MVP exclusion.**
|
|
4
|
+
|
|
5
|
+
| ID | Question or work | Appropriate phase | Revisit when | Decision refs |
|
|
6
|
+
|---|---|---|---|---|
|
|
7
|
+
| P01 | | modelling / design / architecture / pilot / commercial | | D01 |
|
|
8
|
+
|
|
9
|
+
## Review findings (R)
|
|
10
|
+
|
|
11
|
+
Findings from a lens review and where each was routed (decision register, follow-up, interview question,
|
|
12
|
+
rejected). A rejected recommendation is recorded with the reason, never silently dropped.
|
|
13
|
+
|
|
14
|
+
| ID | Finding | Route | Disposition |
|
|
15
|
+
|---|---|---|---|
|
|
16
|
+
| R01 | | | |
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
# Journey inventory
|
|
2
|
+
|
|
3
|
+
A journey passes this test: **it tells the story of a user performing a set of actions to achieve an outcome.**
|
|
4
|
+
A feature, a screen or a technical layer does not pass it. Do this before cutting any backlog.
|
|
5
|
+
|
|
6
|
+
| ID | User and starting need | Actions | Outcome |
|
|
7
|
+
|---|---|---|---|
|
|
8
|
+
| J01 | | | |
|
|
9
|
+
|
|
10
|
+
## Per journey
|
|
11
|
+
|
|
12
|
+
- Alternative outcomes (what happens when it goes wrong).
|
|
13
|
+
- The decision ids it rests on.
|
|
14
|
+
- Approval is of a specific revision, recorded with person and date; drafting is not approval.
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
# Terminology
|
|
2
|
+
|
|
3
|
+
Words used in the brief, register and journeys, each with one meaning. Label a document with the sense of a
|
|
4
|
+
word it uses when the word has more than one.
|
|
5
|
+
|
|
6
|
+
| Term | Meaning | Not to be confused with |
|
|
7
|
+
|---|---|---|
|
|
8
|
+
| Deferred item | Scheduled to a later phase | Exclusion |
|
|
9
|
+
| Readiness | Declared checks pass | Human approval of a specific revision |
|
|
10
|
+
| Lens | An advisory reviewing perspective | A review by the person it is named for |
|
|
@@ -10,11 +10,11 @@ export function phaseGuide(phase: Phase): string {
|
|
|
10
10
|
case "planning":
|
|
11
11
|
return "Planning. Produce the artifacts the sizing asked for, write task records (check each with devsys_task_check), and when the user approves the plan call devsys_begin_work.";
|
|
12
12
|
case "implementing":
|
|
13
|
-
return "Implementing one slice. Write a failing test first, make it pass with the least code, run the tests and read the result before claiming anything. Push
|
|
13
|
+
return "Implementing one slice. Write a failing test first, make it pass with the least code, run the tests and read the result before claiming anything. Push when the slice is complete and reviewed: a push of a clean tree closes it (devsys_finish_slice closes it explicitly).";
|
|
14
14
|
case "reviewing":
|
|
15
|
-
return "Reviewing. Call devsys_review_start for a fresh-context review, pass its packet and diffDigest to devsys_review_record, fix every blocking and should-fix finding, and repeat until the
|
|
15
|
+
return "Reviewing. Call devsys_review_start for a fresh-context review, pass its packet and diffDigest to devsys_review_record, fix every blocking and should-fix finding (red-first still applies), and repeat until the review is satisfied; then the slice moves to delivering.";
|
|
16
16
|
case "delivering":
|
|
17
|
-
return "Delivering. Commit with a rationale, push
|
|
17
|
+
return "Delivering. The review is satisfied. Commit with a rationale, push, and release; a push of a clean tree closes the slice (devsys_finish_slice closes it explicitly, or abandons it with a reason). Editing source reopens implementing. A red trunk (CI) is repaired first.";
|
|
18
18
|
default:
|
|
19
19
|
return assertNever(phase);
|
|
20
20
|
}
|
|
@@ -149,17 +149,23 @@ export function registerTurnVerifier(deps: TurnVerifierDeps): void {
|
|
|
149
149
|
let evidence: ToolEvidence[] = [];
|
|
150
150
|
let corrected = 0;
|
|
151
151
|
let justCorrected = false;
|
|
152
|
+
let inFlightThisRun = false;
|
|
152
153
|
|
|
153
154
|
deps.pi.on("session_start", () => {
|
|
154
155
|
corrected = 0;
|
|
155
156
|
justCorrected = false;
|
|
157
|
+
inFlightThisRun = false;
|
|
156
158
|
evidence = [];
|
|
157
159
|
});
|
|
158
160
|
deps.pi.on("agent_start", () => {
|
|
161
|
+
// A push in this run may close the slice; its delivery report is still the claim to check.
|
|
162
|
+
inFlightThisRun = VERIFIED_PHASES.has(deps.state.get().phase);
|
|
159
163
|
evidence = [];
|
|
160
164
|
justCorrected = false;
|
|
161
165
|
});
|
|
162
166
|
deps.pi.on("tool_result", (event) => {
|
|
167
|
+
// The slice can start inside this run (intake, begin) and be closed by its own push; note it was open.
|
|
168
|
+
if (VERIFIED_PHASES.has(deps.state.get().phase)) inFlightThisRun = true;
|
|
163
169
|
evidence = [...evidence, evidenceOf(event)].slice(-MAX_EVIDENCE);
|
|
164
170
|
});
|
|
165
171
|
|
|
@@ -167,7 +173,7 @@ export function registerTurnVerifier(deps: TurnVerifierDeps): void {
|
|
|
167
173
|
// An aborted or errored turn is dropped by pi; judging it would delay the abort and spend a correction.
|
|
168
174
|
if (event.outcome !== "completed") return undefined;
|
|
169
175
|
const state = deps.state.get();
|
|
170
|
-
if (!VERIFIED_PHASES.has(state.phase)) return undefined;
|
|
176
|
+
if (!(VERIFIED_PHASES.has(state.phase) || inFlightThisRun)) return undefined;
|
|
171
177
|
const parts = assistantParts(event.message);
|
|
172
178
|
if (parts === undefined || parts.callsTools || parts.text.trim() === "") return undefined;
|
|
173
179
|
if (justCorrected) {
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
import type { ReviewAction } from "./review.ts";
|
|
2
|
+
import type { DevsysState, SliceRef } from "./types.ts";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* How a slice moves through the phases once it is implementing. Pure: the tools and guards that observe the
|
|
6
|
+
* events (a review round, an edit, a push) call these and write the result to state.
|
|
7
|
+
*
|
|
8
|
+
* implementing → reviewing a review round starts or is recorded without being satisfied
|
|
9
|
+
* reviewing → delivering the review is satisfied on the current diff
|
|
10
|
+
* delivering → reviewing a later round finds something again
|
|
11
|
+
* delivering → implementing production source is edited (the review no longer covers it)
|
|
12
|
+
* any open → idle the slice is delivered or abandoned (`closeSlice`)
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
const IN_FLIGHT = new Set<DevsysState["phase"]>(["implementing", "reviewing", "delivering"]);
|
|
16
|
+
|
|
17
|
+
/** The state after a review round was started or recorded; `action` is `nextAction` for the diff as it is now. */
|
|
18
|
+
export function afterReviewRound(
|
|
19
|
+
state: DevsysState,
|
|
20
|
+
action: ReviewAction,
|
|
21
|
+
slice: SliceRef | undefined = state.activeSlice,
|
|
22
|
+
): DevsysState {
|
|
23
|
+
if (state.activeSlice === undefined || slice !== state.activeSlice) return state;
|
|
24
|
+
if (!IN_FLIGHT.has(state.phase)) return state;
|
|
25
|
+
return { ...state, phase: action === "done" ? "delivering" : "reviewing" };
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/** Editing production source after review reopens the slice, so red-first applies to the edit again. */
|
|
29
|
+
export const afterSourceEdit = (state: DevsysState): DevsysState =>
|
|
30
|
+
state.phase === "delivering" ? { ...state, phase: "implementing" } : state;
|
|
31
|
+
|
|
32
|
+
/** Back to idle: the slice, its sizing and its slice-scoped departures are done. Review history is kept. */
|
|
33
|
+
export function closeSlice(state: DevsysState): DevsysState {
|
|
34
|
+
const { activeSlice, sizing: _sizing, ...rest } = state;
|
|
35
|
+
return {
|
|
36
|
+
...rest,
|
|
37
|
+
phase: "idle",
|
|
38
|
+
openDepartures: state.openDepartures.filter(
|
|
39
|
+
(d) => d.scope.kind !== "slice" || d.scope.slice !== activeSlice,
|
|
40
|
+
),
|
|
41
|
+
};
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/** A recorded `review.unsatisfied` departure covers this slice, so no review is owed before it ships. */
|
|
45
|
+
export const reviewWaived = (state: DevsysState): boolean =>
|
|
46
|
+
state.openDepartures.some(
|
|
47
|
+
(d) =>
|
|
48
|
+
d.gate === "review.unsatisfied" &&
|
|
49
|
+
(d.scope.kind === "session" ||
|
|
50
|
+
(d.scope.kind === "slice" && d.scope.slice === state.activeSlice)),
|
|
51
|
+
);
|
|
@@ -17,6 +17,7 @@ import { resolveGit } from "../core/git-invocations.ts";
|
|
|
17
17
|
import { DECISION_LOG, reviewGap } from "../core/review-flow.ts";
|
|
18
18
|
import { type GateId, isParseError, parseGateId } from "../core/types.ts";
|
|
19
19
|
import type { Jev } from "../jev/client.ts";
|
|
20
|
+
import { ARCHITECTURE_THRESHOLD, judgeArchitectureShaping } from "../jev/questions/architecture.ts";
|
|
20
21
|
import { judgeCommit, MIX_THRESHOLD, RATIONALE_FLOOR } from "../jev/questions/commit.ts";
|
|
21
22
|
import { snapshotDiff } from "../review/digest.ts";
|
|
22
23
|
import type { SessionState } from "../state/session-state.ts";
|
|
@@ -40,6 +41,13 @@ const gateId = (id: string): GateId => {
|
|
|
40
41
|
const RATIONALE = gateId("commit.rationale");
|
|
41
42
|
const MIXED = gateId("commit.mixed-change");
|
|
42
43
|
const REVIEW = gateId("review.unsatisfied");
|
|
44
|
+
const ADR_MISSING = gateId("adr.missing");
|
|
45
|
+
|
|
46
|
+
/** A new `docs/adr/NNNN-*.md` among the pending changes; editing an older ADR is not recording a new decision. */
|
|
47
|
+
const addsAdr = (diff: string): boolean =>
|
|
48
|
+
/^diff --git a\/docs\/adr\/\d{4}-\S+\.md b\/\S+\n(?:new file mode|rename from |similarity index)/m.test(
|
|
49
|
+
diff,
|
|
50
|
+
);
|
|
43
51
|
|
|
44
52
|
const readMessageFile = (cwd: string, path: string): string | undefined => {
|
|
45
53
|
try {
|
|
@@ -72,25 +80,85 @@ const stagesEverything = (command: string): boolean =>
|
|
|
72
80
|
command,
|
|
73
81
|
);
|
|
74
82
|
|
|
83
|
+
/**
|
|
84
|
+
* The pathspecs a `git add` in the command stages new files under: `.` for `-A` with no path, else the named
|
|
85
|
+
* paths. Empty when nothing in the command can add an untracked file (`-u` stages only tracked files, `-n` nothing).
|
|
86
|
+
*/
|
|
87
|
+
const addedPathspecs = (command: string): string[] =>
|
|
88
|
+
[...command.matchAll(/\bgit\s+add\b([^;&|\n]*)/g)].flatMap((m) => {
|
|
89
|
+
const words = (m[1] ?? "")
|
|
90
|
+
.trim()
|
|
91
|
+
.split(/\s+/)
|
|
92
|
+
.filter(Boolean)
|
|
93
|
+
.map((w) => w.replace(/^(["'])(.*)\1$/, "$2")); // `git add "docs/adr/0005-x.md"` names the same file
|
|
94
|
+
if (words.some((w) => ["-u", "--update", "-n", "--dry-run"].includes(w))) return [];
|
|
95
|
+
const paths = words.filter((w) => !w.startsWith("-"));
|
|
96
|
+
if (paths.length === 0) return words.some((w) => w === "-A" || w === "--all") ? ["."] : [];
|
|
97
|
+
return paths.map((p) => p.replace(/^\.\//, "").replace(/\/$/, "") || ".");
|
|
98
|
+
});
|
|
99
|
+
|
|
100
|
+
type PendingDiff = { stat: string; diff: string; untrackedAdr: boolean };
|
|
101
|
+
|
|
75
102
|
/** What is about to be committed: the staged changes, or every tracked change when the commit stages them itself. */
|
|
76
103
|
async function pendingDiff(
|
|
77
104
|
exec: Exec,
|
|
78
105
|
cwd: string,
|
|
79
106
|
command: string,
|
|
80
|
-
): Promise<
|
|
107
|
+
): Promise<PendingDiff | undefined> {
|
|
81
108
|
try {
|
|
82
|
-
const
|
|
83
|
-
const
|
|
109
|
+
const widened = stagesEverything(command);
|
|
110
|
+
const specs = addedPathspecs(command);
|
|
111
|
+
// Fixed prefixes and no external driver: `addsAdr` reads the headers, and git config can change them.
|
|
112
|
+
const fixed = ["--no-ext-diff", "--no-color", "--src-prefix=a/", "--dst-prefix=b/"];
|
|
113
|
+
const base = widened ? ["diff", ...fixed, "HEAD"] : ["diff", ...fixed, "--cached"];
|
|
114
|
+
const [stat, diff, untracked] = await Promise.all([
|
|
84
115
|
exec("git", [...base, "--stat"], { cwd, timeout: 10_000 }),
|
|
85
116
|
exec("git", base, { cwd, timeout: 10_000 }),
|
|
117
|
+
// `git diff HEAD` leaves out files git does not track yet, such as a new module or an ADR just created.
|
|
118
|
+
specs.length > 0
|
|
119
|
+
? exec("git", ["ls-files", "--others", "--exclude-standard", "--", ...new Set(specs)], {
|
|
120
|
+
cwd,
|
|
121
|
+
timeout: 10_000,
|
|
122
|
+
})
|
|
123
|
+
: Promise.resolve(undefined),
|
|
86
124
|
]);
|
|
87
|
-
if (stat.code !== 0 || diff.code !== 0
|
|
88
|
-
|
|
125
|
+
if (stat.code !== 0 || diff.code !== 0) return undefined;
|
|
126
|
+
const added = untracked?.code === 0 ? untracked.stdout.trim() : "";
|
|
127
|
+
if (diff.stdout.trim() === "" && added === "") return undefined;
|
|
128
|
+
return {
|
|
129
|
+
stat:
|
|
130
|
+
added === ""
|
|
131
|
+
? stat.stdout
|
|
132
|
+
: // First, because Jev clips the stat: a long list of tracked files must not push new files out of view.
|
|
133
|
+
`New files this commit adds (untracked):\n${added}\n\n${stat.stdout}`,
|
|
134
|
+
diff: diff.stdout,
|
|
135
|
+
untrackedAdr: /^docs\/adr\/\d{4}-\S+\.md$/m.test(added),
|
|
136
|
+
};
|
|
89
137
|
} catch {
|
|
90
138
|
return undefined;
|
|
91
139
|
}
|
|
92
140
|
}
|
|
93
141
|
|
|
142
|
+
/** Non-negotiable 9 as a soft gate: the judgement is probabilistic, so a departure can answer it. */
|
|
143
|
+
async function adrNeeds(
|
|
144
|
+
deps: CommitGuardDeps,
|
|
145
|
+
ctx: ExtensionContext,
|
|
146
|
+
pending: PendingDiff,
|
|
147
|
+
): Promise<Need[]> {
|
|
148
|
+
if (addsAdr(pending.diff) || pending.untrackedAdr) return [];
|
|
149
|
+
const judged = await judgeArchitectureShaping(deps.jev(ctx), {
|
|
150
|
+
diffStat: pending.stat,
|
|
151
|
+
diff: pending.diff,
|
|
152
|
+
});
|
|
153
|
+
if (!judged.ok || judged.value < ARCHITECTURE_THRESHOLD) return [];
|
|
154
|
+
return [
|
|
155
|
+
{
|
|
156
|
+
gate: ADR_MISSING,
|
|
157
|
+
why: "Jev reads this diff as an architecture-shaping decision (a boundary, dependency, data format or protocol) and it adds no ADR",
|
|
158
|
+
},
|
|
159
|
+
];
|
|
160
|
+
}
|
|
161
|
+
|
|
94
162
|
async function jevNeeds(
|
|
95
163
|
deps: CommitGuardDeps,
|
|
96
164
|
ctx: ExtensionContext,
|
|
@@ -100,13 +168,13 @@ async function jevNeeds(
|
|
|
100
168
|
): Promise<Need[]> {
|
|
101
169
|
const pending = await pendingDiff(deps.exec, ctx.cwd, command);
|
|
102
170
|
if (pending === undefined) return [];
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
diffStat: pending.stat,
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
171
|
+
// Both ask Jev; run them together so a hung provider costs one timeout, not two.
|
|
172
|
+
const [judged, adr] = await Promise.all([
|
|
173
|
+
judgeCommit(deps.jev(ctx), { message, diffStat: pending.stat, diff: pending.diff }),
|
|
174
|
+
adrNeeds(deps, ctx, pending),
|
|
175
|
+
]);
|
|
176
|
+
const needs: Need[] = adr;
|
|
177
|
+
if (!judged.ok) return needs;
|
|
110
178
|
if (bodyPresent && judged.value.rationale < RATIONALE_FLOOR) {
|
|
111
179
|
needs.push({ gate: RATIONALE, why: "Jev reads the body as restating what changed, not why" });
|
|
112
180
|
}
|
|
@@ -126,7 +194,8 @@ async function reviewNeeds(
|
|
|
126
194
|
command: string,
|
|
127
195
|
): Promise<Need[]> {
|
|
128
196
|
const { phase, activeSlice } = deps.state.get();
|
|
129
|
-
|
|
197
|
+
const inFlight = phase === "implementing" || phase === "reviewing" || phase === "delivering";
|
|
198
|
+
if (!inFlight || activeSlice === undefined) return [];
|
|
130
199
|
const snap = await snapshotDiff(deps.exec, ctx.cwd, "HEAD");
|
|
131
200
|
// An unreadable diff cannot be compared; the gate asks for a departure rather than guessing.
|
|
132
201
|
const gap = reviewGap(
|
|
@@ -247,13 +316,22 @@ const reviewReason = (need: Need): string =>
|
|
|
247
316
|
"and repeat until the review is satisfied. If skipping review is deliberate call devsys_record_departure " +
|
|
248
317
|
`with gate "${need.gate}", what you are doing instead, why, and the cost if wrong; then retry.`;
|
|
249
318
|
|
|
250
|
-
const
|
|
251
|
-
need.gate
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
319
|
+
const adrReason = (need: Need): string =>
|
|
320
|
+
`${need.gate}: ${need.why}. Hard-to-reverse decisions are recorded as an ADR in the same change. ` +
|
|
321
|
+
"Call devsys_adr_new with the decision's title, fill in the ADR and stage it. If this diff is not " +
|
|
322
|
+
`architecture-shaping, call devsys_record_departure with gate "${need.gate}", what you are doing instead, ` +
|
|
323
|
+
"why, and the cost if wrong; then retry.";
|
|
324
|
+
|
|
325
|
+
const commitReason = (need: Need): string =>
|
|
326
|
+
`${need.gate}: ${need.why}. Commit messages carry their rationale and structural and behavioural ` +
|
|
327
|
+
"changes go in separate commits. Fix the commit (split it, or write the why in the body), or if " +
|
|
328
|
+
`departing is deliberate call devsys_record_departure with gate "${need.gate}", what you are doing ` +
|
|
329
|
+
"instead, why, and the cost if wrong; then retry.";
|
|
330
|
+
|
|
331
|
+
function blockReason(need: Need): string {
|
|
332
|
+
if (need.gate === REVIEW) return reviewReason(need);
|
|
333
|
+
return need.gate === ADR_MISSING ? adrReason(need) : commitReason(need);
|
|
334
|
+
}
|
|
257
335
|
|
|
258
336
|
const forbiddenReason = (found: readonly string[]): string =>
|
|
259
337
|
`commit.forbidden-trailer: this commit carries an AI attribution (${found.join("; ")}). ` +
|
|
@@ -2,6 +2,7 @@ import { readFileSync } from "node:fs";
|
|
|
2
2
|
import { homedir } from "node:os";
|
|
3
3
|
import { resolve } from "node:path";
|
|
4
4
|
import { type ExtensionAPI, isToolCallEventType } from "@earendil-works/pi-coding-agent";
|
|
5
|
+
import { afterSourceEdit } from "../core/lifecycle.ts";
|
|
5
6
|
import { classifyPath } from "../core/path-class.ts";
|
|
6
7
|
import { normalizeRepoPath } from "../core/test-paths.ts";
|
|
7
8
|
import type { SessionState } from "../state/session-state.ts";
|
|
@@ -57,7 +58,7 @@ const JUDGED_EXEMPTIONS =
|
|
|
57
58
|
"straightforward CI scripting, a simple dev-environment utility, or a behaviour-preserving refactor with green coverage";
|
|
58
59
|
|
|
59
60
|
/**
|
|
60
|
-
* Soft gate `tdd.red-first`: while implementing, production source is edited only after a failing
|
|
61
|
+
* Soft gate `tdd.red-first`: while a slice is in flight (implementing, reviewing, delivering), production source is edited only after a failing
|
|
61
62
|
* test run has been observed. Test, docs, config and generated files are exempt by path; the
|
|
62
63
|
* judged exemptions go through one recorded departure, which covers its whole slice.
|
|
63
64
|
*/
|
|
@@ -69,11 +70,13 @@ export function registerRedFirstGuard(deps: RedFirstGuardDeps): void {
|
|
|
69
70
|
deps.pi.on("tool_call", (event, ctx) => {
|
|
70
71
|
if (!(isToolCallEventType("edit", event) || isToolCallEventType("write", event)))
|
|
71
72
|
return undefined;
|
|
72
|
-
const { phase,
|
|
73
|
-
if (phase !== "implementing"
|
|
74
|
-
|
|
73
|
+
const { phase, activeSlice } = deps.state.get();
|
|
74
|
+
if (phase !== "implementing" && phase !== "reviewing" && phase !== "delivering")
|
|
75
|
+
return undefined;
|
|
75
76
|
const path = normalizeRepoPath(ctx.cwd, event.input.path, homedir());
|
|
76
77
|
if (classifyPath(path) !== "source") return undefined;
|
|
78
|
+
const { lastTestRun } = deps.state.get();
|
|
79
|
+
if (lastTestRun !== undefined && lastTestRun.exitCode !== 0) return undefined;
|
|
77
80
|
if (touchesInlineTest(event.input, () => existing(ctx.cwd, path))) return undefined;
|
|
78
81
|
if (departure.covers()) return undefined;
|
|
79
82
|
const seen =
|
|
@@ -90,4 +93,17 @@ export function registerRedFirstGuard(deps: RedFirstGuardDeps): void {
|
|
|
90
93
|
`departure covers the rest of the ${scope}.`,
|
|
91
94
|
};
|
|
92
95
|
});
|
|
96
|
+
|
|
97
|
+
// The review covered the code as it was: source that was really written puts the slice back to implementing.
|
|
98
|
+
deps.pi.on("tool_result", (event, ctx) => {
|
|
99
|
+
if (event.isError || (event.toolName !== "edit" && event.toolName !== "write"))
|
|
100
|
+
return undefined;
|
|
101
|
+
if (deps.state.get().phase !== "delivering") return undefined;
|
|
102
|
+
const given = (event.input as { path?: unknown }).path;
|
|
103
|
+
if (typeof given !== "string") return undefined;
|
|
104
|
+
if (classifyPath(normalizeRepoPath(ctx.cwd, given, homedir())) === "source") {
|
|
105
|
+
deps.state.update(afterSourceEdit);
|
|
106
|
+
}
|
|
107
|
+
return undefined;
|
|
108
|
+
});
|
|
93
109
|
}
|