@jwilger/pi-development-system 0.83.0 → 0.85.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -19,8 +19,20 @@ Everything the system offers is reachable without remembering a command. Describ
19
19
  plain words: when a prompt asks for new work, a fix or a review, the system adds a guideline
20
20
  naming the tool to use (`devsys_intake`, `devsys_review_start`). A `devsys` tool always shows
21
21
  what the current phase expects. When you say a slice is finished with no review round recorded,
22
- it tells you to start one. The slash commands (`/devsys-start`, `/devsys-plan`, `/devsys-review`)
23
- are shortcuts to the same tools.
22
+ it tells you to start one. The slash commands (`/devsys-start`, `/devsys-plan`, `/devsys-review`,
23
+ `/devsys-lens-review`, `/devsys-adr`) are shortcuts to the same tools.
24
+
25
+ Product planning has a skill (`product-planning`: brief, decision register, follow-ups,
26
+ terminology, journeys). `devsys_lens_review` plans a review of the brief by five product lenses and
27
+ writes the packets to `docs/product/reviews/`; `devsys_adr_new` creates the next numbered ADR, and
28
+ a commit that shapes the architecture without one is stopped by the soft gate `adr.missing`.
29
+
30
+ A slice has a life cycle: `implementing` → `reviewing` (when a review round starts) → `delivering`
31
+ (review satisfied) → `idle`. Editing production source while delivering reopens `implementing`,
32
+ and red-first stays on throughout. A push of a clean tree closes a delivered slice by itself (a commit
33
+ in `local-only` mode; CI is not awaited), and `devsys_finish_slice` closes it explicitly or, with a
34
+ reason, abandons it. A plan's increments each end in a push, so each increment is its own slice: the
35
+ next one starts with `devsys_intake`.
24
36
 
25
37
  With `codemode` enabled (`"defaultTools": ["+codemode"]` in pi settings), rarely used tools are
26
38
  reached through scripts and the `judge_*` Jev wrappers exist for scripts only; without codemode
@@ -27,9 +27,12 @@ import { createRequestApprovalTool } from "../src/gates/request-approval-tool.ts
27
27
  import { registerTestGuard } from "../src/gates/test-guard.ts";
28
28
  import { createJevHolder } from "../src/jev/holder.ts";
29
29
  import { createJudgeTools } from "../src/jev/judge-tools.ts";
30
+ import { createAdrNewTool } from "../src/planning/adr-tool.ts";
30
31
  import { createBeginWorkTool } from "../src/planning/begin-tool.ts";
31
32
  import { createIntakeTool } from "../src/planning/intake-tool.ts";
33
+ import { createFinishSliceTool, registerSliceClose } from "../src/planning/slice-close.ts";
32
34
  import { createTaskCheckTool } from "../src/planning/task-check-tool.ts";
35
+ import { createLensReviewTool } from "../src/review/lens-review-tool.ts";
33
36
  import { createReviewRecordTool, createReviewStartTool } from "../src/review/review-tools.ts";
34
37
  import { registerCiCommand } from "../src/state/ci-command.ts";
35
38
  import { loadConfig } from "../src/state/config.ts";
@@ -156,6 +159,7 @@ export function createDevelopmentSystem(pi: ExtensionAPI) {
156
159
  },
157
160
  });
158
161
  registerRedFirstGuard({ pi, state });
162
+ registerSliceClose({ pi, state, exec });
159
163
  registerLintSuppressionGuard({ pi, state });
160
164
  pi.registerTool(createRecordDepartureTool({ pi, state }));
161
165
  pi.registerTool(createRequestApprovalTool({ pi, approvals }));
@@ -174,6 +178,11 @@ export function createDevelopmentSystem(pi: ExtensionAPI) {
174
178
  }
175
179
  pi.registerTool(createPhaseTool({ state }));
176
180
  pi.registerTool(createBeginWorkTool({ state }));
181
+ pi.registerTool(createAdrNewTool({ now: () => new Date() }));
182
+ pi.registerTool(createFinishSliceTool({ state, exec }));
183
+ pi.registerTool(
184
+ createLensReviewTool({ state, jev: (ctx) => jevHolder.forContext(ctx), now: () => new Date() }),
185
+ );
177
186
  pi.registerTool(createIntakeTool({ pi, state, jev: (ctx) => jevHolder.forContext(ctx) }));
178
187
  const reviewDeps = {
179
188
  state,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@jwilger/pi-development-system",
3
- "version": "0.83.0",
3
+ "version": "0.85.0",
4
4
  "description": "A pi extension package representing a seasoned approach to software development using a full AI SDLC.",
5
5
  "keywords": [
6
6
  "pi-package"
@@ -0,0 +1,12 @@
1
+ ---
2
+ description: Record a hard-to-reverse technical decision as the next numbered ADR
3
+ argument-hint: "<decision title>"
4
+ ---
5
+ Record an architecture decision with the product-planning skill.
6
+
7
+ 1. Call `devsys_adr_new` with the title: $ARGUMENTS. It creates `docs/adr/NNNN-<slug>.md` from the template and returns the path.
8
+ 2. Fill in Context (the forces and the problem), Decision (stated plainly), Consequences (positive and negative), Alternatives (each with why it was rejected) and Revisit when (a concrete condition), then link related ADRs and plan sections.
9
+ 3. Set the status to `accepted` only when the user has agreed; leave it `proposed` otherwise.
10
+ 4. Report the path and a one-line summary of the decision.
11
+
12
+ Use an ADR for a boundary, dependency, data format or protocol that is hard to reverse. Product decisions belong in the decision register instead.
@@ -0,0 +1,12 @@
1
+ ---
2
+ description: Run the five product lenses over the brief and record the packets
3
+ argument-hint: "[brief path] [round 1|2]"
4
+ ---
5
+ Review the product brief with the product-planning skill.
6
+
7
+ 1. Call `devsys_lens_review` (brief path and round: $ARGUMENTS; omit either for `docs/product/brief.md` and round 1).
8
+ 2. Run the codemode script it returns, unchanged. It spawns the lens agents, waits, writes the packets to `docs/product/reviews/<date>-round<n>.md` and returns only a verdict per lens and the path. If codemode is unavailable, use the `agent_spawn` payloads it lists.
9
+ 3. Read the review file, not the agents' full output. Report each lens verdict and the findings with their `path:line`.
10
+ 4. After round 2, write the synthesis (R-table and a one-question agenda) from the template in the reply, then ask the user that one question.
11
+
12
+ Agreement among agents is useful critique, not customer evidence.
@@ -9,4 +9,4 @@ Write the plan for the current work with the work-intake-and-slicing skill.
9
9
  3. Write each task so that someone who sees only that task could finish it. Never write `TBD`.
10
10
  4. Run `devsys_task_check` on every task record you wrote and fix each one until it says ready. Split any that come back too-big.
11
11
  5. Stop and ask the user to review the plan before any implementation starts. Do not start a goal yet: a goal tool may continue on its own and begin implementing before the review.
12
- 6. Only after the user approves the plan: call `devsys_begin_work` (so the review and red-first gates are on while the plan is built; it does nothing if the work is already implementing). Then, if a `create_goal` tool exists in this session, call it with the objective "complete docs/plan/<slug>.md honouring its Constraints and Progress checklist, one increment at a time, stopping after each increment's Release step". Otherwise tell the user the plan path so they can start a goal on it. Do not depend on any goal tool's file format.
12
+ 6. Only after the user approves the plan: call `devsys_begin_work` (so the review and red-first gates are on while the increment is built; it does nothing if the work is already implementing). A push of a clean, reviewed tree closes the slice, so each later increment starts with `devsys_intake`. Then, if a `create_goal` tool exists in this session, call it with the objective "complete docs/plan/<slug>.md honouring its Constraints and Progress checklist, one increment at a time, stopping after each increment's Release step". Otherwise tell the user the plan path so they can start a goal on it. Do not depend on any goal tool's file format.
@@ -0,0 +1,86 @@
1
+ ---
2
+ name: product-planning
3
+ description: Plan a product or capability before building it - discovery interview, brief, decision register, follow-ups, terminology, journey inventory, lens review and ADRs. Use when sizing says capability or product, when asked for a brief or a discovery interview, when an answer must be recorded as a decision, or when a plan needs a product review.
4
+ ---
5
+
6
+ # Product planning
7
+
8
+ Planning here finds out what is worth building. It is not a spec for code. The artifacts are a small set of
9
+ linked documents; each has one job and a template under `references/`.
10
+
11
+ | Artifact | Job | Template |
12
+ |---|---|---|
13
+ | Brief | The one narrative: outcome, customers, four risks, assumptions, scope, non-goals | [brief](references/brief.md) |
14
+ | Decision register | The spine: D (decisions), Q (questions), F (scheduling and exclusions) | [decisions](references/decisions.md) |
15
+ | Follow-ups | P items and R review findings: not lost, not now | [followups](references/followups.md) |
16
+ | Terminology | One meaning per word | [terminology](references/terminology.md) |
17
+ | Journeys | J items: a user, actions, an outcome | [journeys](references/journeys.md) |
18
+
19
+ Put them under `docs/product/` (`brief.md`, `decisions.md`, `followups.md`, `terminology.md`, `journeys.md`).
20
+ Copy a template only when the work needs that artifact: `capability` work gets a short brief and journeys,
21
+ `product` work gets the lot. Skipping one is a recorded departure (`artifact.skipped`).
22
+
23
+ ## The interview loop
24
+
25
+ Quote this rule and follow it exactly:
26
+
27
+ > Update docs after each answer, ask one next question, yield.
28
+
29
+ 1. Ask one question. Make it the most decision-relevant open Q item.
30
+ 2. Stop. Wait for the answer. Do not ask a second question or guess ahead.
31
+ 3. Record the answer as a D-item in the register, in the owner's words, and add a dated line to the interview log.
32
+ 4. Edit the brief **only where the answered question touches it**. Bump its version. Nothing else changes.
33
+ 5. Ask the next question and yield again.
34
+
35
+ Every answer becomes a D-item, including "no" and "not now" (a Deferred item with a revisit point).
36
+
37
+ ## Do not over-edit the brief
38
+
39
+ The commonest failure is an agent that rewrites the brief between answers. It misframed the team, moved the
40
+ timing boundary, let later-phase ideas leak in and relocated the business case, and the owner had to find and
41
+ undo each one. So:
42
+
43
+ - An edit is limited to the answered question. If you notice something else that is wrong, add a Q item, do not fix it silently.
44
+ - Ideas for later go to follow-ups. Nothing speculative goes in the brief.
45
+ - Show the brief diff after each answer so the owner can see exactly what moved.
46
+ - The owner owns judgements about the domain; you own process and where things are filed.
47
+
48
+ ## Outcome, risks, assumptions
49
+
50
+ - Open with an **outcome** (direction plus target), not a feature. State the problem to solve, not the solution to build.
51
+ - Assess the four risks (value, usability, feasibility, viability). "Unassessed" is a legal, recorded state.
52
+ - Write assumptions specifically; the smaller the assumption, the smaller the test. Test the riskiest first
53
+ (critical to success, least evidence). Compare at least two solutions, and record why when you only had one.
54
+ - Say whether the work builds to learn (disposable, fewer gates) or builds to earn (full gates). That is a decision, never an inference.
55
+
56
+ ## Deferral is not exclusion
57
+
58
+ Postponing the details of a capability does not remove it from scope. Only an explicit scope decision does.
59
+ A deferred item records who deferred it, why, the interim constraint and when to revisit. A non-goal is
60
+ something we decided not to do.
61
+
62
+ ## Journeys
63
+
64
+ Cut the backlog from journeys, not features. A journey passes if it tells the story of a user performing a
65
+ set of actions to achieve an outcome. Decide journeys before generating tasks; feature-shaped backlogs had to
66
+ be thrown away and redone.
67
+
68
+ ## Lens review
69
+
70
+ Run `devsys_lens_review` (or `/devsys-lens-review`) on the brief when the outcome, scope or risks are settled
71
+ enough to be wrong in an interesting way. Five fresh agents read it, each through one lens (Cagan, Torres,
72
+ Pichler, Perri, Rumelt). Round one is independent; round two is a peer exchange; you then synthesise an R table
73
+ and a one-question agenda, and the interview loop resumes. **Agreement among agents is useful critique, not
74
+ customer evidence.** Record rejected recommendations in the register instead of dropping them.
75
+
76
+ ## Decisions that need an ADR
77
+
78
+ A decision about how the code is built that is hard to reverse (a boundary, a dependency, a data format, a
79
+ protocol) gets an ADR through `devsys_adr_new`. Product decisions stay in the register. Do not write an ADR
80
+ for something the register already holds, or the other way round.
81
+
82
+ ## Status words
83
+
84
+ Everything produced here is "proposed" or "advisory" until a person approves a specific revision. Say
85
+ "readiness" for declared checks, "approval" for a person's decision on a named revision, and never let one
86
+ stand in for the other.
@@ -0,0 +1,48 @@
1
+ # Brief: <product or change name>
2
+
3
+ - **Context:** <which product or process this brief is about>
4
+ - **Status:** v0.1 · <date> · links: [decisions](decisions.md), [follow-ups](followups.md)
5
+ - **Not an authorisation:** this brief records what is understood so far. It does not approve building anything.
6
+
7
+ ## Outcome
8
+
9
+ One sentence: the change in customer behaviour or business result we want, with a direction and a target
10
+ (for example "first-session success from 22% to 25%"). An output ("ship X") is not an outcome.
11
+
12
+ ## Customers and the problem
13
+
14
+ Who has the problem, what they do today, and the evidence for it (link the source, date it). Say plainly what
15
+ is evidence and what is belief.
16
+
17
+ ## Four risks
18
+
19
+ | Risk | What could be wrong | Evidence so far | Cheapest test |
20
+ |---|---|---|---|
21
+ | Value | Nobody wants or uses it | unassessed | |
22
+ | Usability | People cannot work it out | unassessed | |
23
+ | Feasibility | We cannot build it | unassessed | |
24
+ | Viability | It does not work for the business | unassessed | |
25
+
26
+ "Unassessed" is a legal state. Writing it is better than guessing.
27
+
28
+ ## Assumptions
29
+
30
+ Each assumption is specific, tagged with a risk category and a test status. Riskiest first (critical to
31
+ success, little evidence). Every assumption gets a D-item in the register once answered.
32
+
33
+ | ID | Assumption | Category | Test status |
34
+ |---|---|---|---|
35
+ | A01 | | desirability / viability / feasibility / usability / ethical | untested |
36
+
37
+ ## Scope
38
+
39
+ What the first release does, in the customer's words.
40
+
41
+ ## Non-goals
42
+
43
+ Things we have decided not to do. Only an explicit scope decision belongs here.
44
+ **Deferral is not exclusion:** an item that is merely later goes to [follow-ups](followups.md), not here.
45
+
46
+ ## Open questions
47
+
48
+ Pointers only (Q-ids in [decisions](decisions.md)). The question text lives in the register, not the brief.
@@ -0,0 +1,33 @@
1
+ # Decision register
2
+
3
+ The referential spine: every brief, review and journey cites these ids.
4
+
5
+ ## Closure rules
6
+
7
+ - **Resolved:** the owner answered or accepted it.
8
+ - **Deferred:** postponed to a later phase. State who decided, why, the interim constraint and the revisit
9
+ point. Deferring the specification does not defer the capability; only an explicit scope decision does.
10
+ - **Open:** not yet answered.
11
+ - Superseding keeps the old row and points at the new one (`D06 → D56`). Rows are never deleted.
12
+
13
+ ## Decisions (D)
14
+
15
+ | ID | Decision | Status | Source |
16
+ |---|---|---|---|
17
+ | D01 | | Resolved | <answer, date> |
18
+
19
+ ## Questions (Q)
20
+
21
+ | ID | Question | Status | Needed for closure |
22
+ |---|---|---|---|
23
+ | Q01 | | Open | |
24
+
25
+ ## Scheduling, conditional scope and exclusions (F)
26
+
27
+ | ID | Item | Disposition | Rationale | Revisit when |
28
+ |---|---|---|---|---|
29
+ | F01 | | true exclusion / conditional / scheduled | | |
30
+
31
+ ## Interview log
32
+
33
+ One dated bullet per answer: the question, the answer in the owner's words, and "Recorded as D<nn>".
@@ -0,0 +1,16 @@
1
+ # Planning follow-ups
2
+
3
+ Where "do not lose it, do not do it now" goes. **Scheduled planning work is not an MVP exclusion.**
4
+
5
+ | ID | Question or work | Appropriate phase | Revisit when | Decision refs |
6
+ |---|---|---|---|---|
7
+ | P01 | | modelling / design / architecture / pilot / commercial | | D01 |
8
+
9
+ ## Review findings (R)
10
+
11
+ Findings from a lens review and where each was routed (decision register, follow-up, interview question,
12
+ rejected). A rejected recommendation is recorded with the reason, never silently dropped.
13
+
14
+ | ID | Finding | Route | Disposition |
15
+ |---|---|---|---|
16
+ | R01 | | | |
@@ -0,0 +1,14 @@
1
+ # Journey inventory
2
+
3
+ A journey passes this test: **it tells the story of a user performing a set of actions to achieve an outcome.**
4
+ A feature, a screen or a technical layer does not pass it. Do this before cutting any backlog.
5
+
6
+ | ID | User and starting need | Actions | Outcome |
7
+ |---|---|---|---|
8
+ | J01 | | | |
9
+
10
+ ## Per journey
11
+
12
+ - Alternative outcomes (what happens when it goes wrong).
13
+ - The decision ids it rests on.
14
+ - Approval is of a specific revision, recorded with person and date; drafting is not approval.
@@ -0,0 +1,10 @@
1
+ # Terminology
2
+
3
+ Words used in the brief, register and journeys, each with one meaning. Label a document with the sense of a
4
+ word it uses when the word has more than one.
5
+
6
+ | Term | Meaning | Not to be confused with |
7
+ |---|---|---|
8
+ | Deferred item | Scheduled to a later phase | Exclusion |
9
+ | Readiness | Declared checks pass | Human approval of a specific revision |
10
+ | Lens | An advisory reviewing perspective | A review by the person it is named for |
@@ -10,11 +10,11 @@ export function phaseGuide(phase: Phase): string {
10
10
  case "planning":
11
11
  return "Planning. Produce the artifacts the sizing asked for, write task records (check each with devsys_task_check), and when the user approves the plan call devsys_begin_work.";
12
12
  case "implementing":
13
- return "Implementing one slice. Write a failing test first, make it pass with the least code, run the tests and read the result before claiming anything. Push small, verified increments.";
13
+ return "Implementing one slice. Write a failing test first, make it pass with the least code, run the tests and read the result before claiming anything. Push when the slice is complete and reviewed: a push of a clean tree closes it (devsys_finish_slice closes it explicitly).";
14
14
  case "reviewing":
15
- return "Reviewing. Call devsys_review_start for a fresh-context review, pass its packet and diffDigest to devsys_review_record, fix every blocking and should-fix finding, and repeat until the round is clean.";
15
+ return "Reviewing. Call devsys_review_start for a fresh-context review, pass its packet and diffDigest to devsys_review_record, fix every blocking and should-fix finding (red-first still applies), and repeat until the review is satisfied; then the slice moves to delivering.";
16
16
  case "delivering":
17
- return "Delivering. Commit with a rationale, push to trunk, wait for CI to go green, and release before starting the next slice. A red trunk is repaired first.";
17
+ return "Delivering. The review is satisfied. Commit with a rationale, push, and release; a push of a clean tree closes the slice (devsys_finish_slice closes it explicitly, or abandons it with a reason). Editing source reopens implementing. A red trunk (CI) is repaired first.";
18
18
  default:
19
19
  return assertNever(phase);
20
20
  }
@@ -149,17 +149,23 @@ export function registerTurnVerifier(deps: TurnVerifierDeps): void {
149
149
  let evidence: ToolEvidence[] = [];
150
150
  let corrected = 0;
151
151
  let justCorrected = false;
152
+ let inFlightThisRun = false;
152
153
 
153
154
  deps.pi.on("session_start", () => {
154
155
  corrected = 0;
155
156
  justCorrected = false;
157
+ inFlightThisRun = false;
156
158
  evidence = [];
157
159
  });
158
160
  deps.pi.on("agent_start", () => {
161
+ // A push in this run may close the slice; its delivery report is still the claim to check.
162
+ inFlightThisRun = VERIFIED_PHASES.has(deps.state.get().phase);
159
163
  evidence = [];
160
164
  justCorrected = false;
161
165
  });
162
166
  deps.pi.on("tool_result", (event) => {
167
+ // The slice can start inside this run (intake, begin) and be closed by its own push; note it was open.
168
+ if (VERIFIED_PHASES.has(deps.state.get().phase)) inFlightThisRun = true;
163
169
  evidence = [...evidence, evidenceOf(event)].slice(-MAX_EVIDENCE);
164
170
  });
165
171
 
@@ -167,7 +173,7 @@ export function registerTurnVerifier(deps: TurnVerifierDeps): void {
167
173
  // An aborted or errored turn is dropped by pi; judging it would delay the abort and spend a correction.
168
174
  if (event.outcome !== "completed") return undefined;
169
175
  const state = deps.state.get();
170
- if (!VERIFIED_PHASES.has(state.phase)) return undefined;
176
+ if (!(VERIFIED_PHASES.has(state.phase) || inFlightThisRun)) return undefined;
171
177
  const parts = assistantParts(event.message);
172
178
  if (parts === undefined || parts.callsTools || parts.text.trim() === "") return undefined;
173
179
  if (justCorrected) {
@@ -0,0 +1,51 @@
1
+ import type { ReviewAction } from "./review.ts";
2
+ import type { DevsysState, SliceRef } from "./types.ts";
3
+
4
+ /**
5
+ * How a slice moves through the phases once it is implementing. Pure: the tools and guards that observe the
6
+ * events (a review round, an edit, a push) call these and write the result to state.
7
+ *
8
+ * implementing → reviewing a review round starts or is recorded without being satisfied
9
+ * reviewing → delivering the review is satisfied on the current diff
10
+ * delivering → reviewing a later round finds something again
11
+ * delivering → implementing production source is edited (the review no longer covers it)
12
+ * any open → idle the slice is delivered or abandoned (`closeSlice`)
13
+ */
14
+
15
+ const IN_FLIGHT = new Set<DevsysState["phase"]>(["implementing", "reviewing", "delivering"]);
16
+
17
+ /** The state after a review round was started or recorded; `action` is `nextAction` for the diff as it is now. */
18
+ export function afterReviewRound(
19
+ state: DevsysState,
20
+ action: ReviewAction,
21
+ slice: SliceRef | undefined = state.activeSlice,
22
+ ): DevsysState {
23
+ if (state.activeSlice === undefined || slice !== state.activeSlice) return state;
24
+ if (!IN_FLIGHT.has(state.phase)) return state;
25
+ return { ...state, phase: action === "done" ? "delivering" : "reviewing" };
26
+ }
27
+
28
+ /** Editing production source after review reopens the slice, so red-first applies to the edit again. */
29
+ export const afterSourceEdit = (state: DevsysState): DevsysState =>
30
+ state.phase === "delivering" ? { ...state, phase: "implementing" } : state;
31
+
32
+ /** Back to idle: the slice, its sizing and its slice-scoped departures are done. Review history is kept. */
33
+ export function closeSlice(state: DevsysState): DevsysState {
34
+ const { activeSlice, sizing: _sizing, ...rest } = state;
35
+ return {
36
+ ...rest,
37
+ phase: "idle",
38
+ openDepartures: state.openDepartures.filter(
39
+ (d) => d.scope.kind !== "slice" || d.scope.slice !== activeSlice,
40
+ ),
41
+ };
42
+ }
43
+
44
+ /** A recorded `review.unsatisfied` departure covers this slice, so no review is owed before it ships. */
45
+ export const reviewWaived = (state: DevsysState): boolean =>
46
+ state.openDepartures.some(
47
+ (d) =>
48
+ d.gate === "review.unsatisfied" &&
49
+ (d.scope.kind === "session" ||
50
+ (d.scope.kind === "slice" && d.scope.slice === state.activeSlice)),
51
+ );
@@ -17,6 +17,7 @@ import { resolveGit } from "../core/git-invocations.ts";
17
17
  import { DECISION_LOG, reviewGap } from "../core/review-flow.ts";
18
18
  import { type GateId, isParseError, parseGateId } from "../core/types.ts";
19
19
  import type { Jev } from "../jev/client.ts";
20
+ import { ARCHITECTURE_THRESHOLD, judgeArchitectureShaping } from "../jev/questions/architecture.ts";
20
21
  import { judgeCommit, MIX_THRESHOLD, RATIONALE_FLOOR } from "../jev/questions/commit.ts";
21
22
  import { snapshotDiff } from "../review/digest.ts";
22
23
  import type { SessionState } from "../state/session-state.ts";
@@ -40,6 +41,13 @@ const gateId = (id: string): GateId => {
40
41
  const RATIONALE = gateId("commit.rationale");
41
42
  const MIXED = gateId("commit.mixed-change");
42
43
  const REVIEW = gateId("review.unsatisfied");
44
+ const ADR_MISSING = gateId("adr.missing");
45
+
46
+ /** A new `docs/adr/NNNN-*.md` among the pending changes; editing an older ADR is not recording a new decision. */
47
+ const addsAdr = (diff: string): boolean =>
48
+ /^diff --git a\/docs\/adr\/\d{4}-\S+\.md b\/\S+\n(?:new file mode|rename from |similarity index)/m.test(
49
+ diff,
50
+ );
43
51
 
44
52
  const readMessageFile = (cwd: string, path: string): string | undefined => {
45
53
  try {
@@ -72,25 +80,85 @@ const stagesEverything = (command: string): boolean =>
72
80
  command,
73
81
  );
74
82
 
83
+ /**
84
+ * The pathspecs a `git add` in the command stages new files under: `.` for `-A` with no path, else the named
85
+ * paths. Empty when nothing in the command can add an untracked file (`-u` stages only tracked files, `-n` nothing).
86
+ */
87
+ const addedPathspecs = (command: string): string[] =>
88
+ [...command.matchAll(/\bgit\s+add\b([^;&|\n]*)/g)].flatMap((m) => {
89
+ const words = (m[1] ?? "")
90
+ .trim()
91
+ .split(/\s+/)
92
+ .filter(Boolean)
93
+ .map((w) => w.replace(/^(["'])(.*)\1$/, "$2")); // `git add "docs/adr/0005-x.md"` names the same file
94
+ if (words.some((w) => ["-u", "--update", "-n", "--dry-run"].includes(w))) return [];
95
+ const paths = words.filter((w) => !w.startsWith("-"));
96
+ if (paths.length === 0) return words.some((w) => w === "-A" || w === "--all") ? ["."] : [];
97
+ return paths.map((p) => p.replace(/^\.\//, "").replace(/\/$/, "") || ".");
98
+ });
99
+
100
+ type PendingDiff = { stat: string; diff: string; untrackedAdr: boolean };
101
+
75
102
  /** What is about to be committed: the staged changes, or every tracked change when the commit stages them itself. */
76
103
  async function pendingDiff(
77
104
  exec: Exec,
78
105
  cwd: string,
79
106
  command: string,
80
- ): Promise<{ stat: string; diff: string } | undefined> {
107
+ ): Promise<PendingDiff | undefined> {
81
108
  try {
82
- const base = stagesEverything(command) ? ["diff", "HEAD"] : ["diff", "--cached"];
83
- const [stat, diff] = await Promise.all([
109
+ const widened = stagesEverything(command);
110
+ const specs = addedPathspecs(command);
111
+ // Fixed prefixes and no external driver: `addsAdr` reads the headers, and git config can change them.
112
+ const fixed = ["--no-ext-diff", "--no-color", "--src-prefix=a/", "--dst-prefix=b/"];
113
+ const base = widened ? ["diff", ...fixed, "HEAD"] : ["diff", ...fixed, "--cached"];
114
+ const [stat, diff, untracked] = await Promise.all([
84
115
  exec("git", [...base, "--stat"], { cwd, timeout: 10_000 }),
85
116
  exec("git", base, { cwd, timeout: 10_000 }),
117
+ // `git diff HEAD` leaves out files git does not track yet, such as a new module or an ADR just created.
118
+ specs.length > 0
119
+ ? exec("git", ["ls-files", "--others", "--exclude-standard", "--", ...new Set(specs)], {
120
+ cwd,
121
+ timeout: 10_000,
122
+ })
123
+ : Promise.resolve(undefined),
86
124
  ]);
87
- if (stat.code !== 0 || diff.code !== 0 || diff.stdout.trim() === "") return undefined;
88
- return { stat: stat.stdout, diff: diff.stdout };
125
+ if (stat.code !== 0 || diff.code !== 0) return undefined;
126
+ const added = untracked?.code === 0 ? untracked.stdout.trim() : "";
127
+ if (diff.stdout.trim() === "" && added === "") return undefined;
128
+ return {
129
+ stat:
130
+ added === ""
131
+ ? stat.stdout
132
+ : // First, because Jev clips the stat: a long list of tracked files must not push new files out of view.
133
+ `New files this commit adds (untracked):\n${added}\n\n${stat.stdout}`,
134
+ diff: diff.stdout,
135
+ untrackedAdr: /^docs\/adr\/\d{4}-\S+\.md$/m.test(added),
136
+ };
89
137
  } catch {
90
138
  return undefined;
91
139
  }
92
140
  }
93
141
 
142
+ /** Non-negotiable 9 as a soft gate: the judgement is probabilistic, so a departure can answer it. */
143
+ async function adrNeeds(
144
+ deps: CommitGuardDeps,
145
+ ctx: ExtensionContext,
146
+ pending: PendingDiff,
147
+ ): Promise<Need[]> {
148
+ if (addsAdr(pending.diff) || pending.untrackedAdr) return [];
149
+ const judged = await judgeArchitectureShaping(deps.jev(ctx), {
150
+ diffStat: pending.stat,
151
+ diff: pending.diff,
152
+ });
153
+ if (!judged.ok || judged.value < ARCHITECTURE_THRESHOLD) return [];
154
+ return [
155
+ {
156
+ gate: ADR_MISSING,
157
+ why: "Jev reads this diff as an architecture-shaping decision (a boundary, dependency, data format or protocol) and it adds no ADR",
158
+ },
159
+ ];
160
+ }
161
+
94
162
  async function jevNeeds(
95
163
  deps: CommitGuardDeps,
96
164
  ctx: ExtensionContext,
@@ -100,13 +168,13 @@ async function jevNeeds(
100
168
  ): Promise<Need[]> {
101
169
  const pending = await pendingDiff(deps.exec, ctx.cwd, command);
102
170
  if (pending === undefined) return [];
103
- const judged = await judgeCommit(deps.jev(ctx), {
104
- message,
105
- diffStat: pending.stat,
106
- diff: pending.diff,
107
- });
108
- if (!judged.ok) return [];
109
- const needs: Need[] = [];
171
+ // Both ask Jev; run them together so a hung provider costs one timeout, not two.
172
+ const [judged, adr] = await Promise.all([
173
+ judgeCommit(deps.jev(ctx), { message, diffStat: pending.stat, diff: pending.diff }),
174
+ adrNeeds(deps, ctx, pending),
175
+ ]);
176
+ const needs: Need[] = adr;
177
+ if (!judged.ok) return needs;
110
178
  if (bodyPresent && judged.value.rationale < RATIONALE_FLOOR) {
111
179
  needs.push({ gate: RATIONALE, why: "Jev reads the body as restating what changed, not why" });
112
180
  }
@@ -126,7 +194,8 @@ async function reviewNeeds(
126
194
  command: string,
127
195
  ): Promise<Need[]> {
128
196
  const { phase, activeSlice } = deps.state.get();
129
- if ((phase !== "implementing" && phase !== "reviewing") || activeSlice === undefined) return [];
197
+ const inFlight = phase === "implementing" || phase === "reviewing" || phase === "delivering";
198
+ if (!inFlight || activeSlice === undefined) return [];
130
199
  const snap = await snapshotDiff(deps.exec, ctx.cwd, "HEAD");
131
200
  // An unreadable diff cannot be compared; the gate asks for a departure rather than guessing.
132
201
  const gap = reviewGap(
@@ -247,13 +316,22 @@ const reviewReason = (need: Need): string =>
247
316
  "and repeat until the review is satisfied. If skipping review is deliberate call devsys_record_departure " +
248
317
  `with gate "${need.gate}", what you are doing instead, why, and the cost if wrong; then retry.`;
249
318
 
250
- const blockReason = (need: Need): string =>
251
- need.gate === REVIEW
252
- ? reviewReason(need)
253
- : `${need.gate}: ${need.why}. Commit messages carry their rationale and structural and behavioural ` +
254
- "changes go in separate commits. Fix the commit (split it, or write the why in the body), or if " +
255
- `departing is deliberate call devsys_record_departure with gate "${need.gate}", what you are doing ` +
256
- "instead, why, and the cost if wrong; then retry.";
319
+ const adrReason = (need: Need): string =>
320
+ `${need.gate}: ${need.why}. Hard-to-reverse decisions are recorded as an ADR in the same change. ` +
321
+ "Call devsys_adr_new with the decision's title, fill in the ADR and stage it. If this diff is not " +
322
+ `architecture-shaping, call devsys_record_departure with gate "${need.gate}", what you are doing instead, ` +
323
+ "why, and the cost if wrong; then retry.";
324
+
325
+ const commitReason = (need: Need): string =>
326
+ `${need.gate}: ${need.why}. Commit messages carry their rationale and structural and behavioural ` +
327
+ "changes go in separate commits. Fix the commit (split it, or write the why in the body), or if " +
328
+ `departing is deliberate call devsys_record_departure with gate "${need.gate}", what you are doing ` +
329
+ "instead, why, and the cost if wrong; then retry.";
330
+
331
+ function blockReason(need: Need): string {
332
+ if (need.gate === REVIEW) return reviewReason(need);
333
+ return need.gate === ADR_MISSING ? adrReason(need) : commitReason(need);
334
+ }
257
335
 
258
336
  const forbiddenReason = (found: readonly string[]): string =>
259
337
  `commit.forbidden-trailer: this commit carries an AI attribution (${found.join("; ")}). ` +
@@ -2,6 +2,7 @@ import { readFileSync } from "node:fs";
2
2
  import { homedir } from "node:os";
3
3
  import { resolve } from "node:path";
4
4
  import { type ExtensionAPI, isToolCallEventType } from "@earendil-works/pi-coding-agent";
5
+ import { afterSourceEdit } from "../core/lifecycle.ts";
5
6
  import { classifyPath } from "../core/path-class.ts";
6
7
  import { normalizeRepoPath } from "../core/test-paths.ts";
7
8
  import type { SessionState } from "../state/session-state.ts";
@@ -57,7 +58,7 @@ const JUDGED_EXEMPTIONS =
57
58
  "straightforward CI scripting, a simple dev-environment utility, or a behaviour-preserving refactor with green coverage";
58
59
 
59
60
  /**
60
- * Soft gate `tdd.red-first`: while implementing, production source is edited only after a failing
61
+ * Soft gate `tdd.red-first`: while a slice is in flight (implementing, reviewing, delivering), production source is edited only after a failing
61
62
  * test run has been observed. Test, docs, config and generated files are exempt by path; the
62
63
  * judged exemptions go through one recorded departure, which covers its whole slice.
63
64
  */
@@ -69,11 +70,13 @@ export function registerRedFirstGuard(deps: RedFirstGuardDeps): void {
69
70
  deps.pi.on("tool_call", (event, ctx) => {
70
71
  if (!(isToolCallEventType("edit", event) || isToolCallEventType("write", event)))
71
72
  return undefined;
72
- const { phase, lastTestRun, activeSlice } = deps.state.get();
73
- if (phase !== "implementing") return undefined;
74
- if (lastTestRun !== undefined && lastTestRun.exitCode !== 0) return undefined;
73
+ const { phase, activeSlice } = deps.state.get();
74
+ if (phase !== "implementing" && phase !== "reviewing" && phase !== "delivering")
75
+ return undefined;
75
76
  const path = normalizeRepoPath(ctx.cwd, event.input.path, homedir());
76
77
  if (classifyPath(path) !== "source") return undefined;
78
+ const { lastTestRun } = deps.state.get();
79
+ if (lastTestRun !== undefined && lastTestRun.exitCode !== 0) return undefined;
77
80
  if (touchesInlineTest(event.input, () => existing(ctx.cwd, path))) return undefined;
78
81
  if (departure.covers()) return undefined;
79
82
  const seen =
@@ -90,4 +93,17 @@ export function registerRedFirstGuard(deps: RedFirstGuardDeps): void {
90
93
  `departure covers the rest of the ${scope}.`,
91
94
  };
92
95
  });
96
+
97
+ // The review covered the code as it was: source that was really written puts the slice back to implementing.
98
+ deps.pi.on("tool_result", (event, ctx) => {
99
+ if (event.isError || (event.toolName !== "edit" && event.toolName !== "write"))
100
+ return undefined;
101
+ if (deps.state.get().phase !== "delivering") return undefined;
102
+ const given = (event.input as { path?: unknown }).path;
103
+ if (typeof given !== "string") return undefined;
104
+ if (classifyPath(normalizeRepoPath(ctx.cwd, given, homedir())) === "source") {
105
+ deps.state.update(afterSourceEdit);
106
+ }
107
+ return undefined;
108
+ });
93
109
  }