@jwilger/pi-development-system 0.16.0 → 0.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/agents/advisor.md +0 -1
- package/agents/implementer.md +0 -1
- package/agents/lens-cagan.md +0 -1
- package/agents/lens-perri.md +0 -1
- package/agents/lens-pichler.md +0 -1
- package/agents/lens-rumelt.md +0 -1
- package/agents/lens-torres.md +0 -1
- package/agents/reviewer.md +0 -1
- package/extensions/development-system.ts +9 -0
- package/package.json +1 -1
- package/prompts/devsys-review.md +10 -0
- package/skills/code-review/SKILL.md +62 -0
- package/src/context/status.ts +13 -1
- package/src/core/review-flow.ts +121 -0
- package/src/core/review-packet.ts +111 -0
- package/src/core/review.ts +156 -0
- package/src/core/types.ts +4 -0
- package/src/gates/commit-guard.ts +26 -5
- package/src/jev/questions/review.ts +125 -0
- package/src/review/digest.ts +31 -0
- package/src/review/review-tools.ts +253 -0
- package/src/state/session-state.ts +27 -2
- package/src/subagents/VENDORED.md +1 -1
package/agents/advisor.md
CHANGED
|
@@ -1,6 +1,5 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: advisor
|
|
3
|
-
icon: ''
|
|
4
3
|
description: "Read-only decision support for hard design, planning or trade-off questions: returns a recommendation, the alternatives considered, and the cost if the recommendation is wrong. Does not implement."
|
|
5
4
|
thinkingLevel: high
|
|
6
5
|
color: accent
|
package/agents/implementer.md
CHANGED
|
@@ -1,6 +1,5 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: implementer
|
|
3
|
-
icon: ''
|
|
4
3
|
description: "Implements exactly one task record test-first: failing test, minimal code, passing run, then reports Run/Expected evidence. Edits only the files the task names; stops and reports when the task is unclear or needs a decision."
|
|
5
4
|
thinkingLevel: medium
|
|
6
5
|
color: success
|
package/agents/lens-cagan.md
CHANGED
|
@@ -1,6 +1,5 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: lens-cagan
|
|
3
|
-
icon: ''
|
|
4
3
|
description: "Read-only product risk lens: value, usability, feasibility and viability risks; outcome versus output; whether the work builds to learn or builds to earn. Reviews planning and product artifacts as Cagan; advisory critique, never customer evidence."
|
|
5
4
|
thinkingLevel: high
|
|
6
5
|
color: accent
|
package/agents/lens-perri.md
CHANGED
|
@@ -1,6 +1,5 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: lens-perri
|
|
3
|
-
icon: ''
|
|
4
3
|
description: "Read-only anti-feature-factory lens: output versus outcome, roadmap-as-promises, escaping the build trap. Reviews planning and product artifacts as Perri; advisory critique, never customer evidence."
|
|
5
4
|
thinkingLevel: high
|
|
6
5
|
color: accent
|
package/agents/lens-pichler.md
CHANGED
|
@@ -1,6 +1,5 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: lens-pichler
|
|
3
|
-
icon: ''
|
|
4
3
|
description: "Read-only product-strategy and backlog lens: vision, goal-based roadmap, product-model coherence, decisions with owners. Reviews planning and product artifacts as Pichler; advisory critique, never customer evidence."
|
|
5
4
|
thinkingLevel: high
|
|
6
5
|
color: accent
|
package/agents/lens-rumelt.md
CHANGED
|
@@ -1,6 +1,5 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: lens-rumelt
|
|
3
|
-
icon: ''
|
|
4
3
|
description: "Read-only strategy-kernel lens: diagnosis, guiding policy, coherent actions; flags fluff, goals-as-strategy and bad strategy. Reviews planning and product artifacts as Rumelt; advisory critique, never customer evidence."
|
|
5
4
|
thinkingLevel: high
|
|
6
5
|
color: accent
|
package/agents/lens-torres.md
CHANGED
|
@@ -1,6 +1,5 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: lens-torres
|
|
3
|
-
icon: ''
|
|
4
3
|
description: "Read-only continuous-discovery lens: outcome at the root, one opportunity at a time, at least two solutions compared, assumptions made specific and tested. Reviews planning and product artifacts as Torres; advisory critique, never customer evidence."
|
|
5
4
|
thinkingLevel: high
|
|
6
5
|
color: accent
|
package/agents/reviewer.md
CHANGED
|
@@ -1,6 +1,5 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: reviewer
|
|
3
|
-
icon: ''
|
|
4
3
|
description: Fresh-context, non-mutating review of one slice's diff for demonstrable defects, with severity, path:line, and the smallest local repair; emits the devsys review packet. Not for implementing fixes or style quotas.
|
|
5
4
|
thinkingLevel: high
|
|
6
5
|
color: warning
|
|
@@ -14,6 +14,7 @@ import { registerRedFirstGuard } from "../src/gates/red-first-guard.ts";
|
|
|
14
14
|
import { createRequestApprovalTool } from "../src/gates/request-approval-tool.ts";
|
|
15
15
|
import { registerTestGuard } from "../src/gates/test-guard.ts";
|
|
16
16
|
import { createJevHolder } from "../src/jev/holder.ts";
|
|
17
|
+
import { createReviewRecordTool, createReviewStartTool } from "../src/review/review-tools.ts";
|
|
17
18
|
import { registerCiCommand } from "../src/state/ci-command.ts";
|
|
18
19
|
import { loadConfig } from "../src/state/config.ts";
|
|
19
20
|
import { createModelsTool, registerModelsCommand } from "../src/state/models-command.ts";
|
|
@@ -103,6 +104,14 @@ export function createDevelopmentSystem(pi: ExtensionAPI) {
|
|
|
103
104
|
pi.registerTool(createRequestApprovalTool({ pi, approvals }));
|
|
104
105
|
pi.registerTool(createModelsTool());
|
|
105
106
|
pi.registerTool(createRouteTaskTool({ jev: (ctx) => jevHolder.forContext(ctx) }));
|
|
107
|
+
const reviewDeps = {
|
|
108
|
+
state,
|
|
109
|
+
jev: (ctx: ExtensionContext) => jevHolder.forContext(ctx),
|
|
110
|
+
exec: (command: string, args: string[], options?: { cwd?: string; timeout?: number }) =>
|
|
111
|
+
pi.exec(command, args, options),
|
|
112
|
+
} satisfies Parameters<typeof createReviewStartTool>[0];
|
|
113
|
+
pi.registerTool(createReviewStartTool(reviewDeps));
|
|
114
|
+
pi.registerTool(createReviewRecordTool(reviewDeps));
|
|
106
115
|
registerModelsCommand(pi);
|
|
107
116
|
piSubagent(pi);
|
|
108
117
|
|
package/package.json
CHANGED
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: Run a fresh-context review round on the active slice and record the result
|
|
3
|
+
argument-hint: "[slice] [diffRange]"
|
|
4
|
+
---
|
|
5
|
+
Review the work in progress with the code-review skill.
|
|
6
|
+
|
|
7
|
+
1. Call `devsys_review_start` (slice and diff range: $ARGUMENTS; omit either to use the active slice and `HEAD`).
|
|
8
|
+
2. Run the `agent_spawn` payload it returns, unchanged.
|
|
9
|
+
3. Pass the reviewer's packet verbatim to `devsys_review_record`.
|
|
10
|
+
4. Report `review: N/R clean` and the next action. If the next action is `fix-findings`, list each blocking and should-fix finding with its `path:line`.
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: code-review
|
|
3
|
+
description: How to review a slice before it is committed in this development system - fresh-context reviewer per round, Jev-chosen lenses, the review packet format, what resets the clean streak, and driving rounds with devsys_review_start and devsys_review_record. Use before committing a slice, when asked to review, or when interpreting reviewer findings.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Code review
|
|
7
|
+
|
|
8
|
+
Every slice is reviewed by a **fresh-context reviewer** before it is committed.
|
|
9
|
+
The reviewer has not seen your reasoning, so it judges the diff, not your intent.
|
|
10
|
+
Review is a soft gate (`review.unsatisfied`): skipping it needs a recorded departure.
|
|
11
|
+
|
|
12
|
+
## Run a round
|
|
13
|
+
|
|
14
|
+
1. `devsys_review_start` (slice defaults to the active slice; `diffRange` defaults
|
|
15
|
+
to `HEAD`, i.e. everything uncommitted). It computes the diff digest, lets Jev
|
|
16
|
+
choose lenses, and returns an exact `agent_spawn` payload.
|
|
17
|
+
2. Run that `agent_spawn` unchanged. The reviewer is a top-level agent, not a
|
|
18
|
+
child of this conversation, so it does not inherit your context.
|
|
19
|
+
3. Pass the reviewer's packet, verbatim, to `devsys_review_record`
|
|
20
|
+
(`packets: [...]`; one packet per lens reviewer if you split them).
|
|
21
|
+
4. Read the reply: `review: N/R clean` and `next:`.
|
|
22
|
+
- `fix-findings`: fix every blocking and should-fix finding, then start a new round.
|
|
23
|
+
- `review`: the streak is short (or the diff changed with findings); start another round.
|
|
24
|
+
- `stale-diff`: the diff changed after a satisfied review; one more round.
|
|
25
|
+
- `done`: commit.
|
|
26
|
+
|
|
27
|
+
## The packet
|
|
28
|
+
|
|
29
|
+
```markdown
|
|
30
|
+
## Review — <slice> — round <n> — lenses: <a, b>
|
|
31
|
+
### Sources inspected
|
|
32
|
+
- <path:line ranges>
|
|
33
|
+
### Findings
|
|
34
|
+
- [blocking|should-fix|nit] <lens> `<path>:<line>` — <one sentence> — <why it matters>
|
|
35
|
+
### Verdict
|
|
36
|
+
no-blocking | blocking
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
A malformed packet is rejected with the reason; ask the reviewer to resend it.
|
|
40
|
+
The verdict must agree with the findings.
|
|
41
|
+
|
|
42
|
+
## Severity and the clean streak
|
|
43
|
+
|
|
44
|
+
- **blocking**: a demonstrable defect or a broken non-negotiable.
|
|
45
|
+
- **should-fix**: a real defect or missing test with a realistic trigger.
|
|
46
|
+
- **nit**: style or report-only. Nits never reset the streak; they go to
|
|
47
|
+
`docs/decisions/followups.md`.
|
|
48
|
+
- **false-positive**: refuted; ignored.
|
|
49
|
+
|
|
50
|
+
Jev may move a finding's severity when it is at least 0.8 confident; the reply
|
|
51
|
+
says what moved. A round is clean when it has no blocking or should-fix findings.
|
|
52
|
+
The default is three consecutive clean rounds (`review.required_clean_rounds`).
|
|
53
|
+
Only real findings reset the count; a changed diff alone does not, so a small fix
|
|
54
|
+
commit keeps the clean rounds already earned unless findings come back.
|
|
55
|
+
|
|
56
|
+
## Rules of thumb
|
|
57
|
+
|
|
58
|
+
- Fix the findings before the next round. Do not argue a finding away; if it is
|
|
59
|
+
wrong, the reviewer or Jev marks it a false positive with evidence.
|
|
60
|
+
- Do not widen a review into an audit of the tree. The reviewer reads the diff and
|
|
61
|
+
what it depends on.
|
|
62
|
+
- Never edit the packet to make it pass.
|
package/src/context/status.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { reviewLabel, reviewOf } from "../core/review-flow.ts";
|
|
1
2
|
import type { DevsysState } from "../core/types.ts";
|
|
2
3
|
|
|
3
4
|
export const STATUS_KEY = "devsys";
|
|
@@ -12,8 +13,18 @@ const ciLabel = (state: DevsysState): string | undefined => {
|
|
|
12
13
|
return `ci ${ci.status}${sha}`;
|
|
13
14
|
};
|
|
14
15
|
|
|
16
|
+
const activeReview = (state: DevsysState): string | undefined => {
|
|
17
|
+
const review = state.activeSlice === undefined ? undefined : reviewOf(state, state.activeSlice);
|
|
18
|
+
return review === undefined ? undefined : reviewLabel(review);
|
|
19
|
+
};
|
|
20
|
+
|
|
15
21
|
export const renderStatusLine = (state: DevsysState, jevModel?: string): string =>
|
|
16
|
-
[
|
|
22
|
+
[
|
|
23
|
+
`devsys: ${state.phase}`,
|
|
24
|
+
`jev ${jevLabel(state, jevModel)}`,
|
|
25
|
+
ciLabel(state),
|
|
26
|
+
activeReview(state)?.replace("review: ", "review ").replace(" clean", ""),
|
|
27
|
+
]
|
|
17
28
|
.filter((part) => part !== undefined)
|
|
18
29
|
.join(" · ");
|
|
19
30
|
|
|
@@ -26,4 +37,5 @@ export const renderStatus = (state: DevsysState, jevModel?: string): string =>
|
|
|
26
37
|
`- open departures: ${state.openDepartures.length}`,
|
|
27
38
|
`- jev: ${jevLabel(state, jevModel)}`,
|
|
28
39
|
`- ci: ${ciLabel(state)?.slice(3) ?? "unknown"}`,
|
|
40
|
+
...(activeReview(state) === undefined ? [] : [`- ${activeReview(state)}`]),
|
|
29
41
|
].join("\n");
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
import {
|
|
2
|
+
cleanStreak,
|
|
3
|
+
type Finding,
|
|
4
|
+
nextAction,
|
|
5
|
+
type ReviewState,
|
|
6
|
+
type Severity,
|
|
7
|
+
} from "./review.ts";
|
|
8
|
+
import type { ReviewPacket } from "./review-packet.ts";
|
|
9
|
+
import type { DevsysState, SliceRef } from "./types.ts";
|
|
10
|
+
|
|
11
|
+
/** Pure helpers around review state; the tools in src/review do the I/O. */
|
|
12
|
+
|
|
13
|
+
export const reviewOf = (state: DevsysState, slice: SliceRef): ReviewState | undefined =>
|
|
14
|
+
state.reviews?.find((r) => r.slice === slice);
|
|
15
|
+
|
|
16
|
+
export const upsertReview = (state: DevsysState, review: ReviewState): DevsysState => {
|
|
17
|
+
const others = (state.reviews ?? []).filter((r) => r.slice !== review.slice);
|
|
18
|
+
return { ...state, reviews: [...others, review] };
|
|
19
|
+
};
|
|
20
|
+
|
|
21
|
+
export const reviewLabel = (review: ReviewState): string =>
|
|
22
|
+
`review: ${cleanStreak(review)}/${review.required} clean`;
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Why a commit on this slice lacks a satisfied review, or `undefined` when the review is complete
|
|
26
|
+
* on the diff as it is now (the `review.unsatisfied` soft gate is raised from this).
|
|
27
|
+
*/
|
|
28
|
+
export function reviewGap(
|
|
29
|
+
state: DevsysState,
|
|
30
|
+
slice: SliceRef,
|
|
31
|
+
digestNow: string,
|
|
32
|
+
): string | undefined {
|
|
33
|
+
const review = reviewOf(state, slice);
|
|
34
|
+
if (review === undefined || review.rounds.length === 0) {
|
|
35
|
+
return `no review has been recorded for slice ${slice}`;
|
|
36
|
+
}
|
|
37
|
+
switch (nextAction(review, digestNow)) {
|
|
38
|
+
case "done":
|
|
39
|
+
return undefined;
|
|
40
|
+
case "fix-findings":
|
|
41
|
+
return `the last review round of slice ${slice} has blocking or should-fix findings still to fix`;
|
|
42
|
+
case "stale-diff":
|
|
43
|
+
return `the diff has changed since slice ${slice} was last reviewed; review it again`;
|
|
44
|
+
case "review":
|
|
45
|
+
return `slice ${slice} has ${cleanStreak(review)}/${review.required} clean review rounds`;
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/** Packets from several lens reviewers in one round become one round of findings. */
|
|
50
|
+
export function combinePackets(packets: readonly ReviewPacket[]): {
|
|
51
|
+
lenses: string[];
|
|
52
|
+
findings: Finding[];
|
|
53
|
+
} {
|
|
54
|
+
const lenses = [...new Set(packets.flatMap((p) => p.lenses))];
|
|
55
|
+
const findings = packets.flatMap((packet, p) =>
|
|
56
|
+
packet.findings.map((f) => ({ ...f, id: `${f.id}.${p + 1}` })),
|
|
57
|
+
);
|
|
58
|
+
return { lenses, findings };
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/** Jev confidence needed before its severity replaces the reviewer's. */
|
|
62
|
+
export const SEVERITY_ADOPT_CONFIDENCE = 0.8;
|
|
63
|
+
|
|
64
|
+
/** Take Jev's severity only when it is confident and different; say what moved so it is never silent. */
|
|
65
|
+
export function adoptSeverity(
|
|
66
|
+
finding: Finding,
|
|
67
|
+
judged: { severity: Severity; confidence: number } | undefined,
|
|
68
|
+
): { finding: Finding; note?: string } {
|
|
69
|
+
if (
|
|
70
|
+
judged === undefined ||
|
|
71
|
+
judged.confidence < SEVERITY_ADOPT_CONFIDENCE ||
|
|
72
|
+
judged.severity === finding.severity
|
|
73
|
+
) {
|
|
74
|
+
return { finding };
|
|
75
|
+
}
|
|
76
|
+
return {
|
|
77
|
+
finding: { ...finding, severity: judged.severity },
|
|
78
|
+
note: `${finding.id}: Jev moved severity ${finding.severity} → ${judged.severity} (confidence ${judged.confidence.toFixed(2)})`,
|
|
79
|
+
};
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
const where = (f: Finding): string =>
|
|
83
|
+
f.path === undefined ? "" : ` \`${f.path}${f.line === undefined ? "" : `:${f.line}`}\``;
|
|
84
|
+
|
|
85
|
+
/** The nits of a round, as a section for docs/decisions/followups.md. */
|
|
86
|
+
export function renderFollowups(
|
|
87
|
+
slice: SliceRef,
|
|
88
|
+
round: number,
|
|
89
|
+
findings: readonly Finding[],
|
|
90
|
+
): string | undefined {
|
|
91
|
+
const nits = findings.filter((f) => f.severity === "nit");
|
|
92
|
+
if (nits.length === 0) return undefined;
|
|
93
|
+
const lines = nits.map((f) => `- ${f.lens}${where(f)} — ${f.summary}`);
|
|
94
|
+
return `\n## From ${slice} review round ${round}\n\n${lines.join("\n")}\n`;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
/** The task text for a fresh reviewer: slice, lenses, how to see the diff, and the packet to return. */
|
|
98
|
+
export function reviewerTask(input: {
|
|
99
|
+
slice: SliceRef;
|
|
100
|
+
round: number;
|
|
101
|
+
lenses: readonly string[];
|
|
102
|
+
diffRange: string;
|
|
103
|
+
}): string {
|
|
104
|
+
const lenses = input.lenses.join(", ");
|
|
105
|
+
return [
|
|
106
|
+
`Review slice ${input.slice}, round ${input.round}, through these lenses: ${lenses}.`,
|
|
107
|
+
`See the change with \`git diff ${input.diffRange}\` (and \`git diff --stat ${input.diffRange}\`). Read the callers, callees and tests it depends on, no more.`,
|
|
108
|
+
"Do not modify files; report demonstrable defects with a realistic trigger and the smallest local repair.",
|
|
109
|
+
"Return exactly the review packet from your instructions:",
|
|
110
|
+
"",
|
|
111
|
+
`## Review — ${input.slice} — round ${input.round} — lenses: ${lenses}`,
|
|
112
|
+
"### Sources inspected",
|
|
113
|
+
"- <path:line ranges>",
|
|
114
|
+
"### Findings",
|
|
115
|
+
"- [blocking|should-fix|nit] <lens> `<path>:<line>` — <one sentence> — <why it matters>",
|
|
116
|
+
"### Verdict",
|
|
117
|
+
"no-blocking | blocking",
|
|
118
|
+
"",
|
|
119
|
+
"Severity: blocking = demonstrable defect or broken non-negotiable; should-fix = real defect or missing test with a realistic trigger; nit = style or report-only. Zero findings is a valid result.",
|
|
120
|
+
].join("\n");
|
|
121
|
+
}
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
import { type Finding, SEVERITIES, type Severity } from "./review.ts";
|
|
2
|
+
import { isParseError, type ParseError, parseError } from "./types.ts";
|
|
3
|
+
|
|
4
|
+
export type ReviewPacket = {
|
|
5
|
+
readonly slice: string;
|
|
6
|
+
readonly round: number;
|
|
7
|
+
readonly lenses: ReadonlyArray<string>;
|
|
8
|
+
readonly sources: ReadonlyArray<string>;
|
|
9
|
+
readonly findings: ReadonlyArray<Finding>;
|
|
10
|
+
readonly verdict: "no-blocking" | "blocking";
|
|
11
|
+
};
|
|
12
|
+
|
|
13
|
+
const SEP = "[—–-]";
|
|
14
|
+
const HEADER = new RegExp(
|
|
15
|
+
`^##\\s+Review\\s+${SEP}\\s+(.+?)\\s+${SEP}\\s+round\\s+(\\S+)\\s+${SEP}\\s+lenses:\\s*(.*)$`,
|
|
16
|
+
"m",
|
|
17
|
+
);
|
|
18
|
+
const FINDING = /^-\s*\[([^\]]+)\]\s+(\S+)\s+(?:`([^`]+)`\s+)?[—–-]\s+(.+)$/;
|
|
19
|
+
|
|
20
|
+
const section = (text: string, name: string): string | undefined => {
|
|
21
|
+
const match = new RegExp(`^###\\s+${name}\\s*$`, "im").exec(text);
|
|
22
|
+
if (match === null) return undefined;
|
|
23
|
+
const rest = text.slice(match.index + match[0].length);
|
|
24
|
+
const next = /^###?\s/m.exec(rest);
|
|
25
|
+
return next === null ? rest : rest.slice(0, next.index);
|
|
26
|
+
};
|
|
27
|
+
|
|
28
|
+
const isSeverity = (value: string): value is Severity =>
|
|
29
|
+
(SEVERITIES as readonly string[]).includes(value);
|
|
30
|
+
|
|
31
|
+
const locate = (target: string | undefined): { path?: string; line?: number } => {
|
|
32
|
+
if (target === undefined) return {};
|
|
33
|
+
const m = /^(.*?):(\d+)(?:-\d+)?$/.exec(target);
|
|
34
|
+
return m === null ? { path: target } : { path: m[1] ?? target, line: Number(m[2]) };
|
|
35
|
+
};
|
|
36
|
+
|
|
37
|
+
function parseFinding(line: string, lens: string[], index: number): Finding | ParseError {
|
|
38
|
+
const m = FINDING.exec(line);
|
|
39
|
+
if (m === null) {
|
|
40
|
+
return parseError(
|
|
41
|
+
`unparseable finding "${line}": expected "- [severity] <lens> \`path:line\` — <summary> — <why>"`,
|
|
42
|
+
);
|
|
43
|
+
}
|
|
44
|
+
const severity = m[1] ?? "";
|
|
45
|
+
if (!isSeverity(severity)) {
|
|
46
|
+
return parseError(
|
|
47
|
+
`unknown severity "${severity}": use blocking, should-fix, nit or false-positive`,
|
|
48
|
+
);
|
|
49
|
+
}
|
|
50
|
+
const findingLens = m[2] ?? "";
|
|
51
|
+
if (!lens.includes(findingLens)) lens.push(findingLens);
|
|
52
|
+
return {
|
|
53
|
+
id: `${findingLens}-${index}`,
|
|
54
|
+
severity,
|
|
55
|
+
...locate(m[3]),
|
|
56
|
+
summary: (m[4] ?? "").trim(),
|
|
57
|
+
lens: findingLens,
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/** Parse the reviewer packet of plan Appendix A. Strict about structure, tolerant of dash style. */
|
|
62
|
+
export function parseReviewPacket(text: string): ReviewPacket | ParseError {
|
|
63
|
+
const header = HEADER.exec(text);
|
|
64
|
+
if (header === null) {
|
|
65
|
+
return parseError('missing packet header "## Review — <slice> — round <n> — lenses: <a, b>"');
|
|
66
|
+
}
|
|
67
|
+
const round = Number(header[2]);
|
|
68
|
+
if (!Number.isInteger(round) || round < 1) {
|
|
69
|
+
return parseError(`round "${header[2]}" is not a positive integer`);
|
|
70
|
+
}
|
|
71
|
+
const lenses = (header[3] ?? "")
|
|
72
|
+
.split(",")
|
|
73
|
+
.map((l) => l.trim())
|
|
74
|
+
.filter((l) => l !== "");
|
|
75
|
+
const verdictText = section(text, "Verdict")?.trim().split(/\s+/)[0]?.toLowerCase();
|
|
76
|
+
if (verdictText !== "no-blocking" && verdictText !== "blocking") {
|
|
77
|
+
return parseError(
|
|
78
|
+
'missing or invalid verdict: end the packet with "### Verdict" then no-blocking or blocking',
|
|
79
|
+
);
|
|
80
|
+
}
|
|
81
|
+
const findings: Finding[] = [];
|
|
82
|
+
const lines = (section(text, "Findings") ?? "")
|
|
83
|
+
.split("\n")
|
|
84
|
+
.map((l) => l.trim())
|
|
85
|
+
.filter((l) => l !== "" && !/^-?\s*none\.?$/i.test(l));
|
|
86
|
+
for (const [i, line] of lines.entries()) {
|
|
87
|
+
const finding = parseFinding(line, lenses, i + 1);
|
|
88
|
+
if (isParseError(finding)) return finding;
|
|
89
|
+
findings.push(finding);
|
|
90
|
+
}
|
|
91
|
+
const hasBlocking = findings.some(
|
|
92
|
+
(f) => f.severity === "blocking" || f.severity === "should-fix",
|
|
93
|
+
);
|
|
94
|
+
if (hasBlocking !== (verdictText === "blocking")) {
|
|
95
|
+
return parseError(
|
|
96
|
+
`verdict "${verdictText}" contradicts the findings: blocking/should-fix findings mean "blocking", otherwise "no-blocking"`,
|
|
97
|
+
);
|
|
98
|
+
}
|
|
99
|
+
const sources = (section(text, "Sources inspected") ?? "")
|
|
100
|
+
.split("\n")
|
|
101
|
+
.map((l) => l.replace(/^-\s*/, "").trim())
|
|
102
|
+
.filter((l) => l !== "");
|
|
103
|
+
return {
|
|
104
|
+
slice: (header[1] ?? "").trim(),
|
|
105
|
+
round,
|
|
106
|
+
lenses,
|
|
107
|
+
sources,
|
|
108
|
+
findings,
|
|
109
|
+
verdict: verdictText,
|
|
110
|
+
};
|
|
111
|
+
}
|
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
import { type ParseError, parseError, type SliceRef } from "./types.ts";
|
|
2
|
+
|
|
3
|
+
/** Severity of one review finding. Only blocking and should-fix make a round unclean. */
|
|
4
|
+
export type Severity = "blocking" | "should-fix" | "nit" | "false-positive";
|
|
5
|
+
|
|
6
|
+
export const SEVERITIES: readonly Severity[] = ["blocking", "should-fix", "nit", "false-positive"];
|
|
7
|
+
|
|
8
|
+
export type Finding = {
|
|
9
|
+
readonly id: string;
|
|
10
|
+
readonly severity: Severity;
|
|
11
|
+
readonly path?: string;
|
|
12
|
+
readonly line?: number;
|
|
13
|
+
readonly summary: string;
|
|
14
|
+
readonly lens: string;
|
|
15
|
+
};
|
|
16
|
+
|
|
17
|
+
export type ReviewRound = {
|
|
18
|
+
readonly n: number;
|
|
19
|
+
readonly lenses: ReadonlyArray<string>;
|
|
20
|
+
readonly findings: ReadonlyArray<Finding>;
|
|
21
|
+
readonly reviewedAt: string;
|
|
22
|
+
/** Digest of the diff the round reviewed (see `digestOf` in src/review/digest.ts). */
|
|
23
|
+
readonly diffDigest: string;
|
|
24
|
+
};
|
|
25
|
+
|
|
26
|
+
export type ReviewState = {
|
|
27
|
+
readonly slice: SliceRef;
|
|
28
|
+
readonly rounds: ReadonlyArray<ReviewRound>;
|
|
29
|
+
/** Consecutive clean rounds needed (config `review.required_clean_rounds`, at least `min_rounds`). */
|
|
30
|
+
readonly required: number;
|
|
31
|
+
};
|
|
32
|
+
|
|
33
|
+
export type ReviewAction = "review" | "fix-findings" | "done" | "stale-diff";
|
|
34
|
+
|
|
35
|
+
export const isClean = (round: ReviewRound): boolean =>
|
|
36
|
+
round.findings.every((f) => f.severity !== "blocking" && f.severity !== "should-fix");
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Trailing clean rounds. Real findings reset it; a changed diff alone does not, so a trivial fix
|
|
40
|
+
* commit keeps the clean rounds already earned unless findings reappear (decision D11).
|
|
41
|
+
*/
|
|
42
|
+
export function cleanStreak(state: ReviewState): number {
|
|
43
|
+
let streak = 0;
|
|
44
|
+
for (let i = state.rounds.length - 1; i >= 0; i--) {
|
|
45
|
+
const round = state.rounds[i];
|
|
46
|
+
if (round === undefined || !isClean(round)) break;
|
|
47
|
+
streak += 1;
|
|
48
|
+
}
|
|
49
|
+
return streak;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
export const isSatisfied = (state: ReviewState): boolean =>
|
|
53
|
+
state.rounds.length > 0 && cleanStreak(state) >= state.required;
|
|
54
|
+
|
|
55
|
+
/** What the coordinator should do next, given the digest of the diff as it is now. */
|
|
56
|
+
export function nextAction(state: ReviewState, diffDigestNow: string): ReviewAction {
|
|
57
|
+
const last = state.rounds.at(-1);
|
|
58
|
+
if (last === undefined) return "review";
|
|
59
|
+
if (!isClean(last)) return last.diffDigest === diffDigestNow ? "fix-findings" : "review";
|
|
60
|
+
if (!isSatisfied(state)) return "review";
|
|
61
|
+
return last.diffDigest === diffDigestNow ? "done" : "stale-diff";
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
export const startReview = (slice: SliceRef, required: number): ReviewState => ({
|
|
65
|
+
slice,
|
|
66
|
+
required,
|
|
67
|
+
rounds: [],
|
|
68
|
+
});
|
|
69
|
+
|
|
70
|
+
export const addRound = (state: ReviewState, round: Omit<ReviewRound, "n">): ReviewState => ({
|
|
71
|
+
...state,
|
|
72
|
+
rounds: [...state.rounds, { ...round, n: state.rounds.length + 1 }],
|
|
73
|
+
});
|
|
74
|
+
|
|
75
|
+
/** Review lenses Jev can select (plan I6.2). A lens applies at probability ≥ LENS_THRESHOLD. */
|
|
76
|
+
export const LENSES = [
|
|
77
|
+
"security",
|
|
78
|
+
"concurrency",
|
|
79
|
+
"types",
|
|
80
|
+
"tests",
|
|
81
|
+
"api-contract",
|
|
82
|
+
"data-migration",
|
|
83
|
+
"ux",
|
|
84
|
+
"performance",
|
|
85
|
+
] as const;
|
|
86
|
+
export type Lens = (typeof LENSES)[number];
|
|
87
|
+
export const LENS_THRESHOLD = 0.5;
|
|
88
|
+
|
|
89
|
+
/** Lenses that apply when Jev is offline: the ones every change deserves. */
|
|
90
|
+
export const DEFAULT_LENSES: ReadonlyArray<Lens> = ["types", "tests"];
|
|
91
|
+
|
|
92
|
+
export const selectLenses = (probabilities: Readonly<Record<Lens, number>>): Lens[] =>
|
|
93
|
+
LENSES.filter((lens) => probabilities[lens] >= LENS_THRESHOLD);
|
|
94
|
+
|
|
95
|
+
const isRecord = (v: unknown): v is Record<string, unknown> =>
|
|
96
|
+
typeof v === "object" && v !== null && !Array.isArray(v);
|
|
97
|
+
|
|
98
|
+
function parseFinding(input: unknown): Finding | ParseError {
|
|
99
|
+
if (!isRecord(input)) return parseError("finding must be an object");
|
|
100
|
+
const { id, severity, lens, summary, path, line } = input;
|
|
101
|
+
if (typeof id !== "string" || typeof lens !== "string" || typeof summary !== "string") {
|
|
102
|
+
return parseError("finding needs string id, lens and summary");
|
|
103
|
+
}
|
|
104
|
+
if (!SEVERITIES.includes(severity as Severity)) {
|
|
105
|
+
return parseError(`unknown severity: ${String(severity)}`);
|
|
106
|
+
}
|
|
107
|
+
if (path !== undefined && typeof path !== "string")
|
|
108
|
+
return parseError("finding path must be a string");
|
|
109
|
+
if (line !== undefined && typeof line !== "number")
|
|
110
|
+
return parseError("finding line must be a number");
|
|
111
|
+
return {
|
|
112
|
+
id,
|
|
113
|
+
lens,
|
|
114
|
+
summary,
|
|
115
|
+
severity: severity as Severity,
|
|
116
|
+
...(path !== undefined ? { path } : {}),
|
|
117
|
+
...(line !== undefined ? { line } : {}),
|
|
118
|
+
};
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
function parseRound(input: unknown): ReviewRound | ParseError {
|
|
122
|
+
if (!isRecord(input)) return parseError("round must be an object");
|
|
123
|
+
const { n, lenses, findings, reviewedAt, diffDigest } = input;
|
|
124
|
+
if (typeof n !== "number" || typeof reviewedAt !== "string" || typeof diffDigest !== "string") {
|
|
125
|
+
return parseError("round needs number n and string reviewedAt and diffDigest");
|
|
126
|
+
}
|
|
127
|
+
if (!Array.isArray(lenses) || !lenses.every((l) => typeof l === "string")) {
|
|
128
|
+
return parseError("round lenses must be an array of strings");
|
|
129
|
+
}
|
|
130
|
+
if (!Array.isArray(findings)) return parseError("round findings must be an array");
|
|
131
|
+
const parsed: Finding[] = [];
|
|
132
|
+
for (const raw of findings) {
|
|
133
|
+
const finding = parseFinding(raw);
|
|
134
|
+
if ("kind" in finding) return finding;
|
|
135
|
+
parsed.push(finding);
|
|
136
|
+
}
|
|
137
|
+
return { n, lenses: lenses as string[], findings: parsed, reviewedAt, diffDigest };
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/** Boundary parse for a persisted review (one slice's rounds). */
|
|
141
|
+
export function parseReviewState(input: unknown): ReviewState | ParseError {
|
|
142
|
+
if (!isRecord(input)) return parseError("review must be an object");
|
|
143
|
+
const { slice, required, rounds } = input;
|
|
144
|
+
if (typeof slice !== "string" || slice === "") return parseError("review slice must be a string");
|
|
145
|
+
if (typeof required !== "number" || !Number.isInteger(required) || required < 1) {
|
|
146
|
+
return parseError("review required must be an integer of at least 1");
|
|
147
|
+
}
|
|
148
|
+
if (!Array.isArray(rounds)) return parseError("review rounds must be an array");
|
|
149
|
+
const parsed: ReviewRound[] = [];
|
|
150
|
+
for (const raw of rounds) {
|
|
151
|
+
const round = parseRound(raw);
|
|
152
|
+
if ("kind" in round) return round;
|
|
153
|
+
parsed.push(round);
|
|
154
|
+
}
|
|
155
|
+
return { slice: slice as SliceRef, required, rounds: parsed };
|
|
156
|
+
}
|
package/src/core/types.ts
CHANGED
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import type { ReviewState } from "./review.ts";
|
|
2
|
+
|
|
1
3
|
/** Pure domain types for the development system. No I/O lives here. */
|
|
2
4
|
|
|
3
5
|
export type ParseError = { readonly kind: "parse-error"; readonly message: string };
|
|
@@ -82,6 +84,8 @@ export type DevsysState = {
|
|
|
82
84
|
readonly ci?: CiState;
|
|
83
85
|
readonly profiles?: ReadonlyArray<Profile>;
|
|
84
86
|
readonly lastTestRun?: TestRun;
|
|
87
|
+
/** One review record per slice (see src/core/review.ts). */
|
|
88
|
+
readonly reviews?: ReadonlyArray<ReviewState>;
|
|
85
89
|
};
|
|
86
90
|
|
|
87
91
|
export const initialState = (): DevsysState => ({
|
|
@@ -13,9 +13,11 @@ import {
|
|
|
13
13
|
parseConventionalCommit,
|
|
14
14
|
} from "../core/commit-message.ts";
|
|
15
15
|
import type { Exec } from "../core/exec.ts";
|
|
16
|
+
import { reviewGap } from "../core/review-flow.ts";
|
|
16
17
|
import { type GateId, isParseError, parseGateId } from "../core/types.ts";
|
|
17
18
|
import type { Jev } from "../jev/client.ts";
|
|
18
19
|
import { judgeCommit, MIX_THRESHOLD, RATIONALE_FLOOR } from "../jev/questions/commit.ts";
|
|
20
|
+
import { snapshotDiff } from "../review/digest.ts";
|
|
19
21
|
import type { SessionState } from "../state/session-state.ts";
|
|
20
22
|
import { departureUse } from "./departure-use.ts";
|
|
21
23
|
|
|
@@ -36,6 +38,7 @@ const gateId = (id: string): GateId => {
|
|
|
36
38
|
|
|
37
39
|
const RATIONALE = gateId("commit.rationale");
|
|
38
40
|
const MIXED = gateId("commit.mixed-change");
|
|
41
|
+
const REVIEW = gateId("review.unsatisfied");
|
|
39
42
|
|
|
40
43
|
const readMessageFile = (cwd: string, path: string): string | undefined => {
|
|
41
44
|
try {
|
|
@@ -115,11 +118,29 @@ async function jevNeeds(
|
|
|
115
118
|
return needs;
|
|
116
119
|
}
|
|
117
120
|
|
|
121
|
+
/** While work is in flight, committing needs the slice review satisfied on the diff as it is now. */
|
|
122
|
+
async function reviewNeeds(deps: CommitGuardDeps, ctx: ExtensionContext): Promise<Need[]> {
|
|
123
|
+
const { phase, activeSlice } = deps.state.get();
|
|
124
|
+
if ((phase !== "implementing" && phase !== "reviewing") || activeSlice === undefined) return [];
|
|
125
|
+
const snap = await snapshotDiff(deps.exec, ctx.cwd, "HEAD");
|
|
126
|
+
// An unreadable diff cannot be compared; the gate asks for a departure rather than guessing.
|
|
127
|
+
const gap = reviewGap(deps.state.get(), activeSlice, snap.ok ? snap.value.digest : "unknown");
|
|
128
|
+
return gap === undefined ? [] : [{ gate: REVIEW, why: gap }];
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
const reviewReason = (need: Need): string =>
|
|
132
|
+
`${need.gate}: ${need.why}. Each slice is reviewed by a fresh-context reviewer before it is committed. ` +
|
|
133
|
+
"Run devsys_review_start, spawn the reviewer it describes, record the packet with devsys_review_record " +
|
|
134
|
+
"and repeat until the review is satisfied. If skipping review is deliberate call devsys_record_departure " +
|
|
135
|
+
`with gate "${need.gate}", what you are doing instead, why, and the cost if wrong; then retry.`;
|
|
136
|
+
|
|
118
137
|
const blockReason = (need: Need): string =>
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
138
|
+
need.gate === REVIEW
|
|
139
|
+
? reviewReason(need)
|
|
140
|
+
: `${need.gate}: ${need.why}. Commit messages carry their rationale and structural and behavioural ` +
|
|
141
|
+
"changes go in separate commits. Fix the commit (split it, or write the why in the body), or if " +
|
|
142
|
+
`departing is deliberate call devsys_record_departure with gate "${need.gate}", what you are doing ` +
|
|
143
|
+
"instead, why, and the cost if wrong; then retry.";
|
|
123
144
|
|
|
124
145
|
const forbiddenReason = (found: readonly string[]): string =>
|
|
125
146
|
`commit.forbidden-trailer: this commit carries an AI attribution (${found.join("; ")}). ` +
|
|
@@ -162,7 +183,7 @@ export function registerCommitGuard(deps: CommitGuardDeps): void {
|
|
|
162
183
|
if (forbidden.length > 0) return { block: true, reason: forbiddenReason(forbidden) };
|
|
163
184
|
|
|
164
185
|
const known = messages.flatMap((m) => (m === undefined ? [] : [m]));
|
|
165
|
-
const all: Need[] =
|
|
186
|
+
const all: Need[] = await reviewNeeds(deps, ctx);
|
|
166
187
|
// A -F file that cannot be read now (written by this same command, or stdin) cannot be checked.
|
|
167
188
|
if (extractions.some((e, i) => e.kind === "file" && messages[i] === undefined)) {
|
|
168
189
|
all.push({
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
import type { ClassifierBoolQuestion, ClassifierChoiceQuestion } from "@earendil-works/pi-ai";
|
|
2
|
+
import { redactSecrets } from "../../core/redact.ts";
|
|
3
|
+
import { err, ok, type Result } from "../../core/result.ts";
|
|
4
|
+
import { LENSES, type Lens, SEVERITIES, type Severity } from "../../core/review.ts";
|
|
5
|
+
import type { Jev, JevError } from "../client.ts";
|
|
6
|
+
|
|
7
|
+
const lensQuestion = (instructions: string, yes: string, no: string): ClassifierBoolQuestion => ({
|
|
8
|
+
type: "bool",
|
|
9
|
+
instructions,
|
|
10
|
+
criteria: { true: yes, false: no },
|
|
11
|
+
});
|
|
12
|
+
|
|
13
|
+
export const LENS_QUESTIONS: Readonly<Record<Lens, ClassifierBoolQuestion>> = {
|
|
14
|
+
security: lensQuestion(
|
|
15
|
+
"Does the change in `diffSample` (stat in `diffStat`) touch authentication, authorisation, secrets, cryptography, input parsing from untrusted sources, shell or SQL construction, file paths from users, or network exposure?",
|
|
16
|
+
"It handles untrusted input, credentials, permissions or external exposure",
|
|
17
|
+
"It does not touch any security-relevant surface",
|
|
18
|
+
),
|
|
19
|
+
concurrency: lensQuestion(
|
|
20
|
+
"Does the change introduce or alter concurrency: threads, async ordering, locks, shared mutable state, retries with side effects, queues, or race-prone file or process handling?",
|
|
21
|
+
"Ordering, sharing or timing between concurrent actors matters to correctness",
|
|
22
|
+
"The code is sequential and has no shared mutable state at risk",
|
|
23
|
+
),
|
|
24
|
+
types: lensQuestion(
|
|
25
|
+
"Does the change add or alter domain types, parsing of external input, error types, state machines or public function signatures where illegal states could become representable?",
|
|
26
|
+
"It models domain data, parses input, or changes types or errors",
|
|
27
|
+
"It changes no types, parsing or error handling",
|
|
28
|
+
),
|
|
29
|
+
tests: lensQuestion(
|
|
30
|
+
"Does the change alter behaviour that tests should pin, or alter tests themselves (added, edited, loosened, skipped or deleted)?",
|
|
31
|
+
"Behaviour or tests changed and test adequacy deserves a look",
|
|
32
|
+
"Docs, formatting or configuration only; no behaviour or tests changed",
|
|
33
|
+
),
|
|
34
|
+
"api-contract": lensQuestion(
|
|
35
|
+
"Does the change alter a public interface: exported functions, CLI flags, tool schemas, file formats, configuration keys, wire formats or documented behaviour that other code or users rely on?",
|
|
36
|
+
"Something other code or users depend on changed shape or meaning",
|
|
37
|
+
"Only internals changed; no external contract is affected",
|
|
38
|
+
),
|
|
39
|
+
"data-migration": lensQuestion(
|
|
40
|
+
"Does the change alter persisted data shape: schemas, stored entries, migrations, serialised formats, or how existing stored data is read?",
|
|
41
|
+
"Existing stored data must still be read, migrated or preserved correctly",
|
|
42
|
+
"No persisted data shape changes",
|
|
43
|
+
),
|
|
44
|
+
ux: lensQuestion(
|
|
45
|
+
"Does the change alter what a person sees or does: messages, prompts, commands, output formats, error text, documentation of workflow, or interaction flow?",
|
|
46
|
+
"A human-facing message, flow or output changed",
|
|
47
|
+
"Nothing a person reads or interacts with changed",
|
|
48
|
+
),
|
|
49
|
+
performance: lensQuestion(
|
|
50
|
+
"Does the change alter work done per request, per item or per run: loops over large inputs, I/O in hot paths, caching, polling, or algorithmic complexity?",
|
|
51
|
+
"Cost grows with input size or call frequency in a way that matters",
|
|
52
|
+
"No meaningful change to time or memory behaviour",
|
|
53
|
+
),
|
|
54
|
+
};
|
|
55
|
+
|
|
56
|
+
export const SEVERITY_QUESTION: ClassifierChoiceQuestion = {
|
|
57
|
+
type: "choice",
|
|
58
|
+
instructions:
|
|
59
|
+
"Given the review `finding` and the surrounding `diffContext`, how severe is it? Judge by the consequence of shipping it, and by whether the reviewer showed a realistic triggering input. A finding with no demonstrated trigger, or one the code already guards against, is a false positive.",
|
|
60
|
+
criteria: {
|
|
61
|
+
blocking:
|
|
62
|
+
"A demonstrable defect or broken principle: wrong behaviour, data loss, security hole, or a defeated safeguard on a realistic input",
|
|
63
|
+
"should-fix":
|
|
64
|
+
"A real defect or missing test with a realistic trigger, but limited impact or easy workaround",
|
|
65
|
+
nit: "Style, naming, a contrived trigger, or a suggestion with no demonstrated impact",
|
|
66
|
+
"false-positive": "The claim is wrong or the code already handles the case",
|
|
67
|
+
},
|
|
68
|
+
};
|
|
69
|
+
|
|
70
|
+
const DIFF_SAMPLE_MAX = 8000;
|
|
71
|
+
const clip = (text: string, max: number): string =>
|
|
72
|
+
redactSecrets(text.slice(0, max * 2)).slice(0, max);
|
|
73
|
+
|
|
74
|
+
export type LensInput = {
|
|
75
|
+
diffStat: string;
|
|
76
|
+
diffSample: string;
|
|
77
|
+
profiles: readonly string[];
|
|
78
|
+
};
|
|
79
|
+
|
|
80
|
+
/** Probability per lens that it applies. One narrow question per lens, asked together. */
|
|
81
|
+
export async function judgeLenses(
|
|
82
|
+
jev: Jev,
|
|
83
|
+
input: LensInput,
|
|
84
|
+
): Promise<Result<Record<Lens, number>, JevError>> {
|
|
85
|
+
const asked = await jev.ask(
|
|
86
|
+
{
|
|
87
|
+
diffStat: clip(input.diffStat, 2000),
|
|
88
|
+
diffSample: clip(input.diffSample, DIFF_SAMPLE_MAX),
|
|
89
|
+
profiles: input.profiles.slice(0, 5).map((p) => clip(p, 40)),
|
|
90
|
+
},
|
|
91
|
+
LENS_QUESTIONS,
|
|
92
|
+
);
|
|
93
|
+
if (!asked.ok) return asked;
|
|
94
|
+
const out = {} as Record<Lens, number>;
|
|
95
|
+
for (const lens of LENSES) {
|
|
96
|
+
const answer = asked.value[lens];
|
|
97
|
+
if (answer?.type !== "bool" || !Number.isFinite(answer.probability)) {
|
|
98
|
+
return err({ kind: "provider", message: `missing or malformed answer for lens ${lens}` });
|
|
99
|
+
}
|
|
100
|
+
out[lens] = answer.probability;
|
|
101
|
+
}
|
|
102
|
+
return ok(out);
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
export type SeverityInput = { finding: string; diffContext: string };
|
|
106
|
+
|
|
107
|
+
export async function judgeSeverity(
|
|
108
|
+
jev: Jev,
|
|
109
|
+
input: SeverityInput,
|
|
110
|
+
): Promise<Result<{ severity: Severity; confidence: number }, JevError>> {
|
|
111
|
+
const asked = await jev.ask(
|
|
112
|
+
{ finding: clip(input.finding, 2000), diffContext: clip(input.diffContext, 6000) },
|
|
113
|
+
{ severity: SEVERITY_QUESTION },
|
|
114
|
+
);
|
|
115
|
+
if (!asked.ok) return asked;
|
|
116
|
+
const answer = asked.value.severity;
|
|
117
|
+
if (answer?.type !== "choice") {
|
|
118
|
+
return err({ kind: "provider", message: "missing severity answer" });
|
|
119
|
+
}
|
|
120
|
+
const label = SEVERITIES.find((s) => s === answer.choice);
|
|
121
|
+
if (label === undefined) {
|
|
122
|
+
return err({ kind: "provider", message: `unknown severity label "${answer.choice}"` });
|
|
123
|
+
}
|
|
124
|
+
return ok({ severity: label, confidence: answer.confidence });
|
|
125
|
+
}
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
import { createHash } from "node:crypto";
|
|
2
|
+
import type { Exec } from "../core/exec.ts";
|
|
3
|
+
import { err, ok, type Result } from "../core/result.ts";
|
|
4
|
+
|
|
5
|
+
/** A short, stable identity for a diff's text. Pure. */
|
|
6
|
+
export const digestOf = (diff: string): string =>
|
|
7
|
+
createHash("sha256").update(diff).digest("hex").slice(0, 16);
|
|
8
|
+
|
|
9
|
+
export type DiffSnapshot = { digest: string; stat: string; sample: string };
|
|
10
|
+
|
|
11
|
+
/** The diff for `range` (default: everything not yet committed, `HEAD`), with its digest. */
|
|
12
|
+
export async function snapshotDiff(
|
|
13
|
+
exec: Exec,
|
|
14
|
+
cwd: string,
|
|
15
|
+
range: string,
|
|
16
|
+
): Promise<Result<DiffSnapshot, string>> {
|
|
17
|
+
try {
|
|
18
|
+
const [diff, stat] = await Promise.all([
|
|
19
|
+
exec("git", ["diff", range], { cwd, timeout: 15_000 }),
|
|
20
|
+
exec("git", ["diff", "--stat", range], { cwd, timeout: 15_000 }),
|
|
21
|
+
]);
|
|
22
|
+
if (diff.code !== 0) return err(diff.stderr.trim() || `git diff ${range} failed`);
|
|
23
|
+
return ok({
|
|
24
|
+
digest: digestOf(diff.stdout),
|
|
25
|
+
stat: stat.stdout,
|
|
26
|
+
sample: diff.stdout.slice(0, 16_000),
|
|
27
|
+
});
|
|
28
|
+
} catch (cause) {
|
|
29
|
+
return err(cause instanceof Error ? cause.message : String(cause));
|
|
30
|
+
}
|
|
31
|
+
}
|
|
@@ -0,0 +1,253 @@
|
|
|
1
|
+
import { appendFile, mkdir } from "node:fs/promises";
|
|
2
|
+
import { join } from "node:path";
|
|
3
|
+
import type { ExtensionContext, ToolDefinition } from "@earendil-works/pi-coding-agent";
|
|
4
|
+
import { type Static, Type } from "typebox";
|
|
5
|
+
import type { Exec } from "../core/exec.ts";
|
|
6
|
+
import { resolveSlot } from "../core/models.ts";
|
|
7
|
+
import {
|
|
8
|
+
addRound,
|
|
9
|
+
DEFAULT_LENSES,
|
|
10
|
+
type Finding,
|
|
11
|
+
nextAction,
|
|
12
|
+
type ReviewState,
|
|
13
|
+
selectLenses,
|
|
14
|
+
startReview,
|
|
15
|
+
} from "../core/review.ts";
|
|
16
|
+
import {
|
|
17
|
+
adoptSeverity,
|
|
18
|
+
combinePackets,
|
|
19
|
+
renderFollowups,
|
|
20
|
+
reviewerTask,
|
|
21
|
+
reviewLabel,
|
|
22
|
+
reviewOf,
|
|
23
|
+
upsertReview,
|
|
24
|
+
} from "../core/review-flow.ts";
|
|
25
|
+
import { parseReviewPacket, type ReviewPacket } from "../core/review-packet.ts";
|
|
26
|
+
import { isParseError, type SliceRef } from "../core/types.ts";
|
|
27
|
+
import type { Jev } from "../jev/client.ts";
|
|
28
|
+
import { judgeLenses, judgeSeverity } from "../jev/questions/review.ts";
|
|
29
|
+
import { CONFIG_FILE, loadConfig } from "../state/config.ts";
|
|
30
|
+
import { availableModels } from "../state/models-command.ts";
|
|
31
|
+
import type { SessionState } from "../state/session-state.ts";
|
|
32
|
+
import { snapshotDiff } from "./digest.ts";
|
|
33
|
+
|
|
34
|
+
export type ReviewToolDeps = {
|
|
35
|
+
state: SessionState;
|
|
36
|
+
jev: (ctx: ExtensionContext) => Jev;
|
|
37
|
+
exec: Exec;
|
|
38
|
+
now?: () => Date;
|
|
39
|
+
};
|
|
40
|
+
|
|
41
|
+
const reply = (payload: string, isError = false) => ({
|
|
42
|
+
content: [{ type: "text" as const, text: payload }],
|
|
43
|
+
details: undefined,
|
|
44
|
+
isError,
|
|
45
|
+
});
|
|
46
|
+
|
|
47
|
+
const sliceOf = (given: string | undefined, state: SessionState): SliceRef | undefined => {
|
|
48
|
+
const slice = given?.trim() || state.get().activeSlice;
|
|
49
|
+
return slice === undefined || slice === "" ? undefined : (slice as SliceRef);
|
|
50
|
+
};
|
|
51
|
+
|
|
52
|
+
const NO_SLICE =
|
|
53
|
+
"no slice: pass `slice`, or start work on a slice first (there is no active slice)";
|
|
54
|
+
|
|
55
|
+
const StartParameters = Type.Object({
|
|
56
|
+
slice: Type.Optional(
|
|
57
|
+
Type.String({ description: "Slice being reviewed; defaults to the active slice." }),
|
|
58
|
+
),
|
|
59
|
+
diffRange: Type.Optional(
|
|
60
|
+
Type.String({
|
|
61
|
+
description: "Git revision range to review; defaults to HEAD (all uncommitted changes).",
|
|
62
|
+
}),
|
|
63
|
+
),
|
|
64
|
+
});
|
|
65
|
+
|
|
66
|
+
const pathSafe = (slice: string): string => slice.toLowerCase().replace(/[^a-z0-9]+/g, "-");
|
|
67
|
+
|
|
68
|
+
/** `devsys_review_start`: Jev picks lenses for the diff; the reply carries the agent_spawn payload for a fresh reviewer. */
|
|
69
|
+
export function createReviewStartTool(
|
|
70
|
+
deps: ReviewToolDeps,
|
|
71
|
+
): ToolDefinition<typeof StartParameters> {
|
|
72
|
+
return {
|
|
73
|
+
name: "devsys_review_start",
|
|
74
|
+
label: "Start review round",
|
|
75
|
+
description:
|
|
76
|
+
"Begin a review round for a slice: computes the diff digest, chooses review lenses, and returns the agent_spawn payload for a fresh-context reviewer. Run that spawn, then pass the reviewer's packet to devsys_review_record.",
|
|
77
|
+
promptSnippet: "Start a fresh-context review round for a slice",
|
|
78
|
+
parameters: StartParameters,
|
|
79
|
+
exposure: "direct",
|
|
80
|
+
async execute(
|
|
81
|
+
_id,
|
|
82
|
+
params: Static<typeof StartParameters>,
|
|
83
|
+
_signal,
|
|
84
|
+
_onUpdate,
|
|
85
|
+
ctx: ExtensionContext,
|
|
86
|
+
) {
|
|
87
|
+
const slice = sliceOf(params.slice, deps.state);
|
|
88
|
+
if (slice === undefined) return reply(NO_SLICE, true);
|
|
89
|
+
const config = await loadConfig(ctx.cwd);
|
|
90
|
+
if (!config.ok) return reply(`${CONFIG_FILE}: ${config.error.message}`, true);
|
|
91
|
+
const range = params.diffRange?.trim() || "HEAD";
|
|
92
|
+
const snap = await snapshotDiff(deps.exec, ctx.cwd, range);
|
|
93
|
+
if (!snap.ok) return reply(`cannot read the diff for ${range}: ${snap.error}`, true);
|
|
94
|
+
if (snap.value.stat.trim() === "")
|
|
95
|
+
return reply(`the diff for ${range} is empty; nothing to review`, true);
|
|
96
|
+
|
|
97
|
+
const judged = await judgeLenses(deps.jev(ctx), {
|
|
98
|
+
diffStat: snap.value.stat,
|
|
99
|
+
diffSample: snap.value.sample,
|
|
100
|
+
profiles: deps.state.get().profiles ?? [],
|
|
101
|
+
});
|
|
102
|
+
const selected = judged.ok ? selectLenses(judged.value) : [];
|
|
103
|
+
const lenses = selected.length > 0 ? selected : [...DEFAULT_LENSES];
|
|
104
|
+
const basis = judged.ok
|
|
105
|
+
? selected.length > 0
|
|
106
|
+
? "Jev chose the lenses."
|
|
107
|
+
: "Jev found no specific lens; using the defaults."
|
|
108
|
+
: `Jev unavailable (${judged.error.kind}); using the default lenses.`;
|
|
109
|
+
|
|
110
|
+
const required = Math.max(
|
|
111
|
+
config.value.review.requiredCleanRounds,
|
|
112
|
+
config.value.review.minRounds,
|
|
113
|
+
);
|
|
114
|
+
const existing = reviewOf(deps.state.get(), slice) ?? startReview(slice, required);
|
|
115
|
+
const review: ReviewState = { ...existing, required };
|
|
116
|
+
deps.state.update((s) => upsertReview(s, review));
|
|
117
|
+
const action = nextAction(review, snap.value.digest);
|
|
118
|
+
if (action === "done") {
|
|
119
|
+
return reply(
|
|
120
|
+
`${reviewLabel(review)}. Review of ${slice} is already satisfied on this diff; nothing to run.`,
|
|
121
|
+
);
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
const round = review.rounds.length + 1;
|
|
125
|
+
const resolved = resolveSlot(
|
|
126
|
+
config.value.models,
|
|
127
|
+
"reviewer",
|
|
128
|
+
availableModels(ctx.modelRegistry),
|
|
129
|
+
);
|
|
130
|
+
const spawn = {
|
|
131
|
+
path: `/review-${pathSafe(slice)}-r${round}`,
|
|
132
|
+
type: "reviewer",
|
|
133
|
+
task: reviewerTask({ slice, round, lenses, diffRange: range }),
|
|
134
|
+
...(resolved.ok ? { model: resolved.value.model } : {}),
|
|
135
|
+
thinkingLevel: "high",
|
|
136
|
+
wait: true,
|
|
137
|
+
};
|
|
138
|
+
return reply(
|
|
139
|
+
[
|
|
140
|
+
`${reviewLabel(review)}; round ${round}; diff ${snap.value.digest}. ${basis}`,
|
|
141
|
+
`lenses: ${lenses.join(", ")}`,
|
|
142
|
+
"Spawn the reviewer with agent_spawn using exactly this payload, then pass its packet to devsys_review_record:",
|
|
143
|
+
JSON.stringify(spawn),
|
|
144
|
+
].join("\n"),
|
|
145
|
+
);
|
|
146
|
+
},
|
|
147
|
+
};
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
const RecordParameters = Type.Object({
|
|
151
|
+
slice: Type.Optional(
|
|
152
|
+
Type.String({ description: "Slice reviewed; defaults to the active slice." }),
|
|
153
|
+
),
|
|
154
|
+
packets: Type.Array(Type.String(), {
|
|
155
|
+
description: "Reviewer packets, verbatim markdown: one per lens reviewer, all for one round.",
|
|
156
|
+
}),
|
|
157
|
+
diffRange: Type.Optional(
|
|
158
|
+
Type.String({ description: "The range that was reviewed; defaults to HEAD." }),
|
|
159
|
+
),
|
|
160
|
+
});
|
|
161
|
+
|
|
162
|
+
async function adjusted(
|
|
163
|
+
deps: ReviewToolDeps,
|
|
164
|
+
ctx: ExtensionContext,
|
|
165
|
+
findings: readonly Finding[],
|
|
166
|
+
diffContext: string,
|
|
167
|
+
): Promise<{ findings: Finding[]; notes: string[] }> {
|
|
168
|
+
const out: Finding[] = [];
|
|
169
|
+
const notes: string[] = [];
|
|
170
|
+
for (const f of findings) {
|
|
171
|
+
const judged = await judgeSeverity(deps.jev(ctx), {
|
|
172
|
+
finding: `[${f.severity}] ${f.lens} ${f.path ?? ""}${f.line === undefined ? "" : `:${f.line}`} — ${f.summary}`,
|
|
173
|
+
diffContext,
|
|
174
|
+
});
|
|
175
|
+
const next = adoptSeverity(f, judged.ok ? judged.value : undefined);
|
|
176
|
+
out.push(next.finding);
|
|
177
|
+
if (next.note !== undefined) notes.push(next.note);
|
|
178
|
+
}
|
|
179
|
+
return { findings: out, notes };
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
/** `devsys_review_record`: parse packets, let Jev adjust severities, record the round, file the nits. */
|
|
183
|
+
export function createReviewRecordTool(
|
|
184
|
+
deps: ReviewToolDeps,
|
|
185
|
+
): ToolDefinition<typeof RecordParameters> {
|
|
186
|
+
return {
|
|
187
|
+
name: "devsys_review_record",
|
|
188
|
+
label: "Record review round",
|
|
189
|
+
description:
|
|
190
|
+
"Record one review round from the reviewer's packet(s): parses findings, lets Jev adjust severity when confident, files nits in docs/decisions/followups.md, and returns the clean-round count and the next action.",
|
|
191
|
+
promptSnippet: "Record a reviewer packet as a review round",
|
|
192
|
+
parameters: RecordParameters,
|
|
193
|
+
exposure: "direct",
|
|
194
|
+
async execute(
|
|
195
|
+
_id,
|
|
196
|
+
params: Static<typeof RecordParameters>,
|
|
197
|
+
_signal,
|
|
198
|
+
_onUpdate,
|
|
199
|
+
ctx: ExtensionContext,
|
|
200
|
+
) {
|
|
201
|
+
const slice = sliceOf(params.slice, deps.state);
|
|
202
|
+
if (slice === undefined) return reply(NO_SLICE, true);
|
|
203
|
+
if (params.packets.length === 0)
|
|
204
|
+
return reply("no packets: pass the reviewer's packet markdown in `packets`", true);
|
|
205
|
+
const packets: ReviewPacket[] = [];
|
|
206
|
+
for (const [i, raw] of params.packets.entries()) {
|
|
207
|
+
const parsed = parseReviewPacket(raw);
|
|
208
|
+
if (isParseError(parsed))
|
|
209
|
+
return reply(`packet ${i + 1} is malformed: ${parsed.message}`, true);
|
|
210
|
+
packets.push(parsed);
|
|
211
|
+
}
|
|
212
|
+
const config = await loadConfig(ctx.cwd);
|
|
213
|
+
if (!config.ok) return reply(`${CONFIG_FILE}: ${config.error.message}`, true);
|
|
214
|
+
const snap = await snapshotDiff(deps.exec, ctx.cwd, params.diffRange?.trim() || "HEAD");
|
|
215
|
+
if (!snap.ok) return reply(`cannot read the diff: ${snap.error}`, true);
|
|
216
|
+
|
|
217
|
+
const merged = combinePackets(packets);
|
|
218
|
+
const { findings, notes } = await adjusted(deps, ctx, merged.findings, snap.value.sample);
|
|
219
|
+
const required = Math.max(
|
|
220
|
+
config.value.review.requiredCleanRounds,
|
|
221
|
+
config.value.review.minRounds,
|
|
222
|
+
);
|
|
223
|
+
const before = reviewOf(deps.state.get(), slice) ?? startReview(slice, required);
|
|
224
|
+
const review = addRound(
|
|
225
|
+
{ ...before, required },
|
|
226
|
+
{
|
|
227
|
+
lenses: merged.lenses,
|
|
228
|
+
findings,
|
|
229
|
+
reviewedAt: (deps.now?.() ?? new Date()).toISOString(),
|
|
230
|
+
diffDigest: snap.value.digest,
|
|
231
|
+
},
|
|
232
|
+
);
|
|
233
|
+
deps.state.update((s) => upsertReview(s, review));
|
|
234
|
+
|
|
235
|
+
const followups = renderFollowups(slice, review.rounds.length, findings);
|
|
236
|
+
if (followups !== undefined) {
|
|
237
|
+
const dir = join(ctx.cwd, "docs", "decisions");
|
|
238
|
+
await mkdir(dir, { recursive: true });
|
|
239
|
+
await appendFile(join(dir, "followups.md"), followups);
|
|
240
|
+
}
|
|
241
|
+
const counts = (sev: Finding["severity"]) =>
|
|
242
|
+
findings.filter((f) => f.severity === sev).length;
|
|
243
|
+
return reply(
|
|
244
|
+
[
|
|
245
|
+
`Round ${review.rounds.length} recorded: ${counts("blocking")} blocking, ${counts("should-fix")} should-fix, ${counts("nit")} nit, ${counts("false-positive")} false-positive.`,
|
|
246
|
+
reviewLabel(review),
|
|
247
|
+
...notes,
|
|
248
|
+
`next: ${nextAction(review, snap.value.digest)}`,
|
|
249
|
+
].join("\n"),
|
|
250
|
+
);
|
|
251
|
+
},
|
|
252
|
+
};
|
|
253
|
+
}
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
2
2
|
import { parseDeparture } from "../core/departure.ts";
|
|
3
|
+
import { parseReviewState, type ReviewState } from "../core/review.ts";
|
|
3
4
|
import {
|
|
4
5
|
CI_STATUSES,
|
|
5
6
|
type CiState,
|
|
@@ -65,11 +66,32 @@ function parseTestRun(input: unknown): TestRun | ParseError {
|
|
|
65
66
|
return { at, exitCode, summary };
|
|
66
67
|
}
|
|
67
68
|
|
|
69
|
+
function parseReviews(input: unknown): ReviewState[] | ParseError {
|
|
70
|
+
if (!Array.isArray(input)) return parseError("reviews must be an array");
|
|
71
|
+
const out: ReviewState[] = [];
|
|
72
|
+
for (const raw of input) {
|
|
73
|
+
const review = parseReviewState(raw);
|
|
74
|
+
if (isParseError(review)) return review;
|
|
75
|
+
out.push(review);
|
|
76
|
+
}
|
|
77
|
+
return out;
|
|
78
|
+
}
|
|
79
|
+
|
|
68
80
|
/** Boundary parse for persisted state: the only place state is cast from `unknown`. */
|
|
69
81
|
export function parseDevsysState(input: unknown): DevsysState | ParseError {
|
|
70
82
|
if (!isRecord(input)) return parseError("devsys state must be an object");
|
|
71
|
-
const {
|
|
72
|
-
|
|
83
|
+
const {
|
|
84
|
+
phase,
|
|
85
|
+
sizing,
|
|
86
|
+
activeSlice,
|
|
87
|
+
openDepartures,
|
|
88
|
+
jev,
|
|
89
|
+
lastPushAt,
|
|
90
|
+
ci,
|
|
91
|
+
profiles,
|
|
92
|
+
lastTestRun,
|
|
93
|
+
reviews,
|
|
94
|
+
} = input;
|
|
73
95
|
if (!PHASES.includes(phase as Phase)) return parseError(`unknown phase: ${String(phase)}`);
|
|
74
96
|
if (!JEV.includes(jev as JevStatus)) return parseError(`unknown jev status: ${String(jev)}`);
|
|
75
97
|
if (!Array.isArray(openDepartures)) return parseError("openDepartures must be an array");
|
|
@@ -94,6 +116,8 @@ export function parseDevsysState(input: unknown): DevsysState | ParseError {
|
|
|
94
116
|
if (isParseError(parsedProfiles)) return parsedProfiles;
|
|
95
117
|
const parsedRun = lastTestRun === undefined ? undefined : parseTestRun(lastTestRun);
|
|
96
118
|
if (isParseError(parsedRun)) return parsedRun;
|
|
119
|
+
const parsedReviews = reviews === undefined ? undefined : parseReviews(reviews);
|
|
120
|
+
if (isParseError(parsedReviews)) return parsedReviews;
|
|
97
121
|
return {
|
|
98
122
|
phase: phase as Phase,
|
|
99
123
|
jev: jev as JevStatus,
|
|
@@ -104,6 +128,7 @@ export function parseDevsysState(input: unknown): DevsysState | ParseError {
|
|
|
104
128
|
...(parsedCi !== undefined ? { ci: parsedCi } : {}),
|
|
105
129
|
...(parsedProfiles !== undefined ? { profiles: parsedProfiles } : {}),
|
|
106
130
|
...(parsedRun !== undefined ? { lastTestRun: parsedRun } : {}),
|
|
131
|
+
...(parsedReviews !== undefined ? { reviews: parsedReviews } : {}),
|
|
107
132
|
};
|
|
108
133
|
}
|
|
109
134
|
|
|
@@ -39,4 +39,4 @@ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
|
39
39
|
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
40
40
|
SOFTWARE.
|
|
41
41
|
6. Empty scope (I5 review): `orch/runtime.ts` `availableScopedModels` treats an empty `/scoped-models` as no restriction (pi's meaning) instead of "matches nothing", because every devsys agent declares `models:` and would otherwise be unspawnable by default. Pinned models are also re-applied on resume (`resolveInitialSettings`), since a restored session may hold only inherited history.
|
|
42
|
-
7. Bundled agents outside `src/subagents`: `agents/researcher.md` and `agents/reviewer.md`
|
|
42
|
+
7. Bundled agents outside `src/subagents`: `agents/researcher.md` (devsys `models:` family list) and `agents/reviewer.md` (devsys `models:` list and packet format; description rewritten; upstream `icon` and `modelSuggestions` removed) differ from upstream; `agents/{advisor,implementer,lens-*}.md` are new. The agent file format itself is unchanged.
|