@jwilger/pi-development-system 0.87.0 → 0.89.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +118 -3
- package/agents/coder.md +1 -1
- package/agents/reviewer.md +2 -1
- package/agents/tasker.md +2 -0
- package/extensions/development-system.ts +5 -0
- package/package.json +3 -1
- package/prompts/devsys-review.md +1 -1
- package/skills/code-review/SKILL.md +10 -6
- package/skills/profile-design-system/SKILL.md +66 -0
- package/skills/threat-modelling/SKILL.md +76 -0
- package/src/context/phase-guide.ts +1 -1
- package/src/core/review-flow.ts +2 -1
- package/src/core/review-submission.ts +60 -0
- package/src/review/review-tools.ts +55 -15
- package/src/review/submissions.ts +38 -0
- package/src/review/submit-tool.ts +85 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 John Wilger
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
CHANGED
|
@@ -1,9 +1,25 @@
|
|
|
1
1
|
# pi-development-system
|
|
2
2
|
|
|
3
3
|
A [pi](https://pi.dev) extension package representing my seasoned approach to
|
|
4
|
-
software development using a full AI SDLC
|
|
4
|
+
software development using a full AI SDLC: work is sized, planned in proportion, built
|
|
5
|
+
test-first in small slices, reviewed by fresh-context reviewers and delivered by trunk-based
|
|
6
|
+
commits. Guards enforce what must hold; the rest is guidance with a recorded way out.
|
|
5
7
|
|
|
6
|
-
|
|
8
|
+
## Install
|
|
9
|
+
|
|
10
|
+
```sh
|
|
11
|
+
pi install npm:@jwilger/pi-development-system
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
Then `/reload`. To move to a newer release use
|
|
15
|
+
`pi install npm:@jwilger/pi-development-system@<version>`.
|
|
16
|
+
|
|
17
|
+
Jev, the classifier behind the judgements, is optional. Without a credential configured, the checks
|
|
18
|
+
that rest on a judgement do not run or fall back to a default: the motive behind a test change, whether a
|
|
19
|
+
diff needs an ADR, the quality of a commit rationale and whether a commit mixes changes, the intent
|
|
20
|
+
nudges, the verifier's check of the agent's own claims, and the lens and severity choices in review.
|
|
21
|
+
The status line shows `jev offline` when that is so. Everything deterministic holds without Jev: the
|
|
22
|
+
hard stops, the delivery mode, the review streak, red-first and the slice life cycle.
|
|
7
23
|
|
|
8
24
|
## Replaces pi-subagent-manager
|
|
9
25
|
|
|
@@ -22,16 +38,26 @@ what the current phase expects. When you say a slice is finished with no review
|
|
|
22
38
|
it tells you to start one. The slash commands (`/devsys-start`, `/devsys-plan`, `/devsys-review`,
|
|
23
39
|
`/devsys-lens-review`, `/devsys-adr`, `/devsys-event-model`) are shortcuts to the same tools.
|
|
24
40
|
|
|
41
|
+
Skills cover the habits behind the gates: `tdd-canon`, `behaviour-tests`, `semantic-types`,
|
|
42
|
+
`typed-errors`, `functional-core-imperative-shell`, `strict-lints`, `delivery-discipline`, `delegation`,
|
|
43
|
+
`code-review`, the language profiles (`profile-rust`, `profile-typescript`), `profile-design-system` for UI
|
|
44
|
+
work (tokens, then components; build from the design system's materials and log a snowflake),
|
|
45
|
+
`threat-modelling` (proportional; a document only when the risk earns it) and the planning skills below.
|
|
46
|
+
|
|
25
47
|
Product planning has a skill (`product-planning`: brief, decision register, follow-ups,
|
|
26
48
|
terminology, journeys). `devsys_lens_review` plans a review of the brief by five product lenses and
|
|
27
49
|
writes the packets to `docs/product/reviews/`; `devsys_adr_new` creates the next numbered ADR, and
|
|
28
50
|
a commit that shapes the architecture without one is stopped by the soft gate `adr.missing`.
|
|
29
51
|
|
|
52
|
+
A reviewer subagent hands over its result with `devsys_submit_review` (typed arguments; a result that
|
|
53
|
+
contradicts itself or is for the wrong round is refused with an error id), and `devsys_review_record`
|
|
54
|
+
reads it. A markdown packet is still accepted (ADR 0006).
|
|
55
|
+
|
|
30
56
|
Event modelling has a skill (`event-modelling`: three slice patterns, Given/When/Then, completeness).
|
|
31
57
|
`devsys_event_model_check` validates a directory of slice files (schema v1) and renders the swimlane
|
|
32
58
|
Markdown or a Mermaid diagram; each profile's `gwt-tests` reference turns scenarios into failing tests.
|
|
33
59
|
A dedicated extension that offers `event_model_validate`, or `event_model.provider` in
|
|
34
|
-
`.development-system.toml`, replaces the builtin tool (
|
|
60
|
+
`.development-system.toml`, replaces the builtin tool ([contract](https://github.com/jwilger/pi-development-system/blob/main/docs/event-model-extension-contract.md)).
|
|
35
61
|
|
|
36
62
|
A slice has a life cycle: `implementing` → `reviewing` (when a review round starts) → `delivering`
|
|
37
63
|
(review satisfied) → `idle`. Editing production source while delivering reopens `implementing`,
|
|
@@ -48,6 +74,95 @@ With `codemode` enabled (`"defaultTools": ["+codemode"]` in pi settings), rarely
|
|
|
48
74
|
reached through scripts and the `judge_*` Jev wrappers exist for scripts only; without codemode
|
|
49
75
|
the rarely used tools are declared directly and the wrappers are absent.
|
|
50
76
|
|
|
77
|
+
## Configuration
|
|
78
|
+
|
|
79
|
+
`.development-system.toml` at the repository root (version 1). Every key is optional; an unknown
|
|
80
|
+
key is an error that names it. `/devsys-models` writes the `[models]` table for the models this
|
|
81
|
+
machine can use.
|
|
82
|
+
|
|
83
|
+
| Table | Keys | Meaning |
|
|
84
|
+
| --- | --- | --- |
|
|
85
|
+
| `[delivery]` | `mode` (`trunk`, `pull-request`, `local-only`), `trunk`, `remote` | Where work lands; the push guard follows it. `local-only` blocks every push. |
|
|
86
|
+
| `[review]` | `required_clean_rounds` (3), `min_rounds` (1) | The clean streak a slice needs before commit. |
|
|
87
|
+
| `[tracker]` | `kind` (`repo-files`, `github`; `jira` and `linear` are not implemented), `repo` | Backlog for `devsys_work_item`. |
|
|
88
|
+
| `[profiles]` | `override` | Languages to apply (`rust`, `typescript`); empty means detect. |
|
|
89
|
+
| `[models]` | one ordered candidate list per slot | See Model tiers below. |
|
|
90
|
+
| `[routing]` | `"<difficulty>/<risk>" = ["<slot>", "<thinking level>"]` | What `devsys_route_task` recommends for a subagent. |
|
|
91
|
+
| `[verifier]` | `max_per_session` (6) | Cap on Jev checks of the agent's own claims. |
|
|
92
|
+
| `[cadence]` | `push_minutes` (60) | Minutes without a push before the cadence nudge. |
|
|
93
|
+
| `[event_model]` | `provider` (`builtin`) | Another provider replaces the builtin validator. |
|
|
94
|
+
|
|
95
|
+
## Model tiers
|
|
96
|
+
|
|
97
|
+
Models are asked for by slot, never by id. Three capability tiers (`frontier`, `strong`, `fast`)
|
|
98
|
+
and role slots built on them (`planning`, `advisor`, `implementer`, `reviewer`, `lens`,
|
|
99
|
+
`researcher`, `jev`). A slot is an ordered list of candidates: `provider/id`, a family pattern
|
|
100
|
+
such as `provider/gpt-*-sol` (the newest available id wins) or a slot reference such as `@strong`.
|
|
101
|
+
The first candidate this machine has credentials for is used, so one committed file works across
|
|
102
|
+
accounts. Pin an exact id to stop it rolling forward. `devsys_models` shows what each slot
|
|
103
|
+
resolves to, and the system recommends a model for a phase but never switches yours.
|
|
104
|
+
|
|
105
|
+
## Enforcement tiers
|
|
106
|
+
|
|
107
|
+
- **Hard stops** fire for the non-negotiables (`principles/NON-NEGOTIABLES.md`): rewriting
|
|
108
|
+
published history, force-pushing, deleting remote branches, discarding work with `reset --hard`,
|
|
109
|
+
`--no-verify`, forbidden commit trailers, pushing onto a red trunk, breaking the delivery mode.
|
|
110
|
+
Only the user can approve one, once, and only with a UI: a headless run refuses.
|
|
111
|
+
- **Soft gates** guard the defaults (`principles/DEFAULTS.md`): weakening tests, commit rationale,
|
|
112
|
+
mixed commits, red-first, lint suppression, an unreviewed slice, scope, model for the phase, a
|
|
113
|
+
skipped planning artifact, a missing ADR. A soft gate is passed by recording a departure with
|
|
114
|
+
`devsys_record_departure` (what, why, cost if wrong, how long it applies).
|
|
115
|
+
- **Guidance** covers everything else: skills and the phase guide, never enforced.
|
|
116
|
+
|
|
117
|
+
What the tiers do not cover, so you can decide what to trust:
|
|
118
|
+
|
|
119
|
+
- **Subagents run unguarded.** A child session is started without extensions, so none of the guards
|
|
120
|
+
runs inside it. The coordinator commits, pushes and delivers; a subagent's prompt tells it not to,
|
|
121
|
+
and the coordinator reviews what it produced.
|
|
122
|
+
- **The red-trunk stop needs `gh`.** It reads CI through an authenticated `gh`; where `gh` is missing
|
|
123
|
+
or offline the trunk reads as unknown and the push is not stopped on that ground.
|
|
124
|
+
- **A departure is the agent's own call.** `devsys_record_departure` waives a soft gate (also with no
|
|
125
|
+
UI) and is written to the decision log; it is never available for a hard stop.
|
|
126
|
+
- **Gating starts at `devsys_intake`.** With no slice open, the review and red-first gates are off.
|
|
127
|
+
- **Editing gate configuration is not guarded** (`biome.json`, hooks, CI files); review catches it.
|
|
128
|
+
|
|
129
|
+
## Decision log
|
|
130
|
+
|
|
131
|
+
Every departure and approval is appended to `docs/decisions/YYYY-MM.md` in the repository, so the
|
|
132
|
+
reasons survive compaction and are reviewable. Architecture-shaping decisions get an ADR in
|
|
133
|
+
`docs/adr/` (`devsys_adr_new`); product decisions go in the decision register that the
|
|
134
|
+
`product-planning` skill describes.
|
|
135
|
+
|
|
136
|
+
## Commands
|
|
137
|
+
|
|
138
|
+
All of them are shortcuts to tools; none is required.
|
|
139
|
+
|
|
140
|
+
| Command | Does |
|
|
141
|
+
| --- | --- |
|
|
142
|
+
| `/devsys-start` | Size the work and propose the artifacts it needs. |
|
|
143
|
+
| `/devsys-plan` | Plan a capability or product, then begin work once approved. |
|
|
144
|
+
| `/devsys-review` | Run a fresh-context review round. |
|
|
145
|
+
| `/devsys-lens-review` | Five product lenses review the brief. |
|
|
146
|
+
| `/devsys-adr` | Create the next ADR. |
|
|
147
|
+
| `/devsys-event-model` | Validate and render an event model. |
|
|
148
|
+
| `/devsys-models` | Write the model matrix for this machine (`--check` to verify). |
|
|
149
|
+
| `/devsys-ci` | Watch CI for the pushed commit. |
|
|
150
|
+
| `/devsys-status` | Phase, slice, review streak, departures and Jev status. |
|
|
151
|
+
| `/agents` | The vendored subagent manager. |
|
|
152
|
+
|
|
153
|
+
## Agents
|
|
154
|
+
|
|
155
|
+
Spawn with `agent_spawn`; `devsys_route_task` picks the model and thinking level. `advisor` gives
|
|
156
|
+
read-only decision support, `implementer`, `coder` and `tasker` build (one task record each),
|
|
157
|
+
`reviewer` reviews a diff in fresh context and submits through `devsys_submit_review`,
|
|
158
|
+
`researcher` reads and cites, `architect` and `writer` design and draft, and the five `lens-*`
|
|
159
|
+
agents (Cagan, Torres, Pichler, Perri, Rumelt) review a product brief.
|
|
160
|
+
|
|
161
|
+
## Changelog
|
|
162
|
+
|
|
163
|
+
`CHANGELOG.md` is generated from the commit history by `npm run changelog`. A release is the commit that
|
|
164
|
+
changes the version, so run it after that commit exists; the file lists what has been committed so far.
|
|
165
|
+
|
|
51
166
|
## Development
|
|
52
167
|
|
|
53
168
|
```sh
|
package/agents/coder.md
CHANGED
|
@@ -46,7 +46,7 @@ You are a coder. You own a change, whether a feature, refactor, bug fix, or fron
|
|
|
46
46
|
|
|
47
47
|
## Boundaries
|
|
48
48
|
|
|
49
|
-
- bash is not a sandbox. Use it to inspect, build, test, and run project tooling. Do not commit, push, install dependencies, or touch the network unless the task says to. Other agents may be editing this tree, so do not revert or restyle their changes.
|
|
49
|
+
- bash is not a sandbox. Use it to inspect, build, test, and run project tooling. Do not commit, push, install dependencies, or touch the network unless the task says to. In this session no devsys guards run in your session, so the repository's rules are yours to keep: never weaken, skip or delete a test to get green, never add a lint suppression without a stated reason, never commit secrets, and leave commits, pushes and history to the coordinator. Other agents may be editing this tree, so do not revert or restyle their changes.
|
|
50
50
|
- You cannot delegate. Send agent_update only when the plan changes or a failure is not obvious. If a missing decision, missing access, an exhausted budget, or a stalled investigation blocks you, call agent_pause with the evidence and stop.
|
|
51
51
|
|
|
52
52
|
## Handback
|
package/agents/reviewer.md
CHANGED
|
@@ -18,6 +18,7 @@ tools:
|
|
|
18
18
|
- find
|
|
19
19
|
- ls
|
|
20
20
|
- agent_update
|
|
21
|
+
- devsys_submit_review
|
|
21
22
|
- agent_pause
|
|
22
23
|
---
|
|
23
24
|
|
|
@@ -41,7 +42,7 @@ Do not modify files. bash is not a sandbox: use it only for read-only inspection
|
|
|
41
42
|
|
|
42
43
|
## Output
|
|
43
44
|
|
|
44
|
-
|
|
45
|
+
Submit your result by calling `devsys_submit_review` (slice and round exactly as the task gives them): it checks the call and refuses a self-contradicting one with an error id, which you correct and resubmit. After it accepts, end with one short line. Only when that tool is not available, return exactly this packet instead; the coordinator parses it.
|
|
45
46
|
|
|
46
47
|
```markdown
|
|
47
48
|
## Review — <slice> — round <n> — lenses: <a, b>
|
package/agents/tasker.md
CHANGED
|
@@ -56,3 +56,5 @@ Run a proportionate check: the one the task specifies, else a focused test, comm
|
|
|
56
56
|
- **Gaps**: anything unverified or left for another role.
|
|
57
57
|
|
|
58
58
|
Send agent_update only when criteria are met, a check fails, or scope is expanding. If blocked, call agent_pause with the blocker and stop.
|
|
59
|
+
|
|
60
|
+
In this session no devsys guards run in your session, so the repository's rules are yours to keep: never weaken, skip or delete a test to get green, never add a lint suppression without a stated reason, never commit secrets, and leave commits, pushes and history to the coordinator.
|
|
@@ -35,6 +35,8 @@ import { createFinishSliceTool, registerSliceClose } from "../src/planning/slice
|
|
|
35
35
|
import { createTaskCheckTool } from "../src/planning/task-check-tool.ts";
|
|
36
36
|
import { createLensReviewTool } from "../src/review/lens-review-tool.ts";
|
|
37
37
|
import { createReviewRecordTool, createReviewStartTool } from "../src/review/review-tools.ts";
|
|
38
|
+
import { createSubmissionStore } from "../src/review/submissions.ts";
|
|
39
|
+
import { createSubmitReviewTool } from "../src/review/submit-tool.ts";
|
|
38
40
|
import { registerCiCommand } from "../src/state/ci-command.ts";
|
|
39
41
|
import { loadConfig } from "../src/state/config.ts";
|
|
40
42
|
import { createModelsTool, registerModelsCommand } from "../src/state/models-command.ts";
|
|
@@ -186,13 +188,16 @@ export function createDevelopmentSystem(pi: ExtensionAPI) {
|
|
|
186
188
|
createLensReviewTool({ state, jev: (ctx) => jevHolder.forContext(ctx), now: () => new Date() }),
|
|
187
189
|
);
|
|
188
190
|
pi.registerTool(createIntakeTool({ pi, state, jev: (ctx) => jevHolder.forContext(ctx) }));
|
|
191
|
+
const submissions = createSubmissionStore();
|
|
189
192
|
const reviewDeps = {
|
|
190
193
|
state,
|
|
194
|
+
submissions,
|
|
191
195
|
jev: (ctx: ExtensionContext) => jevHolder.forContext(ctx),
|
|
192
196
|
exec,
|
|
193
197
|
} satisfies Parameters<typeof createReviewStartTool>[0];
|
|
194
198
|
pi.registerTool(createReviewStartTool(reviewDeps));
|
|
195
199
|
pi.registerTool(createReviewRecordTool(reviewDeps));
|
|
200
|
+
pi.registerTool(createSubmitReviewTool({ state, submissions }));
|
|
196
201
|
registerModelsCommand(pi);
|
|
197
202
|
piSubagent(pi);
|
|
198
203
|
|
package/package.json
CHANGED
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@jwilger/pi-development-system",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.89.0",
|
|
4
4
|
"description": "A pi extension package representing a seasoned approach to software development using a full AI SDLC.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pi-package"
|
|
7
7
|
],
|
|
8
|
+
"license": "MIT",
|
|
8
9
|
"author": "John Wilger",
|
|
9
10
|
"repository": {
|
|
10
11
|
"type": "git",
|
|
@@ -57,6 +58,7 @@
|
|
|
57
58
|
"scripts": {
|
|
58
59
|
"test": "node --experimental-transform-types --no-warnings --test \"test/**/*.test.ts\"",
|
|
59
60
|
"test:jev": "node --experimental-transform-types --no-warnings --test \"test/live/*.live.ts\"",
|
|
61
|
+
"changelog": "node --experimental-transform-types --no-warnings scripts/changelog.ts",
|
|
60
62
|
"check:jev": "node scripts/run-jev-fixtures.ts",
|
|
61
63
|
"typecheck": "tsc --noEmit",
|
|
62
64
|
"lint": "biome check --error-on-warnings .",
|
package/prompts/devsys-review.md
CHANGED
|
@@ -6,5 +6,5 @@ Review the work in progress with the code-review skill.
|
|
|
6
6
|
|
|
7
7
|
1. Call `devsys_review_start` (slice and diff range: $ARGUMENTS; omit either to use the active slice and `HEAD`; only the `HEAD` range clears the commit gate).
|
|
8
8
|
2. Run the `agent_spawn` payload it returns, unchanged.
|
|
9
|
-
3.
|
|
9
|
+
3. The reviewer submits its result with `devsys_submit_review`; call `devsys_review_record` with the same `slice`, `diffDigest` and `diffRange` the start reply gave (no packets needed). Only if the reviewer returned a markdown packet instead, pass it verbatim in `packets`.
|
|
10
10
|
4. Report `review: N/R clean` and the next action. If the next action is `fix-findings`, list each blocking and should-fix finding with its `path:line`.
|
|
@@ -17,11 +17,13 @@ Review is a soft gate (`review.unsatisfied`): skipping it needs a recorded depar
|
|
|
17
17
|
diff digest, lets Jev choose lenses, and returns an exact `agent_spawn` payload.
|
|
18
18
|
2. Run that `agent_spawn` unchanged. The reviewer is a top-level agent, not a
|
|
19
19
|
child of this conversation, so it does not inherit your context.
|
|
20
|
-
3.
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
20
|
+
3. The reviewer submits its result with `devsys_submit_review` (typed arguments; a
|
|
21
|
+
self-contradicting or wrong-round result is refused to the reviewer with an error id).
|
|
22
|
+
Call `devsys_review_record` with the `slice` and `diffDigest` the start reply gave, and
|
|
23
|
+
the same `diffRange` if you gave one; it reads the submitted results of the round. If a
|
|
24
|
+
reviewer returned a markdown packet instead, pass it verbatim in `packets: [...]` (all
|
|
25
|
+
lens packets of one round in one call). A result is recorded once, in the round it
|
|
26
|
+
names; if the diff changed since the start, the round is refused: start again.
|
|
25
27
|
4. Read the reply: `review: N/R clean` and `next:`.
|
|
26
28
|
- `fix-findings`: fix every blocking and should-fix finding, then start a new round
|
|
27
29
|
(starting again on an unchanged diff is refused).
|
|
@@ -29,7 +31,9 @@ Review is a soft gate (`review.unsatisfied`): skipping it needs a recorded depar
|
|
|
29
31
|
- `stale-diff`: the diff changed after a satisfied review; one more round.
|
|
30
32
|
- `done`: commit.
|
|
31
33
|
|
|
32
|
-
## The
|
|
34
|
+
## The result
|
|
35
|
+
|
|
36
|
+
The fields of `devsys_submit_review` mirror this packet; the markdown form is the fallback (ADR 0006).
|
|
33
37
|
|
|
34
38
|
```markdown
|
|
35
39
|
## Review — <slice> — round <n> — lenses: <a, b>
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: profile-design-system
|
|
3
|
+
description: Design-system discipline for UI work - tokens before components, building only from existing design-system materials, and a logged decision when something does not fit. Use when a slice adds or changes user interface (components, styles, pages), when a repo has a design system or token files (*.tokens.json, .css, .scss, .tsx, .vue, .svelte), or when reviewing UI code for snowflakes and raw values.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Design-system profile
|
|
7
|
+
|
|
8
|
+
Brad Frost's stance, as this system applies it: the AI is deliberately constrained to the
|
|
9
|
+
design system's materials. That constraint is what separates working inside a design
|
|
10
|
+
system from vibe coding. Treat the agent as a smart but sometimes unsophisticated
|
|
11
|
+
junior developer who has read the system's codebase and must follow its conventions.
|
|
12
|
+
|
|
13
|
+
## Order of work: tokens, then components, then pages
|
|
14
|
+
|
|
15
|
+
1. **Tokens.** Three tiers. Raw (`color-brand-green`) feeds semantic
|
|
16
|
+
(`theme-color-primary-background`), which feeds component (`button-primary-background`).
|
|
17
|
+
Components read the component tier; they never reach for a raw value.
|
|
18
|
+
2. **Components.** Structure, behaviour and accessibility live in a structural component
|
|
19
|
+
library, kept apart from the aesthetic token layer. Do not fork a component to change
|
|
20
|
+
how it looks; change tokens.
|
|
21
|
+
3. **Pages.** A page is the test of the system. Build it with real content in more than
|
|
22
|
+
one shape: a 40 and a 340 character headline, one and ten items, empty and error.
|
|
23
|
+
Anything that breaks goes back to the component, not into a page-level patch.
|
|
24
|
+
|
|
25
|
+
## Defaults for a UI slice
|
|
26
|
+
|
|
27
|
+
- **Use an existing component.** Search the design system first and name what you found
|
|
28
|
+
in the task record. The plan lists the states and variants the slice must show
|
|
29
|
+
(default, hover, focus, disabled, loading, empty, error, long content, narrow screen).
|
|
30
|
+
- **No raw values in components.** No hex colours, pixel sizes, z-indexes or font
|
|
31
|
+
stacks inline. Make this a lint (stylelint, a biome or eslint rule, a grep in CI);
|
|
32
|
+
prose alone is a write-only channel and is not enforced.
|
|
33
|
+
- **Build states outside the app first** when the repo has Storybook or a pattern lab:
|
|
34
|
+
write the story with real-content variants, then wire the component into the app.
|
|
35
|
+
- **Accessibility is part of the component**, not a later pass: roles, labels, keyboard
|
|
36
|
+
path, focus order, contrast from the token pair, reduced motion.
|
|
37
|
+
|
|
38
|
+
## When it does not fit: 90 percent and missing components
|
|
39
|
+
|
|
40
|
+
Product pressure will find a way around the system. Decide, do not drift.
|
|
41
|
+
|
|
42
|
+
| Situation | Decision | Record |
|
|
43
|
+
| --- | --- | --- |
|
|
44
|
+
| The component fits | Use it as is | nothing |
|
|
45
|
+
| It fits about 90 percent | Extend through a documented variant or a token; ask the system owner when the variant is not yours to add | a line in the task record |
|
|
46
|
+
| Nothing fits, and the need will recur | Propose a new component to the system, build it there first | an ADR if it changes the component API |
|
|
47
|
+
| Nothing fits, and the need is one of a kind | A snowflake: allowed once, local, labelled | `devsys_record_departure` with gate `scope.expansion:snowflake`, naming the 90 percent reasoning and when to revisit |
|
|
48
|
+
|
|
49
|
+
A snowflake written without that record is a defect to fix in review, not a style
|
|
50
|
+
preference.
|
|
51
|
+
|
|
52
|
+
## Checklist
|
|
53
|
+
|
|
54
|
+
- [ ] The slice names the design-system components it uses, or why none fits.
|
|
55
|
+
- [ ] No raw colour, size or font value in a component; the tier rule has a lint.
|
|
56
|
+
- [ ] Every state and variant in the plan has a story or screenshot with real content.
|
|
57
|
+
- [ ] Keyboard, focus, labels and contrast checked on the new or changed component.
|
|
58
|
+
- [ ] Any snowflake or forked component has its recorded decision.
|
|
59
|
+
|
|
60
|
+
## Do not
|
|
61
|
+
|
|
62
|
+
- Do not invent a component, a token or a variant the system does not have without the
|
|
63
|
+
decision above.
|
|
64
|
+
- Do not copy a component's markup into a page to restyle it.
|
|
65
|
+
- Do not choose a different UI library for one slice.
|
|
66
|
+
- Do not let a generated component through that only looks right in the happy case.
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: threat-modelling
|
|
3
|
+
description: Proportional threat modelling for a change - decide whether it needs one, ask four questions, and record the result only when the risk earns it. Use before designing or reviewing anything that touches authentication, authorisation, secrets, user data, a network boundary, file or process execution, payments, or a new dependency, or when asked for a threat model or security review.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Threat modelling
|
|
7
|
+
|
|
8
|
+
Proportional on purpose. Most changes need thirty seconds of thought and no document.
|
|
9
|
+
A few need a page. None need a framework exercise.
|
|
10
|
+
|
|
11
|
+
## What this system trusts
|
|
12
|
+
|
|
13
|
+
This system runs on the author's single-owner machine. Trusted: the author, their
|
|
14
|
+
working tree, their credentials in pi, the local shell and git. Do not model an attacker
|
|
15
|
+
who already has that. Untrusted: everything that arrives from outside it. That means
|
|
16
|
+
network input, file contents from other people, dependency code, model output when it
|
|
17
|
+
becomes a command or a path, and anything a deployed service receives.
|
|
18
|
+
|
|
19
|
+
## Is a threat model needed?
|
|
20
|
+
|
|
21
|
+
Ask all four and count the yes answers:
|
|
22
|
+
|
|
23
|
+
1. Does the change add or move a **trust boundary** (a new input from outside, a new
|
|
24
|
+
network call, a new account or role)?
|
|
25
|
+
2. Does it handle **secrets, personal data, money or authority** (login, permissions,
|
|
26
|
+
keys, tokens, payments)?
|
|
27
|
+
3. Does it **execute or interpret** something it did not write (a shell command, a
|
|
28
|
+
file path, a template, deserialised data, a plugin)?
|
|
29
|
+
4. Does it add a **dependency class** or change how packages are fetched or run?
|
|
30
|
+
|
|
31
|
+
No yes: no document. Note "no new trust boundary" in the task record and move on. One or
|
|
32
|
+
more yes: do the four questions below.
|
|
33
|
+
|
|
34
|
+
## The four questions
|
|
35
|
+
|
|
36
|
+
1. **What are we protecting?** Name the asset: the data, the authority, the money, the
|
|
37
|
+
machine. If you cannot name one, the change probably does not need this.
|
|
38
|
+
2. **Who or what can reach it, and from where?** List the entry points and who controls
|
|
39
|
+
each: users, other services, files, the model, a dependency.
|
|
40
|
+
3. **What could go wrong?** For each entry point, a line each for: pretending to be
|
|
41
|
+
someone else, changing what should not change, denying that it happened, reading what
|
|
42
|
+
should stay private, making it unavailable, gaining more authority than intended.
|
|
43
|
+
Skip the ones that do not apply and say so.
|
|
44
|
+
4. **What stops it, and what is left?** Name the control that exists (a check, a limit,
|
|
45
|
+
a permission, a test). Where there is none, decide: build one, accept the risk, or
|
|
46
|
+
leave it to a named follow-up. An accepted risk is written down with its cost.
|
|
47
|
+
|
|
48
|
+
## Checklist for the usual suspects
|
|
49
|
+
|
|
50
|
+
- [ ] Input is parsed into a type at the boundary; length, size and shape limits exist.
|
|
51
|
+
- [ ] Secrets never reach logs, commit messages, eval fixtures or subagent prompts
|
|
52
|
+
(non-negotiable 7); they are read from the environment or the credential store.
|
|
53
|
+
- [ ] A path or a command built from input cannot leave its directory or add
|
|
54
|
+
arguments; use argument arrays, not string concatenation.
|
|
55
|
+
- [ ] Authority is checked on the server side of every boundary, not inferred from the
|
|
56
|
+
client.
|
|
57
|
+
- [ ] A failure refuses safely: with no user to ask, the answer is no.
|
|
58
|
+
- [ ] Errors do not reveal more than the caller may know.
|
|
59
|
+
- [ ] New dependencies are pinned, maintained and needed; the install does not run code
|
|
60
|
+
you did not intend.
|
|
61
|
+
- [ ] Each control has a test that fails when the control is removed.
|
|
62
|
+
|
|
63
|
+
## When to write `docs/security/threat-model.md`
|
|
64
|
+
|
|
65
|
+
Write it, or extend it, when the change has at least two yes answers above, or any
|
|
66
|
+
accepted risk you would want a reviewer to find later. Keep it to one page: assets,
|
|
67
|
+
entry points, the table from question 3 for what applies, controls with their test
|
|
68
|
+
names, accepted risks with who decided. Link it from the ADR when the decision shapes
|
|
69
|
+
the architecture. Otherwise the task record line is enough.
|
|
70
|
+
|
|
71
|
+
## Do not
|
|
72
|
+
|
|
73
|
+
- Do not produce a threat model for a change with no trust boundary to satisfy a process.
|
|
74
|
+
- Do not list generic threats that have no entry point in this system.
|
|
75
|
+
- Do not accept a risk silently; an unrecorded acceptance is an unmanaged one.
|
|
76
|
+
- Do not weaken a control to make a test or a gate pass (non-negotiable 2).
|
|
@@ -12,7 +12,7 @@ export function phaseGuide(phase: Phase): string {
|
|
|
12
12
|
case "implementing":
|
|
13
13
|
return "Implementing one slice. Write a failing test first, make it pass with the least code, run the tests and read the result before claiming anything. Push when the slice is complete and reviewed: a push of a clean tree closes it (devsys_finish_slice closes it explicitly).";
|
|
14
14
|
case "reviewing":
|
|
15
|
-
return "Reviewing. Call devsys_review_start for a fresh-context review,
|
|
15
|
+
return "Reviewing. Call devsys_review_start for a fresh-context review, give its diffDigest to devsys_review_record (the reviewer submits through devsys_submit_review), fix every blocking and should-fix finding (red-first still applies), and repeat until the review is satisfied; then the slice moves to delivering.";
|
|
16
16
|
case "delivering":
|
|
17
17
|
return "Delivering. The review is satisfied. Commit with a rationale, push, and release; a push of a clean tree closes the slice (devsys_finish_slice closes it explicitly, or abandons it with a reason). Editing source reopens implementing. A red trunk (CI) is repaired first.";
|
|
18
18
|
default:
|
package/src/core/review-flow.ts
CHANGED
|
@@ -176,7 +176,8 @@ export function reviewerTask(input: {
|
|
|
176
176
|
`Review slice ${input.slice}, round ${input.round}, through these lenses: ${lenses}.`,
|
|
177
177
|
`See the change with \`git diff ${input.diffRange}\` (and \`git diff --stat ${input.diffRange}\`), plus any untracked files listed by \`git ls-files --others --exclude-standard\` (git diff does not show them; read them directly). Read the callers, callees and tests it depends on, no more.`,
|
|
178
178
|
"Do not modify files; report demonstrable defects with a realistic trigger and the smallest local repair.",
|
|
179
|
-
"
|
|
179
|
+
`Report your result by calling devsys_submit_review with slice "${input.slice}", round ${input.round}, lenses ${JSON.stringify(input.lenses)}, your sources, your findings (lens, severity, path, line, summary) and the verdict. If it refuses, for example with verdict-contradicts-findings or wrong-round, correct the call and submit again; when it accepts, end with one short line.`,
|
|
180
|
+
"Only if devsys_submit_review is not available, return the review packet from your instructions instead, with exactly this header (slice and round must not change):",
|
|
180
181
|
"",
|
|
181
182
|
`## Review — ${input.slice} — round ${input.round} — lenses: ${lenses}`,
|
|
182
183
|
"### Sources inspected",
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
import type { Finding, Severity } from "./review.ts";
|
|
2
|
+
import type { ReviewPacket } from "./review-packet.ts";
|
|
3
|
+
|
|
4
|
+
/** A reviewer's result as it arrives from the typed submit tool (see ADR 0006). */
|
|
5
|
+
export type Submission = {
|
|
6
|
+
readonly slice: string;
|
|
7
|
+
readonly round: number;
|
|
8
|
+
readonly lenses: ReadonlyArray<string>;
|
|
9
|
+
readonly sources: ReadonlyArray<string>;
|
|
10
|
+
readonly findings: ReadonlyArray<{
|
|
11
|
+
readonly lens: string;
|
|
12
|
+
readonly severity: Severity;
|
|
13
|
+
readonly path?: string | undefined;
|
|
14
|
+
readonly line?: number | undefined;
|
|
15
|
+
readonly summary: string;
|
|
16
|
+
}>;
|
|
17
|
+
readonly verdict: "no-blocking" | "blocking";
|
|
18
|
+
};
|
|
19
|
+
|
|
20
|
+
/** Why a submission was refused: a stable kebab-case id and a message the reviewer can act on. */
|
|
21
|
+
export type SubmitError = {
|
|
22
|
+
readonly id: "verdict-contradicts-findings" | "finding-lens-not-listed";
|
|
23
|
+
readonly message: string;
|
|
24
|
+
};
|
|
25
|
+
|
|
26
|
+
const counts = (f: { severity: Severity }): boolean =>
|
|
27
|
+
f.severity === "blocking" || f.severity === "should-fix";
|
|
28
|
+
|
|
29
|
+
/** The packet a submission stands for, or the refusal. The schema has already checked its shape. */
|
|
30
|
+
export function packetFromSubmission(input: Submission): ReviewPacket | SubmitError {
|
|
31
|
+
const unlisted = input.findings.find((f) => !input.lenses.includes(f.lens));
|
|
32
|
+
if (unlisted !== undefined) {
|
|
33
|
+
return {
|
|
34
|
+
id: "finding-lens-not-listed",
|
|
35
|
+
message: `a finding names lens "${unlisted.lens}" but the packet's lenses are ${input.lenses.join(", ")}; list every lens you used in \`lenses\``,
|
|
36
|
+
};
|
|
37
|
+
}
|
|
38
|
+
if (input.findings.some(counts) !== (input.verdict === "blocking")) {
|
|
39
|
+
return {
|
|
40
|
+
id: "verdict-contradicts-findings",
|
|
41
|
+
message: `verdict "${input.verdict}" contradicts the findings: blocking/should-fix findings mean "blocking", otherwise "no-blocking"`,
|
|
42
|
+
};
|
|
43
|
+
}
|
|
44
|
+
const findings: Finding[] = input.findings.map((f, i) => ({
|
|
45
|
+
id: `${f.lens}-${i + 1}`,
|
|
46
|
+
severity: f.severity,
|
|
47
|
+
...(f.path === undefined ? {} : { path: f.path }),
|
|
48
|
+
...(f.line === undefined ? {} : { line: f.line }),
|
|
49
|
+
summary: f.summary,
|
|
50
|
+
lens: f.lens,
|
|
51
|
+
}));
|
|
52
|
+
return {
|
|
53
|
+
slice: input.slice,
|
|
54
|
+
round: input.round,
|
|
55
|
+
lenses: input.lenses,
|
|
56
|
+
sources: input.sources,
|
|
57
|
+
findings,
|
|
58
|
+
verdict: input.verdict,
|
|
59
|
+
};
|
|
60
|
+
}
|
|
@@ -30,11 +30,14 @@ import { CONFIG_FILE, loadConfig } from "../state/config.ts";
|
|
|
30
30
|
import { availableModels } from "../state/models-command.ts";
|
|
31
31
|
import type { SessionState } from "../state/session-state.ts";
|
|
32
32
|
import { snapshotDiff } from "./digest.ts";
|
|
33
|
+
import type { SubmissionStore } from "./submissions.ts";
|
|
33
34
|
|
|
34
35
|
export type ReviewToolDeps = {
|
|
35
36
|
state: SessionState;
|
|
36
37
|
jev: (ctx: ExtensionContext) => Jev;
|
|
37
38
|
exec: Exec;
|
|
39
|
+
/** Results reviewers submitted with `devsys_submit_review`, read by `devsys_review_record`. */
|
|
40
|
+
submissions: SubmissionStore;
|
|
38
41
|
now?: () => Date;
|
|
39
42
|
};
|
|
40
43
|
|
|
@@ -55,6 +58,9 @@ const sliceOf = (given: string | undefined, state: SessionState): SliceRef | und
|
|
|
55
58
|
const NO_SLICE =
|
|
56
59
|
"no slice: pass `slice`, or start work on a slice first (there is no active slice)";
|
|
57
60
|
|
|
61
|
+
const NO_RESULT = (slice: string, round: number): string =>
|
|
62
|
+
`no result for round ${round} of "${slice}": the reviewer submits it with devsys_submit_review before you record, or you pass its packet markdown in \`packets\``;
|
|
63
|
+
|
|
58
64
|
const StartParameters = Type.Object({
|
|
59
65
|
slice: Type.Optional(
|
|
60
66
|
Type.String({ description: "Slice being reviewed; defaults to the active slice." }),
|
|
@@ -120,6 +126,7 @@ function spawnPayload(input: {
|
|
|
120
126
|
}
|
|
121
127
|
|
|
122
128
|
type Reply = ReturnType<typeof reply>;
|
|
129
|
+
const replyText = (r: Reply): string => r.content.map((c) => c.text).join("\n");
|
|
123
130
|
type Config = Extract<Awaited<ReturnType<typeof loadConfig>>, { ok: true }>["value"];
|
|
124
131
|
type Snapshot = Extract<Awaited<ReturnType<typeof snapshotDiff>>, { ok: true }>["value"];
|
|
125
132
|
type Prepared = { slice: SliceRef; config: Config; range: string; snap: Snapshot };
|
|
@@ -181,6 +188,8 @@ async function startRound(
|
|
|
181
188
|
const required = requiredRounds(p.config);
|
|
182
189
|
const existing = reviewOf(deps.state.get(), p.slice) ?? startReview(p.slice, required);
|
|
183
190
|
const review: ReviewState = { ...existing, required };
|
|
191
|
+
// A new round only sees what reviewers submit after it began.
|
|
192
|
+
deps.submissions.clear(p.slice);
|
|
184
193
|
deps.state.update((s) => upsertReview(s, review));
|
|
185
194
|
deps.state.update((s) => afterReviewRound(s, nextAction(review, p.snap.digest), p.slice));
|
|
186
195
|
const refusal = startRefusal(review, p);
|
|
@@ -200,7 +209,7 @@ async function startRound(
|
|
|
200
209
|
[
|
|
201
210
|
`${reviewLabel(review)}; round ${round}; diff ${p.snap.digest}. ${basis}`,
|
|
202
211
|
`lenses: ${lenses.join(", ")}`,
|
|
203
|
-
`Spawn the reviewer with agent_spawn using exactly this payload, then
|
|
212
|
+
`Spawn the reviewer with agent_spawn using exactly this payload, then call devsys_review_record with slice "${p.slice}", diffDigest "${p.snap.digest}"${rangeArg}. The reviewer submits its result with devsys_submit_review, so pass no packets (only a markdown packet the reviewer returned instead goes in packets):`,
|
|
204
213
|
JSON.stringify(spawn),
|
|
205
214
|
].join("\n"),
|
|
206
215
|
);
|
|
@@ -214,7 +223,7 @@ export function createReviewStartTool(
|
|
|
214
223
|
name: "devsys_review_start",
|
|
215
224
|
label: "Start review round",
|
|
216
225
|
description:
|
|
217
|
-
"Begin a review round for a slice: computes the diff digest, chooses review lenses, and returns the agent_spawn payload for a fresh-context reviewer. Run that spawn, then
|
|
226
|
+
"Begin a review round for a slice: computes the diff digest, chooses review lenses, and returns the agent_spawn payload for a fresh-context reviewer. Run that spawn, then call devsys_review_record (the reviewer submits its result with devsys_submit_review).",
|
|
218
227
|
promptSnippet: "Start a fresh-context review round for a slice",
|
|
219
228
|
parameters: StartParameters,
|
|
220
229
|
exposure: "model-only",
|
|
@@ -243,9 +252,12 @@ const RecordParameters = Type.Object({
|
|
|
243
252
|
slice: Type.Optional(
|
|
244
253
|
Type.String({ description: "Slice reviewed; defaults to the active slice." }),
|
|
245
254
|
),
|
|
246
|
-
packets: Type.
|
|
247
|
-
|
|
248
|
-
|
|
255
|
+
packets: Type.Optional(
|
|
256
|
+
Type.Array(Type.String(), {
|
|
257
|
+
description:
|
|
258
|
+
"Reviewer packets, verbatim markdown, for a reviewer that could not call devsys_submit_review. Results submitted through that tool for this round are read without being passed.",
|
|
259
|
+
}),
|
|
260
|
+
),
|
|
249
261
|
diffRange: Type.Optional(
|
|
250
262
|
Type.String({ description: "The range that was reviewed; defaults to HEAD." }),
|
|
251
263
|
),
|
|
@@ -275,15 +287,22 @@ async function adjusted(
|
|
|
275
287
|
return { findings: out, notes };
|
|
276
288
|
}
|
|
277
289
|
|
|
290
|
+
const lensKey = (p: ReviewPacket): string => [...p.lenses].sort().join("\u0000");
|
|
291
|
+
|
|
292
|
+
/** Submitted results win over a markdown packet for the same lenses (a reviewer that did both). */
|
|
293
|
+
function mergeResults(
|
|
294
|
+
markdown: readonly ReviewPacket[],
|
|
295
|
+
submitted: readonly ReviewPacket[],
|
|
296
|
+
): { packets: ReviewPacket[]; dropped: number } {
|
|
297
|
+
const taken = new Set(submitted.map(lensKey));
|
|
298
|
+
const kept = markdown.filter((m) => !taken.has(lensKey(m)));
|
|
299
|
+
return { packets: [...kept, ...submitted], dropped: markdown.length - kept.length };
|
|
300
|
+
}
|
|
301
|
+
|
|
278
302
|
/** Parses every packet, or the reply naming the first malformed one. */
|
|
279
303
|
function parsePackets(
|
|
280
304
|
raw: readonly string[],
|
|
281
305
|
): { ok: true; value: ReviewPacket[] } | { ok: false; reply: Reply } {
|
|
282
|
-
if (raw.length === 0)
|
|
283
|
-
return {
|
|
284
|
-
ok: false,
|
|
285
|
-
reply: reply("no packets: pass the reviewer's packet markdown in `packets`", true),
|
|
286
|
-
};
|
|
287
306
|
const packets: ReviewPacket[] = [];
|
|
288
307
|
for (const [i, text] of raw.entries()) {
|
|
289
308
|
const parsed = parseReviewPacket(text);
|
|
@@ -359,7 +378,7 @@ export function createReviewRecordTool(
|
|
|
359
378
|
name: "devsys_review_record",
|
|
360
379
|
label: "Record review round",
|
|
361
380
|
description:
|
|
362
|
-
"Record one review round from the reviewer's
|
|
381
|
+
"Record one review round from the reviewer's submitted result (or markdown packets): reads findings, lets Jev adjust severity when confident, and returns the clean-round count and the next action.",
|
|
363
382
|
promptSnippet: "Record a reviewer packet as a review round",
|
|
364
383
|
parameters: RecordParameters,
|
|
365
384
|
exposure: "direct",
|
|
@@ -370,9 +389,23 @@ export function createReviewRecordTool(
|
|
|
370
389
|
_onUpdate,
|
|
371
390
|
ctx: ExtensionContext,
|
|
372
391
|
) {
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
392
|
+
const slice = sliceOf(params.slice, deps.state);
|
|
393
|
+
if (slice === undefined) return reply(NO_SLICE, true);
|
|
394
|
+
const round = (reviewOf(deps.state.get(), slice)?.rounds.length ?? 0) + 1;
|
|
395
|
+
const markdown = parsePackets(params.packets ?? []);
|
|
396
|
+
if (!markdown.ok) {
|
|
397
|
+
return deps.submissions.forRound(slice, round).length === 0
|
|
398
|
+
? markdown.reply
|
|
399
|
+
: reply(
|
|
400
|
+
`${replyText(markdown.reply)}. A submitted result for round ${round} is waiting: omit \`packets\` to record it.`,
|
|
401
|
+
true,
|
|
402
|
+
);
|
|
403
|
+
}
|
|
404
|
+
const { packets, dropped } = mergeResults(
|
|
405
|
+
markdown.value,
|
|
406
|
+
deps.submissions.forRound(slice, round),
|
|
407
|
+
);
|
|
408
|
+
if (packets.length === 0) return reply(NO_RESULT(slice, round), true);
|
|
376
409
|
const prepared = await prepare(
|
|
377
410
|
deps,
|
|
378
411
|
ctx,
|
|
@@ -386,7 +419,14 @@ export function createReviewRecordTool(
|
|
|
386
419
|
true,
|
|
387
420
|
);
|
|
388
421
|
}
|
|
389
|
-
|
|
422
|
+
const recorded = await recordRound(deps, ctx, prepared.value, packets);
|
|
423
|
+
if (recorded.isError === true) return recorded;
|
|
424
|
+
deps.submissions.clear(slice);
|
|
425
|
+
return dropped === 0
|
|
426
|
+
? recorded
|
|
427
|
+
: reply(
|
|
428
|
+
`${replyText(recorded)}\nNote: ${dropped} markdown packet(s) for lenses already submitted were dropped; the submitted result is the one recorded.`,
|
|
429
|
+
);
|
|
390
430
|
},
|
|
391
431
|
};
|
|
392
432
|
}
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
import type { ReviewPacket } from "../core/review-packet.ts";
|
|
2
|
+
|
|
3
|
+
/** Reviewer results submitted through `devsys_submit_review`, waiting for `devsys_review_record`. */
|
|
4
|
+
export type SubmissionStore = {
|
|
5
|
+
/**
|
|
6
|
+
* Keep a packet. One for the same slice and round that shares a lens replaces the earlier one
|
|
7
|
+
* (a corrected resubmission); packets for disjoint lenses (parallel lens reviewers) are all kept.
|
|
8
|
+
*/
|
|
9
|
+
put(packet: ReviewPacket): void;
|
|
10
|
+
forRound(slice: string, round: number): ReviewPacket[];
|
|
11
|
+
/** Drop every packet of a slice, when a new round starts or one is recorded. */
|
|
12
|
+
clear(slice: string): void;
|
|
13
|
+
};
|
|
14
|
+
|
|
15
|
+
const sameRound = (a: ReviewPacket, b: ReviewPacket): boolean =>
|
|
16
|
+
a.slice === b.slice && a.round === b.round;
|
|
17
|
+
|
|
18
|
+
const sharesLens = (a: ReviewPacket, b: ReviewPacket): boolean =>
|
|
19
|
+
a.lenses.some((lens) => b.lenses.includes(lens));
|
|
20
|
+
|
|
21
|
+
/** In memory on purpose (ADR 0006): a round takes minutes and a lost submission is visible. */
|
|
22
|
+
export function createSubmissionStore(): SubmissionStore {
|
|
23
|
+
let packets: ReviewPacket[] = [];
|
|
24
|
+
return {
|
|
25
|
+
put(packet) {
|
|
26
|
+
packets = [
|
|
27
|
+
...packets.filter((p) => !(sameRound(p, packet) && sharesLens(p, packet))),
|
|
28
|
+
packet,
|
|
29
|
+
];
|
|
30
|
+
},
|
|
31
|
+
forRound(slice, round) {
|
|
32
|
+
return packets.filter((p) => p.slice === slice && p.round === round);
|
|
33
|
+
},
|
|
34
|
+
clear(slice) {
|
|
35
|
+
packets = packets.filter((p) => p.slice !== slice);
|
|
36
|
+
},
|
|
37
|
+
};
|
|
38
|
+
}
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
import type { ToolDefinition } from "@earendil-works/pi-coding-agent";
|
|
2
|
+
import { type Static, Type } from "typebox";
|
|
3
|
+
import { SEVERITIES } from "../core/review.ts";
|
|
4
|
+
import { reviewOf } from "../core/review-flow.ts";
|
|
5
|
+
import { packetFromSubmission } from "../core/review-submission.ts";
|
|
6
|
+
import type { SliceRef } from "../core/types.ts";
|
|
7
|
+
import type { SessionState } from "../state/session-state.ts";
|
|
8
|
+
import type { SubmissionStore } from "./submissions.ts";
|
|
9
|
+
|
|
10
|
+
export const SubmitParameters = Type.Object({
|
|
11
|
+
slice: Type.String({ description: "The slice you were asked to review, exactly as given." }),
|
|
12
|
+
round: Type.Integer({ minimum: 1, description: "The round number you were asked to review." }),
|
|
13
|
+
lenses: Type.Array(Type.String({ minLength: 1 }), {
|
|
14
|
+
minItems: 1,
|
|
15
|
+
description: "Every lens you reviewed through.",
|
|
16
|
+
}),
|
|
17
|
+
sources: Type.Array(Type.String(), {
|
|
18
|
+
description: "What you inspected (path:line ranges), and what you checked and refuted.",
|
|
19
|
+
}),
|
|
20
|
+
findings: Type.Array(
|
|
21
|
+
Type.Object({
|
|
22
|
+
lens: Type.String({ minLength: 1, description: "One of `lenses`." }),
|
|
23
|
+
severity: Type.Union(SEVERITIES.map((s) => Type.Literal(s))),
|
|
24
|
+
path: Type.Optional(Type.String({ description: "File the finding is about." })),
|
|
25
|
+
line: Type.Optional(Type.Integer({ minimum: 1 })),
|
|
26
|
+
summary: Type.String({
|
|
27
|
+
minLength: 1,
|
|
28
|
+
description: "One sentence: the defect, why it matters, and the smallest repair.",
|
|
29
|
+
}),
|
|
30
|
+
}),
|
|
31
|
+
{ description: "Empty when there is nothing to report." },
|
|
32
|
+
),
|
|
33
|
+
verdict: Type.Union([Type.Literal("no-blocking"), Type.Literal("blocking")], {
|
|
34
|
+
description: "`blocking` when any finding is blocking or should-fix, otherwise `no-blocking`.",
|
|
35
|
+
}),
|
|
36
|
+
});
|
|
37
|
+
|
|
38
|
+
export type SubmitDeps = { state: SessionState; submissions: SubmissionStore };
|
|
39
|
+
|
|
40
|
+
const reply = (text: string, isError = false) => ({
|
|
41
|
+
content: [{ type: "text" as const, text }],
|
|
42
|
+
details: undefined,
|
|
43
|
+
isError,
|
|
44
|
+
});
|
|
45
|
+
|
|
46
|
+
function submit(deps: SubmitDeps, params: Static<typeof SubmitParameters>) {
|
|
47
|
+
const state = deps.state.get();
|
|
48
|
+
const slice = params.slice as SliceRef;
|
|
49
|
+
const review = reviewOf(state, slice);
|
|
50
|
+
if (review === undefined && state.activeSlice !== slice) {
|
|
51
|
+
return reply(
|
|
52
|
+
`unknown-slice: "${params.slice}" is neither the active slice nor under review; use the slice name exactly as given in your task`,
|
|
53
|
+
true,
|
|
54
|
+
);
|
|
55
|
+
}
|
|
56
|
+
const expected = (review?.rounds.length ?? 0) + 1;
|
|
57
|
+
if (params.round !== expected) {
|
|
58
|
+
return reply(
|
|
59
|
+
`wrong-round: this is for round ${params.round}, but round ${expected} of "${params.slice}" is the one being reviewed; use the round number from your task`,
|
|
60
|
+
true,
|
|
61
|
+
);
|
|
62
|
+
}
|
|
63
|
+
const packet = packetFromSubmission(params);
|
|
64
|
+
if ("id" in packet) return reply(`${packet.id}: ${packet.message}`, true);
|
|
65
|
+
deps.submissions.put(packet);
|
|
66
|
+
return reply(
|
|
67
|
+
`submitted: ${params.slice} round ${params.round}, lenses ${params.lenses.join(", ")}, verdict ${params.verdict}, ${params.findings.length} finding(s). End with one short line.`,
|
|
68
|
+
);
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/** `devsys_submit_review`: a reviewer hands over its packet as validated arguments (ADR 0006). */
|
|
72
|
+
export function createSubmitReviewTool(deps: SubmitDeps): ToolDefinition<typeof SubmitParameters> {
|
|
73
|
+
return {
|
|
74
|
+
name: "devsys_submit_review",
|
|
75
|
+
label: "Submit review result",
|
|
76
|
+
description:
|
|
77
|
+
"For reviewer subagents only: submit your review result as structured arguments. A result that contradicts itself or is for another round is refused with an error id; fix it and submit again. Then end with one short line.",
|
|
78
|
+
promptSnippet: "Submit a review result (reviewer subagents)",
|
|
79
|
+
parameters: SubmitParameters,
|
|
80
|
+
exposure: "direct",
|
|
81
|
+
execute(_id, params: Static<typeof SubmitParameters>) {
|
|
82
|
+
return Promise.resolve(submit(deps, params));
|
|
83
|
+
},
|
|
84
|
+
};
|
|
85
|
+
}
|