@pmelab/gtd 16.0.0 → 17.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1 @@
1
+ { "modules": ["../claude/hooks/register.tsx"] }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@pmelab/gtd",
3
- "version": "16.0.0",
3
+ "version": "17.2.0",
4
4
  "private": false,
5
5
  "description": "Git-aware CLI that emits the next prompt for an autonomous coding agent based on the current repository state",
6
6
  "bin": {
@@ -16,7 +16,14 @@
16
16
  "src/workflows/",
17
17
  "README.md",
18
18
  "LICENSE",
19
- "schema.json"
19
+ "schema.json",
20
+ ".claude-plugin/plugin.json",
21
+ "hooks/",
22
+ "bin/",
23
+ "claude/hooks/",
24
+ "claude/types/",
25
+ "!claude/**/*.test.ts",
26
+ "skills/"
20
27
  ],
21
28
  "publishConfig": {
22
29
  "access": "public",
package/schema.json CHANGED
@@ -5,7 +5,7 @@
5
5
  "properties": {
6
6
  "vars": {
7
7
  "type": "object",
8
- "description": "Flat name -> scalar map merged into the workflow's vars. Scalars are coerced to strings.",
8
+ "description": "Flat name -> scalar map merged into the workflow's process settings (defaults), pinned for the whole process at its start. Scalars are coerced to strings.",
9
9
  "additionalProperties": {
10
10
  "type": [
11
11
  "string",
@@ -14,6 +14,38 @@
14
14
  ]
15
15
  }
16
16
  },
17
+ "env": {
18
+ "type": "object",
19
+ "description": "Flat name -> scalar map merged into the workflow's environment settings (envDefaults) — values that change how a step runs on this machine, like the test command or a model hint. Read fresh on every gtd call, never pinned to a process. GTD_<NAME> environment variables override these entries. Scalars are coerced to strings.",
20
+ "additionalProperties": {
21
+ "type": [
22
+ "string",
23
+ "number",
24
+ "boolean"
25
+ ]
26
+ }
27
+ },
28
+ "judge": {
29
+ "type": "object",
30
+ "required": [],
31
+ "properties": {
32
+ "provider": {
33
+ "type": "string",
34
+ "enum": [
35
+ "fixed",
36
+ "jev",
37
+ "llm"
38
+ ],
39
+ "description": "Judge provider `gtd judge run` uses when no --provider flag and no GTD_JUDGE_PROVIDER is set. Absent means auto: jev when TYPESAFE_API_KEY is set, otherwise llm."
40
+ },
41
+ "model": {
42
+ "type": "string",
43
+ "description": "Model for provider llm, used when no --model flag and no GTD_JUDGE_MODEL is set. Pairing it with another provider is an error."
44
+ }
45
+ },
46
+ "additionalProperties": false,
47
+ "description": "Which judge `gtd judge run` uses, when no flag or GTD_JUDGE_* variable says."
48
+ },
17
49
  "modes": {
18
50
  "type": "object",
19
51
  "description": "Steering-file modes a workflow step's mode: may name. Each entry declares at least one of format/validate: shell commands gtd runs via bash with $GTD_FILE set to the steering file's path. format rewrites the file in place; validate exits 0 when valid, non-zero with findings on stdout/stderr otherwise. The halves layer independently, so naming a built-in mode (qa/review) and declaring only format: adds formatting while keeping gtd's own validation. gtd ships no formatter — bring your own (prettier, dprint, a script).",
@@ -0,0 +1,248 @@
1
+ ---
2
+ name: authoring
3
+ description: >-
4
+ Write or edit a gtd workflow (the repository's `gtd.config.ts`). Use when the
5
+ user asks to create a custom gtd workflow, customize or change their
6
+ workflow's shape, add/remove/rename a step, add a gate/phase/review step,
7
+ change what an agent is prompted to do, adjust fix caps, models, or steering
8
+ files, or otherwise change the flow gtd runs.
9
+ ---
10
+
11
+ # Authoring a gtd workflow
12
+
13
+ A gtd workflow is **plain async TypeScript**: a `gtd.config.ts` at the
14
+ repository root default-exports the **flow**, one async function that awaits
15
+ **steps** built from `@pmelab/gtd/flows`; optional `defaults` (process
16
+ settings), `envDefaults` (environment settings), `summary`, `base` and
17
+ `steering` (steering file → mode, for the LSP) exports sit beside it, and any
18
+ other export is a helper gtd ignores. Every step is a commit; gtd finds where a
19
+ process rests by **replaying** the flow over the episode's commits, so the git
20
+ history IS the state and nothing is stored anywhere else.
21
+
22
+ Your job is to produce or edit that module so it loads cleanly and does what the
23
+ user wants. Driving a workflow once it exists is a separate concern — that is
24
+ what a driver does.
25
+
26
+ **Trust:** gtd evaluates `gtd.config.ts` on every command that resolves workflow
27
+ state (`gtd next` and `gtd lsp` included). It is code the user's repository
28
+ runs; write it with the same care as a build script.
29
+
30
+ ## Golden rule: start from the bundled default, edit incrementally
31
+
32
+ Do **not** write a workflow from a blank page unless the user wants something
33
+ tiny. gtd ships one known-good workflow and runs it when no `gtd.config.ts` is
34
+ found, and publishes it as `@pmelab/gtd/workflow`: its default export is that
35
+ flow, and every phase and single step it is built from is a named export. Start
36
+ by importing what you keep and writing only what changes:
37
+
38
+ ```ts
39
+ import { start } from "@pmelab/gtd/flows"
40
+ import bundled, { afterTail, buildTail } from "@pmelab/gtd/workflow"
41
+
42
+ export {
43
+ defaults,
44
+ envDefaults,
45
+ summary,
46
+ base,
47
+ steering,
48
+ } from "@pmelab/gtd/workflow"
49
+
50
+ export default async ({ entry }) =>
51
+ entry === "hotfix"
52
+ ? afterTail(await buildTail(true, start()))
53
+ : bundled({ entry })
54
+ ```
55
+
56
+ To change a phase itself, read its source in the npm package
57
+ (`node_modules/@pmelab/gtd/src/workflows/`, or under `$(npm root -g)` for a
58
+ global install) and write your own version in `gtd.config.ts`, reusing its
59
+ single steps.
60
+
61
+ There is no `extends`/merge: the innermost `gtd.config.ts` walking up from the
62
+ current directory is the whole workflow. If one already exists, read it and edit
63
+ it in place.
64
+
65
+ Prefer the bundled workflow's own parts over re-implementing them — `healthy`,
66
+ `escalation`, `gate`, `design`, `architecturePass`, `packages`, `specReview`,
67
+ `qualityLap`, `review`, `buildTail`, and single steps like `triage` or `fix`.
68
+ Their full step names are versioned API.
69
+
70
+ Make one small change, **verify it loads** (see "Verify"), then make the next. A
71
+ workflow that fails to load breaks every gtd command in the repository.
72
+
73
+ ## The step API
74
+
75
+ | Call | Actor | Rest content | Resolves to |
76
+ | ---------------------------- | ------- | ------------ | -------------------------------------- |
77
+ | `agent(name, prompt, opts?)` | `agent` | `prompt` | `void`, once an agent turn landed |
78
+ | `human(name, opts?)` | `human` | `message` | `void`, once a person landed |
79
+ | `run(name, body, opts?)` | `check` | `script` | `void`, once the run's tree landed |
80
+ | `judge(name, spec)` | `judge` | `message` | `{ answers, truncated }` |
81
+ | `restart()` | — | — | never: ends the episode from any depth |
82
+
83
+ - `run` body: a POSIX `sh` string the driver runs verbatim. Decide in flow code,
84
+ then render the script from those values — `check(name, command, …)` and the
85
+ exported `checkScript`, `revertScript`, `restoreScript`, `removeScript`,
86
+ `moveScript` and `quote` do that for the common cases. **A run's outcome is
87
+ what it leaves in the tree** — read it back with `changes()` and `read()`.
88
+ - `judge` takes `{ questions, evidence, message?, label? }`. Questions are
89
+ `{ id, primitive: "noul" | "choice" | "score", instructions, criteria }`;
90
+ `evidence` is an object of strings, the `judgeBudgetBytes` var split evenly
91
+ across its keys. It resolves to `answers` — one `{ answer, p }` per question
92
+ id (a `noul` reads back as `"yes"`/`"no"`), `undefined` when the verdict left
93
+ it out — and `truncated`, the evidence keys the budget cut. Compare answers
94
+ with plain `if`s; landing with no verdict leaves every answer `undefined`, so
95
+ make `undefined` take the conservative branch.
96
+ - A person or a driver sees a rest as one of the five content kinds `capture` (a
97
+ dirty tree at a human step), `message`, `script`, `prompt`, `stalled`. Every
98
+ step you add must fit one of them; there is no sixth.
99
+
100
+ Options (all optional): `label`, `file` (a `.gtd/` path), `mode` (needs `file`;
101
+ `qa`, `review`, or a `.gtdrc` `modes:` name), `message` (human/judge), `model`,
102
+ `system` (agent), `allowEmpty` (agent), `acceptClean` (human), `base` (the
103
+ commit the step reviews since — what `gtd base` prints).
104
+
105
+ Helpers — pure reads of the commit replay stands on (the tree the last step
106
+ left, never the live working tree): `read(path)`, `glob(pattern)`,
107
+ `changes(glob?)` (what the last step changed: `{ path, status, before, after }`
108
+ per path, `status` one of `"added"`/`"modified"`/`"deleted"`, plus `paths` and
109
+ `get(path)`), `sections(text)` (`## ` headings), `openQuestions(text)` (a `qa`
110
+ document's unanswered questions), `vars` (process settings, pinned at process
111
+ start — safe to branch on), `env` (environment settings, read live — only for
112
+ prompt text, step options and `run()` bodies, never a branch), `head()` (the
113
+ commit the flow stands on) and `start()` (the process's diff base). State a flow
114
+ needs across steps — a counter, the previous report, a review round's base —
115
+ lives in local variables; replay rebuilds them.
116
+
117
+ Composition: `scope(name, fn)` prefixes step names (`build.fix`) and sets their
118
+ **memory scope** (one scope = one agent conversation = one model/system — mixing
119
+ them inside a scope fails the process); `scope({ name?, model, system }, fn)`
120
+ also sets defaults for agent steps inside. `refuse(message)` refuses the pending
121
+ landing — call it right after the step whose turn you reject.
122
+ `@pmelab/gtd/flows` exports `requireProgress(file)`, `requireAnswers(file)` and
123
+ `requireRevert(edited, base)`, three such checks ready-made.
124
+
125
+ ## Names, commits and history
126
+
127
+ - The step name is the `<to>` in `gtd(<actor>): <from> → <to>`, and every
128
+ landing carries `Gtd-Step: <name>#<n>`. It is also the memory scope key (up to
129
+ the last dot). Rename a step and every process resting on it diverges.
130
+ - An episode ends when the flow returns or calls `restart()`; the next starts at
131
+ the flow's first step on an ordinary start — that step is where a finished
132
+ process waits (the bundled one is `human("idle", …)`).
133
+ - `gtd --entry <name>` starts a process with the flow's `{ entry }` argument set
134
+ to `<name>` (`undefined` on an ordinary start). Branch on it, and `refuse()`
135
+ names you don't accept; a flow that never reads `entry` accepts none. An
136
+ `export const base = (entry, vars) => commitish | undefined` fixes an entered
137
+ process's diff base. `--var <name>=<value>` only pins process settings: names
138
+ the workflow's `defaults` or `.gtdrc` `vars:` declare — never an environment
139
+ setting.
140
+
141
+ ## Landing rules you are designing for
142
+
143
+ - **Agent turn changed something** → the step completes.
144
+ - **Agent turn changed nothing** → an **attempt**: an empty commit, the process
145
+ stays; the next dispatch is a **stall**. Pass `allowEmpty: true` when "nothing
146
+ to change" is a legitimate result (a reviewer approving by writing nothing).
147
+ - **Human landing changed nothing** → a no-op, the gate keeps waiting — unless
148
+ `acceptClean: true`, which makes "change nothing" mean "accept as-is".
149
+ - **Run landed a clean tree** → the step completes; if replay comes straight
150
+ back to the same step, the landing is **settled** (the driver stops).
151
+ - **Nothing the flow branches on explains the turn** → call `refuse(message)`:
152
+ nothing lands, `gtd land` exits 1.
153
+
154
+ Branch on what the step left, not on who acted:
155
+
156
+ ```ts
157
+ await run(
158
+ "check",
159
+ `${env.testCommand} > .gtd/FEEDBACK.md 2>&1 && rm -f .gtd/FEEDBACK.md`,
160
+ )
161
+ if (read(".gtd/FEEDBACK.md") !== undefined) {
162
+ await agent("fix", "Fix what .gtd/FEEDBACK.md reports, then delete it.", {
163
+ file: ".gtd/FEEDBACK.md",
164
+ })
165
+ }
166
+ ```
167
+
168
+ Keep `.gtd/` clean across processes: a steering file should be deleted by the
169
+ step that consumes it. A workflow that accumulates files in `.gtd/` is almost
170
+ certainly a bug.
171
+
172
+ ## Rules for flow code
173
+
174
+ Flow code is replayed on every command, so it must reach the same steps every
175
+ time it sees the same history. `run()` bodies and module top-level code are
176
+ exempt. gtd does not read the source ahead of time: breaking a rule shows up
177
+ when replay runs, as an error or, for nondeterminism, as a divergence later.
178
+
179
+ - No IO or nondeterminism in flow code — no clock, randomness, environment,
180
+ network or filesystem. Read the tree through the helpers; do IO inside a
181
+ `run()` body.
182
+ - Await only a step, `scope()`, or a function that steps. Anything else fails
183
+ with `the flow awaited something that is not a step`.
184
+ - One call site per step name — wrap a reused helper in two different
185
+ `scope()`s.
186
+ - No `try`/`catch` around a step: `restart()` and refusals travel as exceptions.
187
+ - Only the options a step accepts; an unknown key fails naming the step.
188
+ - The default export is a flow that reaches a step on an ordinary start — the
189
+ same step whatever the repository's files hold; every `mode` must exist.
190
+
191
+ ## Verify (after every change)
192
+
193
+ 1. **`gtd next`** — loads the workflow (printing every load error at once) and
194
+ shows the resolved rest: step, actor, label, file. It never mutates, so run
195
+ it as often as you like.
196
+ 2. **A scratch repository** with at least one commit — make the change a step
197
+ expects, run `gtd land --json=script | sh`, then `gtd next` to see where it
198
+ went. A flow is code, so walking it is the only way to see its branches.
199
+
200
+ `gtd validate` is NOT for this — it validates a **steering file**, not the
201
+ workflow.
202
+
203
+ ## Worked example: add an approval gate before building
204
+
205
+ In the bundled default, `planAndBuild` runs the design phase, the architecture
206
+ pass (which writes `.gtd/packages/`), then builds the packages. Add a human
207
+ sign-off between the two:
208
+
209
+ ```ts
210
+ const planAndBuild = async (): Promise<void> => {
211
+ for (;;) {
212
+ await design()
213
+ await architecturePass()
214
+ await human("approve-plan", {
215
+ message:
216
+ "The packages under .gtd/packages/ are ready. Edit them to adjust the plan, or change nothing — then run `gtd land` to start building.",
217
+ label: "Approve the plan",
218
+ acceptClean: true,
219
+ })
220
+ await packages()
221
+ if ((await buildTail(false)) === "signoff") return
222
+ await reUnwind()
223
+ }
224
+ }
225
+ ```
226
+
227
+ `acceptClean: true` is what makes an untouched landing approve; without it the
228
+ gate would wait for an edit. `approve-plan` sits in the `root` scope and is a
229
+ new, unique name. Verify: `gtd next` loads without errors, and in a scratch
230
+ repository a landing at `architecture.decompose` (or `architecture-promote`) now
231
+ leads to `approve-plan`, and an untouched landing there to
232
+ `packages.item.building`.
233
+
234
+ ## No migration
235
+
236
+ A process's commits only make sense to the workflow that made them. If the
237
+ workflow changes under an in-flight process so its history no longer replays to
238
+ the steps its commits name, gtd refuses with a divergence error telling you to
239
+ run `gtd abandon`. Tell the user to finish or `gtd abandon` any in-flight
240
+ process before switching to the edited workflow.
241
+
242
+ ## Notes
243
+
244
+ - This skill is versioned in the gtd repository, not auto-installed. When gtd is
245
+ upgraded, re-copy it from the new version's `skills/authoring/SKILL.md`.
246
+ - Where this file and the code disagree, the code wins: the step API's own doc
247
+ comments in `@pmelab/gtd/flows`, and the bundled workflow under
248
+ `src/workflows/`.
@@ -136,6 +136,7 @@ export interface FlowContext {
136
136
  readonly threads: (text: string) => readonly ThreadInfo[]
137
137
  readonly codeThreads: () => readonly CodeThreadInfo[]
138
138
  readonly vars: Readonly<Record<string, string>>
139
+ readonly env: Readonly<Record<string, string>>
139
140
  readonly head: () => string
140
141
  readonly start: () => string
141
142
  /**
@@ -256,19 +257,31 @@ export const start = (): string => ctx().start()
256
257
  export const skillsFor = (localName: string, ownSkills?: readonly string[]): readonly string[] =>
257
258
  ctx().skillsFor(localName, ownSkills)
258
259
 
259
- /** The merged workflow variables. */
260
- export const vars: Readonly<Record<string, string>> = new Proxy(
261
- {},
262
- {
263
- get: (_target, key) => (typeof key === "string" ? ctx().vars[key] : undefined),
264
- has: (_target, key) => typeof key === "string" && key in ctx().vars,
265
- ownKeys: () => Object.keys(ctx().vars),
266
- getOwnPropertyDescriptor: (_target, key) =>
267
- typeof key === "string" && key in ctx().vars
268
- ? { enumerable: true, configurable: true, value: ctx().vars[key] }
269
- : undefined,
270
- },
271
- )
260
+ const settingsProxy = (
261
+ read: () => Readonly<Record<string, string>>,
262
+ ): Readonly<Record<string, string>> =>
263
+ new Proxy(
264
+ {},
265
+ {
266
+ get: (_target, key) => (typeof key === "string" ? read()[key] : undefined),
267
+ has: (_target, key) => typeof key === "string" && key in read(),
268
+ ownKeys: () => Object.keys(read()),
269
+ getOwnPropertyDescriptor: (_target, key) =>
270
+ typeof key === "string" && key in read()
271
+ ? { enumerable: true, configurable: true, value: read()[key] }
272
+ : undefined,
273
+ },
274
+ )
275
+
276
+ /** The process settings: pinned at process start, so flow code may branch on them. */
277
+ export const vars: Readonly<Record<string, string>> = settingsProxy(() => ctx().vars)
278
+
279
+ /**
280
+ * The environment settings: resolved live on every invocation. Flow code must
281
+ * read them only where they cannot change the next step (step options, script
282
+ * bodies) — replay cannot enforce this.
283
+ */
284
+ export const env: Readonly<Record<string, string>> = settingsProxy(() => ctx().env)
272
285
 
273
286
  // ── Text utilities ──────────────────────────────────────────────────────────
274
287
 
@@ -343,6 +356,7 @@ export interface SummaryContext {
343
356
  readonly processCost: number
344
357
  readonly processCostByModel: readonly { readonly model: string; readonly cost: number }[]
345
358
  readonly vars: Readonly<Record<string, string>>
359
+ readonly env: Readonly<Record<string, string>>
346
360
  }
347
361
 
348
362
  /** `gtd summary`'s prompt — a workflow module's optional `summary` export. */
@@ -1,6 +1,7 @@
1
1
  import {
2
2
  answered,
3
3
  check,
4
+ env,
4
5
  human,
5
6
  judge,
6
7
  numeric,
@@ -17,7 +18,7 @@ export const FIX_CAP = 3
17
18
 
18
19
  /** Run the suite as step `name`; resolves `true` when it passed. A failure is in `.gtd/FEEDBACK.md`. */
19
20
  export const baseline = (name: string, label = "Checking the baseline"): Promise<boolean> =>
20
- check(name, vars.testCommand ?? "", { report: FEEDBACK, label })
21
+ check(name, env.testCommand ?? "", { report: FEEDBACK, label })
21
22
 
22
23
  /** How many escalation rounds a run of red checks has spent — reset once the suite goes green. */
23
24
  export interface EscalationCount {
@@ -95,7 +96,7 @@ export const healthy = async (
95
96
  let fixes = options.fixesSoFar ?? 0
96
97
  let previous: string | undefined
97
98
  for (;;) {
98
- const green = await check("health.check", vars.testCommand ?? "", {
99
+ const green = await check("health.check", env.testCommand ?? "", {
99
100
  report: FEEDBACK,
100
101
  label: "Running checks",
101
102
  // Swept only on green: an unresolved analysis survives every retry.
@@ -68,7 +68,8 @@ deliberately separate from whoever wrote the code, with no attachment
68
68
  to it. Write a structured review document grouping a diff into
69
69
  chunks; classify a round of the human's feedback as actionable or
70
70
  just approving; answer the human's questions inline in the review;
71
- and, when asked, fix the small nits the human flagged in one batch.
71
+ and, when asked, fix the small nits the human flagged and the risks
72
+ you marked yourself, each in one batch.
72
73
  Beyond that you never fix or build anything yourself.`
73
74
 
74
75
  export const specReviewerPersona = `You are the adversarial spec-conformance checker in gtd's build
@@ -1,6 +1,7 @@
1
1
  import { afterEach, describe, expect, it } from "vitest"
2
2
  import { installContext, type Change, type JudgeAnswer, type StepRequest } from "../flows/index.js"
3
3
  import { review, type ReviewOutcome } from "./review.js"
4
+ import { unified } from "./index.js"
4
5
  import { fixtureContext } from "./text.fixture.js"
5
6
 
6
7
  afterEach(() => installContext(undefined))
@@ -30,6 +31,8 @@ interface Drive {
30
31
  readonly vars?: Readonly<Record<string, string>>
31
32
  /** The REVIEW.md the human leaves; defaults to `notes` laid over the baseline. */
32
33
  readonly after?: string
34
+ /** The REVIEW.md each `review.reviewing` turn writes, in order; also raises the review-turn stop to one past the last. */
35
+ readonly reviewDocs?: readonly (string | undefined)[]
33
36
  }
34
37
 
35
38
  interface Run {
@@ -51,7 +54,10 @@ const drive = async (d: Drive): Promise<Run> => {
51
54
  let reviews = 0
52
55
  const effects: Record<string, () => void> = {
53
56
  "review.reviewing": () => {
54
- if (++reviews > 1) throw new Stop()
57
+ const docs = d.reviewDocs
58
+ if (++reviews > (docs?.length ?? 0) + (docs === undefined ? 1 : 0)) throw new Stop()
59
+ const written = docs?.[reviews - 1]
60
+ if (written !== undefined) files.set(REVIEW, written)
55
61
  },
56
62
  "review.await-review": () => {
57
63
  if (++awaits > 1) throw new Stop()
@@ -104,6 +110,65 @@ const drive = async (d: Drive): Promise<Run> => {
104
110
 
105
111
  const verdict = (answer: string, p = 0.9): JudgeAnswer => ({ answer, p })
106
112
 
113
+ describe("the risk-fix pass", () => {
114
+ const risky = doc(["Risk: drops the carry", "sub"])
115
+ const clean = doc(["add — fine", "sub"])
116
+
117
+ it("a marked risk is fixed, kept green, re-reviewed, then rests at the gate", async () => {
118
+ const run = await drive({ notes: [], reviewDocs: [risky, clean] })
119
+ expect(run.log.slice(0, 5)).toEqual([
120
+ "review.reviewing",
121
+ "review.fix-risks",
122
+ "health.check",
123
+ "review.reviewing",
124
+ "review.await-review",
125
+ ])
126
+ expect(run.prompts.get("review.fix-risks")).toContain("risk-1")
127
+ expect(run.prompts.get("review.fix-risks")).toContain("Risk: drops the carry")
128
+ expect(run.prompts.get("review.fix-risks")).toContain("Leave `.gtd/REVIEW.md` untouched")
129
+ })
130
+
131
+ it("a risk the re-review still marks goes to the gate with no second fix", async () => {
132
+ const run = await drive({ notes: [], reviewDocs: [risky, risky] })
133
+ expect(run.log.filter((n) => n === "review.fix-risks")).toHaveLength(1)
134
+ expect(run.log.slice(0, 5)).toEqual([
135
+ "review.reviewing",
136
+ "review.fix-risks",
137
+ "health.check",
138
+ "review.reviewing",
139
+ "review.await-review",
140
+ ])
141
+ })
142
+
143
+ it("no marker means no fix-risks step", async () => {
144
+ const run = await drive({ notes: [], reviewDocs: [clean] })
145
+ expect(run.log).not.toContain("review.fix-risks")
146
+ expect(run.log[0]).toBe("review.reviewing")
147
+ expect(run.log[1]).toBe("review.await-review")
148
+ })
149
+
150
+ it("a nit re-review round gets its own single pass", async () => {
151
+ const run = await drive({
152
+ notes: ["add — typo", "sub", "mul"],
153
+ answers: { "note-1": verdict("nit") },
154
+ reviewDocs: [undefined, risky, clean],
155
+ })
156
+ expect(run.log).toEqual([
157
+ "review.reviewing",
158
+ "review.await-review",
159
+ "review.triage",
160
+ "review.fix-nits",
161
+ "health.check",
162
+ "review.closing",
163
+ "review.reviewing",
164
+ "review.fix-risks",
165
+ "health.check",
166
+ "review.reviewing",
167
+ "review.await-review",
168
+ ])
169
+ })
170
+ })
171
+
107
172
  describe("review verdict routing", () => {
108
173
  const notes = ["add — typo", "sub", "mul"]
109
174
 
@@ -245,3 +310,24 @@ describe("review verdict routing", () => {
245
310
  expect(run.result).toMatchObject({ verdict: "feedback", edited: [] })
246
311
  })
247
312
  })
313
+
314
+ describe("the default quality lenses", () => {
315
+ it("are the six lenses in the settled order", () => {
316
+ expect(unified.defaults.qualityReviews!.split(",").map((l) => l.trim())).toEqual([
317
+ "correctness",
318
+ "owasp-security",
319
+ "ponytail-review",
320
+ "test-audit",
321
+ "conventions",
322
+ "spec-challenge",
323
+ ])
324
+ })
325
+
326
+ it("expose builtInLenses through the public workflow module", () => {
327
+ expect(Object.keys(unified.builtInLenses).sort()).toEqual([
328
+ "conventions",
329
+ "correctness",
330
+ "spec-challenge",
331
+ ])
332
+ })
333
+ })
@@ -17,7 +17,7 @@ import {
17
17
  type Change,
18
18
  type JudgeQuestion,
19
19
  } from "../flows/index.js"
20
- import { reviewNotes, stripCodeThreads, type ReviewNote } from "../steering/index.js"
20
+ import { reviewNotes, reviewRisks, stripCodeThreads, type ReviewNote } from "../steering/index.js"
21
21
  import { escalation, FIX_CAP, healthy, type EscalationCount } from "./health.js"
22
22
  import {
23
23
  answerReviewQuestions,
@@ -26,6 +26,7 @@ import {
26
26
  fix,
27
27
  fixNits,
28
28
  fixQuality,
29
+ fixRisks,
29
30
  QUALITY,
30
31
  REQUIREMENTS,
31
32
  REVIEW,
@@ -43,8 +44,8 @@ export const qualityLenses = (): readonly string[] =>
43
44
  .filter((lens) => lens.length > 0)
44
45
 
45
46
  /**
46
- * One review turn per lens over the whole change, each appending what it
47
- * finds blocking to `.gtd/QUALITY.md`. Resolves `"findings"` when that file
47
+ * One review turn per lens over the whole change, each appending every
48
+ * finding to `.gtd/QUALITY.md`. Resolves `"findings"` when that file
48
49
  * has any.
49
50
  */
50
51
  export const qualityLap = async (): Promise<"clean" | "findings"> => {
@@ -270,6 +271,20 @@ const finish = async (
270
271
  return routeNotes(notes, { round, escalations, close, outcome, unfolded })
271
272
  }
272
273
 
274
+ /** Write the review; if it marks risks, fix them, keep green, and write it again — once, so the re-review's own risks reach the human unfixed. */
275
+ const reviewOnce = async (
276
+ base: string,
277
+ carry: string | undefined,
278
+ escalations: EscalationCount,
279
+ ): Promise<void> => {
280
+ await reviewing(base, carry)
281
+ const risks = reviewRisks(read(REVIEW) ?? "")
282
+ if (risks.length === 0) return
283
+ await fixRisks(risks)
284
+ await healthy(fix, { escalations })
285
+ await reviewing(base, carry)
286
+ }
287
+
273
288
  /**
274
289
  * A reviewer writes `.gtd/REVIEW.md` over everything since `base`, a human
275
290
  * reviews and signs off or comments, and each note is judged: edits go to
@@ -282,7 +297,7 @@ export const review = async (
282
297
  ): Promise<ReviewOutcome> => {
283
298
  let carry: string | undefined
284
299
  for (;;) {
285
- await reviewing(base, carry)
300
+ await reviewOnce(base, carry, escalations)
286
301
  carry = undefined
287
302
  let reviewed = head()
288
303
  let collectedAt: string | undefined
@@ -26,7 +26,7 @@ import { skills } from "./skills.js"
26
26
  // - build.review.answer-review-questions, build.review.fix-nits — NOT yet
27
27
  // e2e-grounded; only steps.test.ts's preamble checks, which use local names
28
28
  describe("the bundled workflow's skills map", () => {
29
- it("declares exactly the sixteen bundled agent steps, by full name", () => {
29
+ it("declares exactly the seventeen bundled agent steps, by full name", () => {
30
30
  expect(Object.keys(skills).sort()).toEqual(
31
31
  [
32
32
  "design.triage",
@@ -44,6 +44,7 @@ describe("the bundled workflow's skills map", () => {
44
44
  "build.review.reviewing",
45
45
  "build.review.answer-review-questions",
46
46
  "build.review.fix-nits",
47
+ "build.review.fix-risks",
47
48
  "build.review.collecting",
48
49
  ].sort(),
49
50
  )
@@ -33,5 +33,6 @@ export const skills: Readonly<Record<string, readonly string[]>> = {
33
33
  "build.review.reviewing": ["code-review-and-quality"],
34
34
  "build.review.answer-review-questions": ["code-review-and-quality"],
35
35
  "build.review.fix-nits": ["incremental-implementation", "code-simplification"],
36
+ "build.review.fix-risks": ["debugging-and-error-recovery", "incremental-implementation"],
36
37
  "build.review.collecting": ["code-review-and-quality"],
37
38
  }