@pmelab/gtd 17.0.0 → 17.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1 @@
1
+ { "modules": ["../claude/hooks/register.tsx"] }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@pmelab/gtd",
3
- "version": "17.0.0",
3
+ "version": "17.3.0",
4
4
  "private": false,
5
5
  "description": "Git-aware CLI that emits the next prompt for an autonomous coding agent based on the current repository state",
6
6
  "bin": {
@@ -16,7 +16,14 @@
16
16
  "src/workflows/",
17
17
  "README.md",
18
18
  "LICENSE",
19
- "schema.json"
19
+ "schema.json",
20
+ ".claude-plugin/plugin.json",
21
+ "hooks/",
22
+ "bin/",
23
+ "claude/hooks/",
24
+ "claude/types/",
25
+ "!claude/**/*.test.ts",
26
+ "skills/"
20
27
  ],
21
28
  "publishConfig": {
22
29
  "access": "public",
@@ -0,0 +1,248 @@
1
+ ---
2
+ name: authoring
3
+ description: >-
4
+ Write or edit a gtd workflow (the repository's `gtd.config.ts`). Use when the
5
+ user asks to create a custom gtd workflow, customize or change their
6
+ workflow's shape, add/remove/rename a step, add a gate/phase/review step,
7
+ change what an agent is prompted to do, adjust fix caps, models, or steering
8
+ files, or otherwise change the flow gtd runs.
9
+ ---
10
+
11
+ # Authoring a gtd workflow
12
+
13
+ A gtd workflow is **plain async TypeScript**: a `gtd.config.ts` at the
14
+ repository root default-exports the **flow**, one async function that awaits
15
+ **steps** built from `@pmelab/gtd/flows`; optional `defaults` (process
16
+ settings), `envDefaults` (environment settings), `summary`, `base` and
17
+ `steering` (steering file → mode, for the LSP) exports sit beside it, and any
18
+ other export is a helper gtd ignores. Every step is a commit; gtd finds where a
19
+ process rests by **replaying** the flow over the episode's commits, so the git
20
+ history IS the state and nothing is stored anywhere else.
21
+
22
+ Your job is to produce or edit that module so it loads cleanly and does what the
23
+ user wants. Driving a workflow once it exists is a separate concern — that is
24
+ what a driver does.
25
+
26
+ **Trust:** gtd evaluates `gtd.config.ts` on every command that resolves workflow
27
+ state (`gtd next` and `gtd lsp` included). It is code the user's repository
28
+ runs; write it with the same care as a build script.
29
+
30
+ ## Golden rule: start from the bundled default, edit incrementally
31
+
32
+ Do **not** write a workflow from a blank page unless the user wants something
33
+ tiny. gtd ships one known-good workflow and runs it when no `gtd.config.ts` is
34
+ found, and publishes it as `@pmelab/gtd/workflow`: its default export is that
35
+ flow, and every phase and single step it is built from is a named export. Start
36
+ by importing what you keep and writing only what changes:
37
+
38
+ ```ts
39
+ import { start } from "@pmelab/gtd/flows"
40
+ import bundled, { afterTail, buildTail } from "@pmelab/gtd/workflow"
41
+
42
+ export {
43
+ defaults,
44
+ envDefaults,
45
+ summary,
46
+ base,
47
+ steering,
48
+ } from "@pmelab/gtd/workflow"
49
+
50
+ export default async ({ entry }) =>
51
+ entry === "hotfix"
52
+ ? afterTail(await buildTail(true, start()))
53
+ : bundled({ entry })
54
+ ```
55
+
56
+ To change a phase itself, read its source in the npm package
57
+ (`node_modules/@pmelab/gtd/src/workflows/`, or under `$(npm root -g)` for a
58
+ global install) and write your own version in `gtd.config.ts`, reusing its
59
+ single steps.
60
+
61
+ There is no `extends`/merge: the innermost `gtd.config.ts` walking up from the
62
+ current directory is the whole workflow. If one already exists, read it and edit
63
+ it in place.
64
+
65
+ Prefer the bundled workflow's own parts over re-implementing them — `healthy`,
66
+ `escalation`, `gate`, `design`, `architecturePass`, `packages`, `specReview`,
67
+ `qualityLap`, `review`, `buildTail`, and single steps like `triage` or `fix`.
68
+ Their full step names are versioned API.
69
+
70
+ Make one small change, **verify it loads** (see "Verify"), then make the next. A
71
+ workflow that fails to load breaks every gtd command in the repository.
72
+
73
+ ## The step API
74
+
75
+ | Call | Actor | Rest content | Resolves to |
76
+ | ---------------------------- | ------- | ------------ | -------------------------------------- |
77
+ | `agent(name, prompt, opts?)` | `agent` | `prompt` | `void`, once an agent turn landed |
78
+ | `human(name, opts?)` | `human` | `message` | `void`, once a person landed |
79
+ | `run(name, body, opts?)` | `check` | `script` | `void`, once the run's tree landed |
80
+ | `judge(name, spec)` | `judge` | `message` | `{ answers, truncated }` |
81
+ | `restart()` | — | — | never: ends the episode from any depth |
82
+
83
+ - `run` body: a POSIX `sh` string the driver runs verbatim. Decide in flow code,
84
+ then render the script from those values — `check(name, command, …)` and the
85
+ exported `checkScript`, `revertScript`, `restoreScript`, `removeScript`,
86
+ `moveScript` and `quote` do that for the common cases. **A run's outcome is
87
+ what it leaves in the tree** — read it back with `changes()` and `read()`.
88
+ - `judge` takes `{ questions, evidence, message?, label? }`. Questions are
89
+ `{ id, primitive: "noul" | "choice" | "score", instructions, criteria }`;
90
+ `evidence` is an object of strings, the `judgeBudgetBytes` var split evenly
91
+ across its keys. It resolves to `answers` — one `{ answer, p }` per question
92
+ id (a `noul` reads back as `"yes"`/`"no"`), `undefined` when the verdict left
93
+ it out — and `truncated`, the evidence keys the budget cut. Compare answers
94
+ with plain `if`s; landing with no verdict leaves every answer `undefined`, so
95
+ make `undefined` take the conservative branch.
96
+ - A person or a driver sees a rest as one of the five content kinds `capture` (a
97
+ dirty tree at a human step), `message`, `script`, `prompt`, `stalled`. Every
98
+ step you add must fit one of them; there is no sixth.
99
+
100
+ Options (all optional): `label`, `file` (a `.gtd/` path), `mode` (needs `file`;
101
+ `qa`, `review`, or a `.gtdrc` `modes:` name), `message` (human/judge), `model`,
102
+ `system` (agent), `allowEmpty` (agent), `acceptClean` (human), `base` (the
103
+ commit the step reviews since — what `gtd base` prints).
104
+
105
+ Helpers — pure reads of the commit replay stands on (the tree the last step
106
+ left, never the live working tree): `read(path)`, `glob(pattern)`,
107
+ `changes(glob?)` (what the last step changed: `{ path, status, before, after }`
108
+ per path, `status` one of `"added"`/`"modified"`/`"deleted"`, plus `paths` and
109
+ `get(path)`), `sections(text)` (`## ` headings), `openQuestions(text)` (a `qa`
110
+ document's unanswered questions), `vars` (process settings, pinned at process
111
+ start — safe to branch on), `env` (environment settings, read live — only for
112
+ prompt text, step options and `run()` bodies, never a branch), `head()` (the
113
+ commit the flow stands on) and `start()` (the process's diff base). State a flow
114
+ needs across steps — a counter, the previous report, a review round's base —
115
+ lives in local variables; replay rebuilds them.
116
+
117
+ Composition: `scope(name, fn)` prefixes step names (`build.fix`) and sets their
118
+ **memory scope** (one scope = one agent conversation = one model/system — mixing
119
+ them inside a scope fails the process); `scope({ name?, model, system }, fn)`
120
+ also sets defaults for agent steps inside. `refuse(message)` refuses the pending
121
+ landing — call it right after the step whose turn you reject.
122
+ `@pmelab/gtd/flows` exports `requireProgress(file)`, `requireAnswers(file)` and
123
+ `requireRevert(edited, base)`, three such checks ready-made.
124
+
125
+ ## Names, commits and history
126
+
127
+ - The step name is the `<to>` in `gtd(<actor>): <from> → <to>`, and every
128
+ landing carries `Gtd-Step: <name>#<n>`. It is also the memory scope key (up to
129
+ the last dot). Rename a step and every process resting on it diverges.
130
+ - An episode ends when the flow returns or calls `restart()`; the next starts at
131
+ the flow's first step on an ordinary start — that step is where a finished
132
+ process waits (the bundled one is `human("idle", …)`).
133
+ - `gtd --entry <name>` starts a process with the flow's `{ entry }` argument set
134
+ to `<name>` (`undefined` on an ordinary start). Branch on it, and `refuse()`
135
+ names you don't accept; a flow that never reads `entry` accepts none. An
136
+ `export const base = (entry, vars) => commitish | undefined` fixes an entered
137
+ process's diff base. `--var <name>=<value>` only pins process settings: names
138
+ the workflow's `defaults` or `.gtdrc` `vars:` declare — never an environment
139
+ setting.
140
+
141
+ ## Landing rules you are designing for
142
+
143
+ - **Agent turn changed something** → the step completes.
144
+ - **Agent turn changed nothing** → an **attempt**: an empty commit, the process
145
+ stays; the next dispatch is a **stall**. Pass `allowEmpty: true` when "nothing
146
+ to change" is a legitimate result (a reviewer approving by writing nothing).
147
+ - **Human landing changed nothing** → a no-op, the gate keeps waiting — unless
148
+ `acceptClean: true`, which makes "change nothing" mean "accept as-is".
149
+ - **Run landed a clean tree** → the step completes; if replay comes straight
150
+ back to the same step, the landing is **settled** (the driver stops).
151
+ - **Nothing the flow branches on explains the turn** → call `refuse(message)`:
152
+ nothing lands, `gtd land` exits 1.
153
+
154
+ Branch on what the step left, not on who acted:
155
+
156
+ ```ts
157
+ await run(
158
+ "check",
159
+ `${env.testCommand} > .gtd/FEEDBACK.md 2>&1 && rm -f .gtd/FEEDBACK.md`,
160
+ )
161
+ if (read(".gtd/FEEDBACK.md") !== undefined) {
162
+ await agent("fix", "Fix what .gtd/FEEDBACK.md reports, then delete it.", {
163
+ file: ".gtd/FEEDBACK.md",
164
+ })
165
+ }
166
+ ```
167
+
168
+ Keep `.gtd/` clean across processes: a steering file should be deleted by the
169
+ step that consumes it. A workflow that accumulates files in `.gtd/` is almost
170
+ certainly a bug.
171
+
172
+ ## Rules for flow code
173
+
174
+ Flow code is replayed on every command, so it must reach the same steps every
175
+ time it sees the same history. `run()` bodies and module top-level code are
176
+ exempt. gtd does not read the source ahead of time: breaking a rule shows up
177
+ when replay runs, as an error or, for nondeterminism, as a divergence later.
178
+
179
+ - No IO or nondeterminism in flow code — no clock, randomness, environment,
180
+ network or filesystem. Read the tree through the helpers; do IO inside a
181
+ `run()` body.
182
+ - Await only a step, `scope()`, or a function that steps. Anything else fails
183
+ with `the flow awaited something that is not a step`.
184
+ - One call site per step name — wrap a reused helper in two different
185
+ `scope()`s.
186
+ - No `try`/`catch` around a step: `restart()` and refusals travel as exceptions.
187
+ - Only the options a step accepts; an unknown key fails naming the step.
188
+ - The default export is a flow that reaches a step on an ordinary start — the
189
+ same step whatever the repository's files hold; every `mode` must exist.
190
+
191
+ ## Verify (after every change)
192
+
193
+ 1. **`gtd next`** — loads the workflow (printing every load error at once) and
194
+ shows the resolved rest: step, actor, label, file. It never mutates, so run
195
+ it as often as you like.
196
+ 2. **A scratch repository** with at least one commit — make the change a step
197
+ expects, run `gtd land --json=script | sh`, then `gtd next` to see where it
198
+ went. A flow is code, so walking it is the only way to see its branches.
199
+
200
+ `gtd validate` is NOT for this — it validates a **steering file**, not the
201
+ workflow.
202
+
203
+ ## Worked example: add an approval gate before building
204
+
205
+ In the bundled default, `planAndBuild` runs the design phase, the architecture
206
+ pass (which writes `.gtd/packages/`), then builds the packages. Add a human
207
+ sign-off between the two:
208
+
209
+ ```ts
210
+ const planAndBuild = async (): Promise<void> => {
211
+ for (;;) {
212
+ await design()
213
+ await architecturePass()
214
+ await human("approve-plan", {
215
+ message:
216
+ "The packages under .gtd/packages/ are ready. Edit them to adjust the plan, or change nothing — then run `gtd land` to start building.",
217
+ label: "Approve the plan",
218
+ acceptClean: true,
219
+ })
220
+ await packages()
221
+ if ((await buildTail(false)) === "signoff") return
222
+ await reUnwind()
223
+ }
224
+ }
225
+ ```
226
+
227
+ `acceptClean: true` is what makes an untouched landing approve; without it the
228
+ gate would wait for an edit. `approve-plan` sits in the `root` scope and is a
229
+ new, unique name. Verify: `gtd next` loads without errors, and in a scratch
230
+ repository a landing at `architecture.decompose` (or `architecture-promote`) now
231
+ leads to `approve-plan`, and an untouched landing there to
232
+ `packages.item.building`.
233
+
234
+ ## No migration
235
+
236
+ A process's commits only make sense to the workflow that made them. If the
237
+ workflow changes under an in-flight process so its history no longer replays to
238
+ the steps its commits name, gtd refuses with a divergence error telling you to
239
+ run `gtd abandon`. Tell the user to finish or `gtd abandon` any in-flight
240
+ process before switching to the edited workflow.
241
+
242
+ ## Notes
243
+
244
+ - This skill is versioned in the gtd repository, not auto-installed. When gtd is
245
+ upgraded, re-copy it from the new version's `skills/authoring/SKILL.md`.
246
+ - Where this file and the code disagree, the code wins: the step API's own doc
247
+ comments in `@pmelab/gtd/flows`, and the bundled workflow under
248
+ `src/workflows/`.
@@ -68,7 +68,8 @@ deliberately separate from whoever wrote the code, with no attachment
68
68
  to it. Write a structured review document grouping a diff into
69
69
  chunks; classify a round of the human's feedback as actionable or
70
70
  just approving; answer the human's questions inline in the review;
71
- and, when asked, fix the small nits the human flagged in one batch.
71
+ and, when asked, fix the small nits the human flagged and the risks
72
+ you marked yourself, each in one batch.
72
73
  Beyond that you never fix or build anything yourself.`
73
74
 
74
75
  export const specReviewerPersona = `You are the adversarial spec-conformance checker in gtd's build
@@ -179,13 +180,20 @@ export const questionBarReturn = `- This lap continues the same goal as the firs
179
180
  checkboxes
180
181
  - \`## Answered Questions\` is always the last \`##\` section — a moved
181
182
  question lands there, never wherever \`## Open Questions\` used to sit
182
- - Never re-raise a deleted question, and never re-open a settled
183
- \`## Answered Questions\` entry
183
+ - Never re-raise a deleted question. A settled \`## Answered Questions\`
184
+ entry stays settled, except one carrying a human footnote on its
185
+ \`### \` heading: that is the only entry that may change. Fold the note
186
+ in, then either rewrite the recorded answer in place (still under
187
+ \`## Answered Questions\`) or, when the note leaves the point genuinely
188
+ open, move it back to \`## Open Questions\` as a fresh fork with
189
+ options and the \`_your answer_\` slot. A note that only asks a
190
+ question follows the footnote rule: one \`- A:\` reply, answer unchanged
184
191
  - An answer may earn a follow-up: if it opens a genuinely new fork above
185
192
  the bar — one the answer itself created — raise it as a fresh \`##
186
193
  Open Questions\` entry on this same lap. Never restate a question
187
194
  already asked, and never treat this as licence to re-open a question
188
- already settled under \`## Answered Questions\`
195
+ already settled under \`## Answered Questions\` that carries no
196
+ footnote on its heading
189
197
  - Recognise a silent lap from \`## Open Questions\` still present with
190
198
  nothing ticked and nothing else changed — the human's way of saying
191
199
  the gap is already closed. That lap ends the questions, whatever the
@@ -1,6 +1,7 @@
1
1
  import { afterEach, describe, expect, it } from "vitest"
2
2
  import { installContext, type Change, type JudgeAnswer, type StepRequest } from "../flows/index.js"
3
3
  import { review, type ReviewOutcome } from "./review.js"
4
+ import { unified } from "./index.js"
4
5
  import { fixtureContext } from "./text.fixture.js"
5
6
 
6
7
  afterEach(() => installContext(undefined))
@@ -30,6 +31,8 @@ interface Drive {
30
31
  readonly vars?: Readonly<Record<string, string>>
31
32
  /** The REVIEW.md the human leaves; defaults to `notes` laid over the baseline. */
32
33
  readonly after?: string
34
+ /** The REVIEW.md each `review.reviewing` turn writes, in order; also raises the review-turn stop to one past the last. */
35
+ readonly reviewDocs?: readonly (string | undefined)[]
33
36
  }
34
37
 
35
38
  interface Run {
@@ -51,7 +54,10 @@ const drive = async (d: Drive): Promise<Run> => {
51
54
  let reviews = 0
52
55
  const effects: Record<string, () => void> = {
53
56
  "review.reviewing": () => {
54
- if (++reviews > 1) throw new Stop()
57
+ const docs = d.reviewDocs
58
+ if (++reviews > (docs?.length ?? 0) + (docs === undefined ? 1 : 0)) throw new Stop()
59
+ const written = docs?.[reviews - 1]
60
+ if (written !== undefined) files.set(REVIEW, written)
55
61
  },
56
62
  "review.await-review": () => {
57
63
  if (++awaits > 1) throw new Stop()
@@ -104,6 +110,65 @@ const drive = async (d: Drive): Promise<Run> => {
104
110
 
105
111
  const verdict = (answer: string, p = 0.9): JudgeAnswer => ({ answer, p })
106
112
 
113
+ describe("the risk-fix pass", () => {
114
+ const risky = doc(["Risk: drops the carry", "sub"])
115
+ const clean = doc(["add — fine", "sub"])
116
+
117
+ it("a marked risk is fixed, kept green, re-reviewed, then rests at the gate", async () => {
118
+ const run = await drive({ notes: [], reviewDocs: [risky, clean] })
119
+ expect(run.log.slice(0, 5)).toEqual([
120
+ "review.reviewing",
121
+ "review.fix-risks",
122
+ "health.check",
123
+ "review.reviewing",
124
+ "review.await-review",
125
+ ])
126
+ expect(run.prompts.get("review.fix-risks")).toContain("risk-1")
127
+ expect(run.prompts.get("review.fix-risks")).toContain("Risk: drops the carry")
128
+ expect(run.prompts.get("review.fix-risks")).toContain("Leave `.gtd/REVIEW.md` untouched")
129
+ })
130
+
131
+ it("a risk the re-review still marks goes to the gate with no second fix", async () => {
132
+ const run = await drive({ notes: [], reviewDocs: [risky, risky] })
133
+ expect(run.log.filter((n) => n === "review.fix-risks")).toHaveLength(1)
134
+ expect(run.log.slice(0, 5)).toEqual([
135
+ "review.reviewing",
136
+ "review.fix-risks",
137
+ "health.check",
138
+ "review.reviewing",
139
+ "review.await-review",
140
+ ])
141
+ })
142
+
143
+ it("no marker means no fix-risks step", async () => {
144
+ const run = await drive({ notes: [], reviewDocs: [clean] })
145
+ expect(run.log).not.toContain("review.fix-risks")
146
+ expect(run.log[0]).toBe("review.reviewing")
147
+ expect(run.log[1]).toBe("review.await-review")
148
+ })
149
+
150
+ it("a nit re-review round gets its own single pass", async () => {
151
+ const run = await drive({
152
+ notes: ["add — typo", "sub", "mul"],
153
+ answers: { "note-1": verdict("nit") },
154
+ reviewDocs: [undefined, risky, clean],
155
+ })
156
+ expect(run.log).toEqual([
157
+ "review.reviewing",
158
+ "review.await-review",
159
+ "review.triage",
160
+ "review.fix-nits",
161
+ "health.check",
162
+ "review.closing",
163
+ "review.reviewing",
164
+ "review.fix-risks",
165
+ "health.check",
166
+ "review.reviewing",
167
+ "review.await-review",
168
+ ])
169
+ })
170
+ })
171
+
107
172
  describe("review verdict routing", () => {
108
173
  const notes = ["add — typo", "sub", "mul"]
109
174
 
@@ -245,3 +310,24 @@ describe("review verdict routing", () => {
245
310
  expect(run.result).toMatchObject({ verdict: "feedback", edited: [] })
246
311
  })
247
312
  })
313
+
314
+ describe("the default quality lenses", () => {
315
+ it("are the six lenses in the settled order", () => {
316
+ expect(unified.defaults.qualityReviews!.split(",").map((l) => l.trim())).toEqual([
317
+ "correctness",
318
+ "owasp-security",
319
+ "ponytail-review",
320
+ "test-audit",
321
+ "conventions",
322
+ "spec-challenge",
323
+ ])
324
+ })
325
+
326
+ it("expose builtInLenses through the public workflow module", () => {
327
+ expect(Object.keys(unified.builtInLenses).sort()).toEqual([
328
+ "conventions",
329
+ "correctness",
330
+ "spec-challenge",
331
+ ])
332
+ })
333
+ })
@@ -17,7 +17,7 @@ import {
17
17
  type Change,
18
18
  type JudgeQuestion,
19
19
  } from "../flows/index.js"
20
- import { reviewNotes, stripCodeThreads, type ReviewNote } from "../steering/index.js"
20
+ import { reviewNotes, reviewRisks, stripCodeThreads, type ReviewNote } from "../steering/index.js"
21
21
  import { escalation, FIX_CAP, healthy, type EscalationCount } from "./health.js"
22
22
  import {
23
23
  answerReviewQuestions,
@@ -26,6 +26,7 @@ import {
26
26
  fix,
27
27
  fixNits,
28
28
  fixQuality,
29
+ fixRisks,
29
30
  QUALITY,
30
31
  REQUIREMENTS,
31
32
  REVIEW,
@@ -43,8 +44,8 @@ export const qualityLenses = (): readonly string[] =>
43
44
  .filter((lens) => lens.length > 0)
44
45
 
45
46
  /**
46
- * One review turn per lens over the whole change, each appending what it
47
- * finds blocking to `.gtd/QUALITY.md`. Resolves `"findings"` when that file
47
+ * One review turn per lens over the whole change, each appending every
48
+ * finding to `.gtd/QUALITY.md`. Resolves `"findings"` when that file
48
49
  * has any.
49
50
  */
50
51
  export const qualityLap = async (): Promise<"clean" | "findings"> => {
@@ -270,6 +271,20 @@ const finish = async (
270
271
  return routeNotes(notes, { round, escalations, close, outcome, unfolded })
271
272
  }
272
273
 
274
+ /** Write the review; if it marks risks, fix them, keep green, and write it again — once, so the re-review's own risks reach the human unfixed. */
275
+ const reviewOnce = async (
276
+ base: string,
277
+ carry: string | undefined,
278
+ escalations: EscalationCount,
279
+ ): Promise<void> => {
280
+ await reviewing(base, carry)
281
+ const risks = reviewRisks(read(REVIEW) ?? "")
282
+ if (risks.length === 0) return
283
+ await fixRisks(risks)
284
+ await healthy(fix, { escalations })
285
+ await reviewing(base, carry)
286
+ }
287
+
273
288
  /**
274
289
  * A reviewer writes `.gtd/REVIEW.md` over everything since `base`, a human
275
290
  * reviews and signs off or comments, and each note is judged: edits go to
@@ -282,7 +297,7 @@ export const review = async (
282
297
  ): Promise<ReviewOutcome> => {
283
298
  let carry: string | undefined
284
299
  for (;;) {
285
- await reviewing(base, carry)
300
+ await reviewOnce(base, carry, escalations)
286
301
  carry = undefined
287
302
  let reviewed = head()
288
303
  let collectedAt: string | undefined
@@ -26,7 +26,7 @@ import { skills } from "./skills.js"
26
26
  // - build.review.answer-review-questions, build.review.fix-nits — NOT yet
27
27
  // e2e-grounded; only steps.test.ts's preamble checks, which use local names
28
28
  describe("the bundled workflow's skills map", () => {
29
- it("declares exactly the sixteen bundled agent steps, by full name", () => {
29
+ it("declares exactly the seventeen bundled agent steps, by full name", () => {
30
30
  expect(Object.keys(skills).sort()).toEqual(
31
31
  [
32
32
  "design.triage",
@@ -44,6 +44,7 @@ describe("the bundled workflow's skills map", () => {
44
44
  "build.review.reviewing",
45
45
  "build.review.answer-review-questions",
46
46
  "build.review.fix-nits",
47
+ "build.review.fix-risks",
47
48
  "build.review.collecting",
48
49
  ].sort(),
49
50
  )
@@ -33,5 +33,6 @@ export const skills: Readonly<Record<string, readonly string[]>> = {
33
33
  "build.review.reviewing": ["code-review-and-quality"],
34
34
  "build.review.answer-review-questions": ["code-review-and-quality"],
35
35
  "build.review.fix-nits": ["incremental-implementation", "code-simplification"],
36
+ "build.review.fix-risks": ["debugging-and-error-recovery", "incremental-implementation"],
36
37
  "build.review.collecting": ["code-review-and-quality"],
37
38
  }
@@ -91,6 +91,46 @@ describe("the bundled workflow's steps declare skills — a bundled step's rende
91
91
  expect(prompt).toContain("owasp-security")
92
92
  })
93
93
 
94
+ it("correctness loads code-review-and-quality and carries all four trace points", async () => {
95
+ const request = await capture(() => steps.reviewQuality("correctness"))
96
+ expect(agentSkillsOption(request)).toEqual(["code-review-and-quality"])
97
+ const body = JSON.stringify(request)
98
+ for (const point of [
99
+ "partial-failure",
100
+ "format, not just its presence",
101
+ "parallel write paths",
102
+ "contract test",
103
+ ])
104
+ expect(body).toContain(point)
105
+ })
106
+
107
+ it.each(["conventions", "spec-challenge"])(
108
+ "%s sends no skills and carries its brief",
109
+ async (lens) => {
110
+ const request = await capture(() => steps.reviewQuality(lens))
111
+ expect(agentSkillsOption(request)).toEqual([])
112
+ expect(JSON.stringify(request)).toContain(
113
+ steps.builtInLenses[lens]!.brief.split("\n")[0]!.slice(0, 40),
114
+ )
115
+ },
116
+ )
117
+
118
+ it("an unknown lens loads itself as a skill, with no brief", async () => {
119
+ const request = await capture(() => steps.reviewQuality("my-lens"))
120
+ expect(agentSkillsOption(request)).toEqual(["my-lens"])
121
+ expect(JSON.stringify(request)).not.toContain("This lens's brief")
122
+ })
123
+
124
+ it("a configured build.quality.reviewing entry replaces a built-in lens's skills; the brief stays", async () => {
125
+ const prompt = agentPrompt(
126
+ await capture(() => steps.reviewQuality("conventions"), {
127
+ "quality.reviewing": ["my-org-checklist"],
128
+ }),
129
+ )
130
+ expect(prompt).toContain("my-org-checklist")
131
+ expect(prompt).toContain("AGENTS.md")
132
+ })
133
+
94
134
  it("fixQuality carries build.fix-quality's bundled skills", async () => {
95
135
  const prompt = agentPrompt(await capture(() => steps.fixQuality()))
96
136
  expect(prompt).toContain("incremental-implementation, code-simplification")
@@ -143,6 +183,19 @@ describe("the bundled workflow's steps declare skills — a bundled step's rende
143
183
  expect(request.prompt).toContain("Leave `.gtd/REVIEW.md` untouched")
144
184
  })
145
185
 
186
+ it("fixRisks fixes every risk, tolerates an empty turn, and leaves .gtd/REVIEW.md alone", async () => {
187
+ const request = await capture(() =>
188
+ steps.fixRisks([{ id: "risk-1", anchor: "calc ./a.ts#1-1", text: "Risk: drops it" }]),
189
+ )
190
+ if (request.kind !== "agent") throw new Error("unreachable")
191
+ expect(request.name).toBe("review.fix-risks")
192
+ expect(request.options).toMatchObject({ allowEmpty: true })
193
+ expect(request.prompt).toContain("debugging-and-error-recovery, incremental-implementation")
194
+ expect(request.prompt).toContain("risk-1 — calc ./a.ts#1-1")
195
+ expect(request.prompt).toContain("Risk: drops it")
196
+ expect(request.prompt).toContain("Leave `.gtd/REVIEW.md` untouched")
197
+ })
198
+
146
199
  it("reviewing names the carry-over commit only when given one", async () => {
147
200
  const bare = await capture(() => steps.reviewing("base"))
148
201
  const carried = await capture(() => steps.reviewing("base", "abc1234"))
@@ -134,6 +134,18 @@ export const escalationExhausted = (): Promise<void> =>
134
134
 
135
135
  // ── Quality and review ──────────────────────────────────────────────────────
136
136
 
137
+ export interface BuiltInLens {
138
+ readonly skills: readonly string[]
139
+ readonly brief: string
140
+ }
141
+
142
+ /** Lenses the workflow defines itself, not bundled skills: `skills/` is not in the npm package, and a missing lens skill burns a turn silently. */
143
+ export const builtInLenses: Readonly<Record<string, BuiltInLens>> = {
144
+ correctness: { skills: ["code-review-and-quality"], brief: t.correctnessBrief },
145
+ conventions: { skills: [], brief: t.conventionsBrief },
146
+ "spec-challenge": { skills: [], brief: t.specChallengeBrief },
147
+ }
148
+
137
149
  /**
138
150
  * One quality review, through the skill `lens`. `lens` rides as this turn's
139
151
  * own `skills` option — a `.gtdrc` `build.quality.reviewing` entry still
@@ -141,14 +153,18 @@ export const escalationExhausted = (): Promise<void> =>
141
153
  * but absent one the lens itself is what the turn loads by default.
142
154
  */
143
155
  export const reviewQuality = (lens: string): Promise<void> =>
144
- t.agentWithSkills("quality.reviewing", t.buildQualityReviewingPrompt(lens), {
145
- label: "Reviewing (one quality lens)",
146
- file: QUALITY,
147
- model: planner(),
148
- system: t.reviewerSystem(),
149
- allowEmpty: true,
150
- skills: [lens],
151
- })
156
+ t.agentWithSkills(
157
+ "quality.reviewing",
158
+ t.buildQualityReviewingPrompt(lens, builtInLenses[lens]?.brief),
159
+ {
160
+ label: "Reviewing (one quality lens)",
161
+ file: QUALITY,
162
+ model: planner(),
163
+ system: t.reviewerSystem(),
164
+ allowEmpty: true,
165
+ skills: builtInLenses[lens]?.skills ?? [lens],
166
+ },
167
+ )
152
168
 
153
169
  export const fixQuality = (): Promise<void> =>
154
170
  t.agentWithSkills("fix-quality", t.buildFixQualityPrompt(), {
@@ -192,6 +208,16 @@ export const fixNits = (notes: readonly t.NoteInput[]): Promise<void> =>
192
208
  system: t.reviewerSystem(),
193
209
  })
194
210
 
211
+ /** Fix the `Risk:`-marked notes the reviewer named. Shares the `build.review` conversation like `fixNits`; an empty turn means the risk was judged false. */
212
+ export const fixRisks = (notes: readonly t.NoteInput[]): Promise<void> =>
213
+ t.agentWithSkills("review.fix-risks", t.buildReviewFixRisksPrompt(notes), {
214
+ label: "Fixing the reviewer's risks",
215
+ file: REVIEW,
216
+ model: planner(),
217
+ system: t.reviewerSystem(),
218
+ allowEmpty: true,
219
+ })
220
+
195
221
  export const awaitReview = (base: string): Promise<void> =>
196
222
  human("review.await-review", {
197
223
  message: t.buildReviewAwaitReviewMessage(base),