@pmelab/gtd 17.0.0 → 17.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -17,7 +17,7 @@ import {
17
17
  type Change,
18
18
  type JudgeQuestion,
19
19
  } from "../flows/index.js"
20
- import { reviewNotes, stripCodeThreads, type ReviewNote } from "../steering/index.js"
20
+ import { reviewNotes, reviewRisks, stripCodeThreads, type ReviewNote } from "../steering/index.js"
21
21
  import { escalation, FIX_CAP, healthy, type EscalationCount } from "./health.js"
22
22
  import {
23
23
  answerReviewQuestions,
@@ -26,6 +26,7 @@ import {
26
26
  fix,
27
27
  fixNits,
28
28
  fixQuality,
29
+ fixRisks,
29
30
  QUALITY,
30
31
  REQUIREMENTS,
31
32
  REVIEW,
@@ -43,8 +44,8 @@ export const qualityLenses = (): readonly string[] =>
43
44
  .filter((lens) => lens.length > 0)
44
45
 
45
46
  /**
46
- * One review turn per lens over the whole change, each appending what it
47
- * finds blocking to `.gtd/QUALITY.md`. Resolves `"findings"` when that file
47
+ * One review turn per lens over the whole change, each appending every
48
+ * finding to `.gtd/QUALITY.md`. Resolves `"findings"` when that file
48
49
  * has any.
49
50
  */
50
51
  export const qualityLap = async (): Promise<"clean" | "findings"> => {
@@ -270,6 +271,20 @@ const finish = async (
270
271
  return routeNotes(notes, { round, escalations, close, outcome, unfolded })
271
272
  }
272
273
 
274
+ /** Write the review; if it marks risks, fix them, keep green, and write it again — once, so the re-review's own risks reach the human unfixed. */
275
+ const reviewOnce = async (
276
+ base: string,
277
+ carry: string | undefined,
278
+ escalations: EscalationCount,
279
+ ): Promise<void> => {
280
+ await reviewing(base, carry)
281
+ const risks = reviewRisks(read(REVIEW) ?? "")
282
+ if (risks.length === 0) return
283
+ await fixRisks(risks)
284
+ await healthy(fix, { escalations })
285
+ await reviewing(base, carry)
286
+ }
287
+
273
288
  /**
274
289
  * A reviewer writes `.gtd/REVIEW.md` over everything since `base`, a human
275
290
  * reviews and signs off or comments, and each note is judged: edits go to
@@ -282,7 +297,7 @@ export const review = async (
282
297
  ): Promise<ReviewOutcome> => {
283
298
  let carry: string | undefined
284
299
  for (;;) {
285
- await reviewing(base, carry)
300
+ await reviewOnce(base, carry, escalations)
286
301
  carry = undefined
287
302
  let reviewed = head()
288
303
  let collectedAt: string | undefined
@@ -26,7 +26,7 @@ import { skills } from "./skills.js"
26
26
  // - build.review.answer-review-questions, build.review.fix-nits — NOT yet
27
27
  // e2e-grounded; only steps.test.ts's preamble checks, which use local names
28
28
  describe("the bundled workflow's skills map", () => {
29
- it("declares exactly the sixteen bundled agent steps, by full name", () => {
29
+ it("declares exactly the seventeen bundled agent steps, by full name", () => {
30
30
  expect(Object.keys(skills).sort()).toEqual(
31
31
  [
32
32
  "design.triage",
@@ -44,6 +44,7 @@ describe("the bundled workflow's skills map", () => {
44
44
  "build.review.reviewing",
45
45
  "build.review.answer-review-questions",
46
46
  "build.review.fix-nits",
47
+ "build.review.fix-risks",
47
48
  "build.review.collecting",
48
49
  ].sort(),
49
50
  )
@@ -33,5 +33,6 @@ export const skills: Readonly<Record<string, readonly string[]>> = {
33
33
  "build.review.reviewing": ["code-review-and-quality"],
34
34
  "build.review.answer-review-questions": ["code-review-and-quality"],
35
35
  "build.review.fix-nits": ["incremental-implementation", "code-simplification"],
36
+ "build.review.fix-risks": ["debugging-and-error-recovery", "incremental-implementation"],
36
37
  "build.review.collecting": ["code-review-and-quality"],
37
38
  }
@@ -91,6 +91,46 @@ describe("the bundled workflow's steps declare skills — a bundled step's rende
91
91
  expect(prompt).toContain("owasp-security")
92
92
  })
93
93
 
94
+ it("correctness loads code-review-and-quality and carries all four trace points", async () => {
95
+ const request = await capture(() => steps.reviewQuality("correctness"))
96
+ expect(agentSkillsOption(request)).toEqual(["code-review-and-quality"])
97
+ const body = JSON.stringify(request)
98
+ for (const point of [
99
+ "partial-failure",
100
+ "format, not just its presence",
101
+ "parallel write paths",
102
+ "contract test",
103
+ ])
104
+ expect(body).toContain(point)
105
+ })
106
+
107
+ it.each(["conventions", "spec-challenge"])(
108
+ "%s sends no skills and carries its brief",
109
+ async (lens) => {
110
+ const request = await capture(() => steps.reviewQuality(lens))
111
+ expect(agentSkillsOption(request)).toEqual([])
112
+ expect(JSON.stringify(request)).toContain(
113
+ steps.builtInLenses[lens]!.brief.split("\n")[0]!.slice(0, 40),
114
+ )
115
+ },
116
+ )
117
+
118
+ it("an unknown lens loads itself as a skill, with no brief", async () => {
119
+ const request = await capture(() => steps.reviewQuality("my-lens"))
120
+ expect(agentSkillsOption(request)).toEqual(["my-lens"])
121
+ expect(JSON.stringify(request)).not.toContain("This lens's brief")
122
+ })
123
+
124
+ it("a configured build.quality.reviewing entry replaces a built-in lens's skills; the brief stays", async () => {
125
+ const prompt = agentPrompt(
126
+ await capture(() => steps.reviewQuality("conventions"), {
127
+ "quality.reviewing": ["my-org-checklist"],
128
+ }),
129
+ )
130
+ expect(prompt).toContain("my-org-checklist")
131
+ expect(prompt).toContain("AGENTS.md")
132
+ })
133
+
94
134
  it("fixQuality carries build.fix-quality's bundled skills", async () => {
95
135
  const prompt = agentPrompt(await capture(() => steps.fixQuality()))
96
136
  expect(prompt).toContain("incremental-implementation, code-simplification")
@@ -143,6 +183,19 @@ describe("the bundled workflow's steps declare skills — a bundled step's rende
143
183
  expect(request.prompt).toContain("Leave `.gtd/REVIEW.md` untouched")
144
184
  })
145
185
 
186
+ it("fixRisks fixes every risk, tolerates an empty turn, and leaves .gtd/REVIEW.md alone", async () => {
187
+ const request = await capture(() =>
188
+ steps.fixRisks([{ id: "risk-1", anchor: "calc ./a.ts#1-1", text: "Risk: drops it" }]),
189
+ )
190
+ if (request.kind !== "agent") throw new Error("unreachable")
191
+ expect(request.name).toBe("review.fix-risks")
192
+ expect(request.options).toMatchObject({ allowEmpty: true })
193
+ expect(request.prompt).toContain("debugging-and-error-recovery, incremental-implementation")
194
+ expect(request.prompt).toContain("risk-1 — calc ./a.ts#1-1")
195
+ expect(request.prompt).toContain("Risk: drops it")
196
+ expect(request.prompt).toContain("Leave `.gtd/REVIEW.md` untouched")
197
+ })
198
+
146
199
  it("reviewing names the carry-over commit only when given one", async () => {
147
200
  const bare = await capture(() => steps.reviewing("base"))
148
201
  const carried = await capture(() => steps.reviewing("base", "abc1234"))
@@ -134,6 +134,18 @@ export const escalationExhausted = (): Promise<void> =>
134
134
 
135
135
  // ── Quality and review ──────────────────────────────────────────────────────
136
136
 
137
+ export interface BuiltInLens {
138
+ readonly skills: readonly string[]
139
+ readonly brief: string
140
+ }
141
+
142
+ /** Lenses the workflow defines itself, not bundled skills: `skills/` is not in the npm package, and a missing lens skill burns a turn silently. */
143
+ export const builtInLenses: Readonly<Record<string, BuiltInLens>> = {
144
+ correctness: { skills: ["code-review-and-quality"], brief: t.correctnessBrief },
145
+ conventions: { skills: [], brief: t.conventionsBrief },
146
+ "spec-challenge": { skills: [], brief: t.specChallengeBrief },
147
+ }
148
+
137
149
  /**
138
150
  * One quality review, through the skill `lens`. `lens` rides as this turn's
139
151
  * own `skills` option — a `.gtdrc` `build.quality.reviewing` entry still
@@ -141,14 +153,18 @@ export const escalationExhausted = (): Promise<void> =>
141
153
  * but absent one the lens itself is what the turn loads by default.
142
154
  */
143
155
  export const reviewQuality = (lens: string): Promise<void> =>
144
- t.agentWithSkills("quality.reviewing", t.buildQualityReviewingPrompt(lens), {
145
- label: "Reviewing (one quality lens)",
146
- file: QUALITY,
147
- model: planner(),
148
- system: t.reviewerSystem(),
149
- allowEmpty: true,
150
- skills: [lens],
151
- })
156
+ t.agentWithSkills(
157
+ "quality.reviewing",
158
+ t.buildQualityReviewingPrompt(lens, builtInLenses[lens]?.brief),
159
+ {
160
+ label: "Reviewing (one quality lens)",
161
+ file: QUALITY,
162
+ model: planner(),
163
+ system: t.reviewerSystem(),
164
+ allowEmpty: true,
165
+ skills: builtInLenses[lens]?.skills ?? [lens],
166
+ },
167
+ )
152
168
 
153
169
  export const fixQuality = (): Promise<void> =>
154
170
  t.agentWithSkills("fix-quality", t.buildFixQualityPrompt(), {
@@ -192,6 +208,16 @@ export const fixNits = (notes: readonly t.NoteInput[]): Promise<void> =>
192
208
  system: t.reviewerSystem(),
193
209
  })
194
210
 
211
+ /** Fix the `Risk:`-marked notes the reviewer named. Shares the `build.review` conversation like `fixNits`; an empty turn means the risk was judged false. */
212
+ export const fixRisks = (notes: readonly t.NoteInput[]): Promise<void> =>
213
+ t.agentWithSkills("review.fix-risks", t.buildReviewFixRisksPrompt(notes), {
214
+ label: "Fixing the reviewer's risks",
215
+ file: REVIEW,
216
+ model: planner(),
217
+ system: t.reviewerSystem(),
218
+ allowEmpty: true,
219
+ })
220
+
195
221
  export const awaitReview = (base: string): Promise<void> =>
196
222
  human("review.await-review", {
197
223
  message: t.buildReviewAwaitReviewMessage(base),
@@ -1,9 +1,11 @@
1
1
  import { describe, expect, it } from "vitest"
2
2
  import {
3
+ buildReviewReviewingPrompt,
3
4
  agentWithSkills,
4
5
  architectureAuthorPrompt,
5
6
  architectureGateAnswerMessage,
6
7
  buildFixQualityPrompt,
8
+ buildQualityReviewingPrompt,
7
9
  buildReviewAwaitReviewMessage,
8
10
  buildReviewCollectingPrompt,
9
11
  designGateAnswerMessage,
@@ -116,6 +118,37 @@ describe("buildFixQualityPrompt", () => {
116
118
  })
117
119
  })
118
120
 
121
+ describe("buildFixQualityPrompt scope", () => {
122
+ it("fixes every finding, blocking or not", () => {
123
+ const prompt = renderText(() => buildFixQualityPrompt())
124
+ expect(prompt).toContain("fix every finding in every chunk, blocking or not")
125
+ expect(prompt).not.toContain("found something blocking")
126
+ })
127
+ })
128
+
129
+ describe("buildQualityReviewingPrompt", () => {
130
+ const prompt = renderText(() => buildQualityReviewingPrompt("x"))
131
+ it("traces, reads touched paths on a large change, and judges tests by reasoning", () => {
132
+ expect(prompt).toContain("Trace, do not skim")
133
+ expect(prompt).toContain("order of external calls against")
134
+ expect(prompt).toContain("read the touched code")
135
+ expect(prompt).toContain("would fail with the guarded")
136
+ expect(prompt).toContain("never by running a")
137
+ expect(prompt).toContain("pins a bug is a finding")
138
+ })
139
+ it("writes every finding and approves only when nothing was found", () => {
140
+ expect(prompt).toContain("APPEND every finding you have, blocking or not")
141
+ expect(prompt).toContain("only when this lens found nothing at all")
142
+ expect(prompt).not.toContain("Write nothing when nothing is blocking")
143
+ })
144
+ it("adds a brief block only when given one", () => {
145
+ expect(prompt).not.toContain("This lens's brief")
146
+ expect(renderText(() => buildQualityReviewingPrompt("x", "BRIEF-TEXT"))).toContain(
147
+ "This lens's brief:\nBRIEF-TEXT",
148
+ )
149
+ })
150
+ })
151
+
119
152
  describe("summaryPrompt", () => {
120
153
  it("puts the total and every model's cost on a line of its own", () => {
121
154
  const prompt = summaryPrompt({
@@ -198,3 +231,13 @@ describe("code threads in prompts", () => {
198
231
  }
199
232
  })
200
233
  })
234
+
235
+ describe("buildReviewReviewingPrompt risk marking", () => {
236
+ it("states the Risk: marker rule and the re-review rule", () => {
237
+ const prompt = renderText(() => buildReviewReviewingPrompt("base"))
238
+ expect(prompt).toContain("Open a hunk's note with `Risk:` only for a concrete defect")
239
+ expect(prompt).toContain("never a style remark")
240
+ expect(prompt).toContain("describe what the fix changed under the")
241
+ expect(prompt).toContain("re-mark only a risk the fix did not resolve")
242
+ })
243
+ })
@@ -505,28 +505,70 @@ ${agentConduct}`
505
505
 
506
506
  export const buildFixQualityPrompt = (): string =>
507
507
  `${stateFileRules}
508
- - Read \`.gtd/QUALITY.md\` — one \`## \` chunk per quality dimension
509
- that found something blocking. Merge duplicate findings across
510
- dimensions FIRST, then fix every chunk
508
+ - Read \`.gtd/QUALITY.md\` — one \`## \` chunk per finding a quality
509
+ dimension wrote. Merge duplicate findings across dimensions
510
+ FIRST, then fix every finding in every chunk, blocking or not
511
511
  - When findings conflict, missing test signal beats line count —
512
512
  a test is never deleted to satisfy a simplification finding
513
513
  - Delete \`.gtd/QUALITY.md\` once every finding is resolved
514
514
  - Leave everything else uncommitted and finish your turn
515
515
  `
516
516
 
517
- export const buildQualityReviewingPrompt = (lens: string): string =>
517
+ /** Brief for a lens the workflow defines itself; a lens with none is just a skill of that name. */
518
+ export const correctnessBrief = `- Trace partial-failure and retry paths: what a step that fails
519
+ after saving an id leaves behind, and whether the retry resumes
520
+ it or starts over
521
+ - Check validation done before an external side effect — its
522
+ format, not just its presence
523
+ - Check invariants that parallel write paths share: every path
524
+ that writes the same data must enforce what the main path
525
+ enforces
526
+ - New mock behaviour without a contract test against the live
527
+ behaviour (mock/live drift) is a finding`
528
+
529
+ export const conventionsBrief = `- Read every \`AGENTS.md\` and \`CLAUDE.md\` in the repository root
530
+ and in each touched file's directory ancestry, end to end, plus
531
+ every file they pull in by \`@path\`
532
+ - Every violation of them in the change is a finding — quote the
533
+ rule it breaks`
534
+
535
+ export const specChallengeBrief = `- Find this process's planning documents in history:
536
+ \`git log <start>..HEAD\` (\`<start>\` is the commit above) over the steering directory
537
+ (\`.gtd/\`), then \`git show\` the last version of the
538
+ requirements, architecture and package files
539
+ - Flag a spec decision that conflicts with a system invariant — an
540
+ existing test, a documented constraint, or a data invariant
541
+ other code relies on — naming the decision and the invariant
542
+ - With no planning documents in history, write nothing`
543
+
544
+ export const buildQualityReviewingPrompt = (lens: string, brief?: string): string =>
518
545
  `${stateFileRules}
519
546
  - The only state file this turn writes is \`.gtd/QUALITY.md\` — no
520
547
  other files for notes or output
521
548
  - Review the whole assembled change, from \`${start()}\`
522
549
  to the working tree, through this ONE quality lens only —
523
- \`${lens}\`, already loaded as this turn's own skill
524
- - Where you find something blocking, APPEND a \`## \` chunk to
525
- \`.gtd/QUALITY.md\` describing it — never overwrite what an
526
- earlier dimension already wrote there
527
- - Write nothing when nothing is blocking under this lens — a
550
+ \`${lens}\`
551
+ - Trace, do not skim: follow the order of external calls against
552
+ the resume/retry logic. On a large change, read the touched code
553
+ paths, not just the diff hunks
554
+ - A test counts as coverage only if it would fail with the guarded
555
+ behaviour removed — decide that by reasoning, never by running a
556
+ mutation-testing tool. A test that pins a bug is a finding, not
557
+ praise
558
+ - APPEND every finding you have, blocking or not, as a \`## \`
559
+ chunk to \`.gtd/QUALITY.md\` — never overwrite what an earlier
560
+ dimension already wrote there
561
+ - Write nothing only when this lens found nothing at all — then a
528
562
  clean turn IS this dimension's approval
529
- - Touch no other state file, and leave everything uncommitted
563
+ ${
564
+ brief
565
+ ? `
566
+ This lens's brief:
567
+ ${brief}
568
+
569
+ `
570
+ : ""
571
+ }- Touch no other state file, and leave everything uncommitted
530
572
  `
531
573
 
532
574
  export const reviewerSystem = (): string =>
@@ -642,6 +684,17 @@ The nit notes are:
642
684
 
643
685
  ${notesCapture(notes)}`
644
686
 
687
+ export const buildReviewFixRisksPrompt = (notes: readonly NoteInput[]): string =>
688
+ `${stateFileRules}
689
+ - Fix every risk below in this one turn, all together
690
+ - Where a risk is behavioural, add a test that fails without the fix
691
+ - Leave \`.gtd/REVIEW.md\` untouched — the re-review rewrites it
692
+ - Leave everything uncommitted and finish your turn
693
+
694
+ The risk notes are:
695
+
696
+ ${notesCapture(notes)}`
697
+
645
698
  export const reviewEditNotesCapture = (
646
699
  commit: string,
647
700
  edits: readonly NoteInput[],
@@ -767,6 +820,14 @@ review the changes:
767
820
  - [ ] ./path/to/file.ts#42-70 — what this hunk does
768
821
  and here is more detail, continued below it
769
822
 
823
+ Open a hunk's note with \`Risk:\` only for a concrete defect
824
+ the change introduces, never a style remark — each such note is
825
+ fixed automatically before the human sees the review. On the
826
+ re-review after a fix, describe what the fix changed under the
827
+ hunk it touched, and re-mark only a risk the fix did not resolve:
828
+
829
+ - [ ] ./path/to/file.ts#42-70 — Risk: what is wrong
830
+
770
831
  A note sitting entirely on the line(s) beneath the pointer is
771
832
  also valid. Either way, the note must never start with a bare \`./path\` token
772
833
  — that parses as a second pointer, not a note
@@ -16,7 +16,8 @@ export const defaults: Readonly<Record<string, string>> = {
16
16
  // `./review.ts`) rather than naming one step's skill list, so it is a
17
17
  // setting — pairing with `build.quality.reviewing`'s entry in
18
18
  // `./skills.ts`, which replaces the lens on every one of those turns.
19
- qualityReviews: "owasp-security, ponytail-review, test-audit",
19
+ qualityReviews:
20
+ "correctness, owasp-security, ponytail-review, test-audit, conventions, spec-challenge",
20
21
  }
21
22
 
22
23
  /** Environment settings. */