@pmelab/gtd 16.0.0 → 17.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -91,6 +91,46 @@ describe("the bundled workflow's steps declare skills — a bundled step's rende
91
91
  expect(prompt).toContain("owasp-security")
92
92
  })
93
93
 
94
+ it("correctness loads code-review-and-quality and carries all four trace points", async () => {
95
+ const request = await capture(() => steps.reviewQuality("correctness"))
96
+ expect(agentSkillsOption(request)).toEqual(["code-review-and-quality"])
97
+ const body = JSON.stringify(request)
98
+ for (const point of [
99
+ "partial-failure",
100
+ "format, not just its presence",
101
+ "parallel write paths",
102
+ "contract test",
103
+ ])
104
+ expect(body).toContain(point)
105
+ })
106
+
107
+ it.each(["conventions", "spec-challenge"])(
108
+ "%s sends no skills and carries its brief",
109
+ async (lens) => {
110
+ const request = await capture(() => steps.reviewQuality(lens))
111
+ expect(agentSkillsOption(request)).toEqual([])
112
+ expect(JSON.stringify(request)).toContain(
113
+ steps.builtInLenses[lens]!.brief.split("\n")[0]!.slice(0, 40),
114
+ )
115
+ },
116
+ )
117
+
118
+ it("an unknown lens loads itself as a skill, with no brief", async () => {
119
+ const request = await capture(() => steps.reviewQuality("my-lens"))
120
+ expect(agentSkillsOption(request)).toEqual(["my-lens"])
121
+ expect(JSON.stringify(request)).not.toContain("This lens's brief")
122
+ })
123
+
124
+ it("a configured build.quality.reviewing entry replaces a built-in lens's skills; the brief stays", async () => {
125
+ const prompt = agentPrompt(
126
+ await capture(() => steps.reviewQuality("conventions"), {
127
+ "quality.reviewing": ["my-org-checklist"],
128
+ }),
129
+ )
130
+ expect(prompt).toContain("my-org-checklist")
131
+ expect(prompt).toContain("AGENTS.md")
132
+ })
133
+
94
134
  it("fixQuality carries build.fix-quality's bundled skills", async () => {
95
135
  const prompt = agentPrompt(await capture(() => steps.fixQuality()))
96
136
  expect(prompt).toContain("incremental-implementation, code-simplification")
@@ -143,6 +183,19 @@ describe("the bundled workflow's steps declare skills — a bundled step's rende
143
183
  expect(request.prompt).toContain("Leave `.gtd/REVIEW.md` untouched")
144
184
  })
145
185
 
186
+ it("fixRisks fixes every risk, tolerates an empty turn, and leaves .gtd/REVIEW.md alone", async () => {
187
+ const request = await capture(() =>
188
+ steps.fixRisks([{ id: "risk-1", anchor: "calc ./a.ts#1-1", text: "Risk: drops it" }]),
189
+ )
190
+ if (request.kind !== "agent") throw new Error("unreachable")
191
+ expect(request.name).toBe("review.fix-risks")
192
+ expect(request.options).toMatchObject({ allowEmpty: true })
193
+ expect(request.prompt).toContain("debugging-and-error-recovery, incremental-implementation")
194
+ expect(request.prompt).toContain("risk-1 — calc ./a.ts#1-1")
195
+ expect(request.prompt).toContain("Risk: drops it")
196
+ expect(request.prompt).toContain("Leave `.gtd/REVIEW.md` untouched")
197
+ })
198
+
146
199
  it("reviewing names the carry-over commit only when given one", async () => {
147
200
  const bare = await capture(() => steps.reviewing("base"))
148
201
  const carried = await capture(() => steps.reviewing("base", "abc1234"))
@@ -1,4 +1,4 @@
1
- import { human, vars } from "../flows/index.js"
1
+ import { env, human } from "../flows/index.js"
2
2
  import * as t from "./text.js"
3
3
 
4
4
  // The bundled workflow's single steps. A step's name is relative to the
@@ -14,8 +14,8 @@ export const REVIEW = ".gtd/REVIEW.md"
14
14
  export const QUALITY = ".gtd/QUALITY.md"
15
15
  export const SPEC_FEEDBACK = ".gtd/SPEC_FEEDBACK.md"
16
16
 
17
- const planner = (): string => vars.plannerModel ?? ""
18
- const coder = (): string => vars.coderModel ?? ""
17
+ const planner = (): string => env.plannerModel ?? ""
18
+ const coder = (): string => env.coderModel ?? ""
19
19
 
20
20
  // ── Planning ────────────────────────────────────────────────────────────────
21
21
 
@@ -134,6 +134,18 @@ export const escalationExhausted = (): Promise<void> =>
134
134
 
135
135
  // ── Quality and review ──────────────────────────────────────────────────────
136
136
 
137
+ export interface BuiltInLens {
138
+ readonly skills: readonly string[]
139
+ readonly brief: string
140
+ }
141
+
142
+ /** Lenses the workflow defines itself, not bundled skills: `skills/` is not in the npm package, and a missing lens skill burns a turn silently. */
143
+ export const builtInLenses: Readonly<Record<string, BuiltInLens>> = {
144
+ correctness: { skills: ["code-review-and-quality"], brief: t.correctnessBrief },
145
+ conventions: { skills: [], brief: t.conventionsBrief },
146
+ "spec-challenge": { skills: [], brief: t.specChallengeBrief },
147
+ }
148
+
137
149
  /**
138
150
  * One quality review, through the skill `lens`. `lens` rides as this turn's
139
151
  * own `skills` option — a `.gtdrc` `build.quality.reviewing` entry still
@@ -141,14 +153,18 @@ export const escalationExhausted = (): Promise<void> =>
141
153
  * but absent one the lens itself is what the turn loads by default.
142
154
  */
143
155
  export const reviewQuality = (lens: string): Promise<void> =>
144
- t.agentWithSkills("quality.reviewing", t.buildQualityReviewingPrompt(lens), {
145
- label: "Reviewing (one quality lens)",
146
- file: QUALITY,
147
- model: planner(),
148
- system: t.reviewerSystem(),
149
- allowEmpty: true,
150
- skills: [lens],
151
- })
156
+ t.agentWithSkills(
157
+ "quality.reviewing",
158
+ t.buildQualityReviewingPrompt(lens, builtInLenses[lens]?.brief),
159
+ {
160
+ label: "Reviewing (one quality lens)",
161
+ file: QUALITY,
162
+ model: planner(),
163
+ system: t.reviewerSystem(),
164
+ allowEmpty: true,
165
+ skills: builtInLenses[lens]?.skills ?? [lens],
166
+ },
167
+ )
152
168
 
153
169
  export const fixQuality = (): Promise<void> =>
154
170
  t.agentWithSkills("fix-quality", t.buildFixQualityPrompt(), {
@@ -192,6 +208,16 @@ export const fixNits = (notes: readonly t.NoteInput[]): Promise<void> =>
192
208
  system: t.reviewerSystem(),
193
209
  })
194
210
 
211
+ /** Fix the `Risk:`-marked notes the reviewer named. Shares the `build.review` conversation like `fixNits`; an empty turn means the risk was judged false. */
212
+ export const fixRisks = (notes: readonly t.NoteInput[]): Promise<void> =>
213
+ t.agentWithSkills("review.fix-risks", t.buildReviewFixRisksPrompt(notes), {
214
+ label: "Fixing the reviewer's risks",
215
+ file: REVIEW,
216
+ model: planner(),
217
+ system: t.reviewerSystem(),
218
+ allowEmpty: true,
219
+ })
220
+
195
221
  export const awaitReview = (base: string): Promise<void> =>
196
222
  human("review.await-review", {
197
223
  message: t.buildReviewAwaitReviewMessage(base),
@@ -5,10 +5,11 @@ import {
5
5
  type StepRequest,
6
6
  } from "../flows/index.js"
7
7
  import { skills as bundledSkills } from "./skills.js"
8
- import { defaults } from "./vars.js"
8
+ import { defaults, envDefaults } from "./vars.js"
9
9
 
10
10
  export interface TextContext {
11
11
  readonly vars?: Readonly<Record<string, string>>
12
+ readonly env?: Readonly<Record<string, string>>
12
13
  readonly head?: string
13
14
  readonly start?: string
14
15
  readonly codeThreads?: readonly CodeThreadInfo[]
@@ -73,6 +74,7 @@ export const fixtureContext = (
73
74
  threads: () => [],
74
75
  codeThreads: () => context.codeThreads ?? [],
75
76
  vars: { ...defaults, ...context.vars },
77
+ env: { ...envDefaults, ...context.env },
76
78
  head: () => context.head ?? "",
77
79
  start: () => context.start ?? "",
78
80
  skillsFor: skillsForOf(context),
@@ -1,9 +1,11 @@
1
1
  import { describe, expect, it } from "vitest"
2
2
  import {
3
+ buildReviewReviewingPrompt,
3
4
  agentWithSkills,
4
5
  architectureAuthorPrompt,
5
6
  architectureGateAnswerMessage,
6
7
  buildFixQualityPrompt,
8
+ buildQualityReviewingPrompt,
7
9
  buildReviewAwaitReviewMessage,
8
10
  buildReviewCollectingPrompt,
9
11
  designGateAnswerMessage,
@@ -116,6 +118,37 @@ describe("buildFixQualityPrompt", () => {
116
118
  })
117
119
  })
118
120
 
121
+ describe("buildFixQualityPrompt scope", () => {
122
+ it("fixes every finding, blocking or not", () => {
123
+ const prompt = renderText(() => buildFixQualityPrompt())
124
+ expect(prompt).toContain("fix every finding in every chunk, blocking or not")
125
+ expect(prompt).not.toContain("found something blocking")
126
+ })
127
+ })
128
+
129
+ describe("buildQualityReviewingPrompt", () => {
130
+ const prompt = renderText(() => buildQualityReviewingPrompt("x"))
131
+ it("traces, reads touched paths on a large change, and judges tests by reasoning", () => {
132
+ expect(prompt).toContain("Trace, do not skim")
133
+ expect(prompt).toContain("order of external calls against")
134
+ expect(prompt).toContain("read the touched code")
135
+ expect(prompt).toContain("would fail with the guarded")
136
+ expect(prompt).toContain("never by running a")
137
+ expect(prompt).toContain("pins a bug is a finding")
138
+ })
139
+ it("writes every finding and approves only when nothing was found", () => {
140
+ expect(prompt).toContain("APPEND every finding you have, blocking or not")
141
+ expect(prompt).toContain("only when this lens found nothing at all")
142
+ expect(prompt).not.toContain("Write nothing when nothing is blocking")
143
+ })
144
+ it("adds a brief block only when given one", () => {
145
+ expect(prompt).not.toContain("This lens's brief")
146
+ expect(renderText(() => buildQualityReviewingPrompt("x", "BRIEF-TEXT"))).toContain(
147
+ "This lens's brief:\nBRIEF-TEXT",
148
+ )
149
+ })
150
+ })
151
+
119
152
  describe("summaryPrompt", () => {
120
153
  it("puts the total and every model's cost on a line of its own", () => {
121
154
  const prompt = summaryPrompt({
@@ -129,6 +162,7 @@ describe("summaryPrompt", () => {
129
162
  { model: "base", cost: 7 },
130
163
  ],
131
164
  vars: {},
165
+ env: {},
132
166
  })
133
167
  expect(prompt).toContain("Token cost: 12\n- smart: 5\n- base: 7\n\nPrint the closing message")
134
168
  })
@@ -197,3 +231,13 @@ describe("code threads in prompts", () => {
197
231
  }
198
232
  })
199
233
  })
234
+
235
+ describe("buildReviewReviewingPrompt risk marking", () => {
236
+ it("states the Risk: marker rule and the re-review rule", () => {
237
+ const prompt = renderText(() => buildReviewReviewingPrompt("base"))
238
+ expect(prompt).toContain("Open a hunk's note with `Risk:` only for a concrete defect")
239
+ expect(prompt).toContain("never a style remark")
240
+ expect(prompt).toContain("describe what the fix changed under the")
241
+ expect(prompt).toContain("re-mark only a risk the fix did not resolve")
242
+ })
243
+ })
@@ -1,10 +1,10 @@
1
1
  import {
2
2
  agent,
3
3
  codeThreads,
4
+ env,
4
5
  head,
5
6
  skillsFor,
6
7
  start,
7
- vars,
8
8
  type AgentOptions,
9
9
  type SummaryContext,
10
10
  } from "../flows/index.js"
@@ -195,7 +195,7 @@ ${questionBarReturn}
195
195
  never re-open one. A genuinely open PRODUCT point may still
196
196
  raise a fresh \`## Open Questions\` entry: the product gate sits
197
197
  on this path exactly as on the first lap
198
- - Before grouping anything, run \`${vars.testCommand}\`
198
+ - Before grouping anything, run \`${env.testCommand}\`
199
199
  yourself — a loop-back runs no green-baseline gate, so the
200
200
  tree may already be red (reverting the edit can undo a fix it
201
201
  made). If red, make that breakage the first concern, ahead of
@@ -505,28 +505,70 @@ ${agentConduct}`
505
505
 
506
506
  export const buildFixQualityPrompt = (): string =>
507
507
  `${stateFileRules}
508
- - Read \`.gtd/QUALITY.md\` — one \`## \` chunk per quality dimension
509
- that found something blocking. Merge duplicate findings across
510
- dimensions FIRST, then fix every chunk
508
+ - Read \`.gtd/QUALITY.md\` — one \`## \` chunk per finding a quality
509
+ dimension wrote. Merge duplicate findings across dimensions
510
+ FIRST, then fix every finding in every chunk, blocking or not
511
511
  - When findings conflict, missing test signal beats line count —
512
512
  a test is never deleted to satisfy a simplification finding
513
513
  - Delete \`.gtd/QUALITY.md\` once every finding is resolved
514
514
  - Leave everything else uncommitted and finish your turn
515
515
  `
516
516
 
517
- export const buildQualityReviewingPrompt = (lens: string): string =>
517
+ /** Brief for a lens the workflow defines itself; a lens with none is just a skill of that name. */
518
+ export const correctnessBrief = `- Trace partial-failure and retry paths: what a step that fails
519
+ after saving an id leaves behind, and whether the retry resumes
520
+ it or starts over
521
+ - Check validation done before an external side effect — its
522
+ format, not just its presence
523
+ - Check invariants that parallel write paths share: every path
524
+ that writes the same data must enforce what the main path
525
+ enforces
526
+ - New mock behaviour without a contract test against the live
527
+ behaviour (mock/live drift) is a finding`
528
+
529
+ export const conventionsBrief = `- Read every \`AGENTS.md\` and \`CLAUDE.md\` in the repository root
530
+ and in each touched file's directory ancestry, end to end, plus
531
+ every file they pull in by \`@path\`
532
+ - Every violation of them in the change is a finding — quote the
533
+ rule it breaks`
534
+
535
+ export const specChallengeBrief = `- Find this process's planning documents in history:
536
+ \`git log <start>..HEAD\` (\`<start>\` is the commit above) over the steering directory
537
+ (\`.gtd/\`), then \`git show\` the last version of the
538
+ requirements, architecture and package files
539
+ - Flag a spec decision that conflicts with a system invariant — an
540
+ existing test, a documented constraint, or a data invariant
541
+ other code relies on — naming the decision and the invariant
542
+ - With no planning documents in history, write nothing`
543
+
544
+ export const buildQualityReviewingPrompt = (lens: string, brief?: string): string =>
518
545
  `${stateFileRules}
519
546
  - The only state file this turn writes is \`.gtd/QUALITY.md\` — no
520
547
  other files for notes or output
521
548
  - Review the whole assembled change, from \`${start()}\`
522
549
  to the working tree, through this ONE quality lens only —
523
- \`${lens}\`, already loaded as this turn's own skill
524
- - Where you find something blocking, APPEND a \`## \` chunk to
525
- \`.gtd/QUALITY.md\` describing it — never overwrite what an
526
- earlier dimension already wrote there
527
- - Write nothing when nothing is blocking under this lens — a
550
+ \`${lens}\`
551
+ - Trace, do not skim: follow the order of external calls against
552
+ the resume/retry logic. On a large change, read the touched code
553
+ paths, not just the diff hunks
554
+ - A test counts as coverage only if it would fail with the guarded
555
+ behaviour removed — decide that by reasoning, never by running a
556
+ mutation-testing tool. A test that pins a bug is a finding, not
557
+ praise
558
+ - APPEND every finding you have, blocking or not, as a \`## \`
559
+ chunk to \`.gtd/QUALITY.md\` — never overwrite what an earlier
560
+ dimension already wrote there
561
+ - Write nothing only when this lens found nothing at all — then a
528
562
  clean turn IS this dimension's approval
529
- - Touch no other state file, and leave everything uncommitted
563
+ ${
564
+ brief
565
+ ? `
566
+ This lens's brief:
567
+ ${brief}
568
+
569
+ `
570
+ : ""
571
+ }- Touch no other state file, and leave everything uncommitted
530
572
  `
531
573
 
532
574
  export const reviewerSystem = (): string =>
@@ -642,6 +684,17 @@ The nit notes are:
642
684
 
643
685
  ${notesCapture(notes)}`
644
686
 
687
+ export const buildReviewFixRisksPrompt = (notes: readonly NoteInput[]): string =>
688
+ `${stateFileRules}
689
+ - Fix every risk below in this one turn, all together
690
+ - Where a risk is behavioural, add a test that fails without the fix
691
+ - Leave \`.gtd/REVIEW.md\` untouched — the re-review rewrites it
692
+ - Leave everything uncommitted and finish your turn
693
+
694
+ The risk notes are:
695
+
696
+ ${notesCapture(notes)}`
697
+
645
698
  export const reviewEditNotesCapture = (
646
699
  commit: string,
647
700
  edits: readonly NoteInput[],
@@ -767,6 +820,14 @@ review the changes:
767
820
  - [ ] ./path/to/file.ts#42-70 — what this hunk does
768
821
  and here is more detail, continued below it
769
822
 
823
+ Open a hunk's note with \`Risk:\` only for a concrete defect
824
+ the change introduces, never a style remark — each such note is
825
+ fixed automatically before the human sees the review. On the
826
+ re-review after a fix, describe what the fix changed under the
827
+ hunk it touched, and re-mark only a risk the fix did not resolve:
828
+
829
+ - [ ] ./path/to/file.ts#42-70 — Risk: what is wrong
830
+
770
831
  A note sitting entirely on the line(s) beneath the pointer is
771
832
  also valid. Either way, the note must never start with a bare \`./path\` token
772
833
  — that parses as a second pointer, not a note
@@ -31,7 +31,7 @@ import * as t from "./text.js"
31
31
  // re-exported below.
32
32
 
33
33
  export { threads, type ThreadInfo } from "../flows/index.js"
34
- export { defaults } from "./vars.js"
34
+ export { defaults, envDefaults } from "./vars.js"
35
35
  export { skills } from "./skills.js"
36
36
  export { agentWithSkills } from "./text.js"
37
37
  export * from "./steps.js"
@@ -1,21 +1,28 @@
1
- // The bundled workflow's variable defaults. `.gtdrc` `vars:`, `--var` and
2
- // `GTD_<NAME>` override any of them.
1
+ // The bundled workflow's settings, in two kinds. A process setting changes
2
+ // which step comes next, so it is pinned for the whole process at its start;
3
+ // an environment setting only changes how a step runs on this machine, so it
4
+ // is read live on every invocation. `.gtdrc` `vars:`/`env:`, `--var` (process
5
+ // settings only) and `GTD_<NAME>` override either.
6
+
7
+ /** Process settings. */
3
8
  export const defaults: Readonly<Record<string, string>> = {
4
- testCommand: "npm test",
5
- plannerModel: "smart",
6
- coderModel: "base",
7
9
  reviewBase: "",
8
10
  judgeBudgetBytes: "32768",
9
11
  judgeIdenticalMinP: "0.7",
10
12
  specPreJudge: "0.9",
11
13
  reviewNoteActionable: "0.7",
12
14
  architectureSkipMinP: "0.85",
13
- // The per-step skill lists formerly declared here as `*Skills` vars now
14
- // live in `./skills.ts`, addressed per step by `.gtdrc` `skills:`.
15
- // `qualityReviews` is the one exception: it fans out into one turn per
16
- // entry (`qualityLenses`, in `./review.ts`) rather than naming one step's
17
- // skill list, so it stays a var — pairing with `build.quality.reviewing`'s
18
- // entry in `./skills.ts`, which replaces the lens on every one of those
19
- // turns.
20
- qualityReviews: "owasp-security, ponytail-review, test-audit",
15
+ // `qualityReviews` fans out into one turn per entry (`qualityLenses`, in
16
+ // `./review.ts`) rather than naming one step's skill list, so it is a
17
+ // setting — pairing with `build.quality.reviewing`'s entry in
18
+ // `./skills.ts`, which replaces the lens on every one of those turns.
19
+ qualityReviews:
20
+ "correctness, owasp-security, ponytail-review, test-audit, conventions, spec-challenge",
21
+ }
22
+
23
+ /** Environment settings. */
24
+ export const envDefaults: Readonly<Record<string, string>> = {
25
+ testCommand: "npm test",
26
+ plannerModel: "smart",
27
+ coderModel: "base",
21
28
  }