@pmelab/gtd 18.1.0 → 20.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,6 +3,7 @@ import {
3
3
  codeThreads,
4
4
  env,
5
5
  head,
6
+ quote,
6
7
  accessFor,
7
8
  skillsFor,
8
9
  start,
@@ -18,7 +19,6 @@ import {
18
19
  designPersona,
19
20
  architectPersona,
20
21
  reviewerPersona,
21
- specReviewerPersona,
22
22
  builderPersona,
23
23
  finisherPersona,
24
24
  escalationPersona,
@@ -82,22 +82,12 @@ What each change does next (then run \`gtd land\`):
82
82
  - **Continue** — having undone the sketch by hand, check the test baseline is green and start triage (**start-gate.check**).
83
83
  `
84
84
 
85
- export const architecturePreMessage = (): string =>
86
- `Judging whether \`.gtd/REQUIREMENTS.md\`'s settled concerns need a
87
- dedicated architecture pass — real structural decisions, multiple
88
- integration points, or a non-obvious tradeoff — before packages
89
- are written. Run \`gtd judge answer\` and pipe a verdict for
90
- \`architectureWarranted\` — or land untouched to run the full pass
91
- (the conservative default; a skipped judgment never suppresses
92
- it).
93
- `
94
-
95
85
  export const startGateBlockedMessage = (): string =>
96
86
  `The test baseline is red — gtd will not start new work on a broken suite.
97
87
  \`.gtd/FEEDBACK.md\` holds the failing output.
98
88
 
99
89
  What each change does next (then run \`gtd land\`):
100
- - **Retry check** — edit the code and/or \`.gtd/FEEDBACK.md\` to fix the failing tests (**start-gate.check**). To repair the baseline as its own separate reviewed commit instead, abandon this start and run \`gtd --entry fix-precheck\` from a clean \`idle\`.
90
+ - **Retry check** — edit the code and/or \`.gtd/FEEDBACK.md\` to fix the failing tests (**start-gate.check**). To repair the baseline as its own separate reviewed commit instead, abandon this start and run \`gtd --workflow fix\` from a clean \`idle\`.
101
91
  `
102
92
 
103
93
  export const reviewGateBlockedMessage = (): string =>
@@ -269,22 +259,46 @@ ${footnoteFoldIn}${codeThreadReplies()}
269
259
  authority is to merge only, never to split — the whole problem
270
260
  is over-granularity
271
261
  - Record every merge under \`## Merged Concerns\`,
272
- carrying both merged requirements verbatim so spec review
273
- still covers each independently
262
+ carrying both merged requirements verbatim so each requirement
263
+ stays traceable to its tests
274
264
  - A merge raises no open question and stops for no
275
265
  human — do not route it to \`architecture.gate\` for a veto; the
276
- human sees it when reviewing the plan, and spec review is the
277
- real safety net
266
+ human sees it when reviewing the plan
278
267
  - Prefer fewer, larger packages — the smallest independently
279
268
  valuable change, not the smallest change that compiles
280
269
  - Every open point here is TECHNICAL — triage already resolved
281
270
  the product ones, one phase earlier. PERMISSIVE: answer it
282
271
  yourself unless you genuinely cannot defend a default; a wrong
283
- technical call is still caught at spec review
272
+ technical call is still caught by the full run and the quality lap
284
273
  - The narrow exception: a simplification that drops something
285
274
  \`.gtd/REQUIREMENTS.md\` mentions is an open question, not a
286
275
  silent default
287
276
 
277
+ - \`.gtd/ARCHITECTURE.md\` is the acceptance spec the human signs
278
+ off. It carries these sections, in this order:
279
+ 1. \`## Open Questions\` (qa format, first, only when present)
280
+ 2. \`## Interfaces\` — new and changed signatures/contracts
281
+ only, as fenced TypeScript (or the repository's language)
282
+ 3. \`## Call Stacks\` — seam level, entry point → module
283
+ boundary; one line per hop, never per line of code
284
+ 4. \`## E2E Scenarios\` — Gherkin in fenced \`gherkin\` blocks
285
+ (whatever e2e framework the repository uses), each preceded
286
+ by a line \`- e2e: <path>\`. With no user-visible
287
+ behaviour change, the literal line \`No e2e change.\` plus a
288
+ one-line reason — never empty or missing
289
+ 5. \`## Unit Tests\` — one \`### <interface>\` per entry of
290
+ \`## Interfaces\`, each test a line
291
+ \`- unit: <path> — <behaviour> (covers <concern>)\`;
292
+ never against an internal helper. A requirement no test can
293
+ cover (docs, README, deletions, config) goes under a
294
+ \`### Chores\` heading in this section as
295
+ \`- chore: <path> — <what> (covers <concern>)\`
296
+ 6. \`## Merged Concerns\` — kept, when any merge happened
297
+ 7. \`## Answered Questions\` — last
298
+ - Coverage rule: every concern in \`.gtd/REQUIREMENTS.md\` appears
299
+ in at least one \`(covers …)\` tail, as a test or a chore. Each
300
+ test entry carries its file path and level (unit / e2e)
301
+
288
302
  ${questionBar}
289
303
  ## Return lap
290
304
 
@@ -300,26 +314,41 @@ export const architectureDecomposePrompt = (): string =>
300
314
 
301
315
  ${stateFileRules}
302
316
  - The only state files this turn touches are the package files
303
- under \`.gtd/packages/\` and \`.gtd/ARCHITECTURE.md\` (deleted) —
304
- no other files for notes or output
317
+ under \`.gtd/packages/\` — no other files for notes or output.
318
+ Never delete \`.gtd/ARCHITECTURE.md\`: the build tail's full run
319
+ still needs it, and package files may reference it
305
320
  - Work from \`.gtd/ARCHITECTURE.md\` if you wrote it earlier this
306
321
  conversation, otherwise read it (the converged technical
307
- plan). It already lists an ordered set of concerns with every
308
- merge/split judgement made — this turn is a mechanical
309
- write-out, not a planning one. A \`## Merged Concerns\` heading
310
- there records those merges, never a concern of its own: write
311
- no package file for it
312
- - Write one package file per concern, in the settled order,
313
- under \`.gtd/packages/\` (e.g. \`.gtd/packages/01-name.md\`,
314
- \`02-name.md\`, ...), each carrying that concern's
315
- requirement(s) — both, independently, if merged — its
316
- independent tasks, and each task's acceptance criteria as
317
- \`- [ ]\` checkboxes and relevant paths
318
- - Do not merge or split concerns here — that judgement already
319
- happened; carry the settled grouping over verbatim. No
320
- package file may reference any other \`.gtd/\` file
321
- - Once written, delete \`.gtd/ARCHITECTURE.md\`. Leave everything
322
- uncommitted and finish
322
+ plan). It already lists the concerns, interfaces, e2e scenarios
323
+ and unit tests — this turn groups them into packages. A
324
+ \`## Merged Concerns\` heading there records merges, never a
325
+ concern of its own: write no package file for it
326
+ - A package is a consecutive run of \`## Unit Tests\` entries that go
327
+ green together on the fast suite, at most ~8 unit-test entries
328
+ each. Keep the merge rule: concerns whose footprints centre on the
329
+ same files stay in one package unless the later one only consumes
330
+ an interface the earlier one creates. Prefer fewer, larger
331
+ packages. Order them so each stays green on the fast suite —
332
+ interface-introducing packages first
333
+ - Write the packages under \`.gtd/packages/\`, each carrying its
334
+ requirement(s) — both, independently, if merged — its tasks, and
335
+ each task's acceptance criteria as \`- [ ]\` checkboxes and
336
+ relevant paths:
337
+ - \`.gtd/packages/00-e2e-scenarios.md\` — package 0: it writes
338
+ \`## E2E Scenarios\` as tests in the repository's own e2e
339
+ framework and leaves them red. Write no such file when
340
+ \`## E2E Scenarios\` says "No e2e change."
341
+ - \`01-…\` onward — the unit-test packages. A requirement no test
342
+ covers (docs, README, deletions, config) becomes a chore
343
+ package, or a task of one; chore packages declare no tests
344
+ - the last numbered package is the wiring package: CLI entry,
345
+ workflow composition, config. A red e2e after it is a bug, not
346
+ missing work
347
+ - Every package with tests has a \`## Tests\` section, one line per
348
+ declared test, copied from the architecture entries:
349
+ \`- unit: \\\`<path>\\\`\` or \`- e2e: \\\`<path>\\\`\`. Chore packages have
350
+ none
351
+ - Leave everything uncommitted and finish
323
352
  `
324
353
 
325
354
  export const architectureGateAnswerMessage = (): string =>
@@ -352,7 +381,7 @@ What each change does next (then run \`gtd land\`):
352
381
  export const packagesItemBuildingPrompt = (pkg: string): string =>
353
382
  `${stateFileRules}
354
383
  - The only state file this turn may write is \`.gtd/SATISFIED.md\`;
355
- never delete the package file (the spec-review gate reads it
384
+ never delete the package file (the declared-tests guard reads it
356
385
  after you)
357
386
  - The package to implement is \`${pkg}\`
358
387
  - First check its acceptance criteria against the current tree —
@@ -361,6 +390,13 @@ export const packagesItemBuildingPrompt = (pkg: string): string =>
361
390
  with each criterion's concrete evidence (commit, file, or
362
391
  symbol), change nothing else, and finish. Otherwise implement
363
392
  normally and skip that file
393
+ - Write the tests the package declares under \`## Tests\` first,
394
+ then the code that makes them pass; the turn is refused unless
395
+ every declared test is in your diff
396
+ - Package 0 (\`.gtd/packages/00-e2e-scenarios.md\`) writes the scenarios verbatim
397
+ from \`## E2E Scenarios\` of \`.gtd/ARCHITECTURE.md\`, with only
398
+ step code that typechecks, and leaves them red — no
399
+ implementation
364
400
  - Implement every task the package describes, no more, no less,
365
401
  fanning independent ones out to parallel subagents where your
366
402
  harness supports it; leave other package files untouched
@@ -382,17 +418,6 @@ ${fixFeedbackPrompt}
382
418
  - Leave everything uncommitted and finish your turn — do not commit
383
419
  `
384
420
 
385
- export const packagesItemFixSpecPrompt = (pkg: string): string =>
386
- `${stateFileRules}
387
- - The only state file this turn touches is
388
- \`.gtd/SPEC_FEEDBACK.md\` — address it, then delete it
389
- - Read it (the reviewer's concerns) and the package spec
390
- (\`${pkg}\`), then fix the code to
391
- resolve every concern
392
- - Delete \`.gtd/SPEC_FEEDBACK.md\` once resolved; leave everything
393
- else uncommitted and finish your turn
394
- `
395
-
396
421
  export const healthJudgeMessage = (): string =>
397
422
  `The check is still red, and this isn't the first round —
398
423
  \`.gtd/FEEDBACK.md\` holds this round's output, and the previous round's
@@ -440,59 +465,34 @@ turn, or land untouched to give it one more attempt at the same
440
465
  analysis.
441
466
  `
442
467
 
443
- export const packagesItemSpecPreMessage = (): string =>
444
- `Judging whether the code already satisfies each requirement in the
445
- package spec, before spending a full review turn. Run \`gtd judge
446
- answer\` and pipe a verdict per section — or land untouched to run
447
- the full review (the conservative default; a skipped judgment
448
- never suppresses anything).
449
- `
450
-
451
- /** The sections a pre-judge could not clear, when it cleared the rest. */
452
- const specScope = (failing: readonly string[]): string =>
453
- failing.length > 0
454
- ? `- A pre-judge already found the other sections satisfied. Confine
455
- your review to only these sections:
456
- ${failing.map((title) => ` - ${title}\n`).join("")}`
457
- : ""
458
-
459
- export const packagesItemSpecReviewPrompt = (
460
- pkg: string,
461
- failing: readonly string[] = [],
462
- ): string =>
463
- `${styleBlock}
464
-
465
- You are reviewing a freshly-built work package against its own
466
- spec.
467
-
468
- ${stateFileRules}
469
- - The only state file this turn touches is
470
- \`.gtd/SPEC_FEEDBACK.md\` — write it only when you find problems
471
- - The package spec is \`${pkg}\`
472
- ${specScope(failing)}- Verify the implementation against it: tasks done, criteria
473
- met, code sound and consistent with the codebase. No diff is
474
- given — read the range yourself, from \`${start()}\`
475
- to the working tree, process-wide (it can span earlier
476
- packages)
477
- - You own that bar; nothing downstream re-weighs your findings
478
- - Write nothing when the package fully satisfies its spec —
479
- silence is your approval. Otherwise write
480
- \`.gtd/SPEC_FEEDBACK.md\` listing what would violate the spec if
481
- it shipped unaddressed, specific enough to act on, each as its
482
- own \`## \` heading
483
- - Never fix anything yourself and never delete the package
484
- file — a later step owns that
485
- `
486
-
487
- export const specReviewerSystem = (): string =>
488
- `${specReviewerPersona}
468
+ /** What the tail's full-run fix needs of a `BuiltPlan` (structural: `packages.ts` imports this module). */
469
+ export interface BuildContext {
470
+ readonly ranges: readonly { readonly pkg: string; readonly from: string; readonly to: string }[]
471
+ readonly scenarios: {
472
+ readonly added: readonly string[]
473
+ readonly changed: readonly string[]
474
+ }
475
+ }
489
476
 
490
- ${agentConduct}`
477
+ const buildContextPrompt = (built: BuildContext): string => {
478
+ const list = (paths: readonly string[]): string =>
479
+ paths.length === 0 ? " (none)" : paths.map((path) => ` - ${path}`).join("\n")
480
+ return `- Read \`.gtd/ARCHITECTURE.md\` — the plan every package was built from
481
+ - A red scenario that is NEW in this plan likely means missing wiring
482
+ between packages; new scenarios:
483
+ ${list(built.scenarios.added)}
484
+ - A red CHANGED or existing scenario or test is a regression a package
485
+ introduced; scenarios this plan changed:
486
+ ${list(built.scenarios.changed)}
487
+ - Each package's commits, to bisect a break:
488
+ ${built.ranges.map((r) => ` git log --oneline ${r.from}..${r.to} # ${r.pkg}`).join("\n")}
489
+ `
490
+ }
491
491
 
492
- export const buildFixPrompt = (): string =>
492
+ export const buildFixPrompt = (built?: BuildContext): string =>
493
493
  `${stateFileRules}
494
494
  ${fixFeedbackPrompt}
495
- - Leave everything uncommitted — do not commit
495
+ ${built === undefined ? "" : buildContextPrompt(built)}- Leave everything uncommitted — do not commit
496
496
  `
497
497
 
498
498
  export const finisherSystem = (): string =>
@@ -883,3 +883,25 @@ ${it.processCostByModel.map((m) => `- ${m.model}: ${m.cost}\n`).join("")}
883
883
  Print the closing message and stop — this writes nothing itself.
884
884
  `
885
885
  }
886
+
887
+ export interface WordingDrift {
888
+ readonly path: string
889
+ /** `- `/`+ `-prefixed, in file order. */
890
+ readonly lines: readonly string[]
891
+ }
892
+
893
+ export const scenarioWordingMessage = (at: string, drifted: readonly WordingDrift[]): string =>
894
+ `The e2e scenario wording was frozen when package 0 landed, and the last
895
+ turn changed it. Removed lines are marked \`-\`, added ones \`+\`:
896
+
897
+ ${drifted.map(({ path, lines }) => `${path}\n${lines.join("\n")}`).join("\n\n")}
898
+
899
+ To accept the change, land untouched.
900
+
901
+ To reject it, run
902
+
903
+ git checkout ${at} -- ${drifted.map(({ path }) => quote(path)).join(" ")}
904
+
905
+ and land. Restoring only some of the files is a partial accept: every file
906
+ still differing from the frozen wording is accepted.
907
+ `
@@ -9,9 +9,7 @@ export const defaults: Readonly<Record<string, string>> = {
9
9
  reviewBase: "",
10
10
  judgeBudgetBytes: "32768",
11
11
  judgeIdenticalMinP: "0.7",
12
- specPreJudge: "0.9",
13
12
  reviewNoteActionable: "0.7",
14
- architectureSkipMinP: "0.85",
15
13
  // Decides the turn count: one `build.quality.<lens>` scope per entry, each
16
14
  // keyed in `./skills.ts`.
17
15
  qualityReviews:
@@ -21,6 +19,7 @@ export const defaults: Readonly<Record<string, string>> = {
21
19
  /** Environment settings. */
22
20
  export const envDefaults: Readonly<Record<string, string>> = {
23
21
  testCommand: "npm test",
22
+ fastTestCommand: "",
24
23
  plannerModel: "smart",
25
24
  coderModel: "base",
26
25
  }
@@ -1,34 +0,0 @@
1
- // The workflow's side doors: `fix` repairs a red baseline as its own reviewed
2
- // commit, `review` reviews everything since a base, straight to the review
3
- // tail. `gtd --entry` prints the script that starts the process; running it
4
- // is the driver's job, as with every git write gtd plans.
5
-
6
- import type { ShipIo, Shipped } from "./ship"
7
-
8
- const fail = (text: string): Shipped => ({ ok: false, text })
9
-
10
- export async function enter(io: ShipIo, door: "fix" | "review", base?: string): Promise<Shipped> {
11
- const args = door === "fix" ? ["--entry", "fix-precheck"] : await reviewArgs(io, base)
12
- if (typeof args === "string") return fail(args)
13
- const planned = await io.run(["gtd", ...args])
14
- if (planned.code !== 0)
15
- return fail(planned.err.trim() || `gtd ${args.join(" ")} exited ${planned.code}`)
16
- const started = await io.run(["sh", "-c", planned.out])
17
- if (started.code !== 0) return fail(`The entry script failed.\n${started.err.trim()}`)
18
- return {
19
- ok: true,
20
- text: door === "fix" ? "Started a fix." : `Started a review since ${args.at(-1)?.slice(-7)}.`,
21
- }
22
- }
23
-
24
- // Reviews the branch since it left `base`, the default branch unless named.
25
- async function reviewArgs(io: ShipIo, base: string | undefined) {
26
- const git = async (...a: string[]) => (await io.run(["git", ...a])).out.trim()
27
- const from =
28
- base || (await git("symbolic-ref", "--quiet", "--short", "refs/remotes/origin/HEAD")) || "main"
29
- const mergeBase = await git("merge-base", from, "HEAD")
30
- if (!mergeBase) return `There is no common ancestor of ${from} and HEAD to review from.`
31
- if (mergeBase === (await git("rev-parse", "HEAD")))
32
- return `Nothing to review: HEAD has no commits beyond ${from}.`
33
- return ["--entry", "review-gate.check", "--var", `reviewBase=${mergeBase}`]
34
- }
@@ -1,115 +0,0 @@
1
- import { describe, expect, it } from "vitest"
2
- import type { Change } from "../flows/index.js"
3
- import { filterChanges, packageDiff } from "./diff.js"
4
-
5
- const change = (over: Partial<Change> & Pick<Change, "path" | "status">): Change => ({
6
- before: undefined,
7
- after: undefined,
8
- ...over,
9
- })
10
-
11
- // A cap high enough that nothing is ever dropped, for cases exercising the
12
- // rendering pipeline rather than the byte-budget behavior itself.
13
- const NO_CAP = 10_000
14
-
15
- describe("packageDiff / rendering", () => {
16
- it("renders a normal edit as hunks with three lines of context", () => {
17
- const before = Array.from({ length: 10 }, (_, i) => `line ${i + 1}`).join("\n")
18
- const after = before.replace("line 5", "line five")
19
- const text = packageDiff([change({ path: "a.ts", status: "modified", before, after })], NO_CAP)
20
- expect(text).toContain("--- a/a.ts")
21
- expect(text).toContain("+++ b/a.ts")
22
- expect(text).toContain("-line 5")
23
- expect(text).toContain("+line five")
24
- // three lines of context on either side of the one changed line
25
- expect(text).toContain(" line 2")
26
- expect(text).toContain(" line 8")
27
- expect(text).not.toContain(" line 1\n")
28
- })
29
-
30
- it("degrades a 1500+ line change to its summary line", () => {
31
- const before = Array.from({ length: 2000 }, (_, i) => `line ${i}`).join("\n")
32
- const after = Array.from({ length: 2000 }, (_, i) => `changed ${i}`).join("\n")
33
- const text = packageDiff(
34
- [change({ path: "big.ts", status: "modified", before, after })],
35
- NO_CAP,
36
- )
37
- expect(text).toBe("big.ts: +2000/-2000 lines, too large to inline")
38
- })
39
-
40
- it("renders files in path order for byte-identical output across replays", () => {
41
- const a = change({ path: "z.ts", status: "added", after: "z" })
42
- const b = change({ path: "a.ts", status: "added", after: "a" })
43
- expect(packageDiff([a, b], NO_CAP)).toBe(packageDiff([b, a], NO_CAP))
44
- expect(packageDiff([a, b], NO_CAP).indexOf("a.ts")).toBeLessThan(
45
- packageDiff([a, b], NO_CAP).indexOf("z.ts"),
46
- )
47
- })
48
- })
49
-
50
- describe("filterChanges / exclusion list", () => {
51
- it("excludes a lockfile change entirely", () => {
52
- const c = change({ path: "package-lock.json", status: "modified", before: "a", after: "b" })
53
- expect(filterChanges([c])).toEqual([])
54
- expect(packageDiff([c], NO_CAP)).toBe("")
55
- })
56
-
57
- it("excludes .gtd/** paths", () => {
58
- const c = change({ path: ".gtd/PLAN.md", status: "added", after: "x" })
59
- expect(filterChanges([c])).toEqual([])
60
- })
61
-
62
- it("excludes a binary file by extension", () => {
63
- const c = change({ path: "logo.png", status: "added", after: "binary-ish" })
64
- expect(filterChanges([c])).toEqual([])
65
- })
66
-
67
- it("excludes a binary file by content, separately from extension", () => {
68
- const c = change({ path: "weird.txt", status: "added", after: "abc\0def" })
69
- expect(filterChanges([c])).toEqual([])
70
- })
71
-
72
- it("keeps an ordinary source file", () => {
73
- const c = change({ path: "src/a.ts", status: "added", after: "export const a = 1\n" })
74
- expect(filterChanges([c])).toEqual([c])
75
- })
76
- })
77
-
78
- describe("packageDiff", () => {
79
- it("drops whole files from the end and names them in a trailer", () => {
80
- const a = change({ path: "a.ts", status: "added", after: "export const a = 1\n" })
81
- const b = change({
82
- path: "b.ts",
83
- status: "added",
84
- after: Array.from({ length: 50 }, (_, i) => `line ${i}`).join("\n"),
85
- })
86
- const text = packageDiff([a, b], 120)
87
- expect(text).toContain("--- /dev/null")
88
- expect(text).toContain("+++ b/a.ts")
89
- expect(text).not.toContain("b.ts\n")
90
- expect(text).toContain("1 file(s) omitted for the judge's byte budget: b.ts")
91
- })
92
-
93
- it("counts the omission trailer itself against the cap, so many dropped paths never push the total past it", () => {
94
- const kept = change({ path: "a.ts", status: "added", after: "export const a = 1\n" })
95
- const rest = Array.from({ length: 6 }, (_, i) =>
96
- change({ path: `f${i}.ts`, status: "added", after: "y" }),
97
- )
98
- // A naive cap check (size the kept text alone, append the trailer
99
- // afterwards) would keep a.ts plus two of the small files here — 150
100
- // bytes of text — then tack on an unbudgeted trailer for the other four,
101
- // landing well past capBytes. Budgeting the trailer itself must instead
102
- // drop enough files that the total, trailer included, stays at or under it.
103
- const capBytes = 150
104
- const text = packageDiff([kept, ...rest], capBytes)
105
- expect(Buffer.byteLength(text, "utf8")).toBeLessThanOrEqual(capBytes)
106
- expect(text).toContain("omitted")
107
- })
108
-
109
- it("keeps every file when the cap is not exceeded", () => {
110
- const a = change({ path: "a.ts", status: "added", after: "x" })
111
- const text = packageDiff([a], 10_000)
112
- expect(text).not.toContain("omitted")
113
- expect(text).toContain("a.ts")
114
- })
115
- })