@pmelab/gtd 16.0.0 → 17.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +9 -0
- package/README.md +65 -25
- package/bin/gtd +3 -0
- package/claude/hooks/drive.ts +169 -0
- package/claude/hooks/entry.ts +34 -0
- package/claude/hooks/handoff.ts +174 -0
- package/claude/hooks/judge.ts +99 -0
- package/claude/hooks/models.ts +19 -0
- package/claude/hooks/register.tsx +828 -0
- package/claude/hooks/ship.ts +450 -0
- package/claude/hooks/wording.ts +85 -0
- package/claude/types/index.d.ts +33 -0
- package/dist/gtd.bundle.mjs +448 -127
- package/hooks/hooks.json +1 -0
- package/package.json +9 -2
- package/schema.json +33 -1
- package/skills/authoring/SKILL.md +248 -0
- package/src/flows/runtime.ts +27 -13
- package/src/workflows/health.ts +3 -2
- package/src/workflows/prose.ts +2 -1
- package/src/workflows/review.test.ts +87 -1
- package/src/workflows/review.ts +19 -4
- package/src/workflows/skills.test.ts +2 -1
- package/src/workflows/skills.ts +1 -0
- package/src/workflows/steps.test.ts +53 -0
- package/src/workflows/steps.ts +37 -11
- package/src/workflows/text.fixture.ts +3 -1
- package/src/workflows/text.test.ts +44 -0
- package/src/workflows/text.ts +73 -12
- package/src/workflows/unified.ts +1 -1
- package/src/workflows/vars.ts +20 -13
package/hooks/hooks.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{ "modules": ["../claude/hooks/register.tsx"] }
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@pmelab/gtd",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "17.2.0",
|
|
4
4
|
"private": false,
|
|
5
5
|
"description": "Git-aware CLI that emits the next prompt for an autonomous coding agent based on the current repository state",
|
|
6
6
|
"bin": {
|
|
@@ -16,7 +16,14 @@
|
|
|
16
16
|
"src/workflows/",
|
|
17
17
|
"README.md",
|
|
18
18
|
"LICENSE",
|
|
19
|
-
"schema.json"
|
|
19
|
+
"schema.json",
|
|
20
|
+
".claude-plugin/plugin.json",
|
|
21
|
+
"hooks/",
|
|
22
|
+
"bin/",
|
|
23
|
+
"claude/hooks/",
|
|
24
|
+
"claude/types/",
|
|
25
|
+
"!claude/**/*.test.ts",
|
|
26
|
+
"skills/"
|
|
20
27
|
],
|
|
21
28
|
"publishConfig": {
|
|
22
29
|
"access": "public",
|
package/schema.json
CHANGED
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
"properties": {
|
|
6
6
|
"vars": {
|
|
7
7
|
"type": "object",
|
|
8
|
-
"description": "Flat name -> scalar map merged into the workflow's
|
|
8
|
+
"description": "Flat name -> scalar map merged into the workflow's process settings (defaults), pinned for the whole process at its start. Scalars are coerced to strings.",
|
|
9
9
|
"additionalProperties": {
|
|
10
10
|
"type": [
|
|
11
11
|
"string",
|
|
@@ -14,6 +14,38 @@
|
|
|
14
14
|
]
|
|
15
15
|
}
|
|
16
16
|
},
|
|
17
|
+
"env": {
|
|
18
|
+
"type": "object",
|
|
19
|
+
"description": "Flat name -> scalar map merged into the workflow's environment settings (envDefaults) — values that change how a step runs on this machine, like the test command or a model hint. Read fresh on every gtd call, never pinned to a process. GTD_<NAME> environment variables override these entries. Scalars are coerced to strings.",
|
|
20
|
+
"additionalProperties": {
|
|
21
|
+
"type": [
|
|
22
|
+
"string",
|
|
23
|
+
"number",
|
|
24
|
+
"boolean"
|
|
25
|
+
]
|
|
26
|
+
}
|
|
27
|
+
},
|
|
28
|
+
"judge": {
|
|
29
|
+
"type": "object",
|
|
30
|
+
"required": [],
|
|
31
|
+
"properties": {
|
|
32
|
+
"provider": {
|
|
33
|
+
"type": "string",
|
|
34
|
+
"enum": [
|
|
35
|
+
"fixed",
|
|
36
|
+
"jev",
|
|
37
|
+
"llm"
|
|
38
|
+
],
|
|
39
|
+
"description": "Judge provider `gtd judge run` uses when no --provider flag and no GTD_JUDGE_PROVIDER is set. Absent means auto: jev when TYPESAFE_API_KEY is set, otherwise llm."
|
|
40
|
+
},
|
|
41
|
+
"model": {
|
|
42
|
+
"type": "string",
|
|
43
|
+
"description": "Model for provider llm, used when no --model flag and no GTD_JUDGE_MODEL is set. Pairing it with another provider is an error."
|
|
44
|
+
}
|
|
45
|
+
},
|
|
46
|
+
"additionalProperties": false,
|
|
47
|
+
"description": "Which judge `gtd judge run` uses, when no flag or GTD_JUDGE_* variable says."
|
|
48
|
+
},
|
|
17
49
|
"modes": {
|
|
18
50
|
"type": "object",
|
|
19
51
|
"description": "Steering-file modes a workflow step's mode: may name. Each entry declares at least one of format/validate: shell commands gtd runs via bash with $GTD_FILE set to the steering file's path. format rewrites the file in place; validate exits 0 when valid, non-zero with findings on stdout/stderr otherwise. The halves layer independently, so naming a built-in mode (qa/review) and declaring only format: adds formatting while keeping gtd's own validation. gtd ships no formatter — bring your own (prettier, dprint, a script).",
|
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: authoring
|
|
3
|
+
description: >-
|
|
4
|
+
Write or edit a gtd workflow (the repository's `gtd.config.ts`). Use when the
|
|
5
|
+
user asks to create a custom gtd workflow, customize or change their
|
|
6
|
+
workflow's shape, add/remove/rename a step, add a gate/phase/review step,
|
|
7
|
+
change what an agent is prompted to do, adjust fix caps, models, or steering
|
|
8
|
+
files, or otherwise change the flow gtd runs.
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
# Authoring a gtd workflow
|
|
12
|
+
|
|
13
|
+
A gtd workflow is **plain async TypeScript**: a `gtd.config.ts` at the
|
|
14
|
+
repository root default-exports the **flow**, one async function that awaits
|
|
15
|
+
**steps** built from `@pmelab/gtd/flows`; optional `defaults` (process
|
|
16
|
+
settings), `envDefaults` (environment settings), `summary`, `base` and
|
|
17
|
+
`steering` (steering file → mode, for the LSP) exports sit beside it, and any
|
|
18
|
+
other export is a helper gtd ignores. Every step is a commit; gtd finds where a
|
|
19
|
+
process rests by **replaying** the flow over the episode's commits, so the git
|
|
20
|
+
history IS the state and nothing is stored anywhere else.
|
|
21
|
+
|
|
22
|
+
Your job is to produce or edit that module so it loads cleanly and does what the
|
|
23
|
+
user wants. Driving a workflow once it exists is a separate concern — that is
|
|
24
|
+
what a driver does.
|
|
25
|
+
|
|
26
|
+
**Trust:** gtd evaluates `gtd.config.ts` on every command that resolves workflow
|
|
27
|
+
state (`gtd next` and `gtd lsp` included). It is code the user's repository
|
|
28
|
+
runs; write it with the same care as a build script.
|
|
29
|
+
|
|
30
|
+
## Golden rule: start from the bundled default, edit incrementally
|
|
31
|
+
|
|
32
|
+
Do **not** write a workflow from a blank page unless the user wants something
|
|
33
|
+
tiny. gtd ships one known-good workflow and runs it when no `gtd.config.ts` is
|
|
34
|
+
found, and publishes it as `@pmelab/gtd/workflow`: its default export is that
|
|
35
|
+
flow, and every phase and single step it is built from is a named export. Start
|
|
36
|
+
by importing what you keep and writing only what changes:
|
|
37
|
+
|
|
38
|
+
```ts
|
|
39
|
+
import { start } from "@pmelab/gtd/flows"
|
|
40
|
+
import bundled, { afterTail, buildTail } from "@pmelab/gtd/workflow"
|
|
41
|
+
|
|
42
|
+
export {
|
|
43
|
+
defaults,
|
|
44
|
+
envDefaults,
|
|
45
|
+
summary,
|
|
46
|
+
base,
|
|
47
|
+
steering,
|
|
48
|
+
} from "@pmelab/gtd/workflow"
|
|
49
|
+
|
|
50
|
+
export default async ({ entry }) =>
|
|
51
|
+
entry === "hotfix"
|
|
52
|
+
? afterTail(await buildTail(true, start()))
|
|
53
|
+
: bundled({ entry })
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
To change a phase itself, read its source in the npm package
|
|
57
|
+
(`node_modules/@pmelab/gtd/src/workflows/`, or under `$(npm root -g)` for a
|
|
58
|
+
global install) and write your own version in `gtd.config.ts`, reusing its
|
|
59
|
+
single steps.
|
|
60
|
+
|
|
61
|
+
There is no `extends`/merge: the innermost `gtd.config.ts` walking up from the
|
|
62
|
+
current directory is the whole workflow. If one already exists, read it and edit
|
|
63
|
+
it in place.
|
|
64
|
+
|
|
65
|
+
Prefer the bundled workflow's own parts over re-implementing them — `healthy`,
|
|
66
|
+
`escalation`, `gate`, `design`, `architecturePass`, `packages`, `specReview`,
|
|
67
|
+
`qualityLap`, `review`, `buildTail`, and single steps like `triage` or `fix`.
|
|
68
|
+
Their full step names are versioned API.
|
|
69
|
+
|
|
70
|
+
Make one small change, **verify it loads** (see "Verify"), then make the next. A
|
|
71
|
+
workflow that fails to load breaks every gtd command in the repository.
|
|
72
|
+
|
|
73
|
+
## The step API
|
|
74
|
+
|
|
75
|
+
| Call | Actor | Rest content | Resolves to |
|
|
76
|
+
| ---------------------------- | ------- | ------------ | -------------------------------------- |
|
|
77
|
+
| `agent(name, prompt, opts?)` | `agent` | `prompt` | `void`, once an agent turn landed |
|
|
78
|
+
| `human(name, opts?)` | `human` | `message` | `void`, once a person landed |
|
|
79
|
+
| `run(name, body, opts?)` | `check` | `script` | `void`, once the run's tree landed |
|
|
80
|
+
| `judge(name, spec)` | `judge` | `message` | `{ answers, truncated }` |
|
|
81
|
+
| `restart()` | — | — | never: ends the episode from any depth |
|
|
82
|
+
|
|
83
|
+
- `run` body: a POSIX `sh` string the driver runs verbatim. Decide in flow code,
|
|
84
|
+
then render the script from those values — `check(name, command, …)` and the
|
|
85
|
+
exported `checkScript`, `revertScript`, `restoreScript`, `removeScript`,
|
|
86
|
+
`moveScript` and `quote` do that for the common cases. **A run's outcome is
|
|
87
|
+
what it leaves in the tree** — read it back with `changes()` and `read()`.
|
|
88
|
+
- `judge` takes `{ questions, evidence, message?, label? }`. Questions are
|
|
89
|
+
`{ id, primitive: "noul" | "choice" | "score", instructions, criteria }`;
|
|
90
|
+
`evidence` is an object of strings, the `judgeBudgetBytes` var split evenly
|
|
91
|
+
across its keys. It resolves to `answers` — one `{ answer, p }` per question
|
|
92
|
+
id (a `noul` reads back as `"yes"`/`"no"`), `undefined` when the verdict left
|
|
93
|
+
it out — and `truncated`, the evidence keys the budget cut. Compare answers
|
|
94
|
+
with plain `if`s; landing with no verdict leaves every answer `undefined`, so
|
|
95
|
+
make `undefined` take the conservative branch.
|
|
96
|
+
- A person or a driver sees a rest as one of the five content kinds `capture` (a
|
|
97
|
+
dirty tree at a human step), `message`, `script`, `prompt`, `stalled`. Every
|
|
98
|
+
step you add must fit one of them; there is no sixth.
|
|
99
|
+
|
|
100
|
+
Options (all optional): `label`, `file` (a `.gtd/` path), `mode` (needs `file`;
|
|
101
|
+
`qa`, `review`, or a `.gtdrc` `modes:` name), `message` (human/judge), `model`,
|
|
102
|
+
`system` (agent), `allowEmpty` (agent), `acceptClean` (human), `base` (the
|
|
103
|
+
commit the step reviews since — what `gtd base` prints).
|
|
104
|
+
|
|
105
|
+
Helpers — pure reads of the commit replay stands on (the tree the last step
|
|
106
|
+
left, never the live working tree): `read(path)`, `glob(pattern)`,
|
|
107
|
+
`changes(glob?)` (what the last step changed: `{ path, status, before, after }`
|
|
108
|
+
per path, `status` one of `"added"`/`"modified"`/`"deleted"`, plus `paths` and
|
|
109
|
+
`get(path)`), `sections(text)` (`## ` headings), `openQuestions(text)` (a `qa`
|
|
110
|
+
document's unanswered questions), `vars` (process settings, pinned at process
|
|
111
|
+
start — safe to branch on), `env` (environment settings, read live — only for
|
|
112
|
+
prompt text, step options and `run()` bodies, never a branch), `head()` (the
|
|
113
|
+
commit the flow stands on) and `start()` (the process's diff base). State a flow
|
|
114
|
+
needs across steps — a counter, the previous report, a review round's base —
|
|
115
|
+
lives in local variables; replay rebuilds them.
|
|
116
|
+
|
|
117
|
+
Composition: `scope(name, fn)` prefixes step names (`build.fix`) and sets their
|
|
118
|
+
**memory scope** (one scope = one agent conversation = one model/system — mixing
|
|
119
|
+
them inside a scope fails the process); `scope({ name?, model, system }, fn)`
|
|
120
|
+
also sets defaults for agent steps inside. `refuse(message)` refuses the pending
|
|
121
|
+
landing — call it right after the step whose turn you reject.
|
|
122
|
+
`@pmelab/gtd/flows` exports `requireProgress(file)`, `requireAnswers(file)` and
|
|
123
|
+
`requireRevert(edited, base)`, three such checks ready-made.
|
|
124
|
+
|
|
125
|
+
## Names, commits and history
|
|
126
|
+
|
|
127
|
+
- The step name is the `<to>` in `gtd(<actor>): <from> → <to>`, and every
|
|
128
|
+
landing carries `Gtd-Step: <name>#<n>`. It is also the memory scope key (up to
|
|
129
|
+
the last dot). Rename a step and every process resting on it diverges.
|
|
130
|
+
- An episode ends when the flow returns or calls `restart()`; the next starts at
|
|
131
|
+
the flow's first step on an ordinary start — that step is where a finished
|
|
132
|
+
process waits (the bundled one is `human("idle", …)`).
|
|
133
|
+
- `gtd --entry <name>` starts a process with the flow's `{ entry }` argument set
|
|
134
|
+
to `<name>` (`undefined` on an ordinary start). Branch on it, and `refuse()`
|
|
135
|
+
names you don't accept; a flow that never reads `entry` accepts none. An
|
|
136
|
+
`export const base = (entry, vars) => commitish | undefined` fixes an entered
|
|
137
|
+
process's diff base. `--var <name>=<value>` only pins process settings: names
|
|
138
|
+
the workflow's `defaults` or `.gtdrc` `vars:` declare — never an environment
|
|
139
|
+
setting.
|
|
140
|
+
|
|
141
|
+
## Landing rules you are designing for
|
|
142
|
+
|
|
143
|
+
- **Agent turn changed something** → the step completes.
|
|
144
|
+
- **Agent turn changed nothing** → an **attempt**: an empty commit, the process
|
|
145
|
+
stays; the next dispatch is a **stall**. Pass `allowEmpty: true` when "nothing
|
|
146
|
+
to change" is a legitimate result (a reviewer approving by writing nothing).
|
|
147
|
+
- **Human landing changed nothing** → a no-op, the gate keeps waiting — unless
|
|
148
|
+
`acceptClean: true`, which makes "change nothing" mean "accept as-is".
|
|
149
|
+
- **Run landed a clean tree** → the step completes; if replay comes straight
|
|
150
|
+
back to the same step, the landing is **settled** (the driver stops).
|
|
151
|
+
- **Nothing the flow branches on explains the turn** → call `refuse(message)`:
|
|
152
|
+
nothing lands, `gtd land` exits 1.
|
|
153
|
+
|
|
154
|
+
Branch on what the step left, not on who acted:
|
|
155
|
+
|
|
156
|
+
```ts
|
|
157
|
+
await run(
|
|
158
|
+
"check",
|
|
159
|
+
`${env.testCommand} > .gtd/FEEDBACK.md 2>&1 && rm -f .gtd/FEEDBACK.md`,
|
|
160
|
+
)
|
|
161
|
+
if (read(".gtd/FEEDBACK.md") !== undefined) {
|
|
162
|
+
await agent("fix", "Fix what .gtd/FEEDBACK.md reports, then delete it.", {
|
|
163
|
+
file: ".gtd/FEEDBACK.md",
|
|
164
|
+
})
|
|
165
|
+
}
|
|
166
|
+
```
|
|
167
|
+
|
|
168
|
+
Keep `.gtd/` clean across processes: a steering file should be deleted by the
|
|
169
|
+
step that consumes it. A workflow that accumulates files in `.gtd/` is almost
|
|
170
|
+
certainly a bug.
|
|
171
|
+
|
|
172
|
+
## Rules for flow code
|
|
173
|
+
|
|
174
|
+
Flow code is replayed on every command, so it must reach the same steps every
|
|
175
|
+
time it sees the same history. `run()` bodies and module top-level code are
|
|
176
|
+
exempt. gtd does not read the source ahead of time: breaking a rule shows up
|
|
177
|
+
when replay runs, as an error or, for nondeterminism, as a divergence later.
|
|
178
|
+
|
|
179
|
+
- No IO or nondeterminism in flow code — no clock, randomness, environment,
|
|
180
|
+
network or filesystem. Read the tree through the helpers; do IO inside a
|
|
181
|
+
`run()` body.
|
|
182
|
+
- Await only a step, `scope()`, or a function that steps. Anything else fails
|
|
183
|
+
with `the flow awaited something that is not a step`.
|
|
184
|
+
- One call site per step name — wrap a reused helper in two different
|
|
185
|
+
`scope()`s.
|
|
186
|
+
- No `try`/`catch` around a step: `restart()` and refusals travel as exceptions.
|
|
187
|
+
- Only the options a step accepts; an unknown key fails naming the step.
|
|
188
|
+
- The default export is a flow that reaches a step on an ordinary start — the
|
|
189
|
+
same step whatever the repository's files hold; every `mode` must exist.
|
|
190
|
+
|
|
191
|
+
## Verify (after every change)
|
|
192
|
+
|
|
193
|
+
1. **`gtd next`** — loads the workflow (printing every load error at once) and
|
|
194
|
+
shows the resolved rest: step, actor, label, file. It never mutates, so run
|
|
195
|
+
it as often as you like.
|
|
196
|
+
2. **A scratch repository** with at least one commit — make the change a step
|
|
197
|
+
expects, run `gtd land --json=script | sh`, then `gtd next` to see where it
|
|
198
|
+
went. A flow is code, so walking it is the only way to see its branches.
|
|
199
|
+
|
|
200
|
+
`gtd validate` is NOT for this — it validates a **steering file**, not the
|
|
201
|
+
workflow.
|
|
202
|
+
|
|
203
|
+
## Worked example: add an approval gate before building
|
|
204
|
+
|
|
205
|
+
In the bundled default, `planAndBuild` runs the design phase, the architecture
|
|
206
|
+
pass (which writes `.gtd/packages/`), then builds the packages. Add a human
|
|
207
|
+
sign-off between the two:
|
|
208
|
+
|
|
209
|
+
```ts
|
|
210
|
+
const planAndBuild = async (): Promise<void> => {
|
|
211
|
+
for (;;) {
|
|
212
|
+
await design()
|
|
213
|
+
await architecturePass()
|
|
214
|
+
await human("approve-plan", {
|
|
215
|
+
message:
|
|
216
|
+
"The packages under .gtd/packages/ are ready. Edit them to adjust the plan, or change nothing — then run `gtd land` to start building.",
|
|
217
|
+
label: "Approve the plan",
|
|
218
|
+
acceptClean: true,
|
|
219
|
+
})
|
|
220
|
+
await packages()
|
|
221
|
+
if ((await buildTail(false)) === "signoff") return
|
|
222
|
+
await reUnwind()
|
|
223
|
+
}
|
|
224
|
+
}
|
|
225
|
+
```
|
|
226
|
+
|
|
227
|
+
`acceptClean: true` is what makes an untouched landing approve; without it the
|
|
228
|
+
gate would wait for an edit. `approve-plan` sits in the `root` scope and is a
|
|
229
|
+
new, unique name. Verify: `gtd next` loads without errors, and in a scratch
|
|
230
|
+
repository a landing at `architecture.decompose` (or `architecture-promote`) now
|
|
231
|
+
leads to `approve-plan`, and an untouched landing there to
|
|
232
|
+
`packages.item.building`.
|
|
233
|
+
|
|
234
|
+
## No migration
|
|
235
|
+
|
|
236
|
+
A process's commits only make sense to the workflow that made them. If the
|
|
237
|
+
workflow changes under an in-flight process so its history no longer replays to
|
|
238
|
+
the steps its commits name, gtd refuses with a divergence error telling you to
|
|
239
|
+
run `gtd abandon`. Tell the user to finish or `gtd abandon` any in-flight
|
|
240
|
+
process before switching to the edited workflow.
|
|
241
|
+
|
|
242
|
+
## Notes
|
|
243
|
+
|
|
244
|
+
- This skill is versioned in the gtd repository, not auto-installed. When gtd is
|
|
245
|
+
upgraded, re-copy it from the new version's `skills/authoring/SKILL.md`.
|
|
246
|
+
- Where this file and the code disagree, the code wins: the step API's own doc
|
|
247
|
+
comments in `@pmelab/gtd/flows`, and the bundled workflow under
|
|
248
|
+
`src/workflows/`.
|
package/src/flows/runtime.ts
CHANGED
|
@@ -136,6 +136,7 @@ export interface FlowContext {
|
|
|
136
136
|
readonly threads: (text: string) => readonly ThreadInfo[]
|
|
137
137
|
readonly codeThreads: () => readonly CodeThreadInfo[]
|
|
138
138
|
readonly vars: Readonly<Record<string, string>>
|
|
139
|
+
readonly env: Readonly<Record<string, string>>
|
|
139
140
|
readonly head: () => string
|
|
140
141
|
readonly start: () => string
|
|
141
142
|
/**
|
|
@@ -256,19 +257,31 @@ export const start = (): string => ctx().start()
|
|
|
256
257
|
export const skillsFor = (localName: string, ownSkills?: readonly string[]): readonly string[] =>
|
|
257
258
|
ctx().skillsFor(localName, ownSkills)
|
|
258
259
|
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
260
|
+
const settingsProxy = (
|
|
261
|
+
read: () => Readonly<Record<string, string>>,
|
|
262
|
+
): Readonly<Record<string, string>> =>
|
|
263
|
+
new Proxy(
|
|
264
|
+
{},
|
|
265
|
+
{
|
|
266
|
+
get: (_target, key) => (typeof key === "string" ? read()[key] : undefined),
|
|
267
|
+
has: (_target, key) => typeof key === "string" && key in read(),
|
|
268
|
+
ownKeys: () => Object.keys(read()),
|
|
269
|
+
getOwnPropertyDescriptor: (_target, key) =>
|
|
270
|
+
typeof key === "string" && key in read()
|
|
271
|
+
? { enumerable: true, configurable: true, value: read()[key] }
|
|
272
|
+
: undefined,
|
|
273
|
+
},
|
|
274
|
+
)
|
|
275
|
+
|
|
276
|
+
/** The process settings: pinned at process start, so flow code may branch on them. */
|
|
277
|
+
export const vars: Readonly<Record<string, string>> = settingsProxy(() => ctx().vars)
|
|
278
|
+
|
|
279
|
+
/**
|
|
280
|
+
* The environment settings: resolved live on every invocation. Flow code must
|
|
281
|
+
* read them only where they cannot change the next step (step options, script
|
|
282
|
+
* bodies) — replay cannot enforce this.
|
|
283
|
+
*/
|
|
284
|
+
export const env: Readonly<Record<string, string>> = settingsProxy(() => ctx().env)
|
|
272
285
|
|
|
273
286
|
// ── Text utilities ──────────────────────────────────────────────────────────
|
|
274
287
|
|
|
@@ -343,6 +356,7 @@ export interface SummaryContext {
|
|
|
343
356
|
readonly processCost: number
|
|
344
357
|
readonly processCostByModel: readonly { readonly model: string; readonly cost: number }[]
|
|
345
358
|
readonly vars: Readonly<Record<string, string>>
|
|
359
|
+
readonly env: Readonly<Record<string, string>>
|
|
346
360
|
}
|
|
347
361
|
|
|
348
362
|
/** `gtd summary`'s prompt — a workflow module's optional `summary` export. */
|
package/src/workflows/health.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import {
|
|
2
2
|
answered,
|
|
3
3
|
check,
|
|
4
|
+
env,
|
|
4
5
|
human,
|
|
5
6
|
judge,
|
|
6
7
|
numeric,
|
|
@@ -17,7 +18,7 @@ export const FIX_CAP = 3
|
|
|
17
18
|
|
|
18
19
|
/** Run the suite as step `name`; resolves `true` when it passed. A failure is in `.gtd/FEEDBACK.md`. */
|
|
19
20
|
export const baseline = (name: string, label = "Checking the baseline"): Promise<boolean> =>
|
|
20
|
-
check(name,
|
|
21
|
+
check(name, env.testCommand ?? "", { report: FEEDBACK, label })
|
|
21
22
|
|
|
22
23
|
/** How many escalation rounds a run of red checks has spent — reset once the suite goes green. */
|
|
23
24
|
export interface EscalationCount {
|
|
@@ -95,7 +96,7 @@ export const healthy = async (
|
|
|
95
96
|
let fixes = options.fixesSoFar ?? 0
|
|
96
97
|
let previous: string | undefined
|
|
97
98
|
for (;;) {
|
|
98
|
-
const green = await check("health.check",
|
|
99
|
+
const green = await check("health.check", env.testCommand ?? "", {
|
|
99
100
|
report: FEEDBACK,
|
|
100
101
|
label: "Running checks",
|
|
101
102
|
// Swept only on green: an unresolved analysis survives every retry.
|
package/src/workflows/prose.ts
CHANGED
|
@@ -68,7 +68,8 @@ deliberately separate from whoever wrote the code, with no attachment
|
|
|
68
68
|
to it. Write a structured review document grouping a diff into
|
|
69
69
|
chunks; classify a round of the human's feedback as actionable or
|
|
70
70
|
just approving; answer the human's questions inline in the review;
|
|
71
|
-
and, when asked, fix the small nits the human flagged
|
|
71
|
+
and, when asked, fix the small nits the human flagged and the risks
|
|
72
|
+
you marked yourself, each in one batch.
|
|
72
73
|
Beyond that you never fix or build anything yourself.`
|
|
73
74
|
|
|
74
75
|
export const specReviewerPersona = `You are the adversarial spec-conformance checker in gtd's build
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { afterEach, describe, expect, it } from "vitest"
|
|
2
2
|
import { installContext, type Change, type JudgeAnswer, type StepRequest } from "../flows/index.js"
|
|
3
3
|
import { review, type ReviewOutcome } from "./review.js"
|
|
4
|
+
import { unified } from "./index.js"
|
|
4
5
|
import { fixtureContext } from "./text.fixture.js"
|
|
5
6
|
|
|
6
7
|
afterEach(() => installContext(undefined))
|
|
@@ -30,6 +31,8 @@ interface Drive {
|
|
|
30
31
|
readonly vars?: Readonly<Record<string, string>>
|
|
31
32
|
/** The REVIEW.md the human leaves; defaults to `notes` laid over the baseline. */
|
|
32
33
|
readonly after?: string
|
|
34
|
+
/** The REVIEW.md each `review.reviewing` turn writes, in order; also raises the review-turn stop to one past the last. */
|
|
35
|
+
readonly reviewDocs?: readonly (string | undefined)[]
|
|
33
36
|
}
|
|
34
37
|
|
|
35
38
|
interface Run {
|
|
@@ -51,7 +54,10 @@ const drive = async (d: Drive): Promise<Run> => {
|
|
|
51
54
|
let reviews = 0
|
|
52
55
|
const effects: Record<string, () => void> = {
|
|
53
56
|
"review.reviewing": () => {
|
|
54
|
-
|
|
57
|
+
const docs = d.reviewDocs
|
|
58
|
+
if (++reviews > (docs?.length ?? 0) + (docs === undefined ? 1 : 0)) throw new Stop()
|
|
59
|
+
const written = docs?.[reviews - 1]
|
|
60
|
+
if (written !== undefined) files.set(REVIEW, written)
|
|
55
61
|
},
|
|
56
62
|
"review.await-review": () => {
|
|
57
63
|
if (++awaits > 1) throw new Stop()
|
|
@@ -104,6 +110,65 @@ const drive = async (d: Drive): Promise<Run> => {
|
|
|
104
110
|
|
|
105
111
|
const verdict = (answer: string, p = 0.9): JudgeAnswer => ({ answer, p })
|
|
106
112
|
|
|
113
|
+
describe("the risk-fix pass", () => {
|
|
114
|
+
const risky = doc(["Risk: drops the carry", "sub"])
|
|
115
|
+
const clean = doc(["add — fine", "sub"])
|
|
116
|
+
|
|
117
|
+
it("a marked risk is fixed, kept green, re-reviewed, then rests at the gate", async () => {
|
|
118
|
+
const run = await drive({ notes: [], reviewDocs: [risky, clean] })
|
|
119
|
+
expect(run.log.slice(0, 5)).toEqual([
|
|
120
|
+
"review.reviewing",
|
|
121
|
+
"review.fix-risks",
|
|
122
|
+
"health.check",
|
|
123
|
+
"review.reviewing",
|
|
124
|
+
"review.await-review",
|
|
125
|
+
])
|
|
126
|
+
expect(run.prompts.get("review.fix-risks")).toContain("risk-1")
|
|
127
|
+
expect(run.prompts.get("review.fix-risks")).toContain("Risk: drops the carry")
|
|
128
|
+
expect(run.prompts.get("review.fix-risks")).toContain("Leave `.gtd/REVIEW.md` untouched")
|
|
129
|
+
})
|
|
130
|
+
|
|
131
|
+
it("a risk the re-review still marks goes to the gate with no second fix", async () => {
|
|
132
|
+
const run = await drive({ notes: [], reviewDocs: [risky, risky] })
|
|
133
|
+
expect(run.log.filter((n) => n === "review.fix-risks")).toHaveLength(1)
|
|
134
|
+
expect(run.log.slice(0, 5)).toEqual([
|
|
135
|
+
"review.reviewing",
|
|
136
|
+
"review.fix-risks",
|
|
137
|
+
"health.check",
|
|
138
|
+
"review.reviewing",
|
|
139
|
+
"review.await-review",
|
|
140
|
+
])
|
|
141
|
+
})
|
|
142
|
+
|
|
143
|
+
it("no marker means no fix-risks step", async () => {
|
|
144
|
+
const run = await drive({ notes: [], reviewDocs: [clean] })
|
|
145
|
+
expect(run.log).not.toContain("review.fix-risks")
|
|
146
|
+
expect(run.log[0]).toBe("review.reviewing")
|
|
147
|
+
expect(run.log[1]).toBe("review.await-review")
|
|
148
|
+
})
|
|
149
|
+
|
|
150
|
+
it("a nit re-review round gets its own single pass", async () => {
|
|
151
|
+
const run = await drive({
|
|
152
|
+
notes: ["add — typo", "sub", "mul"],
|
|
153
|
+
answers: { "note-1": verdict("nit") },
|
|
154
|
+
reviewDocs: [undefined, risky, clean],
|
|
155
|
+
})
|
|
156
|
+
expect(run.log).toEqual([
|
|
157
|
+
"review.reviewing",
|
|
158
|
+
"review.await-review",
|
|
159
|
+
"review.triage",
|
|
160
|
+
"review.fix-nits",
|
|
161
|
+
"health.check",
|
|
162
|
+
"review.closing",
|
|
163
|
+
"review.reviewing",
|
|
164
|
+
"review.fix-risks",
|
|
165
|
+
"health.check",
|
|
166
|
+
"review.reviewing",
|
|
167
|
+
"review.await-review",
|
|
168
|
+
])
|
|
169
|
+
})
|
|
170
|
+
})
|
|
171
|
+
|
|
107
172
|
describe("review verdict routing", () => {
|
|
108
173
|
const notes = ["add — typo", "sub", "mul"]
|
|
109
174
|
|
|
@@ -245,3 +310,24 @@ describe("review verdict routing", () => {
|
|
|
245
310
|
expect(run.result).toMatchObject({ verdict: "feedback", edited: [] })
|
|
246
311
|
})
|
|
247
312
|
})
|
|
313
|
+
|
|
314
|
+
describe("the default quality lenses", () => {
|
|
315
|
+
it("are the six lenses in the settled order", () => {
|
|
316
|
+
expect(unified.defaults.qualityReviews!.split(",").map((l) => l.trim())).toEqual([
|
|
317
|
+
"correctness",
|
|
318
|
+
"owasp-security",
|
|
319
|
+
"ponytail-review",
|
|
320
|
+
"test-audit",
|
|
321
|
+
"conventions",
|
|
322
|
+
"spec-challenge",
|
|
323
|
+
])
|
|
324
|
+
})
|
|
325
|
+
|
|
326
|
+
it("expose builtInLenses through the public workflow module", () => {
|
|
327
|
+
expect(Object.keys(unified.builtInLenses).sort()).toEqual([
|
|
328
|
+
"conventions",
|
|
329
|
+
"correctness",
|
|
330
|
+
"spec-challenge",
|
|
331
|
+
])
|
|
332
|
+
})
|
|
333
|
+
})
|
package/src/workflows/review.ts
CHANGED
|
@@ -17,7 +17,7 @@ import {
|
|
|
17
17
|
type Change,
|
|
18
18
|
type JudgeQuestion,
|
|
19
19
|
} from "../flows/index.js"
|
|
20
|
-
import { reviewNotes, stripCodeThreads, type ReviewNote } from "../steering/index.js"
|
|
20
|
+
import { reviewNotes, reviewRisks, stripCodeThreads, type ReviewNote } from "../steering/index.js"
|
|
21
21
|
import { escalation, FIX_CAP, healthy, type EscalationCount } from "./health.js"
|
|
22
22
|
import {
|
|
23
23
|
answerReviewQuestions,
|
|
@@ -26,6 +26,7 @@ import {
|
|
|
26
26
|
fix,
|
|
27
27
|
fixNits,
|
|
28
28
|
fixQuality,
|
|
29
|
+
fixRisks,
|
|
29
30
|
QUALITY,
|
|
30
31
|
REQUIREMENTS,
|
|
31
32
|
REVIEW,
|
|
@@ -43,8 +44,8 @@ export const qualityLenses = (): readonly string[] =>
|
|
|
43
44
|
.filter((lens) => lens.length > 0)
|
|
44
45
|
|
|
45
46
|
/**
|
|
46
|
-
* One review turn per lens over the whole change, each appending
|
|
47
|
-
*
|
|
47
|
+
* One review turn per lens over the whole change, each appending every
|
|
48
|
+
* finding to `.gtd/QUALITY.md`. Resolves `"findings"` when that file
|
|
48
49
|
* has any.
|
|
49
50
|
*/
|
|
50
51
|
export const qualityLap = async (): Promise<"clean" | "findings"> => {
|
|
@@ -270,6 +271,20 @@ const finish = async (
|
|
|
270
271
|
return routeNotes(notes, { round, escalations, close, outcome, unfolded })
|
|
271
272
|
}
|
|
272
273
|
|
|
274
|
+
/** Write the review; if it marks risks, fix them, keep green, and write it again — once, so the re-review's own risks reach the human unfixed. */
|
|
275
|
+
const reviewOnce = async (
|
|
276
|
+
base: string,
|
|
277
|
+
carry: string | undefined,
|
|
278
|
+
escalations: EscalationCount,
|
|
279
|
+
): Promise<void> => {
|
|
280
|
+
await reviewing(base, carry)
|
|
281
|
+
const risks = reviewRisks(read(REVIEW) ?? "")
|
|
282
|
+
if (risks.length === 0) return
|
|
283
|
+
await fixRisks(risks)
|
|
284
|
+
await healthy(fix, { escalations })
|
|
285
|
+
await reviewing(base, carry)
|
|
286
|
+
}
|
|
287
|
+
|
|
273
288
|
/**
|
|
274
289
|
* A reviewer writes `.gtd/REVIEW.md` over everything since `base`, a human
|
|
275
290
|
* reviews and signs off or comments, and each note is judged: edits go to
|
|
@@ -282,7 +297,7 @@ export const review = async (
|
|
|
282
297
|
): Promise<ReviewOutcome> => {
|
|
283
298
|
let carry: string | undefined
|
|
284
299
|
for (;;) {
|
|
285
|
-
await
|
|
300
|
+
await reviewOnce(base, carry, escalations)
|
|
286
301
|
carry = undefined
|
|
287
302
|
let reviewed = head()
|
|
288
303
|
let collectedAt: string | undefined
|
|
@@ -26,7 +26,7 @@ import { skills } from "./skills.js"
|
|
|
26
26
|
// - build.review.answer-review-questions, build.review.fix-nits — NOT yet
|
|
27
27
|
// e2e-grounded; only steps.test.ts's preamble checks, which use local names
|
|
28
28
|
describe("the bundled workflow's skills map", () => {
|
|
29
|
-
it("declares exactly the
|
|
29
|
+
it("declares exactly the seventeen bundled agent steps, by full name", () => {
|
|
30
30
|
expect(Object.keys(skills).sort()).toEqual(
|
|
31
31
|
[
|
|
32
32
|
"design.triage",
|
|
@@ -44,6 +44,7 @@ describe("the bundled workflow's skills map", () => {
|
|
|
44
44
|
"build.review.reviewing",
|
|
45
45
|
"build.review.answer-review-questions",
|
|
46
46
|
"build.review.fix-nits",
|
|
47
|
+
"build.review.fix-risks",
|
|
47
48
|
"build.review.collecting",
|
|
48
49
|
].sort(),
|
|
49
50
|
)
|
package/src/workflows/skills.ts
CHANGED
|
@@ -33,5 +33,6 @@ export const skills: Readonly<Record<string, readonly string[]>> = {
|
|
|
33
33
|
"build.review.reviewing": ["code-review-and-quality"],
|
|
34
34
|
"build.review.answer-review-questions": ["code-review-and-quality"],
|
|
35
35
|
"build.review.fix-nits": ["incremental-implementation", "code-simplification"],
|
|
36
|
+
"build.review.fix-risks": ["debugging-and-error-recovery", "incremental-implementation"],
|
|
36
37
|
"build.review.collecting": ["code-review-and-quality"],
|
|
37
38
|
}
|