@pmelab/gtd 18.0.0 → 19.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/README.md +34 -30
- package/claude/hooks/access.ts +58 -0
- package/claude/hooks/drive.ts +42 -4
- package/claude/hooks/register.tsx +12 -0
- package/dist/gtd.bundle.mjs +779 -668
- package/package.json +2 -1
- package/schema.json +22 -0
- package/skills/authoring/SKILL.md +6 -7
- package/src/flows/helpers.ts +4 -2
- package/src/flows/runtime.ts +20 -0
- package/src/flows/scripts.test.ts +28 -1
- package/src/flows/scripts.ts +3 -0
- package/src/workflows/access.test.ts +40 -0
- package/src/workflows/access.ts +22 -0
- package/src/workflows/health.test.ts +21 -1
- package/src/workflows/health.ts +47 -8
- package/src/workflows/packages.test.ts +91 -0
- package/src/workflows/packages.ts +96 -76
- package/src/workflows/planning.ts +5 -49
- package/src/workflows/prose.ts +18 -9
- package/src/workflows/review.test.ts +88 -1
- package/src/workflows/review.ts +41 -16
- package/src/workflows/scenarios.test.ts +256 -0
- package/src/workflows/scenarios.ts +137 -0
- package/src/workflows/skills.test.ts +0 -4
- package/src/workflows/skills.ts +0 -2
- package/src/workflows/steps.test.ts +0 -10
- package/src/workflows/steps.ts +2 -20
- package/src/workflows/text.fixture.ts +26 -0
- package/src/workflows/text.test.ts +101 -0
- package/src/workflows/text.ts +125 -97
- package/src/workflows/unified.ts +9 -4
- package/src/workflows/vars.ts +1 -2
- package/src/workflows/diff.test.ts +0 -115
- package/src/workflows/diff.ts +0 -306
package/README.md
CHANGED
|
@@ -34,6 +34,16 @@ own addressable `.gtdrc` `skills:` entry — see
|
|
|
34
34
|
[Configuration](https://github.com/pmelab/gtd/blob/main/docs/configuration.md#the-skills-key)
|
|
35
35
|
to repoint one to a set your own harness has instead.
|
|
36
36
|
|
|
37
|
+
Each scope also declares which files its agent turns may read and write;
|
|
38
|
+
`gtd land` refuses a turn that wrote outside its scope, and a driver can pass
|
|
39
|
+
the read side to its agent. `read` is not a security boundary unless your driver
|
|
40
|
+
enforces it at the OS level. See
|
|
41
|
+
[Configuration](https://github.com/pmelab/gtd/blob/main/docs/configuration.md#file-access).
|
|
42
|
+
|
|
43
|
+
> **`fastTestCommand` is required, with no fallback to `testCommand`.** Set it
|
|
44
|
+
> (everything but e2e) under `env:` in `.gtdrc` or as `GTD_FASTTESTCOMMAND`; gtd
|
|
45
|
+
> stops at the start gate, writing `.gtd/SETUP.md`, until it is.
|
|
46
|
+
|
|
37
47
|
> **A repository's `gtd.config.ts` is code, and gtd runs it.** A custom workflow
|
|
38
48
|
> is a TypeScript module, and every gtd command that looks at workflow state —
|
|
39
49
|
> `gtd next` and `gtd lsp` included, not just `gtd land` — evaluates it. Treat
|
|
@@ -349,7 +359,7 @@ lists the open ones (editor-only — the phone UI does not show them).
|
|
|
349
359
|
|
|
350
360
|
One built-in workflow drives all of that. From where you sit, it has four
|
|
351
361
|
moments — everything between them runs without you, with the judged exceptions
|
|
352
|
-
noted in steps
|
|
362
|
+
noted in steps 3 and 4 below.
|
|
353
363
|
|
|
354
364
|
1. **You sketch.** Change anything, or write the idea into `.gtd/TODO.md`. Rough
|
|
355
365
|
is fine; it is treated as a sketch, not as work.
|
|
@@ -362,25 +372,22 @@ noted in steps 2, 3, and 4 below.
|
|
|
362
372
|
the same gate again. Close a thread by replying with a conclusion or deleting
|
|
363
373
|
it. While a thread's last entry is the agent's, moving on is refused. Leave
|
|
364
374
|
the file untouched and start the loop to accept the plan as-is, unanswered
|
|
365
|
-
questions and all.
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
plan shown to you at all.
|
|
371
|
-
|
|
372
|
-
The reference driver answers this judgment itself (`gtd judge run`, auto
|
|
373
|
-
selection); if that fails it shows you the message and stops, same as any
|
|
374
|
-
other question. **The `llm` provider's `p` is self-reported by the model, not
|
|
375
|
-
a measured probability, so a confidently wrong haiku verdict can skip a
|
|
376
|
-
question you would have asked.**
|
|
375
|
+
questions and all. Every plan gets the technical pass: its document has four
|
|
376
|
+
sections, in order — `## Interfaces`, `## Call Stacks`, `## E2E Scenarios`,
|
|
377
|
+
`## Unit Tests` — behind a leading `## Open Questions` when there are any.
|
|
378
|
+
`## E2E Scenarios` is never empty: it holds the scenarios, or, when nothing
|
|
379
|
+
user-visible changes, the line `No e2e change.` with a one-line reason.
|
|
377
380
|
|
|
378
381
|
3. **You wait.** The work is split into packages and built one at a time, each
|
|
379
|
-
one
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
382
|
+
one starting from the unit tests it declares (a build turn missing one is
|
|
383
|
+
refused), checked against the fast suite and fixed until it passes before
|
|
384
|
+
moving on. A package that rewords a frozen `.feature` step stops at a wording
|
|
385
|
+
gate: accept the change, or reject it and the original is restored. After the
|
|
386
|
+
last package a full run (e2e included) has its own fix loop before the
|
|
387
|
+
quality lap. Two points along the process are judged rather than always
|
|
388
|
+
asking you outright — each stops and hands you a verdict to make
|
|
389
|
+
(`gtd judge answer`, or land with a clean tree to accept the conservative
|
|
390
|
+
default, which never skips work;
|
|
384
391
|
`gtd judge run --provider fixed --answers <path>` — or the
|
|
385
392
|
`GTD_JUDGE_ANSWERS` env var, inline JSON — answers one from a file, piped
|
|
386
393
|
between `gtd judge --json` and `gtd judge answer`;
|
|
@@ -397,17 +404,15 @@ noted in steps 2, 3, and 4 below.
|
|
|
397
404
|
stdout when they cannot answer every question):
|
|
398
405
|
- Every red round after the first: was the failure identical, new, or
|
|
399
406
|
progress?
|
|
400
|
-
-
|
|
401
|
-
|
|
402
|
-
- After a review turn raises concerns: would each one actually violate the
|
|
403
|
-
spec if left unaddressed, or is it a nit?
|
|
407
|
+
- After you review: is each of your notes an edit, a question, a nit or
|
|
408
|
+
praise?
|
|
404
409
|
|
|
405
410
|
The reference driver answers these itself (`gtd judge run`, auto selection:
|
|
406
411
|
jev when `TYPESAFE_API_KEY` is set, else `llm` via `claude`, default model
|
|
407
412
|
haiku, `--model <name>` overrides); if that fails it shows you the message
|
|
408
413
|
and stops. **The `llm` provider's `p` is self-reported by the model, not a
|
|
409
|
-
measured probability, so a confidently wrong verdict can clear the 0.
|
|
410
|
-
|
|
414
|
+
measured probability, so a confidently wrong verdict can clear the 0.7 floor
|
|
415
|
+
and skip a gate unattended.**
|
|
411
416
|
|
|
412
417
|
Once the last package is built, the whole change goes through a qualitative
|
|
413
418
|
review lap before you see anything: six lenses, one turn each, in order —
|
|
@@ -415,8 +420,7 @@ noted in steps 2, 3, and 4 below.
|
|
|
415
420
|
`conventions`, `spec-challenge`. Each traces the change from its own angle
|
|
416
421
|
and records every finding, blocking or not; one fix turn then fixes ALL of
|
|
417
422
|
them once, with no re-review after the fix. A clean turn means approval only
|
|
418
|
-
when that lens found nothing at all.
|
|
419
|
-
that package against its own spec; this lap is where code quality is looked
|
|
423
|
+
when that lens found nothing at all. This lap is where code quality is looked
|
|
420
424
|
at, and every round pays for it. It never replaces step 4 — your review stays
|
|
421
425
|
the final gate, and nothing here skips it. The `gtd --entry fix-precheck`
|
|
422
426
|
side door (below) repairs a red baseline through this same lap.
|
|
@@ -432,10 +436,10 @@ noted in steps 2, 3, and 4 below.
|
|
|
432
436
|
|
|
433
437
|
4. **You review.** You get a review document listing what changed and what to
|
|
434
438
|
look at. Before you see it, an automatic risk-fix pass
|
|
435
|
-
(`build.review.fix
|
|
436
|
-
with `Risk:`) is fixed first, the suite kept green, and the review
|
|
437
|
-
— once per review round, so a risk the rewrite still names reaches
|
|
438
|
-
unfixed; risk: a fix lands with no check that the risk was real. Tick the
|
|
439
|
+
(`build.review.fix.risks.fixing`) runs. Any risk the reviewer names (a note
|
|
440
|
+
opening with `Risk:`) is fixed first, the suite kept green, and the review
|
|
441
|
+
rewritten — once per review round, so a risk the rewrite still names reaches
|
|
442
|
+
you unfixed; risk: a fix lands with no check that the risk was real. Tick the
|
|
439
443
|
boxes to approve, or write what is wrong. Approving ends the process;
|
|
440
444
|
feedback is judged note by note, each as `edit`, `question`, `nit` or
|
|
441
445
|
`praise`. An `edit` sends the process back to step 2 for a fresh plan — it
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
import { isAbsolute, relative, resolve } from "node:path"
|
|
2
|
+
|
|
3
|
+
import { globMatches } from "../../src/replay/Glob.js"
|
|
4
|
+
|
|
5
|
+
export type AccessDef = { read: string[] | null; write: string[] | null }
|
|
6
|
+
|
|
7
|
+
const READ = new Set(["Read"])
|
|
8
|
+
const WRITE: Record<string, string> = {
|
|
9
|
+
Edit: "file_path",
|
|
10
|
+
MultiEdit: "file_path",
|
|
11
|
+
Write: "file_path",
|
|
12
|
+
NotebookEdit: "notebook_path",
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
// Repo-relative form of a path; undefined when it lies outside the repo.
|
|
16
|
+
const inRepo = (p: string, root: string): string | undefined => {
|
|
17
|
+
const rel = relative(root, isAbsolute(p) ? p : resolve(root, p))
|
|
18
|
+
return rel === ".." || rel.startsWith("../") || isAbsolute(rel) ? undefined : rel
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
const deny = (verb: string, path: string, globs: string[]) =>
|
|
22
|
+
`gtd access: ${verb} of ${path} is outside this step's ${verb} access (${globs.join(", ") || "none"})`
|
|
23
|
+
|
|
24
|
+
// Bash is deliberately not inspected: access is best effort at the tool level,
|
|
25
|
+
// and `gtd land` is the backstop for writes.
|
|
26
|
+
export function accessDenial(
|
|
27
|
+
tool: string,
|
|
28
|
+
input: Record<string, unknown>,
|
|
29
|
+
access: AccessDef,
|
|
30
|
+
root: string,
|
|
31
|
+
): string | undefined {
|
|
32
|
+
if (READ.has(tool)) return check("read", input.file_path, access.read, root)
|
|
33
|
+
const key = WRITE[tool]
|
|
34
|
+
if (key) return check("write", input[key], access.write, root)
|
|
35
|
+
if (tool === "Glob" || tool === "Grep") return searchDenial(input.path, access.read, root)
|
|
36
|
+
return undefined
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
function check(verb: string, raw: unknown, globs: string[] | null, root: string) {
|
|
40
|
+
if (globs === null || typeof raw !== "string") return undefined
|
|
41
|
+
const rel = inRepo(raw, root)
|
|
42
|
+
if (rel === undefined || globs.some((g) => globMatches(rel, g))) return undefined
|
|
43
|
+
return deny(verb, rel, globs)
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
// A search is allowed only from a root that sits inside the literal directory
|
|
47
|
+
// prefix of some read glob, so it cannot enumerate files outside the access.
|
|
48
|
+
function searchDenial(raw: unknown, globs: string[] | null, root: string) {
|
|
49
|
+
if (globs === null) return undefined
|
|
50
|
+
const rel = typeof raw === "string" && raw ? inRepo(raw, root) : ""
|
|
51
|
+
if (rel === undefined) return undefined
|
|
52
|
+
const inside = globs.some((g) => {
|
|
53
|
+
const prefix = g.split("*")[0]!
|
|
54
|
+
const dir = prefix.slice(0, prefix.lastIndexOf("/") + 1).replace(/\/$/, "")
|
|
55
|
+
return dir === "" || rel === dir || rel.startsWith(`${dir}/`)
|
|
56
|
+
})
|
|
57
|
+
return inside ? undefined : deny("read", rel || ".", globs)
|
|
58
|
+
}
|
package/claude/hooks/drive.ts
CHANGED
|
@@ -1,4 +1,6 @@
|
|
|
1
|
+
import { ACCESS_REFUSAL } from "../../src/wire/constants.js"
|
|
1
2
|
import type { Stop } from "../types"
|
|
3
|
+
import type { AccessDef } from "./access"
|
|
2
4
|
|
|
3
5
|
// The subset of `gtd next --json` this driver reads. Absent optional fields
|
|
4
6
|
// are simply missing; booleans may arrive as strings.
|
|
@@ -15,6 +17,7 @@ export type Beat = {
|
|
|
15
17
|
system?: string
|
|
16
18
|
skills?: string[]
|
|
17
19
|
validate?: string
|
|
20
|
+
access?: AccessDef
|
|
18
21
|
judge?: unknown
|
|
19
22
|
changes?: { status: string; path: string }[]
|
|
20
23
|
session?: { id?: string; resume?: boolean | string }
|
|
@@ -36,6 +39,7 @@ export type Turn = {
|
|
|
36
39
|
system?: string
|
|
37
40
|
skills: readonly string[]
|
|
38
41
|
scope: string
|
|
42
|
+
access?: AccessDef
|
|
39
43
|
}
|
|
40
44
|
|
|
41
45
|
export type TurnEnd = { ok: true } | { ok: false; why: string }
|
|
@@ -83,7 +87,7 @@ export async function drive(io: Io, landTurn?: string): Promise<Stop> {
|
|
|
83
87
|
const acted = await act(io, b, beat, judged, beat === 1 ? landTurn : undefined)
|
|
84
88
|
if (acted.stop) return acted.stop
|
|
85
89
|
judged = acted.verdict ? b.state : undefined
|
|
86
|
-
const landed = await land(io, b, acted.verdict)
|
|
90
|
+
const landed = await land(io, b, acted.verdict, true)
|
|
87
91
|
if (landed) return landed
|
|
88
92
|
}
|
|
89
93
|
}
|
|
@@ -123,7 +127,7 @@ async function message(io: Io, b: Beat, beat: number, judged: string | undefined
|
|
|
123
127
|
}
|
|
124
128
|
|
|
125
129
|
async function agent(io: Io, b: Beat, landTurn: string | undefined): Promise<Acted> {
|
|
126
|
-
const memory = b
|
|
130
|
+
const memory = memoryOf(b)
|
|
127
131
|
const t =
|
|
128
132
|
landTurn === memory
|
|
129
133
|
? ({ ok: true } as const)
|
|
@@ -136,6 +140,7 @@ async function agent(io: Io, b: Beat, landTurn: string | undefined): Promise<Act
|
|
|
136
140
|
system: b.system || undefined,
|
|
137
141
|
skills: b.skills ?? [],
|
|
138
142
|
scope: memory.split("#")[0]!,
|
|
143
|
+
access: b.access,
|
|
139
144
|
})
|
|
140
145
|
if (!t.ok) return failed(b, t.why)
|
|
141
146
|
return b.validate ? validated(io, b, b.validate, memory) : {}
|
|
@@ -153,8 +158,25 @@ async function validated(io: Io, b: Beat, validate: string, memory: string): Pro
|
|
|
153
158
|
}
|
|
154
159
|
}
|
|
155
160
|
|
|
156
|
-
|
|
157
|
-
|
|
161
|
+
const memoryOf = (b: Beat) => b.memory ?? b.session?.id ?? b.state ?? "root"
|
|
162
|
+
|
|
163
|
+
// A prompt beat whose turn wrote outside its access gets one recovery turn
|
|
164
|
+
// with the refusal text; a second refusal is an error, not another loop.
|
|
165
|
+
async function land(
|
|
166
|
+
io: Io,
|
|
167
|
+
b: Beat,
|
|
168
|
+
verdict: string | undefined,
|
|
169
|
+
mayRecover: boolean,
|
|
170
|
+
): Promise<Stop | undefined> {
|
|
171
|
+
let l: Landing
|
|
172
|
+
try {
|
|
173
|
+
l = await io.land(verdict)
|
|
174
|
+
} catch (e) {
|
|
175
|
+
const text = e instanceof Error ? e.message : String(e)
|
|
176
|
+
if (b.kind !== "prompt" || !text.includes(ACCESS_REFUSAL)) throw e
|
|
177
|
+
if (!mayRecover) return { kind: "error", text, ...where(b) }
|
|
178
|
+
return recover(io, b, verdict, text)
|
|
179
|
+
}
|
|
158
180
|
const code = await io.sh(l.script, b.log)
|
|
159
181
|
if (code !== 0) return { kind: "error", text: `landing script exited ${code}`, ...where(b) }
|
|
160
182
|
if (!isTrue(l.settled)) return undefined
|
|
@@ -162,6 +184,22 @@ async function land(io: Io, b: Beat, verdict: string | undefined): Promise<Stop
|
|
|
162
184
|
return { kind: "done", text, ...where(b) }
|
|
163
185
|
}
|
|
164
186
|
|
|
187
|
+
async function recover(
|
|
188
|
+
io: Io,
|
|
189
|
+
b: Beat,
|
|
190
|
+
verdict: string | undefined,
|
|
191
|
+
text: string,
|
|
192
|
+
): Promise<Stop | undefined> {
|
|
193
|
+
const memory = memoryOf(b)
|
|
194
|
+
const r = await io.resume(memory, text)
|
|
195
|
+
if (!r.ok) return { kind: "error", text: r.why, ...where(b) }
|
|
196
|
+
if (b.validate) {
|
|
197
|
+
const v = await validated(io, b, b.validate, memory)
|
|
198
|
+
if (v.stop) return v.stop
|
|
199
|
+
}
|
|
200
|
+
return land(io, b, verdict, false)
|
|
201
|
+
}
|
|
202
|
+
|
|
165
203
|
// What a reload that cut the loop off does next. Only a turn the agent
|
|
166
204
|
// finished lands; any other end runs it again. A script cut off may have run
|
|
167
205
|
// partly, and running it again is not safe to assume, so a person decides.
|
|
@@ -2,6 +2,8 @@ import { atom, read, update } from "claude-code"
|
|
|
2
2
|
import type { EngineInterface, Register } from "claude-code"
|
|
3
3
|
|
|
4
4
|
import type { Run, Stop } from "../types"
|
|
5
|
+
import { accessDenial } from "./access"
|
|
6
|
+
import type { AccessDef } from "./access"
|
|
5
7
|
import { afterReload, drive, isTrue } from "./drive"
|
|
6
8
|
import type { Beat, Io, Landing, Turn, TurnEnd } from "./drive"
|
|
7
9
|
import { enter } from "./entry"
|
|
@@ -66,6 +68,11 @@ let isShipping = false
|
|
|
66
68
|
// subagents answering into a file: what each may touch (see tool.call)
|
|
67
69
|
const delegates = new Map<string, { file: string; mayRun: RegExp | undefined }>()
|
|
68
70
|
const ours = new Set<string>()
|
|
71
|
+
// What each turn agent may read and write; reset on every spawn and resume,
|
|
72
|
+
// since the steering file can differ per step.
|
|
73
|
+
const accesses = new Map<string, AccessDef>()
|
|
74
|
+
const setAccess = (id: string, a: AccessDef | undefined) =>
|
|
75
|
+
void (a ? accesses.set(id, a) : accesses.delete(id))
|
|
69
76
|
const waiting = new Map<string, (end: TurnEnd) => void>()
|
|
70
77
|
const ended = new Map<string, TurnEnd>()
|
|
71
78
|
const personas = new Set<string>()
|
|
@@ -168,6 +175,7 @@ function io($: $): Io {
|
|
|
168
175
|
const known = resumable(entry, name)
|
|
169
176
|
if (t.resume && entry && !known) $.ui.log(restartLine(t.scope))
|
|
170
177
|
if (t.resume && known) {
|
|
178
|
+
setAccess(known, t.access)
|
|
171
179
|
const end = send($, known, t.prompt)
|
|
172
180
|
await update($, run, (r) => ({ ...r, inflight: { agentId: known, memory: t.memory } }))
|
|
173
181
|
const ended = await end
|
|
@@ -188,6 +196,7 @@ function io($: $): Io {
|
|
|
188
196
|
const id = spawned.agentId
|
|
189
197
|
if (!id) return { ok: false, why: spawned.deny ?? "subagent spawn refused" }
|
|
190
198
|
ours.add(id)
|
|
199
|
+
setAccess(id, t.access)
|
|
191
200
|
await update($, agents, (list) => [...list, id].slice(-200))
|
|
192
201
|
await update($, scopes, (s) => ({ ...s, [t.memory]: { agentId: id, persona: name } }))
|
|
193
202
|
await update($, run, (r) => ({ ...r, inflight: { agentId: id, memory: t.memory } }))
|
|
@@ -744,6 +753,9 @@ export const register: Register = (on) => {
|
|
|
744
753
|
// would still be writing the tree gtd is committing.
|
|
745
754
|
on("tool.call", ($, e, next) => {
|
|
746
755
|
if (!e.agentId || !ours.has(e.agentId)) return next(e)
|
|
756
|
+
const access = accesses.get(e.agentId)
|
|
757
|
+
const denial = access && accessDenial(String(e.tool), e, access, root)
|
|
758
|
+
if (denial) return { deny: denial }
|
|
747
759
|
const delegated = delegates.get(e.agentId)
|
|
748
760
|
if (delegated) {
|
|
749
761
|
if (e.tool === "Write" && e.file_path !== delegated.file) {
|