@afokapu/atdd-bun 0.9.2 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/delivery.ts CHANGED
@@ -6,22 +6,28 @@ import { join, resolve } from "node:path";
6
6
  import type { PlanFinding } from "./planner-kernel";
7
7
  import { topologyFor } from "./topology";
8
8
 
9
- /** The delivery profile: the record a tranche leaves of its reviews, checked against the policy in atdd-bun.yaml.
9
+ /** The delivery profile: the record a tranche leaves of who wrote each stage and of its reviews, checked against the
10
+ * policy in atdd-bun.yaml.
10
11
  *
11
- * A tranche is one independently mergeable piece of a program. Its driver appends each review to
12
- * `<root>/<tranche>/evidence.yaml`; this validator judges that record, never the running agents. It holds on
13
- * every run: each reviewer is an allowed model, a fallback names why the preferred one was unavailable, the
14
- * reviewer is independent of the authors, and every finding is fixed, withdrawn after a written dispute, or ruled
15
- * on by a human. A `ready` record must also have every configured stage approved and an approved SHA that exists.
16
- * At the merge gate (a pull request or merge queue in CI), every record the branch changes must be ready and the
17
- * branch head may differ from its approved SHA only under the delivery root.
12
+ * A tranche is one independently mergeable piece of a program. Its driver appends each stage's work and each review
13
+ * to `<root>/<tranche>/evidence.yaml`; this validator judges that record, never the running agents. It holds on every
14
+ * run: each writer and reviewer is an allowed model, a fallback names why the preferred one was unavailable, every
15
+ * reviewer is independent of the writers its review answers for, and every finding is fixed, withdrawn after a written
16
+ * dispute, or ruled on by a human. A `ready` record must also have every written stage recorded, every reviewed stage
17
+ * approved, and an approved SHA that exists. At the merge gate (a pull request or merge queue in CI), every record the
18
+ * branch changes must be ready and the branch head may differ from its approved SHA only under the delivery root.
18
19
  *
19
20
  * The capability is inert until adopted in atdd-bun.yaml: a `delivery:` block, or `delivery` named in `profiles:`. */
20
21
 
21
- export const STAGES = ["plan_review", "test_review", "code_review", "final_review"] as const;
22
+ /** The lifecycle stages, in order. Each names who may write it and who reviews it; either may be absent. */
23
+ export const STAGES = ["plan", "red", "green", "refactor", "final"] as const;
22
24
  export type Stage = (typeof STAGES)[number];
25
+ /** Stage names before 0.10, still read in configs and records: each review stage maps to the stage it reviewed. */
26
+ export const LEGACY_STAGES: Record<string, Stage> = { plan_review: "plan", test_review: "red", code_review: "refactor", final_review: "final" };
27
+ /** A legacy stage that named reviewers but no authors took these package defaults for its authors. */
28
+ const LEGACY_AUTHORS: Record<string, string[]> = { plan_review: ["codex"], test_review: ["glm", "claude"], code_review: ["glm", "claude"], final_review: ["codex"] };
23
29
  export type Independence = "fresh-process" | "different-model";
24
- export type StagePolicy = { authors: string[]; reviewers: string[]; independence: Independence };
30
+ export type StagePolicy = { writer: string[]; reviewer: string[]; independence: Independence };
25
31
  export type DeliveryPolicy = {
26
32
  root: string;
27
33
  independence: Independence;
@@ -32,13 +38,17 @@ export type DeliveryPolicy = {
32
38
  multiplexer: string;
33
39
  };
34
40
 
35
-
36
- const DEFAULT_STAGES: Record<Stage, Omit<StagePolicy, "independence">> = {
37
- plan_review: { authors: ["codex"], reviewers: ["glm", "claude"] },
38
- test_review: { authors: ["glm", "claude"], reviewers: ["codex", "claude"] },
39
- code_review: { authors: ["glm", "claude"], reviewers: ["glm", "claude"] },
40
- final_review: { authors: ["codex"], reviewers: ["codex", "claude"] },
41
+ /** The default operating model: two reviews, the plan and the whole change; the stages between them are written and
42
+ * held by their deterministic gates. Lists are preference orders; a later model is used only as a recorded fallback. */
43
+ const DEFAULT_STAGES: Partial<Record<Stage, { writer?: string[]; reviewer?: string[] }>> = {
44
+ plan: { writer: ["codex", "claude-opus"], reviewer: ["glm", "claude-opus", "codex"] },
45
+ red: { writer: ["glm", "claude-sonnet", "claude-opus", "codex"] },
46
+ green: { writer: ["glm", "claude-sonnet", "claude-opus", "codex"] },
47
+ refactor: { writer: ["glm", "claude-sonnet", "claude-opus", "codex"] },
48
+ final: { reviewer: ["codex", "glm", "claude-opus"] },
41
49
  };
50
+ /** The stage a record or config name refers to: a current name, or a legacy review stage's. */
51
+ export const stageOf = (name: string): Stage | undefined => (STAGES as readonly string[]).includes(name) ? name as Stage : LEGACY_STAGES[name];
42
52
  /** Tranche records live with the program's reasoning, under the docs profile's delivery area; the docs profile leaves
43
53
  * this folder to the delivery profile (records are YAML and data, never authored AsciiDoc). */
44
54
  export const DEFAULT_ROOT = "docs/delivery/tranches";
@@ -69,11 +79,20 @@ export function deliveryPolicy(block: unknown): DeliveryPolicy {
69
79
  const models = (value: unknown, fallback: string[]) => (Array.isArray(value) && value.every(item => typeof item === "string") ? value as string[] : fallback);
70
80
  const mode = (value: unknown, fallback: Independence): Independence => (value === "fresh-process" || value === "different-model" ? value : fallback);
71
81
  const count = (value: unknown, fallback: number) => (Number.isInteger(value) ? value as number : fallback);
72
- const independence = mode(raw.independence, "fresh-process"), given = record(raw.stages), stages: DeliveryPolicy["stages"] = {};
73
- for (const stage of STAGES) {
74
- const entry = given ? record(given[stage]) : DEFAULT_STAGES[stage];
75
- if (!entry) continue;
76
- stages[stage] = { authors: models(entry.authors, DEFAULT_STAGES[stage].authors), reviewers: models(entry.reviewers, DEFAULT_STAGES[stage].reviewers), independence: mode((entry as Record<string, unknown>).independence, independence) };
82
+ const independence = mode(raw.independence, "different-model"), given = record(raw.stages), stages: DeliveryPolicy["stages"] = {};
83
+ // `stages`, when given, replaces the default set. A legacy name (plan_review, …) is read as the stage it reviewed,
84
+ // its authors as the writers and its reviewers as the reviewers; a current name wins over a legacy one.
85
+ const entries: Array<[Stage, Record<string, unknown>, string]> = given
86
+ ? Object.entries(given).flatMap(([name, value]) => { const stage = stageOf(name), entry = record(value); return stage && entry ? [[stage, entry, name] as [Stage, Record<string, unknown>, string]] : []; })
87
+ .sort((a, b) => Number(a[2] === a[0]) - Number(b[2] === b[0]))
88
+ : Object.entries(DEFAULT_STAGES).map(([stage, entry]) => [stage as Stage, entry as Record<string, unknown>, stage]);
89
+ for (const [stage, entry, name] of entries) {
90
+ const legacy = name !== stage;
91
+ stages[stage] = {
92
+ writer: models(legacy ? entry.authors : entry.writer, legacy ? LEGACY_AUTHORS[name] : []),
93
+ reviewer: models(legacy ? entry.reviewers : entry.reviewer, []),
94
+ independence: mode(entry.independence, independence),
95
+ };
77
96
  }
78
97
  const fallback = record(raw.fallback) ?? {};
79
98
  return {
@@ -99,14 +118,17 @@ export function loosenedDelivery(base: unknown, current: unknown): string[] {
99
118
  if (!b) continue;
100
119
  if (!c) { out.push(`delivery.stages drops ${stage}`); continue; }
101
120
  if (b.independence === "different-model" && c.independence === "fresh-process") out.push(`delivery.stages.${stage}.independence different-model → fresh-process`);
102
- for (const role of ["reviewers", "authors"] as const) {
121
+ if (b.reviewer.length && !c.reviewer.length) out.push(`delivery.stages.${stage} is no longer reviewed`);
122
+ for (const role of ["reviewer", "writer"] as const) {
103
123
  const added = c[role].filter(model => !b[role].includes(model));
104
- if (added.length) out.push(`delivery.stages.${stage}.${role} adds ${added.join(", ")}`);
124
+ if (added.length && b[role].length) out.push(`delivery.stages.${stage}.${role} adds ${added.join(", ")}`);
105
125
  // The lists are preference orders: moving a model earlier, by reordering or removing one before it, makes a
106
126
  // fallback model usable without the fallback.
107
127
  const promoted = c[role].filter(model => b[role].includes(model) && c[role].indexOf(model) < b[role].indexOf(model));
108
128
  if (promoted.length) out.push(`delivery.stages.${stage}.${role} [${b[role].join(", ")}] → [${c[role].join(", ")}] promotes ${promoted.join(", ")}`);
109
129
  }
130
+ // A stage that had no writer list now accepting writers is a loosening too: before, nothing could be written there.
131
+ if (!b.writer.length && c.writer.length) out.push(`delivery.stages.${stage}.writer adds ${c.writer.join(", ")}`);
110
132
  }
111
133
  // Moving the root hides every earlier record from the validator and the gate. No exception, not even pinning the 0.8.0
112
134
  // default: from the config alone it cannot be told apart from moving a 0.9 repository's records out of view, so a
@@ -143,9 +165,11 @@ const git = async (cwd: string, args: string[]) => {
143
165
  };
144
166
 
145
167
  type Actor = { model: string; run: string };
146
- type Review = { stage: Stage; sha: string; author?: Actor; reviewer: Actor; fallback?: Array<{ role?: "author" | "reviewer"; from: string; kind: string; failures: number; window: { from: string; to: string }; reason: string }>; verdict: "approve" | "request_changes"; findings?: Finding[]; report?: string };
168
+ type Fallback = { role?: "author" | "writer" | "reviewer"; from: string; kind: string; failures: number; window: { from: string; to: string }; reason: string };
169
+ type Work = { stage: Stage; sha: string; writer: Actor; fallback?: Fallback[] };
170
+ type Review = { stage: Stage; sha: string; author?: Actor; reviewer: Actor; fallback?: Fallback[]; verdict: "approve" | "request_changes"; findings?: Finding[]; report?: string };
147
171
  type Finding = { id: string; severity: string; rebuttal?: string; outcome?: "fixed" | "withdrawn" | "human"; decision?: string };
148
- type Evidence = { tranche: string; status: "open" | "ready"; base_sha: string; approved_sha?: string; reviews: Review[] };
172
+ type Evidence = { tranche: string; status: "open" | "ready"; base_sha: string; approved_sha?: string; work?: Work[]; reviews: Review[] };
149
173
  export type EvidenceFile = { file: string; tranche: string; data: Evidence | null; error?: string };
150
174
 
151
175
  /** Files in the records folder that are neither a tranche's evidence.yaml nor a data file (a report): authored documents
@@ -198,29 +222,38 @@ export async function loadEvidence(root: string, policy: DeliveryPolicy): Promis
198
222
  return out;
199
223
  }
200
224
 
201
- /** Models must come from the stage's list; a model after the first needs a recorded fallback from each one before it. */
202
- function checkModels(file: string, review: Review, policy: StagePolicy, at: string, fallback: DeliveryPolicy["fallback"]): PlanFinding[] {
225
+ /** A model must come from the stage's list; a model after the first needs a recorded fallback from each one before it. */
226
+ function checkModel(file: string, at: string, stage: Stage, role: "writer" | "reviewer", actor: Actor, list: string[], fallbacks: Fallback[], defaultRole: "writer" | "reviewer", fallback: DeliveryPolicy["fallback"]): PlanFinding[] {
227
+ const out: PlanFinding[] = [];
228
+ const index = list.indexOf(actor.model);
229
+ if (index < 0) return [finding("delivery.model-allowed", file, `${at}: ${role} model '${actor.model}' is not in ${stage}.${role} [${list.join(", ")}]`)];
230
+ // `author` is the 0.9 spelling of `writer`.
231
+ const recorded = fallbacks.filter(entry => ((entry.role === "author" ? "writer" : entry.role) ?? defaultRole) === role).map(entry => entry.from);
232
+ for (const skipped of list.slice(0, index)) if (!recorded.includes(skipped)) out.push(finding("delivery.model-allowed", file, `${at}: ${role} '${actor.model}' is a fallback, but no fallback from '${skipped}' records why it was unavailable`));
233
+ for (const from of recorded) if (!list.slice(0, index).includes(from)) out.push(finding("delivery.model-allowed", file, `${at}: ${role} fallback from '${from}' does not precede '${actor.model}' in [${list.join(", ")}]`));
234
+ return out;
235
+ }
236
+
237
+ /** The count and window of every recorded fallback are the driver's claims, but explicit ones: outside the policy is out of policy. */
238
+ function checkFallbacks(file: string, at: string, fallbacks: Fallback[], fallback: DeliveryPolicy["fallback"]): PlanFinding[] {
203
239
  const out: PlanFinding[] = [];
204
- // The count and window are the driver's claims, but explicit ones: a fallback outside the policy is out of policy.
205
- for (const entry of review.fallback ?? []) {
240
+ for (const entry of fallbacks) {
206
241
  if (entry.failures < fallback.after_failures) out.push(finding("delivery.model-allowed", file, `${at}: fallback from '${entry.from}' after ${entry.failures} failure(s); the policy requires ${fallback.after_failures} (delivery.fallback.after_failures)`));
207
242
  const span = (Date.parse(entry.window.to) - Date.parse(entry.window.from)) / 60_000;
208
243
  if (!(span >= 0)) out.push(finding("delivery.model-allowed", file, `${at}: fallback from '${entry.from}' has a window that ends before it starts`));
209
244
  else if (span > fallback.within_minutes) out.push(finding("delivery.model-allowed", file, `${at}: fallback from '${entry.from}' counts failures over ${Math.round(span)} minutes; the policy allows ${fallback.within_minutes} (delivery.fallback.within_minutes)`));
210
245
  }
211
- const role = (name: "author" | "reviewer", actor: Actor | undefined, list: string[]) => {
212
- if (!actor) return;
213
- const index = list.indexOf(actor.model);
214
- if (index < 0) { out.push(finding("delivery.model-allowed", file, `${at}: ${name} model '${actor.model}' is not in ${review.stage}.${name}s [${list.join(", ")}]`)); return; }
215
- const recorded = (review.fallback ?? []).filter(entry => (entry.role ?? "reviewer") === name).map(entry => entry.from);
216
- for (const skipped of list.slice(0, index)) if (!recorded.includes(skipped)) out.push(finding("delivery.model-allowed", file, `${at}: ${name} '${actor.model}' is a fallback, but no fallback from '${skipped}' records why it was unavailable`));
217
- for (const from of recorded) if (!list.slice(0, index).includes(from)) out.push(finding("delivery.model-allowed", file, `${at}: ${name} fallback from '${from}' does not precede '${actor.model}' in [${list.join(", ")}]`));
218
- };
219
- role("author", review.author, policy.authors);
220
- role("reviewer", review.reviewer, policy.reviewers);
221
246
  return out;
222
247
  }
223
248
 
249
+ /** The stages a review of `stage` answers for: its own, and every stage after the previous reviewed one. The final review
250
+ * covers red, green and refactor when none of them is reviewed. */
251
+ function coveredStages(stage: Stage, policy: DeliveryPolicy): Stage[] {
252
+ const position = STAGES.indexOf(stage);
253
+ const previous = STAGES.slice(0, position).findLastIndex(name => (policy.stages[name]?.reviewer.length ?? 0) > 0);
254
+ return STAGES.slice(previous + 1, position + 1);
255
+ }
256
+
224
257
  /** Reports are data, never code: a report path is exempt from drift, so it must not be able to name a source file. */
225
258
  const REPORT_EXTENSION = /\.(json|jsonl|yaml|yml|txt|md|log)$/;
226
259
  /** The reports a record names, read defensively: this runs before schema validation, so a malformed record (reviews not
@@ -244,17 +277,27 @@ const sameCommit = (a: string, b: string) => a.length <= b.length ? b.startsWith
244
277
 
245
278
  /** The judgement over one well-formed record. */
246
279
  function checkEvidence(file: string, evidence: Evidence, policy: DeliveryPolicy): PlanFinding[] {
247
- const out: PlanFinding[] = [], reviews = evidence.reviews;
248
- const authorRuns = new Set(reviews.flatMap(review => review.author ? [review.author.run] : [])), reviewerRuns = new Map<string, number>();
280
+ const out: PlanFinding[] = [], reviews = evidence.reviews, work = evidence.work ?? [];
281
+ const writerRuns = new Set([...work.map(entry => entry.writer.run), ...reviews.flatMap(review => review.author ? [review.author.run] : [])]), reviewerRuns = new Map<string, number>();
282
+ work.forEach((entry, index) => {
283
+ const at = `work[${index}] (${entry.stage} @ ${entry.sha})`, stage = policy.stages[entry.stage];
284
+ if (!stage?.writer.length) { out.push(finding("delivery.evidence-schema", file, `${at}: ${entry.stage} has no writer in delivery.stages`)); return; }
285
+ out.push(...checkFallbacks(file, at, entry.fallback ?? [], policy.fallback), ...checkModel(file, at, entry.stage, "writer", entry.writer, stage.writer, entry.fallback ?? [], "writer", policy.fallback));
286
+ });
249
287
  reviews.forEach((review, index) => {
250
288
  const at = `reviews[${index}] (${review.stage} @ ${review.sha})`, stage = policy.stages[review.stage];
251
- if (!stage) { out.push(finding("delivery.evidence-schema", file, `${at}: ${review.stage} is not a configured stage in delivery.stages`)); return; }
252
- if (!review.author) out.push(finding("delivery.evidence-schema", file, `${at}: names no author; the reviewer's independence cannot be judged without one`));
253
- out.push(...checkModels(file, review, stage, at, policy.fallback));
254
- if (authorRuns.has(review.reviewer.run)) out.push(finding("delivery.reviewer-independent", file, `${at}: reviewer run '${review.reviewer.run}' also authored in this tranche; a reviewer that edits becomes an author`));
289
+ if (!stage?.reviewer.length) { out.push(finding("delivery.evidence-schema", file, `${at}: ${review.stage} has no reviewer in delivery.stages`)); return; }
290
+ const fallbacks = review.fallback ?? [];
291
+ out.push(...checkFallbacks(file, at, fallbacks, policy.fallback), ...checkModel(file, at, review.stage, "reviewer", review.reviewer, stage.reviewer, fallbacks, "reviewer", policy.fallback));
292
+ if (review.author && stage.writer.length) out.push(...checkModel(file, at, review.stage, "writer", review.author, stage.writer, fallbacks, "reviewer", policy.fallback));
293
+ // Independence is judged against everyone who wrote what this review answers for.
294
+ const covered = coveredStages(review.stage, policy), writers = [...work.filter(entry => covered.includes(entry.stage)).map(entry => entry.writer), ...(review.author ? [review.author] : [])];
295
+ if (!writers.length) out.push(finding("delivery.evidence-schema", file, `${at}: names no writer of the work it reviews (${covered.join(", ")}); record each stage's work, or the review's author, so independence can be judged`));
296
+ if (writerRuns.has(review.reviewer.run)) out.push(finding("delivery.reviewer-independent", file, `${at}: reviewer run '${review.reviewer.run}' also wrote in this tranche; a reviewer that edits becomes a writer`));
255
297
  if (reviewerRuns.has(review.reviewer.run)) out.push(finding("delivery.reviewer-independent", file, `${at}: reviewer run '${review.reviewer.run}' already reviewed reviews[${reviewerRuns.get(review.reviewer.run)}]; every review is a fresh process`));
256
298
  else reviewerRuns.set(review.reviewer.run, index);
257
- if (stage.independence === "different-model" && review.author && review.author.model === review.reviewer.model) out.push(finding("delivery.reviewer-independent", file, `${at}: ${review.stage} requires a different model, but '${review.reviewer.model}' reviewed its own model's work`));
299
+ const own = writers.find(writer => writer.model === review.reviewer.model);
300
+ if (stage.independence === "different-model" && own) out.push(finding("delivery.reviewer-independent", file, `${at}: ${review.stage} requires a different model, but '${review.reviewer.model}' reviewed work its own model wrote`));
258
301
  });
259
302
 
260
303
  // Findings: each one on a request-changes review is fixed, withdrawn after one written dispute, or ruled on by a
@@ -284,12 +327,16 @@ function checkEvidence(file: string, evidence: Evidence, policy: DeliveryPolicy)
284
327
  if (evidence.status === "ready") {
285
328
  const configured = STAGES.filter(stage => policy.stages[stage]);
286
329
  for (const stage of configured) {
330
+ // A 0.9 record names writers as the authors of the reviews that covered their stages.
331
+ const written = work.some(entry => entry.stage === stage) || reviews.some(review => review.author && coveredStages(review.stage, policy).includes(stage));
332
+ if (policy.stages[stage]!.writer.length && !written) out.push(finding("delivery.stages-complete", file, `status is ready, but ${stage} records no work; name who wrote it`));
333
+ if (!policy.stages[stage]!.reviewer.length) continue;
287
334
  const last = reviews.filter(review => review.stage === stage).at(-1);
288
335
  if (!last) out.push(finding("delivery.stages-complete", file, `status is ready, but ${stage} has no review`));
289
- else if (last.verdict !== "approve") out.push(finding("delivery.stages-complete", file, `status is ready, but the last ${stage} requests changes`));
336
+ else if (last.verdict !== "approve") out.push(finding("delivery.stages-complete", file, `status is ready, but the last ${stage} review requests changes`));
290
337
  }
291
- const closing = configured.at(-1), approval = closing && reviews.filter(review => review.stage === closing).at(-1);
292
- if (approval && approval.verdict === "approve" && !sameCommit(approval.sha, evidence.approved_sha ?? "")) out.push(finding("delivery.stages-complete", file, `approved_sha ${evidence.approved_sha} is not the SHA the last ${closing} approved (${approval.sha})`));
338
+ const closing = configured.filter(stage => policy.stages[stage]!.reviewer.length).at(-1), approval = closing && reviews.filter(review => review.stage === closing).at(-1);
339
+ if (approval && approval.verdict === "approve" && !sameCommit(approval.sha, evidence.approved_sha ?? "")) out.push(finding("delivery.stages-complete", file, `approved_sha ${evidence.approved_sha} is not the SHA the last ${closing} review approved (${approval.sha})`));
293
340
  }
294
341
  return out;
295
342
  }
@@ -394,6 +441,8 @@ export async function validateDelivery(root = process.cwd(), options: DeliveryOp
394
441
  for (const error of validEvidence.errors ?? []) findings.push(finding("delivery.evidence-schema", entry.file, `${entry.file} violates delivery-evidence.schema.json: ${error.instancePath || "/"} ${error.message ?? error.keyword}`));
395
442
  continue;
396
443
  }
444
+ // Legacy stage names (plan_review, …) are read as the stages they reviewed.
445
+ for (const item of [...entry.data.reviews, ...(entry.data.work ?? [])]) item.stage = stageOf(item.stage) ?? item.stage;
397
446
  if (entry.data.tranche !== entry.tranche) findings.push(finding("delivery.evidence-schema", entry.file, `tranche '${entry.data.tranche}' does not match its folder '${entry.tranche}'`));
398
447
  // A report lives in its tranche's folder: a report path is exempt from drift, so it may never name other files.
399
448
  const folder = `${policy.root}/${entry.tranche}/`;
@@ -411,27 +460,22 @@ export async function validateDelivery(root = process.cwd(), options: DeliveryOp
411
460
  // Ready means auditable: every review's raw output is retained and named.
412
461
  for (const [index, review] of entry.data.reviews.entries()) if (!review.report || !regularFile(join(absolute, review.report))) findings.push(finding("delivery.stages-complete", entry.file, `status is ready, but reviews[${index}] (${review.stage}) ${review.report ? `names report ${review.report}, which is not a regular file` : "retains no report"}`));
413
462
  if ((await git(absolute, ["cat-file", "-e", `${entry.data.approved_sha}^{commit}`])).code) { findings.push(finding("delivery.approved-sha-resolves", entry.file, `approved_sha ${entry.data.approved_sha} is not a commit in this repository's history`)); continue; }
414
- // Every stage's approval names a real commit in the approved history, in lifecycle order.
463
+ // Every stage's approval and every recorded piece of work names a real commit in the approved history, and
464
+ // approvals follow the lifecycle.
465
+ const approved = entry.data.approved_sha!, inHistory = async (sha: string) => !(await git(absolute, ["merge-base", "--is-ancestor", sha, approved])).code;
466
+ for (const [index, item] of (entry.data.work ?? []).entries()) {
467
+ if ((await git(absolute, ["cat-file", "-e", `${item.sha}^{commit}`])).code) findings.push(finding("delivery.approved-sha-resolves", entry.file, `work[${index}] (${item.stage}) names ${item.sha}, which is not a commit in this repository's history`));
468
+ else if (!await inHistory(item.sha)) findings.push(finding("delivery.approved-sha-resolves", entry.file, `work[${index}] (${item.stage}) names ${item.sha.slice(0, 7)}, which is not in the history of approved_sha ${approved.slice(0, 7)}`));
469
+ }
415
470
  let previous: { stage: Stage; sha: string } | null = null;
416
- for (const stage of STAGES.filter(name => policy.stages[name])) {
471
+ for (const stage of STAGES.filter(name => policy.stages[name]?.reviewer.length)) {
417
472
  const last = entry.data.reviews.filter(review => review.stage === stage).at(-1);
418
473
  if (!last || last.verdict !== "approve") continue; // reported by delivery.stages-complete
419
474
  if ((await git(absolute, ["cat-file", "-e", `${last.sha}^{commit}`])).code) { findings.push(finding("delivery.approved-sha-resolves", entry.file, `${stage} approved ${last.sha}, which is not a commit in this repository's history`)); continue; }
420
- if ((await git(absolute, ["merge-base", "--is-ancestor", last.sha, entry.data.approved_sha!])).code) { findings.push(finding("delivery.approved-sha-resolves", entry.file, `${stage} approved ${last.sha.slice(0, 7)}, which is not in the history of approved_sha ${entry.data.approved_sha!.slice(0, 7)}`)); continue; }
475
+ if (!await inHistory(last.sha)) { findings.push(finding("delivery.approved-sha-resolves", entry.file, `${stage} approved ${last.sha.slice(0, 7)}, which is not in the history of approved_sha ${approved.slice(0, 7)}`)); continue; }
421
476
  if (previous && (await git(absolute, ["merge-base", "--is-ancestor", previous.sha, last.sha])).code) findings.push(finding("delivery.approved-sha-resolves", entry.file, `${stage} approved ${last.sha.slice(0, 7)}, which does not contain what ${previous.stage} approved (${previous.sha.slice(0, 7)}); stages follow the lifecycle`));
422
477
  previous = { stage, sha: last.sha };
423
478
  }
424
- // The closing review does not replace the stage before it: code that changed after that stage approved must go
425
- // back through it (driver step 5), so nothing outside the delivery root may differ between the two.
426
- // Anchored on code_review, the stage that approves code: not the position, which would ask plan_review to approve
427
- // the final code under a reduced stage set.
428
- const configured = STAGES.filter(name => policy.stages[name]), closing = configured.at(-1);
429
- const before: Stage | undefined = configured.includes("code_review") && closing !== "code_review" ? "code_review" : undefined;
430
- const earlier = before && entry.data.reviews.filter(review => review.stage === before).at(-1);
431
- if (earlier && earlier.verdict === "approve" && closing && !(await git(absolute, ["cat-file", "-e", `${earlier.sha}^{commit}`])).code) {
432
- const changedSince = (await git(absolute, ["diff", "--no-renames", "--name-only", earlier.sha, entry.data.approved_sha!, "--", ":(top)", `:(exclude)${policy.root}`])).out.split("\n").filter(Boolean);
433
- if (changedSince.length) findings.push(finding("delivery.stages-complete", entry.file, `${changedSince.slice(0, 5).join(", ")}${changedSince.length > 5 ? ` and ${changedSince.length - 5} more` : ""} changed after ${before} approved ${earlier.sha.slice(0, 7)}, and only ${closing} reviewed the change; a code change goes back through ${before}`));
434
- }
435
479
  }
436
480
  }
437
481
  if (mode) findings.push(...await mergeGate(absolute, policy, files, mode, options.base));
@@ -470,7 +514,10 @@ async function mergeGate(root: string, policy: DeliveryPolicy, files: EvidenceFi
470
514
  const named = new Set(changed.flatMap(path => namedReports(files.find(file => file.file === path)?.data)));
471
515
  for (const path of changedHere.filter(path => path.startsWith(`${policy.root}/`) && !isRecord(path) && !named.has(path)))
472
516
  out.push(finding("delivery.merge-gate", path, `${mode === "merge" ? "the branch" : "this push"} changes ${path} under ${policy.root}/, and no record it changes names it as a report; only <tranche>/evidence.yaml records and their reports live there`));
473
- if (policy.require_record && outside.length && !changed.length) out.push(finding("delivery.merge-gate", "atdd-bun.yaml", `${mode === "merge" ? "the branch" : "this push"} changes ${outside.slice(0, 5).join(", ")}${outside.length > 5 ? ` and ${outside.length - 5} more` : ""} with no tranche record under ${policy.root}/; every change merges through a reviewed tranche (delivery.require_record)`));
517
+ // A change to the policy alone needs no tranche: the integrity check reports any loosening for a human to approve,
518
+ // and a tightening needs no review. A policy change that comes with anything else is reviewed with it.
519
+ const reviewable = outside.filter(path => path !== "atdd-bun.yaml");
520
+ if (policy.require_record && reviewable.length && !changed.length) out.push(finding("delivery.merge-gate", "atdd-bun.yaml", `${mode === "merge" ? "the branch" : "this push"} changes ${outside.slice(0, 5).join(", ")}${outside.length > 5 ? ` and ${outside.length - 5} more` : ""} with no tranche record under ${policy.root}/; every change merges through a reviewed tranche (delivery.require_record)`));
474
521
  // Every changed ready record, its reports, and the approved SHAs that may cover each other's files.
475
522
  const readyChanged = changed.map(path => files.find(file => file.file === path)!).filter(entry => entry.data?.status === "ready");
476
523
  const siblings = readyChanged.map(entry => entry.data!.approved_sha!).filter(Boolean);
package/src/hooks.ts CHANGED
@@ -3,6 +3,7 @@ import { existsSync } from "node:fs";
3
3
  import { dirname, join, relative, resolve } from "node:path";
4
4
  import { enabledProfiles, enforce } from "./enforce";
5
5
  import { topologyFor } from "./topology";
6
+ import { JOURNEY_DOCS_DIR } from "./journey-docs";
6
7
 
7
8
  export type WorktreePolicy = { enabled: boolean; root: string; primary_directory: string; primary_branch: string; require_linked_worktree: boolean };
8
9
  export type HookPolicy = { max_staged_files: number; max_staged_changed_lines: number; max_uncommitted_files: number; max_commits_per_push: number; max_registry_removed_lines: number; registry_paths: string[]; protected_branches: string[]; require_plan_reference: boolean; require_traceability: boolean; worktrees: WorktreePolicy };
@@ -20,10 +21,14 @@ const isRegistry = (cfg: HookPolicy, path: string) => cfg.registry_paths.some(pa
20
21
  const renamedTo = (path: string) => { const braced = path.match(/^(.*)\{(.*) => (.*)\}(.*)$/); return braced ? `${braced[1]}${braced[3]}${braced[4]}`.replace(/\/{2,}/g, "/") : path.includes(" => ") ? path.split(" => ")[1]! : path; };
21
22
  /** Lines a commit changes, for the size cap: a moved file counts only the edits it carries, not its whole body twice. */
22
23
  const stagedLineStats = async (root: string) => (await git(root, ["diff", "--cached", "--numstat", "-M"])).out.split("\n").filter(Boolean).map(row => { const [added, removed, ...path] = row.split("\t"); return { added: Number(added) || 0, removed: Number(removed) || 0, path: renamedTo(path.join("\t")) }; });
24
+ /** The journey view atdd-bun generates. Like a registry it is exempt from the size caps, since its index alone can
25
+ * exceed them and it cannot be split; unlike hand-written docs it is never exempt from the docs profile, which refuses
26
+ * a copy that is not byte-for-byte what the plan generates. */
27
+ const isGeneratedView = (path: string) => path.startsWith(`${JOURNEY_DOCS_DIR}/`);
23
28
  const stagedStats = async (root: string) => (await git(root, ["diff", "--cached", "--numstat", "--no-renames"])).out.split("\n").filter(Boolean).map(row => { const [added, removed, ...path] = row.split("\t"); return { added: Number(added) || 0, removed: Number(removed) || 0, path: path.join("\t") }; });
24
29
 
25
30
  export async function policy(root: string): Promise<HookPolicy> { const file = join(root, "atdd-bun.yaml"); if (!existsSync(file)) return defaultHookPolicy; const data = Bun.YAML.parse(await readFile(file, "utf8")) as Record<string, unknown>; const configuredWorktrees = data?.worktrees && typeof data.worktrees === "object" && !Array.isArray(data.worktrees) ? data.worktrees as Record<string, unknown> : {}; return { ...defaultHookPolicy, ...Object.fromEntries(Object.entries(data ?? {}).filter(([key, value]) => key in defaultHookPolicy && key !== "worktrees" && typeof value === typeof (defaultHookPolicy as any)[key])), protected_branches: strings(data?.protected_branches) ?? defaultHookPolicy.protected_branches, registry_paths: strings(data?.registry_paths) ?? defaultHookPolicy.registry_paths, worktrees: { ...defaultWorktreePolicy, ...Object.fromEntries(Object.entries(configuredWorktrees).filter(([key, value]) => key in defaultWorktreePolicy && typeof value === typeof (defaultWorktreePolicy as any)[key])) } }; }
26
- async function validation(root: string, changed: string[], full: boolean, registryChanged = false) { const profiles: any[] = [], topology = await topologyFor(root); if (registryChanged || changed.some(path => path.startsWith(`${topology.planRoot}/`)) || changed.includes("atdd-bun.yaml")) profiles.push("planner", "traceability"); if (changed.some(path => /\.(?:[cm]?[jt]sx?|html)$/.test(path))) profiles.push("coder", "tester"); if (changed.includes("atdd-bun.yaml") || changed.some(path => /\.ya?ml$/.test(path))) profiles.push("topology"); if (!profiles.length) return { ok: true, message: "no affected area" }; try { const enabled = await enabledProfiles(root), selected = full ? ["all" as const] : [...new Set(profiles)].filter(profile => enabled.includes(profile)); if (!selected.length) return { ok: true, message: "no activated profile covers the affected area" }; const findings = await enforce({ root, profiles: selected }); return findings.length ? bad(findings.map(f => `${f.rule_id}: ${f.file}`).join("\n")) : { ok: true, message: "validation clean" }; } catch (error) { return bad(`Bun enforcer resolution failed: ${String(error)}`); } }
31
+ async function validation(root: string, changed: string[], full: boolean, registryChanged = false) { const profiles: any[] = [], topology = await topologyFor(root); if (registryChanged || changed.some(path => path.startsWith(`${topology.planRoot}/`)) || changed.includes("atdd-bun.yaml")) profiles.push("planner", "traceability"); if (changed.some(path => /\.(?:[cm]?[jt]sx?|html)$/.test(path))) profiles.push("coder", "tester"); if (changed.includes("atdd-bun.yaml") || changed.some(path => /\.ya?ml$/.test(path))) profiles.push("topology"); if (changed.some(isGeneratedView)) profiles.push("docs"); if (!profiles.length) return { ok: true, message: "no affected area" }; try { const enabled = await enabledProfiles(root), selected = full ? ["all" as const] : [...new Set(profiles)].filter(profile => enabled.includes(profile)); if (!selected.length) return { ok: true, message: "no activated profile covers the affected area" }; const findings = await enforce({ root, profiles: selected }); return findings.length ? bad(findings.map(f => `${f.rule_id}: ${f.file}`).join("\n")) : { ok: true, message: "validation clean" }; } catch (error) { return bad(`Bun enforcer resolution failed: ${String(error)}`); } }
27
32
 
28
33
  const inside = (parent: string, child: string) => { const path = relative(parent, child); return path === "" || (path !== ".." && !path.startsWith("../") && !path.startsWith("..\\")); };
29
34
  async function primaryRoot(root: string): Promise<string | null> { const result = await git(root, ["rev-parse", "--git-common-dir"]); if (result.code) return null; return dirname(resolve(root, result.out)); }
@@ -35,7 +40,7 @@ export async function runHook(event: HookEvent, root = process.cwd(), args: stri
35
40
  const cfg = await policy(root), branch = (await git(root, ["symbolic-ref", "--quiet", "--short", "HEAD"])).out;
36
41
  if (["pre-commit", "pre-merge-commit"].includes(event) && cfg.protected_branches.includes(branch)) return bad(`protected branch ${branch} is blocked`);
37
42
  if (["pre-commit", "pre-merge-commit"].includes(event)) { const violation = await worktreeCommitPolicy(root, cfg.worktrees); if (violation) return bad(violation); }
38
- if (event === "pre-commit") { const staged = await files(root, ["diff", "--cached", "--name-only"]), dirty = [...new Set([...await files(root, ["diff", "--name-only"]), ...await files(root, ["ls-files", "--others", "--exclude-standard"])])], registries = staged.filter(path => isRegistry(cfg, path)), counted = staged.length - registries.length, lines = (await stagedLineStats(root)).filter(row => !isRegistry(cfg, row.path)).reduce((n, row) => n + row.added + row.removed, 0); if (counted > cfg.max_staged_files) return bad(`staged files ${counted} exceed ${cfg.max_staged_files}`); if (dirty.length > cfg.max_uncommitted_files) return bad(`unstaged or untracked files ${dirty.length} exceed ${cfg.max_uncommitted_files}; the staged commit itself is not counted`); if (lines > cfg.max_staged_changed_lines) return bad(`staged changed lines ${lines} exceed ${cfg.max_staged_changed_lines}`); return cfg.require_traceability || registries.length ? validation(root, staged, false, registries.length > 0) : { ok: true, message: "ok" }; }
43
+ if (event === "pre-commit") { const staged = await files(root, ["diff", "--cached", "--name-only"]), dirty = [...new Set([...await files(root, ["diff", "--name-only"]), ...await files(root, ["ls-files", "--others", "--exclude-standard"])])], registries = staged.filter(path => isRegistry(cfg, path)), generated = staged.filter(isGeneratedView), counted = staged.length - registries.length - generated.length, lines = (await stagedLineStats(root)).filter(row => !isRegistry(cfg, row.path) && !isGeneratedView(row.path)).reduce((n, row) => n + row.added + row.removed, 0); if (counted > cfg.max_staged_files) return bad(`staged files ${counted} exceed ${cfg.max_staged_files}`); if (dirty.length > cfg.max_uncommitted_files) return bad(`unstaged or untracked files ${dirty.length} exceed ${cfg.max_uncommitted_files}; the staged commit itself is not counted`); if (lines > cfg.max_staged_changed_lines) return bad(`staged changed lines ${lines} exceed ${cfg.max_staged_changed_lines}`); return cfg.require_traceability || registries.length || generated.length ? validation(root, staged, false, registries.length > 0) : { ok: true, message: "ok" }; }
39
44
  if (event === "commit-msg") { const deleted = (await files(root, ["diff", "--cached", "--name-only", "--diff-filter=D"])).length, stats = await stagedStats(root), lines = stats.reduce((n, row) => n + row.removed, 0), registryRemoved = stats.filter(row => isRegistry(cfg, row.path)).reduce((n, row) => n + row.removed - row.added, 0), message = args[0] && existsSync(args[0]) ? await readFile(args[0], "utf8") : "", approved = message.includes("[mass-delete-approved]"); if ((deleted > 50 || lines > 10_000) && !approved) return bad("mass delete requires [mass-delete-approved]"); return registryRemoved > cfg.max_registry_removed_lines && !approved ? bad(`registry removal of ${registryRemoved} net lines exceeds ${cfg.max_registry_removed_lines}; requires [mass-delete-approved]`) : { ok: true, message: "ok" }; }
40
45
  if (event === "pre-push") { for (const row of stdin.split("\n").filter(Boolean).map(row => row.split(/\s+/))) { const [,,remote, remoteSha] = row, target = remote?.replace("refs/heads/", ""); if (target && cfg.protected_branches.includes(target)) return bad(`protected branch ${target} is blocked`); const local = row[1]; if (local && !/^0+$/.test(local)) { const range = !remoteSha || /^0+$/.test(remoteSha) ? `${local}^..${local}` : `${remoteSha}..${local}`, count = Number((await git(root, ["rev-list", "--count", range])).out); if (count > cfg.max_commits_per_push) return bad(`commits per push ${count} exceed ${cfg.max_commits_per_push}`); } } return validation(root, await files(root, ["diff", "--name-only", "HEAD~1..HEAD"]), true); }
41
46
  if (event === "post-commit") { const result = await validation(root, await files(root, ["show", "--pretty=format:", "--name-only", "HEAD"]), false); return { ok: true, message: result.ok ? result.message : `advisory: ${result.message}` }; }
@@ -1,77 +1,15 @@
1
1
  ---
2
2
  name: delivery
3
- description: Use when a program is delivered as tranches by a coordinator and persistent drivers, with headless authors and independent reviewers. Covers activation, dispatch, model fallback, the four reviews, disputes and the evidence record the `delivery` profile checks in CI.
3
+ description: Use when a program is delivered as tranches by a coordinator and persistent drivers, with headless writers and independent reviewers. Points to the delivery policy and the conventions that say who writes and reviews each stage, how work is dispatched, and what the evidence record must hold.
4
4
  ---
5
5
  <!-- Generated by @afokapu/atdd-bun {{VERSION}}. Do not edit; run `atdd-bun agent init --replace`. -->
6
6
 
7
- The policy is the `delivery:` block of `atdd-bun.yaml`: per stage, the allowed authors and reviewers in preference order, the independence mode, and the fallback thresholds. Read it; do not hard-code models. The rules are `conventions/delivery/*.convention.yaml` in `node_modules/@afokapu/atdd-bun/`. Gate: `bun run atdd-bun delivery`.
7
+ Read these before acting; they are the operating model, this skill only points to them. All live in `node_modules/@afokapu/atdd-bun/`.
8
8
 
9
- A **tranche** is one independently mergeable piece of the program, on its own branch and worktree, owned end to end by one persistent **driver**. Each tranche runs the ATDD lifecycle (`.agents/skills/atdd/SKILL.md`) with four reviews: `plan_review` after PLAN, `test_review` after RED, `code_review` after GREEN → SMOKE → REFACTOR → TRACE, and `final_review` of the PR head.
9
+ - **Who writes and who reviews each stage** (plan, red, green, refactor, final): the `delivery:` block of `atdd-bun.yaml`. Read it; never hard-code models.
10
+ - **Coordinator, driver, dispatch, fallback, repair, merge, default commands, an example record:** `conventions/delivery/delivery.operating-model.convention.yaml`.
11
+ - **What a reviewer checks and how it answers:** `conventions/delivery/delivery.review.convention.yaml`.
12
+ - **When `delivery.board` is set in `atdd-bun.yaml`, how agents talk (topics, identities, `atdd-bun chat`):** `conventions/delivery/delivery.board.convention.yaml`. Without it there is no board.
13
+ - **What the record must satisfy:** the other `conventions/delivery/*.convention.yaml` rules.
10
14
 
11
- ## Coordinator
12
-
13
- Its job is throughput: every worker slot busy, every tranche moving. It never implements, repairs tests, reviews, or merges a tranche.
14
-
15
- 1. Split the program into tranches with explicit dependencies, and write why the program exists, its scope and how it was split in `docs/delivery/index.adoc` (a docs-profile document; where the docs profile is active, `docs/index.adoc` is needed too, each with `:doc-id:` and `:status:`). Activate a tranche as soon as its own dependencies have merged; do not wait for a whole wave. A tranche whose dependencies are still open may run PLAN and `plan_review` but nothing after; revalidate its plan once they merge.
16
- 2. For each active tranche, create a worktree from the owning repository's workspace and start a driver in it through the multiplexer (see Multiplexer). Send the mandate, submit it, wait 4–6 s, and read the pane: a working indicator or agent output means it landed; an empty prompt or placeholder means retry before waiting on anything.
17
- 3. Wait on the multiplexer's events, not polling loops, and on every driver at once. When a slot frees, give it to the next ready tranche or to planning ahead.
18
- 4. Keep provider health for the whole program. When a driver reports a model unavailable, tell every driver to go straight to the next model in its lists until it recovers, so no tranche spends time rediscovering an outage.
19
- 5. Intervene when a tranche is not BLOCKED yet no worker has run for a while, or when two hours pass with deliverable-shaped changes and no commit: tell the driver to commit, push and open its PR.
20
- 6. A `BLOCKED disputed-finding` goes to the human with both sides. `BLOCKED provider-unavailable` frees the slot for other work.
21
-
22
- ## Multiplexer
23
-
24
- Agents run in panes of the terminal multiplexer named in `delivery.multiplexer` (default `herdr`); say which one you are using when you start. Do not assume its commands: before the first dispatch, read its own help (`<multiplexer> --help`, then `<multiplexer> <group> --help`, and any schema it publishes) and map each operation below to a command. If one is missing, report it instead of scripting around it.
25
-
26
- | Operation | herdr |
27
- |---|---|
28
- | create a worktree under the owning repository's workspace | `herdr worktree create --workspace <repo-ws> --branch <b> --base <sha> --path <p> --label <tranche> --no-focus` |
29
- | start an agent in a pane with a working directory | `herdr agent start <name> --cwd <worktree> --workspace <ws> --no-focus -- codex` |
30
- | send text, then submit it | `herdr agent send <name> "<text>"`, then `herdr pane send-keys <pane> Enter` (send does not press Enter) |
31
- | read recent output | `herdr agent read <pane> --source recent-unwrapped --lines 12` |
32
- | wait for one pane's status or output | `herdr wait agent-status <pane> --status idle`, `herdr wait output <pane> --match "PROGRAM_EVENT" --regex` |
33
- | wait on every pane at once | the socket API's `events.subscribe` (`herdr api schema --json`): `pane_agent_status_changed`, `pane_exited`, `pane_output_changed` |
34
- | inspect a pane's process | `herdr pane process-info --pane <pane>` |
35
-
36
- ## Driver
37
-
38
- 1. Run each stage's author and each review as a separate headless process, a review in its own pane so its output is retained. Build the command from `delivery.commands` or the defaults below. An author runs in the tranche worktree.
39
- 2. Run every review in its own detached worktree at the exact SHA (`git worktree add --detach <path> <sha>`), never in the tranche worktree, and give it `.agents/skills/delivery/review.md`, the stage and the SHA. The reviewer only reads and proposes. Tool allowlists cannot make a CLI fully read-only (`git diff --output=` writes), so isolation does: afterwards `git -C <path> status --porcelain` must be empty and HEAD still the SHA, or the review is a `REVIEWER_FAILURE` and does not count. Remove the worktree after retaining the report.
40
- 3. Fallback: after `fallback.after_failures` failures within `fallback.within_minutes` (outage, rate limit, no auditable report), use the next model in the stage's list and record it with its `kind` (`outage`, `rate_limit`, `no_report`, `timeout`), `failures`, and the `window` from the first to the last counted failure. REQUEST CHANGES is never a failure. With the list exhausted, `when_exhausted: block` emits `BLOCKED provider-unavailable`; `wait` keeps retrying the last model.
41
- 4. For each finding of a REQUEST CHANGES review, either have the author fix it, or write one rebuttal with evidence (a test result, a rule id, file:line). Then run a fresh review of the same stage. If that reviewer upholds a disputed finding, emit `BLOCKED disputed-finding`; never dispute it a second time.
42
- 5. Any commit, regenerated file, conflict fix or rebase after an approval cancels it. A change to the code goes back through `code_review`, then `final_review`.
43
- 6. Append every review to `<root>/<tranche>/evidence.yaml` (`delivery.root`, default `docs/delivery/tranches`) as it happens; never edit an earlier entry, only add each finding's `outcome` (`fixed`, `withdrawn`, or `human` with the `decision`). Keep the raw reviewer output inside the tranche's folder and name it in `report`; nothing else goes in that folder. When `final_review` approves, set `status: ready` and `approved_sha` to that SHA, commit the evidence and reports alone, push, and merge with a merge commit once CI is green. A squash or rebase merge writes a commit no reviewer saw, and CI fails it after the merge.
44
- 7. Emit events the coordinator can wait on, one line each: `PROGRAM_EVENT <tranche> <PLAN|RED|COMMIT <sha>|WORKER_START <role> <model> <sha>|WORKER_END <role> <model> <verdict>|FALLBACK <role> <from>→<to> <reason>|PLAN_REVIEW <sha>|TEST_REVIEW <sha>|CODE_REVIEW <sha>|FINAL_REVIEW <sha>|PR_OPENED <url>|MERGED <sha>|BLOCKED <reason>|HEARTBEAT>`.
45
-
46
- ```yaml
47
- # docs/delivery/tranches/<tranche>/evidence.yaml
48
- tranche: api
49
- status: open # ready once final_review approves
50
- base_sha: 3f2a91c
51
- approved_sha: c77d0a2 # required when ready
52
- reviews:
53
- - stage: code_review
54
- sha: a4c0f11
55
- author: { model: glm, run: glm-green-1 }
56
- reviewer: { model: claude, run: claude-code-1 }
57
- fallback: [{ role: reviewer, from: glm, kind: rate_limit, failures: 3, window: { from: "2026-09-25T09:00:00Z", to: "2026-09-25T09:08:00Z" }, reason: "429 from the provider on 3 attempts in 10 minutes" }]
58
- verdict: request_changes
59
- checked: [ACC-API-001, src/wagons/api, coder.bun.error-response-*]
60
- findings:
61
- - { id: F1, severity: high, evidence: "src/wagons/api/handler.ts:42", invariant: "coded error bodies", affects: [ui], proposed_fix: "return { code: 'API_NOT_FOUND' }" } # outcome added once a fresh code_review confirms the fix
62
- report: docs/delivery/tranches/api/code_review-1.json
63
- ```
64
-
65
- ## Default commands
66
-
67
- `{prompt}` and `{worktree}` are substituted. Override per model under `delivery.commands.<model>.author` / `.review` when a CLI changes.
68
-
69
- | Model | Author | Review (read-only) |
70
- |---|---|---|
71
- | glm | `zcode -p="{prompt}" --cwd {worktree} --mode edit` | `zcode -p="{prompt}" --cwd {worktree} --mode plan` |
72
- | claude | `cd {worktree} && claude -p "{prompt}" --permission-mode acceptEdits --allowedTools "Bash(bun:*)" "Bash(git:*)" "Bash(gh pr:*)" --output-format json` | `cd {worktree} && claude -p "{prompt}" --allowedTools Read Grep Glob "Bash(git show:*)" "Bash(git diff:*)" "Bash(git log:*)" "Bash(bun test:*)" "Bash(bun run atdd-bun all)" "Bash(bun run atdd-bun planner)" "Bash(bun run atdd-bun tester)" "Bash(bun run atdd-bun coder security)" "Bash(bun run atdd-bun traceability)" "Bash(bun run atdd-bun delivery)" --disallowedTools Edit Write NotebookEdit --output-format json` |
73
- | codex | `codex exec --cd {worktree} --sandbox workspace-write "{prompt}"` | `codex exec --cd {worktree} --sandbox read-only "{prompt}"` |
74
-
75
- Before a hosted model receives private repository content, confirm the user or organization authorized it.
76
-
77
- Never edit this skill, `review.md`, the conventions, or loosen the `delivery:` policy to get a tranche through. If the policy must change, stop and ask the human.
15
+ Gate: `bun run atdd-bun delivery`. Never edit this skill or the conventions, and never loosen the policy to get a tranche through; if the policy must change, stop and ask the human.
@@ -1,45 +0,0 @@
1
- <!-- Generated by @afokapu/atdd-bun {{VERSION}}. Do not edit; run `atdd-bun agent init --replace`. -->
2
- # Review contract
3
-
4
- You are an independent reviewer for one stage of one tranche. The driver gave you the stage and the exact SHA.
5
-
6
- **Read only.** You run in a detached worktree at the SHA under review. Do not edit, commit, or run anything that writes to it (including output-file options such as `git diff --output=`); the driver checks it is untouched afterwards. You may read files, search, use `git show`/`git diff`/`git log`, and run the gates (`bun test`, `bun run atdd-bun …`). You judge and propose; the author applies. If you edit, your review does not count.
7
-
8
- ## How to review
9
-
10
- Apply all three, every time:
11
-
12
- - **Systematic.** Work through the stage's checklist below completely. List everything you checked in `checked`, not only what failed, so a skipped area is visible.
13
- - **Systemic.** Look beyond the diff: contracts, other tranches, downstream owners, invariants the change could break elsewhere. Name them in each finding's `affects`.
14
- - **Adversarial.** Assume the work is wrong and try to prove it: inputs that fail, tests that pass for the wrong reason, ways around a gate, rules satisfied in letter only. A finding needs concrete evidence (file:line, a failing command, a counter-example); "might be an issue" is not a finding.
15
-
16
- ## Checklist by stage
17
-
18
- - `plan_review`: the decomposition (wagon → WMBT → acceptance → train → journey → contract) covers the intent; every acceptance is testable and has one observable outcome; every WMBT has a SMOKE acceptance; dependencies and owned files do not overlap other tranches; `bun run atdd-bun planner` passes.
19
- - `test_review`: every acceptance has a RED test bound to it by URN; each test fails for the missing behaviour and would still fail for a wrong implementation; no test asserts on mocks where the acceptance names observable output; `bun run atdd-bun tester` passes.
20
- - `code_review`: every behaviour is correct against its acceptance, including edge and error paths; layering, composition, DTO and error-response rules hold; no security fault; SMOKE runs through the real entry point; `bun test` and `bun run atdd-bun all` pass.
21
- - `final_review`: the whole PR at its head SHA, read as an architecture: coherent with the plan, no drift from what the earlier stages approved, no change a stage did not review, safe for every downstream owner.
22
-
23
- ## Disputes
24
-
25
- If a finding carries the author's `rebuttal`, judge it against the evidence. Withdraw it (leave it out of your findings) if the rebuttal holds. Uphold it (repeat it with the same `id`) if not. You settle it; there is no second round.
26
-
27
- ## Output
28
-
29
- Return exactly one YAML document, nothing else. The driver appends it to the tranche's evidence record.
30
-
31
- ```yaml
32
- stage: code_review # the stage you were given
33
- sha: c77d0a2 # the SHA you reviewed
34
- verdict: request_changes # or approve (approve only with no critical or high finding)
35
- checked: [ACC-API-001, ACC-API-002, src/wagons/api, coder.bun.error-response-*]
36
- findings:
37
- - id: F1 # keep an upheld finding's id
38
- severity: high # critical | high | medium | low
39
- evidence: src/wagons/api/handler.ts:42 returns a bare string
40
- invariant: every error response carries a coded body
41
- affects: [ui]
42
- proposed_fix: return { code "API_NOT_FOUND" } instead of the bare string
43
- ```
44
-
45
- Keep `proposed_fix` to a precise description or a short snippet, never a rewrite.