@selesai/code 0.3.2 → 0.3.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/dist/config.d.ts +1 -1
  2. package/dist/config.d.ts.map +1 -1
  3. package/dist/config.js +24 -1
  4. package/dist/config.js.map +1 -1
  5. package/dist/core/git-command.d.ts +9 -0
  6. package/dist/core/git-command.d.ts.map +1 -0
  7. package/dist/core/git-command.js +49 -0
  8. package/dist/core/git-command.js.map +1 -0
  9. package/dist/core/slash-commands.d.ts.map +1 -1
  10. package/dist/core/slash-commands.js +1 -0
  11. package/dist/core/slash-commands.js.map +1 -1
  12. package/dist/defaults/models.json +18 -0
  13. package/dist/extensions/handoff-new.test.ts +210 -0
  14. package/dist/extensions/handoff-new.ts +172 -0
  15. package/dist/extensions/node_modules/.vite/vitest/da39a3ee5e6b4b0d3255bfef95601890afd80709/results.json +1 -0
  16. package/dist/extensions/package.json +1 -0
  17. package/dist/extensions/test-resolve-hook-impl.mjs +7 -0
  18. package/dist/extensions/test-resolve-hook.mjs +7 -0
  19. package/dist/extensions/workflow/adapter.ts +162 -6
  20. package/dist/extensions/workflow/modes/prototype.ts +42 -17
  21. package/dist/extensions/workflow/modes/quick.ts +40 -17
  22. package/dist/extensions/workflow/state-machine.ts +117 -15
  23. package/dist/extensions/workflow/validators.ts +42 -0
  24. package/dist/modes/interactive/interactive-mode.d.ts +1 -0
  25. package/dist/modes/interactive/interactive-mode.d.ts.map +1 -1
  26. package/dist/modes/interactive/interactive-mode.js +31 -0
  27. package/dist/modes/interactive/interactive-mode.js.map +1 -1
  28. package/dist/package-manager-cli.d.ts.map +1 -1
  29. package/dist/package-manager-cli.js +23 -8
  30. package/dist/package-manager-cli.js.map +1 -1
  31. package/dist/skills/improve-codebase/SKILL.md +2 -2
  32. package/dist/utils/version-check.d.ts +3 -0
  33. package/dist/utils/version-check.d.ts.map +1 -1
  34. package/dist/utils/version-check.js +19 -0
  35. package/dist/utils/version-check.js.map +1 -1
  36. package/docs/workflows.md +21 -24
  37. package/package.json +5 -5
@@ -11,7 +11,7 @@ import type {
11
11
  } from "@selesai/code";
12
12
  import { Text } from "@earendil-works/pi-tui";
13
13
  import { Type } from "typebox";
14
- import { access, mkdir, writeFile } from "node:fs/promises";
14
+ import { access, mkdir, readFile, writeFile } from "node:fs/promises";
15
15
  import { isAbsolute, resolve } from "node:path";
16
16
 
17
17
  import {
@@ -23,6 +23,7 @@ import {
23
23
  type FooterState,
24
24
  type Phase,
25
25
  } from "./state-machine.ts";
26
+ import { MARKERS } from "./validators.ts";
26
27
 
27
28
  function slugify(s: string): string {
28
29
  const slug = s
@@ -59,6 +60,9 @@ async function realFileExists(path: string): Promise<boolean> {
59
60
  // ponytail: phases where one spawned subagent owns the phase artifact. Force
60
61
  // the child `output` path; child processes do not share the parent's workflow
61
62
  // state, so the parent-scoped write_workflow_artifact tool cannot help there.
63
+ // Plan 5: these are also the single-owner phases — parallel/chain calls are
64
+ // blocked here because one agent must own plan.md / reuse.md / handoff.md /
65
+ // review.md. Loop is engine-owned (Plan 3) and may fan out.
62
66
  const FORCE_OUTPUT_PHASES = new Set<Phase>(["plan", "reuse", "handoff", "audit"]);
63
67
 
64
68
  // ponytail: phases where the parent can save subagent text as the artifact
@@ -143,6 +147,22 @@ interface WorkflowController {
143
147
  config: WorkflowConfig;
144
148
  sm: WorkflowStateMachine;
145
149
  deps: WorkflowDeps;
150
+ // ponytail: Plan 3 — engine-owned loop state. Lives on the controller, not
151
+ // the state machine: the SM only cares "does loop-complete.md exist →
152
+ // advance"; the adapter owns iteration counting + marker parsing + driving
153
+ // the next subagent call. Cleared when the phase leaves "loop".
154
+ // Ceiling: not persisted across session restart — a restart re-runs the
155
+ // loop from round 0. Acceptable for a prototype workflow.
156
+ loopState?: LoopState;
157
+ }
158
+
159
+ // ponytail: Plan 3 — loop orchestration state. The parent model does not
160
+ // remember iterations; the adapter counts review rounds, parses the
161
+ // commentator's WORKFLOW_REVIEW_STATUS marker, and drives the next step.
162
+ interface LoopState {
163
+ reviewRound: number;
164
+ maxIterations: number;
165
+ stage: "building" | "reviewing" | "clean" | "maxed";
146
166
  }
147
167
 
148
168
  const WORKFLOW_ARTIFACT_TOOL = "write_workflow_artifact";
@@ -183,6 +203,25 @@ function makeDeps(pi: ExtensionAPI, config: WorkflowConfig): WorkflowDeps {
183
203
  await mkdir(path, { recursive: true });
184
204
  },
185
205
  artifactPathFor: defaultArtifactPathFor,
206
+ // ponytail: Plan 4 — read an artifact for semantic validation. Returns
207
+ // undefined on missing/unreadable so the SM treats it as "not written"
208
+ // rather than an empty-string that would always fail the validator.
209
+ async readArtifact(phase, dir) {
210
+ const file = config.phaseArtifacts[phase];
211
+ if (!file) return undefined;
212
+ try {
213
+ return await readFile(`${dir}/${file}`, "utf8");
214
+ } catch {
215
+ return undefined;
216
+ }
217
+ },
218
+ async readFile(path) {
219
+ try {
220
+ return await readFile(path, "utf8");
221
+ } catch {
222
+ return undefined;
223
+ }
224
+ },
186
225
  };
187
226
  }
188
227
 
@@ -197,6 +236,18 @@ async function isEmptyProject(pi: ExtensionAPI): Promise<boolean> {
197
236
  }
198
237
  }
199
238
 
239
+ // ponytail: Plan 3 — parse the commentator's machine-readable status line.
240
+ // First match wins; case-insensitive; trailing whitespace allowed. Returns
241
+ // undefined when the marker is absent — the loop treats that as blocking so
242
+ // a malformed review never silently advances the workflow.
243
+ const LOOP_STATUS_RE = /WORKFLOW_REVIEW_STATUS\s*:\s*(clean|blocking)\b/i;
244
+ function parseLoopReviewStatus(text: string | undefined): "clean" | "blocking" | undefined {
245
+ if (!text) return undefined;
246
+ const m = text.match(LOOP_STATUS_RE);
247
+ if (!m) return undefined;
248
+ return m[1]!.toLowerCase() as "clean" | "blocking";
249
+ }
250
+
200
251
  function footerText(footer: FooterState, ctx: ExtensionContext): string | undefined {
201
252
  if (!footer.visible) return undefined;
202
253
  return ctx.ui.theme.fg("warning", footer.text);
@@ -250,6 +301,13 @@ function applyEffect(
250
301
  if (eff.promptToQueue) continueAgent(pi, ctx, eff.promptToQueue);
251
302
  return;
252
303
  case "blocked":
304
+ // ponytail: Plan 4 — surface the semantic reason so a subagent-driven
305
+ // phase (plan/handoff/audit) learns which marker is missing instead of
306
+ // the workflow appearing to stall after the artifact was written.
307
+ if (eff.reason) {
308
+ continueAgent(pi, ctx, `Phase ${eff.phase} artifact exists but is not approved: ${eff.reason}. Edit it via write_workflow_artifact to add the required marker, then the workflow will advance.`);
309
+ }
310
+ return;
253
311
  case "terminalReady":
254
312
  case "endBlocked":
255
313
  case "idle":
@@ -296,14 +354,20 @@ function registerSharedArtifactWriter(pi: ExtensionAPI): void {
296
354
  await writeFile(path, params.content, "utf8");
297
355
  const eff = await controller.sm.onArtifactMaybe(controller.deps);
298
356
  applyEffect(pi, ctx, controller.config, eff);
357
+ if (eff.kind === "blocked" && eff.reason) {
358
+ return {
359
+ content: [{ type: "text", text: `Wrote ${path}, but it is not approved: ${eff.reason}. Re-write it via write_workflow_artifact to add the required marker.` }],
360
+ details: { mode: controller.config.mode, phase: snap.phase, path, file, blocked: true, reason: eff.reason },
361
+ };
362
+ }
299
363
  return {
300
364
  content: [{ type: "text", text: `Wrote ${path}.` }],
301
365
  details: { mode: controller.config.mode, phase: snap.phase, path, file },
302
366
  };
303
367
  },
304
368
  renderResult(result, _options, theme) {
305
- const d = result.details as { path?: string; blocked?: boolean };
306
- if (d.blocked) return new Text(theme.fg("warning", "○ workflow artifact not written"), 0, 0);
369
+ const d = result.details as { path?: string; blocked?: boolean; reason?: string };
370
+ if (d.blocked) return new Text(theme.fg("warning", `○ workflow artifact not approved: ${d.reason ?? "missing marker"}`), 0, 0);
307
371
  return new Text(theme.fg(d.path ? "success" : "warning", d.path ? `✓ wrote ${d.path}` : "○ no active workflow"), 0, 0);
308
372
  },
309
373
  } satisfies ToolDefinition);
@@ -404,9 +468,16 @@ export function createWorkflowExtension(
404
468
  terminate: true,
405
469
  };
406
470
  case "endBlocked": {
407
- // ponytail: missing is either a filename (closeArtifacts loop)
408
- // or a phase description like "not at terminal (audit)". Only
409
- // suggest writing when it's an actual artifact file.
471
+ // ponytail: missing is either a filename (existence failure),
472
+ // a phase description like "not at terminal (audit)", or — Plan 4 —
473
+ // a filename whose validator failed (reason set). Only suggest
474
+ // writing when it's an actual artifact file existence miss.
475
+ if (eff.reason) {
476
+ return {
477
+ content: [{ type: "text", text: `Cannot end: ${eff.missing} is incomplete: ${eff.reason}.` }],
478
+ details: { phase: eff.phase, blocked: eff.missing, reason: eff.reason },
479
+ };
480
+ }
410
481
  const isArtifactFile =
411
482
  !eff.missing.includes(" ") && eff.missing.includes(".");
412
483
  const hint = isArtifactFile
@@ -483,6 +554,7 @@ export function createWorkflowExtension(
483
554
  if (tool !== "subagent" && tool !== "write" && tool !== "edit") return;
484
555
  const snap = sm.snapshot;
485
556
  if (!snap.active) return;
557
+ if (tool === "subagent" && typeof event.input?.action === "string") return;
486
558
  const file = config.phaseArtifacts[snap.phase];
487
559
  if (tool === "write" || tool === "edit") {
488
560
  return {
@@ -491,6 +563,30 @@ export function createWorkflowExtension(
491
563
  };
492
564
  }
493
565
  if (!file || !event.input || !FORCE_OUTPUT_PHASES.has(snap.phase)) return;
566
+ // ponytail: Plan 5 — single-owner phases reject parallel/chain. One agent
567
+ // must own the artifact; tasks/chain split ownership and the output
568
+ // injector cannot pin one file per agent. Loop is exempt (engine-owned).
569
+ if (Array.isArray(event.input.tasks) || Array.isArray(event.input.chain)) {
570
+ return {
571
+ block: true,
572
+ reason: `Workflow phase ${snap.phase} expects one subagent owner for ${file}; use a single { agent, task } call, not tasks/chain.`,
573
+ };
574
+ }
575
+ // ponytail: Plan 5 — workflow-spawned subagents run fresh by default so
576
+ // no hidden parent context leaks into the phase owner. Only override
577
+ // when the caller explicitly pins a context.
578
+ if (typeof event.input.context !== "string") {
579
+ event.input.context = "fresh";
580
+ }
581
+ // ponytail: Plan 5 — ban model override on workflow-owned phases. The
582
+ // workflow, not the parent model, picks the agent; a model pin would
583
+ // silently change cost/quality without the workflow knowing.
584
+ if (typeof event.input.model === "string") {
585
+ return {
586
+ block: true,
587
+ reason: `Workflow phase ${snap.phase} does not allow a model override on subagent calls; remove the model parameter.`,
588
+ };
589
+ }
494
590
  const onlyAgents = snap.phase === "audit" ? new Set(["commentator", "reviewer"]) : undefined;
495
591
  forceSubagentOutputToArtifactDir(event.input, snap.artifactDir, file, onlyAgents);
496
592
  });
@@ -522,8 +618,68 @@ export function createWorkflowExtension(
522
618
  }
523
619
  }
524
620
  }
621
+ // ponytail: Plan 3 — engine-owned loop orchestration. The parent model
622
+ // does NOT track iterations; the adapter counts review rounds, parses the
623
+ // commentator's WORKFLOW_REVIEW_STATUS marker, and drives the next step
624
+ // via followUp. Ceiling: there is no pi.callTool() API, so the engine
625
+ // cannot invoke the subagent tool directly — it sends a followUp the
626
+ // model is expected to act on. If the model deviates, the loop stalls
627
+ // until the next tool_result nudges again. loopState is cleared when the
628
+ // phase leaves "loop" (below).
629
+ if (event.toolName === "subagent" && sm.snapshot.active && sm.snapshot.phase === "loop") {
630
+ const dir = sm.snapshot.artifactDir;
631
+ const maxIt = config.loopMaxIterations ?? 3;
632
+ if (!controller.loopState) {
633
+ controller.loopState = { reviewRound: 0, maxIterations: maxIt, stage: "building" };
634
+ }
635
+ const ls = controller.loopState;
636
+ const agent = event.input?.agent;
637
+
638
+ if (agent === "builder") {
639
+ // Build step done → drive the commentator review.
640
+ ls.stage = "reviewing";
641
+ continueAgent(pi, ctx, `Builder finished. Call the subagent tool now with { agent: "commentator", task: "..." } to review the builder's uncommitted diff against ${dir}/plan.md. Craft the review task yourself based on what matters for this task. End the review with exactly one line on its own:
642
+ WORKFLOW_REVIEW_STATUS: clean
643
+ or
644
+ WORKFLOW_REVIEW_STATUS: blocking`);
645
+ } else if (agent === "commentator") {
646
+ ls.reviewRound += 1;
647
+ const status = parseLoopReviewStatus(textFromToolResultContent(event.content));
648
+ if (status === "clean") {
649
+ ls.stage = "clean";
650
+ await mkdir(dir, { recursive: true });
651
+ await writeFile(resolve(dir, config.phaseArtifacts["loop"]!), `Loop complete after ${ls.reviewRound} review round(s).\n${MARKERS.loopComplete}`, "utf8");
652
+ // onArtifactMaybe below sees loop-complete.md and advances to audit.
653
+ } else if (ls.reviewRound >= ls.maxIterations) {
654
+ // Cap reached without a clean review. Notify once, stop driving.
655
+ const wasMaxed = ls.stage === "maxed";
656
+ ls.stage = "maxed";
657
+ if (!wasMaxed) {
658
+ ctx.ui.notify(
659
+ `Workflow loop hit max iterations (${ls.maxIterations}) without a clean review. Last review status: ${status ?? "no marker"}. Resolve issues manually or re-run.`,
660
+ "warning",
661
+ );
662
+ }
663
+ // A genuine clean review on a later round still advances — the
664
+ // clean branch above runs first.
665
+ } else {
666
+ // Blocking (or no marker) → drive a builder fix round.
667
+ ls.stage = "building";
668
+ const reason = status === "blocking"
669
+ ? "the blocking issues listed in the review"
670
+ : "the review (no WORKFLOW_REVIEW_STATUS: clean|blocking marker was found)";
671
+ continueAgent(pi, ctx, `Review round ${ls.reviewRound} was blocking. Call the subagent tool now with { agent: "builder", task: "..." } to fix ${reason}. Give the builder the review feedback and the relevant artifact paths. After the builder returns, the workflow will drive the next review automatically.`);
672
+ }
673
+ }
674
+ // Non-builder/commentator subagent calls in loop fall through to
675
+ // onArtifactMaybe (a no-op unless loop-complete.md exists).
676
+ }
525
677
  const eff = await sm.onArtifactMaybe(deps);
526
678
  applyEffect(pi, ctx, config, eff);
679
+ // Clear loop state once we have left the loop phase (advanced to audit).
680
+ if (sm.snapshot.phase !== "loop") {
681
+ controller.loopState = undefined;
682
+ }
527
683
  });
528
684
 
529
685
  // ── /<command> ──
@@ -4,6 +4,12 @@ import type {
4
4
  WorkflowConfig,
5
5
  WorkflowModeRegistration,
6
6
  } from "../state-machine.ts";
7
+ import {
8
+ handoffValidator,
9
+ loopCompleteValidator,
10
+ planValidator,
11
+ reviewValidator,
12
+ } from "../validators.ts";
7
13
 
8
14
  // ponytail: prototype workflow — full 7-phase tool-driven loop:
9
15
  // grill → research → plan → reuse → handoff → loop → audit.
@@ -74,9 +80,11 @@ Produce the concrete build plan for the actual prototype/output — not a plan f
74
80
 
75
81
  If research was skipped, proceed directly — you already understand the user's intent from grilling.
76
82
 
77
- Spawn the ARCHITECT sub-agent via the subagent tool (subagent_type "architect"). Do NOT pass a model parameter. Craft a tailored prompt from ${artifactDir}/requirements.md and ${artifactDir}/research.md so the architect plans for THIS task. The architect returns the plan; the workflow saves it to ${artifactDir}/plan.md.
83
+ Call the subagent tool with { agent: "architect", task: "..." } (do NOT pass a model parameter). Craft the task from ${artifactDir}/requirements.md and ${artifactDir}/research.md so the architect plans for THIS task. The architect returns the plan; the workflow saves it to ${artifactDir}/plan.md.
78
84
 
79
- The workflow advances once ${artifactDir}/plan.md exists.`,
85
+ The plan MUST end with exactly one machine-readable line on its own:
86
+ WORKFLOW_PLAN_STATUS: ready
87
+ The workflow will NOT advance until ${artifactDir}/plan.md contains that marker.`,
80
88
  reuse: ({ artifactDir }) =>
81
89
  `You are in the REUSE phase (optional).
82
90
 
@@ -84,7 +92,7 @@ Decide whether codebase exploration is useful. Skip if the project is empty, the
84
92
 
85
93
  If unsure, ask the user one focused question: "Should I explore the existing codebase for reusable patterns before implementing?" Then follow their answer.
86
94
 
87
- If YES (or user confirms): spawn the EXPLORER sub-agent via the subagent tool (subagent_type "explorer"). Do NOT pass a model parameter. Craft a specific prompt from ${artifactDir}/requirements.md and ${artifactDir}/plan.md pointing it at relevant areas, patterns, and dependencies (e.g., Tailwind setup, state-machine usage). Synthesize the findings into ${artifactDir}/reuse.md: what is reusable, where, and how to leverage it.
95
+ If YES (or user confirms): call the subagent tool with { agent: "explorer", task: "..." } (do NOT pass a model parameter). Craft the task from ${artifactDir}/requirements.md and ${artifactDir}/plan.md pointing it at relevant areas, patterns, and dependencies (e.g., Tailwind setup, state-machine usage). Synthesize the findings into ${artifactDir}/reuse.md: what is reusable, where, and how to leverage it.
88
96
 
89
97
  If NO: call write_workflow_artifact with a brief skip note explaining why.
90
98
 
@@ -100,29 +108,36 @@ Draw from all prior phases:
100
108
  - ${artifactDir}/plan.md (the build plan)
101
109
  - ${artifactDir}/reuse.md (codebase exploration findings)
102
110
 
103
- Spawn the RECAPPER sub-agent via the subagent tool (subagent_type "recapper"). Do NOT pass a model parameter. Craft a tailored prompt pointing the recapper at all four artifact files and telling it what the prototype is about. The recapper returns the handoff; the workflow saves it to ${artifactDir}/handoff.md.
111
+ Call the subagent tool with { agent: "recapper", task: "..." } (do NOT pass a model parameter). Craft the task pointing the recapper at all four artifact files and telling it what the prototype is about. The recapper returns the handoff; the workflow saves it to ${artifactDir}/handoff.md.
104
112
 
105
- The workflow advances once ${artifactDir}/handoff.md exists.`,
106
- loop: ({ artifactDir }) =>
107
- `You are in the LOOP (orchestration) phase. Delegate implementation to sub-agents — do not write code yourself.
113
+ The handoff MUST end with exactly one machine-readable line on its own:
114
+ WORKFLOW_HANDOFF_STATUS: ready
115
+ The workflow will NOT advance until ${artifactDir}/handoff.md contains that marker.`,
116
+ loop: ({ artifactDir, loopMaxIterations }) =>
117
+ `You are in the LOOP (orchestration) phase. The workflow ENGINE owns the implement→review loop — you do NOT track iterations or decide when the loop is clean.
108
118
 
109
- Use read to inspect ${artifactDir}/plan.md, ${artifactDir}/handoff.md, ${artifactDir}/research.md, and ${artifactDir}/reuse.md. Using that context, GENERATE YOUR OWN delegation prompts (no static template):
110
- 1. Dispatch ONE builder sub-agent via the subagent tool (subagent_type "builder"). Do NOT pass a model parameter. Give it plan + handoff + any research/reuse context, tailored to this task. Instruct it to implement every task in plan.md in order. All code changes go in the workspace, never in ${artifactDir}.
111
- 2. Dispatch ONE commentator sub-agent via the subagent tool (subagent_type "commentator"). Do NOT pass a model parameter. Review the builder's diff against plan.md. Generate the review prompt YOURSELF based on what matters for this task.
112
- 3. If the commentator reports blocking issues, dispatch the builder again with the issues to fix, then re-run the commentator. Repeat until no issues.
113
- 4. Call write_workflow_artifact with a one-line summary of the finished plan.
119
+ Use read to inspect ${artifactDir}/plan.md, ${artifactDir}/handoff.md, ${artifactDir}/research.md, and ${artifactDir}/reuse.md. Using that context, GENERATE YOUR OWN delegation prompt (no static template):
114
120
 
115
- Tailor prompts to the task — not generic. The workflow advances to audit once ${artifactDir}/loop-complete.md exists.`,
121
+ Call the subagent tool with { agent: "builder", task: "..." } (do NOT pass a model parameter). Give it plan + handoff + any research/reuse context, tailored to this task. Instruct it to implement every task in plan.md in order. All code changes go in the workspace, never in ${artifactDir}.
122
+
123
+ After the builder returns, the workflow engine will automatically prompt you to call the commentator. Craft the review task YOURSELF based on what matters for this task. Each commentator review MUST end with exactly one machine-readable line:
124
+ WORKFLOW_REVIEW_STATUS: clean
125
+ OR
126
+ WORKFLOW_REVIEW_STATUS: blocking
127
+
128
+ If a review is blocking, the engine prompts you to call the builder again with the issues. This repeats up to ${loopMaxIterations ?? 3} round(s). When a review is clean, the engine writes loop-complete.md and advances to audit. Do NOT write loop-complete.md yourself.`,
116
129
  audit: ({ artifactDir }) =>
117
130
  `You are in the AUDIT (review) phase.
118
131
 
119
132
  Review uncommitted changes for correctness, plan adherence, and over-engineering. Use ponytail-review style: cut bloat, unnecessary abstractions, dead flexibility, and reinvented stdlib/native behavior.
120
133
 
121
- 1. Spawn the COMMENTATOR sub-agent via the subagent tool (subagent_type "commentator"). Do NOT pass a model parameter. Craft a tailored prompt from the full uncommitted diff and ${artifactDir}/plan.md. The commentator returns its review; the workflow saves it to ${artifactDir}/review.md.
122
- 2. If ${artifactDir}/review.md lists actionable issues, spawn the BUILDER sub-agent via the subagent tool (subagent_type "builder"). Do NOT pass a model parameter. Instruct it to fix every issue in the workspace, never in ${artifactDir}.
123
- 3. Re-dispatch the commentator until ${artifactDir}/review.md says the review is clean and no issues remain.
134
+ 1. Call the subagent tool with { agent: "commentator", task: "..." } (do NOT pass a model parameter). Craft the task from the full uncommitted diff and ${artifactDir}/plan.md. The commentator returns its review; the workflow saves it to ${artifactDir}/review.md. The review MUST end with exactly one machine-readable line on its own:
135
+ WORKFLOW_REVIEW_STATUS: clean
136
+ (use WORKFLOW_REVIEW_STATUS: blocking if actionable issues remain)
137
+ 2. If ${artifactDir}/review.md lists actionable issues, call the subagent tool with { agent: "builder", task: "..." } (do NOT pass a model parameter). Instruct it to fix every issue in the workspace, never in ${artifactDir}.
138
+ 3. Re-dispatch the commentator until ${artifactDir}/review.md ends with WORKFLOW_REVIEW_STATUS: clean.
124
139
 
125
- The workflow closes once ${artifactDir}/review.md exists and the commentator reports no outstanding issues.`,
140
+ The workflow closes only once ${artifactDir}/review.md exists AND ends with the WORKFLOW_REVIEW_STATUS: clean marker.`,
126
141
  };
127
142
 
128
143
  const config: WorkflowConfig = {
@@ -138,7 +153,17 @@ const config: WorkflowConfig = {
138
153
  audit: "review.md",
139
154
  },
140
155
  prompts,
156
+ // ponytail: Plan 4 — semantic gates for the critical phases. Existence
157
+ // is still required, but the file content must also carry its marker line
158
+ // so an empty/stub file cannot silently advance the workflow.
159
+ artifactValidators: {
160
+ plan: planValidator,
161
+ handoff: handoffValidator,
162
+ loop: loopCompleteValidator,
163
+ },
164
+ closeValidators: { "review.md": reviewValidator },
141
165
  closeArtifacts: ["review.md"],
166
+ loopMaxIterations: 3,
142
167
  statusKey: "prototype",
143
168
  entryType: "prototype-phase",
144
169
  footerLabel: "prototype",
@@ -4,6 +4,12 @@ import type {
4
4
  WorkflowConfig,
5
5
  WorkflowModeRegistration,
6
6
  } from "../state-machine.ts";
7
+ import {
8
+ handoffValidator,
9
+ loopCompleteValidator,
10
+ planValidator,
11
+ reviewValidator,
12
+ } from "../validators.ts";
7
13
 
8
14
  // ponytail: quick workflow — same tool-driven loop as prototype, but shorter:
9
15
  // grill (max 4 questions) → plan → reuse → handoff → loop → audit.
@@ -58,9 +64,11 @@ The workflow advances once ${artifactDir}/requirements.md exists.`,
58
64
 
59
65
  This is a QUICK workflow — no separate research phase. Use ${artifactDir}/requirements.md to produce the concrete build plan: what to build, how, in order, components, and what the finished prototype looks like.
60
66
 
61
- Spawn the ARCHITECT sub-agent via the subagent tool (subagent_type "architect"). Do NOT pass a model parameter. Craft a tailored prompt from ${artifactDir}/requirements.md so the architect plans for THIS task. The architect returns the plan; the workflow saves it to ${artifactDir}/plan.md.
67
+ Call the subagent tool with { agent: "architect", task: "..." } (do NOT pass a model parameter). Craft the task from ${artifactDir}/requirements.md so the architect plans for THIS task. The architect returns the plan; the workflow saves it to ${artifactDir}/plan.md.
62
68
 
63
- The workflow advances once ${artifactDir}/plan.md exists.`,
69
+ The plan MUST end with exactly one machine-readable line on its own:
70
+ WORKFLOW_PLAN_STATUS: ready
71
+ The workflow will NOT advance until ${artifactDir}/plan.md contains that marker.`,
64
72
  reuse: ({ artifactDir }) =>
65
73
  `You are in the REUSE phase (optional).
66
74
 
@@ -68,7 +76,7 @@ Decide whether codebase exploration is useful. Skip if the project is empty, the
68
76
 
69
77
  If unsure, ask the user one focused question: "Should I explore the existing codebase for reusable patterns before implementing?" Then follow their answer.
70
78
 
71
- If YES (or user confirms): spawn the EXPLORER sub-agent via the subagent tool (subagent_type "explorer"). Do NOT pass a model parameter. Craft a specific prompt from ${artifactDir}/requirements.md and ${artifactDir}/plan.md pointing it at relevant areas, patterns, and dependencies. Synthesize the findings into ${artifactDir}/reuse.md: what is reusable, where, and how to leverage it.
79
+ If YES (or user confirms): call the subagent tool with { agent: "explorer", task: "..." } (do NOT pass a model parameter). Craft the task from ${artifactDir}/requirements.md and ${artifactDir}/plan.md pointing it at relevant areas, patterns, and dependencies. Synthesize the findings into ${artifactDir}/reuse.md: what is reusable, where, and how to leverage it.
72
80
 
73
81
  If NO: call write_workflow_artifact with a brief skip note explaining why.
74
82
 
@@ -83,29 +91,36 @@ Draw from all prior phases:
83
91
  - ${artifactDir}/plan.md
84
92
  - ${artifactDir}/reuse.md
85
93
 
86
- Spawn the RECAPPER sub-agent via the subagent tool (subagent_type "recapper"). Do NOT pass a model parameter. Craft a tailored prompt pointing the recapper at all three artifact files and telling it what the prototype is about. The recapper returns the handoff; the workflow saves it to ${artifactDir}/handoff.md.
94
+ Call the subagent tool with { agent: "recapper", task: "..." } (do NOT pass a model parameter). Craft the task pointing the recapper at all three artifact files and telling it what the prototype is about. The recapper returns the handoff; the workflow saves it to ${artifactDir}/handoff.md.
87
95
 
88
- The workflow advances once ${artifactDir}/handoff.md exists.`,
89
- loop: ({ artifactDir }) =>
90
- `You are in the LOOP (orchestration) phase. Delegate implementation to sub-agents — do not write code yourself.
96
+ The handoff MUST end with exactly one machine-readable line on its own:
97
+ WORKFLOW_HANDOFF_STATUS: ready
98
+ The workflow will NOT advance until ${artifactDir}/handoff.md contains that marker.`,
99
+ loop: ({ artifactDir, loopMaxIterations }) =>
100
+ `You are in the LOOP (orchestration) phase. The workflow ENGINE owns the implement→review loop — you do NOT track iterations or decide when the loop is clean.
91
101
 
92
- Use read to inspect ${artifactDir}/plan.md, ${artifactDir}/handoff.md, and ${artifactDir}/reuse.md. Using that context, GENERATE YOUR OWN delegation prompts:
93
- 1. Dispatch ONE builder sub-agent via the subagent tool (subagent_type "builder"). Do NOT pass a model parameter. Give it plan + handoff + reuse context, tailored to this task. Instruct it to implement every task in plan.md in order. All code changes go in the workspace, never in ${artifactDir}.
94
- 2. Dispatch ONE commentator sub-agent via the subagent tool (subagent_type "commentator"). Do NOT pass a model parameter. Review the builder's diff against plan.md. Generate the review prompt YOURSELF.
95
- 3. If the commentator reports blocking issues, dispatch the builder again with the issues to fix, then re-run the commentator. Repeat until no issues.
96
- 4. Call write_workflow_artifact with a one-line summary of the finished plan.
102
+ Use read to inspect ${artifactDir}/plan.md, ${artifactDir}/handoff.md, and ${artifactDir}/reuse.md. Using that context, GENERATE YOUR OWN delegation prompt:
97
103
 
98
- The workflow advances to audit once ${artifactDir}/loop-complete.md exists.`,
104
+ Call the subagent tool with { agent: "builder", task: "..." } (do NOT pass a model parameter). Give it plan + handoff + reuse context, tailored to this task. Instruct it to implement every task in plan.md in order. All code changes go in the workspace, never in ${artifactDir}.
105
+
106
+ After the builder returns, the workflow engine will automatically prompt you to call the commentator. Craft the review task YOURSELF. Each commentator review MUST end with exactly one machine-readable line:
107
+ WORKFLOW_REVIEW_STATUS: clean
108
+ OR
109
+ WORKFLOW_REVIEW_STATUS: blocking
110
+
111
+ If a review is blocking, the engine prompts you to call the builder again with the issues. This repeats up to ${loopMaxIterations ?? 3} round(s). When a review is clean, the engine writes loop-complete.md and advances to audit. Do NOT write loop-complete.md yourself.`,
99
112
  audit: ({ artifactDir }) =>
100
113
  `You are in the AUDIT (review) phase.
101
114
 
102
115
  Review uncommitted changes for correctness, plan adherence, and over-engineering. Use ponytail-review style: cut bloat, unnecessary abstractions, dead flexibility, and reinvented stdlib/native behavior.
103
116
 
104
- 1. Spawn the COMMENTATOR sub-agent via the subagent tool (subagent_type "commentator"). Do NOT pass a model parameter. Craft a tailored prompt from the full uncommitted diff and ${artifactDir}/plan.md. The commentator returns its review; the workflow saves it to ${artifactDir}/review.md.
105
- 2. If ${artifactDir}/review.md lists actionable issues, spawn the BUILDER sub-agent via the subagent tool (subagent_type "builder"). Do NOT pass a model parameter. Instruct it to fix every issue in the workspace, never in ${artifactDir}.
106
- 3. Re-dispatch the commentator until ${artifactDir}/review.md says the review is clean and no issues remain.
117
+ 1. Call the subagent tool with { agent: "commentator", task: "..." } (do NOT pass a model parameter). Craft the task from the full uncommitted diff and ${artifactDir}/plan.md. The commentator returns its review; the workflow saves it to ${artifactDir}/review.md. The review MUST end with exactly one machine-readable line on its own:
118
+ WORKFLOW_REVIEW_STATUS: clean
119
+ (use WORKFLOW_REVIEW_STATUS: blocking if actionable issues remain)
120
+ 2. If ${artifactDir}/review.md lists actionable issues, call the subagent tool with { agent: "builder", task: "..." } (do NOT pass a model parameter). Instruct it to fix every issue in the workspace, never in ${artifactDir}.
121
+ 3. Re-run the commentator until ${artifactDir}/review.md ends with WORKFLOW_REVIEW_STATUS: clean.
107
122
 
108
- The workflow closes once ${artifactDir}/review.md exists and the commentator reports no outstanding issues.`,
123
+ The workflow closes only once ${artifactDir}/review.md exists AND ends with the WORKFLOW_REVIEW_STATUS: clean marker.`,
109
124
  };
110
125
 
111
126
  const config: WorkflowConfig = {
@@ -120,7 +135,15 @@ const config: WorkflowConfig = {
120
135
  audit: "review.md",
121
136
  },
122
137
  prompts,
138
+ // ponytail: Plan 4 — semantic gates for the critical phases.
139
+ artifactValidators: {
140
+ plan: planValidator,
141
+ handoff: handoffValidator,
142
+ loop: loopCompleteValidator,
143
+ },
144
+ closeValidators: { "review.md": reviewValidator },
123
145
  closeArtifacts: ["review.md"],
146
+ loopMaxIterations: 3,
124
147
  statusKey: "quick",
125
148
  entryType: "quick-phase",
126
149
  footerLabel: "quick",