@selesai/code 0.9.3 → 0.9.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,13 @@
2
2
 
3
3
  All notable changes to `@selesai/code` will be documented in this file.
4
4
 
5
+ ## [0.9.4] - 2026-08-22
6
+
7
+ ### Changed
8
+ - **Extensible pi-subagents workflows.** The bundled `/workflow-*` extension is now a thin registry-based adapter over pi-subagents orchestration. Modes can use ordered runs, parallel discovery, or scripted conditional loops without adding command plumbing. The prototype workflow now runs external research and codebase exploration in parallel.
9
+ - **Workflow completion contract.** Build/review/fix loops now finish only when the reviewer reports `clean` with no remaining work, preventing a clean review of one slice from ending an incomplete plan.
10
+ - **Workflow documentation.** Replaced stale durable-state-machine documentation with the current pi-subagents launch, recovery, and extension model.
11
+
5
12
  ## [0.9.3] - 2026-08-22
6
13
 
7
14
  ### Fixed
@@ -1,26 +1,26 @@
1
- // ponytail: thin shell. Registers the four /workflow-* commands and launches
2
- // each mode's scripted workflow through pi-subagents' launchSlashSubagent.
1
+ // ponytail: thin slash-command adapter for the pi-subagents workflow runtime.
3
2
 
4
3
  import type { ExtensionAPI } from "@selesai/code";
5
4
  import { launchSlashSubagent } from "../pi-subagents/src/slash/slash-commands.ts";
6
- import { buildLoopScript, buildPrototypeScript, buildQuicktypeScript, buildTaskScript } from "./modes.ts";
5
+ import { WORKFLOW_MODES, type WorkflowMode } from "./modes.ts";
7
6
 
8
7
  export default function workflowModesExtension(pi: ExtensionAPI): void {
9
- const register = (name: string, description: string, build: (goal: string) => string) =>
10
- pi.registerCommand(name, {
11
- description,
12
- handler: async (args, ctx) => {
13
- const goal = args.trim();
14
- if (!goal) {
15
- ctx.ui.notify(`${description}\nUsage: /${name} <goal>`, "info");
16
- return;
17
- }
18
- launchSlashSubagent(pi, ctx, { workflowScript: build(goal), async: true, agentScope: "both", mission: { title: goal } });
19
- },
20
- });
8
+ const register = (mode: WorkflowMode) => pi.registerCommand(mode.command, {
9
+ description: mode.description,
10
+ handler: async (args, ctx) => {
11
+ const goal = args.trim();
12
+ if (!goal) {
13
+ ctx.ui.notify(`${mode.description}\nUsage: /${mode.command} <goal>`, "info");
14
+ return;
15
+ }
16
+ launchSlashSubagent(pi, ctx, {
17
+ ...mode.launch(goal),
18
+ async: true,
19
+ agentScope: "both",
20
+ mission: { title: goal },
21
+ });
22
+ },
23
+ });
21
24
 
22
- register("workflow-task", "Run the task workflow (plan → reuse → handoff → auto build→review→fix loop) as a scripted workflow.", buildTaskScript);
23
- register("workflow-prototype", "Run the full prototype workflow (research → plan → reuse → handoff → auto loop → audit) as a scripted workflow.", buildPrototypeScript);
24
- register("workflow-quicktype", "Run the quicker prototype workflow without research (plan → reuse → handoff → auto loop → audit) as a scripted workflow.", buildQuicktypeScript);
25
- register("workflow-loop", "Run a direct auto build→review→fix loop for an already-agreed plan as a scripted workflow.", buildLoopScript);
25
+ for (const mode of WORKFLOW_MODES) register(mode);
26
26
  }
@@ -1,12 +1,20 @@
1
- // ponytail: four workflow modes as pi-subagents workflowScript builders.
2
- // One-shot-and-sleep: phases run as runs.run steps and the build → review → fix
3
- // round auto-repeats until the reviewer reports clean. A workflowScript has no
4
- // checkpoints, so there are no human gates.
1
+ // ponytail: workflow mode registry over pi-subagents' public workflowScript seam.
2
+ // A mode returns launch parameters; the extension owns slash-command plumbing.
3
+
4
+ import type { SubagentParamsLike } from "../pi-subagents/src/runs/foreground/subagent-executor.ts";
5
+
6
+ export interface WorkflowMode {
7
+ command: string;
8
+ description: string;
9
+ launch(goal: string): Pick<SubagentParamsLike, "workflowScript" | "chain" | "tasks" | "concurrency">;
10
+ }
5
11
 
6
12
  function js(value: string): string {
7
- return JSON.stringify(value);
13
+ return JSON.stringify(value);
8
14
  }
9
15
 
16
+ // A loop depends on the previous review's output, so it belongs in pi-subagents'
17
+ // scripted workflow runtime rather than a fixed native chain.
10
18
  const AUTO_LOOP = String.raw`
11
19
  const autoLoop = async (goal, context, progressFile) => {
12
20
  emit({ phase: 'start', goal });
@@ -25,14 +33,15 @@ const autoLoop = async (goal, context, progressFile) => {
25
33
  timeoutMs: 15 * 60 * 1000,
26
34
  task: 'Independently review the builder work for this round and report concrete evidence (what you inspected and what you ran). Do not modify the workspace.\n\nAcceptance criteria (source of truth):\n' + context + '\n\nProgress file (scope your review to its latest round entry; also re-check the files from the immediately preceding fix entry if one exists; fall back to the full uncommitted diff if it is missing or empty):\n' + progressFile + '\n\nBuilder completion summary:\n' + build.output + '\n\nIf the plan is not yet complete, add a "Remaining work:" section listing the next concrete step(s). End with exactly one line: WORKFLOW_REVIEW_STATUS: clean OR WORKFLOW_REVIEW_STATUS: blocking.',
27
35
  });
28
- if (/WORKFLOW_REVIEW_STATUS\s*:\s*clean/i.test(review.output)) {
36
+ const hasRemainingWork = /Remaining work\s*:\s*\S/i.test(review.output);
37
+ if (/WORKFLOW_REVIEW_STATUS\s*:\s*clean/i.test(review.output) && !hasRemainingWork) {
29
38
  return { result: 'clean', rounds: completed + 1 };
30
39
  }
31
40
  previousReview = review.output;
32
41
  await runs.run('fix-' + round, {
33
42
  agent: 'builder',
34
43
  timeoutMs: 45 * 60 * 1000,
35
- task: 'Address ONLY the findings from the review below. The "Remaining work:" section (if present) is for the next round; do not act on it.\n\nProgress ledger: append a "## Round ' + round + ' fix" entry to the progress file at ' + progressFile + ' before finishing. List every file you changed and a short summary of the fixes.\n\nReviewer findings:\n' + review.output,
44
+ task: 'Address ONLY the findings from the review below. The "Remaining work:" section (if present) is for the next round; do not act on it. If the review is clean but has Remaining work, make no changes and record that fact.\n\nProgress ledger: append a "## Round ' + round + ' fix" entry to the progress file at ' + progressFile + ' before finishing. List every file you changed and a short summary of the fixes.\n\nReviewer findings:\n' + review.output,
36
45
  });
37
46
  completed += 1;
38
47
  round += 1;
@@ -49,13 +58,13 @@ const autoLoop = async (goal, context, progressFile) => {
49
58
  const PROGRESS_DIR = ".pi-subagents/progress/";
50
59
 
51
60
  export function buildLoopScript(goal: string): string {
52
- return String.raw`const goal = ${js(goal)};
61
+ return String.raw`const goal = ${js(goal)};
53
62
  ${AUTO_LOOP}
54
63
  return await autoLoop(goal, goal, ${js(PROGRESS_DIR + "loop.md")});`;
55
64
  }
56
65
 
57
66
  export function buildTaskScript(goal: string): string {
58
- return String.raw`const goal = ${js(goal)};
67
+ return String.raw`const goal = ${js(goal)};
59
68
  const plan = await runs.run('plan', { agent: 'architect', task: 'Produce a concrete implementation plan for: ' + goal + '. Cover what to build, how, in what order, which files and components, and the finished result. Return inline.' });
60
69
  const reuse = await runs.run('reuse', { agent: 'explorer', task: 'Explore the codebase for reusable patterns relevant to: ' + plan.output + '. Point at relevant areas and dependencies; skip cleanly if wholly new. Return inline.' });
61
70
  const handoff = await runs.run('handoff', { agent: 'recapper', task: 'Compile a self-contained handoff from the plan and reuse findings so fresh agents understand the goal, constraints, and acceptance criteria without re-planning.\n\nPlan:\n' + plan.output + '\n\nReuse findings:\n' + reuse.output + '\n\nReturn inline.' });
@@ -64,10 +73,14 @@ return await autoLoop(goal, handoff.output, ${js(PROGRESS_DIR + "task.md")});`;
64
73
  }
65
74
 
66
75
  export function buildPrototypeScript(goal: string): string {
67
- return String.raw`const goal = ${js(goal)};
68
- const research = await runs.run('research', { agent: 'researcher', task: 'Research the external, fast-changing knowledge this task depends on (libraries, SDKs, APIs, unfamiliar alternatives). Task: ' + goal + '. Synthesize actionable findings with sources. Return inline.' });
69
- const plan = await runs.run('plan', { agent: 'architect', task: 'Produce a concrete build plan from the research findings.\n\nResearch:\n' + research.output + '\n\nRequest:\n' + goal + '\n\nReturn inline.' });
70
- const reuse = await runs.run('reuse', { agent: 'explorer', task: 'Explore the codebase for reusable patterns relevant to: ' + plan.output + '. Return inline.' });
76
+ return String.raw`const goal = ${js(goal)};
77
+ const discovery = await runs.all([
78
+ { key: 'research', agent: 'researcher', task: 'Research the external, fast-changing knowledge this task depends on (libraries, SDKs, APIs, unfamiliar alternatives). Task: ' + goal + '. Synthesize actionable findings with sources. Return inline.' },
79
+ { key: 'explore', agent: 'explorer', task: 'Map existing code, dependencies, and reusable patterns relevant to: ' + goal + '. Return inline.' },
80
+ ]);
81
+ const research = discovery.find(result => result.key === 'research');
82
+ const reuse = discovery.find(result => result.key === 'explore');
83
+ const plan = await runs.run('plan', { agent: 'architect', task: 'Produce a concrete build plan from the research and codebase findings.\n\nResearch:\n' + research.output + '\n\nCodebase findings:\n' + reuse.output + '\n\nRequest:\n' + goal + '\n\nReturn inline.' });
71
84
  const handoff = await runs.run('handoff', { agent: 'recapper', task: 'Compile a self-contained handoff from the plan and reuse findings.\n\nPlan:\n' + plan.output + '\n\nReuse:\n' + reuse.output + '\n\nReturn inline.' });
72
85
  ${AUTO_LOOP}
73
86
  const loop = await autoLoop(goal, handoff.output, ${js(PROGRESS_DIR + "prototype.md")});
@@ -76,7 +89,7 @@ return { ...loop, audited: true };`;
76
89
  }
77
90
 
78
91
  export function buildQuicktypeScript(goal: string): string {
79
- return String.raw`const goal = ${js(goal)};
92
+ return String.raw`const goal = ${js(goal)};
80
93
  const plan = await runs.run('plan', { agent: 'architect', task: 'Produce a concrete build plan for: ' + goal + '. Cover what to build, how, in what order, which components, and the finished result. Return inline.' });
81
94
  const reuse = await runs.run('reuse', { agent: 'explorer', task: 'Explore the codebase for reusable patterns relevant to: ' + plan.output + '. Return inline.' });
82
95
  const handoff = await runs.run('handoff', { agent: 'recapper', task: 'Compile a self-contained handoff from the plan and reuse findings.\n\nPlan:\n' + plan.output + '\n\nReuse:\n' + reuse.output + '\n\nReturn inline.' });
@@ -85,3 +98,10 @@ const loop = await autoLoop(goal, handoff.output, ${js(PROGRESS_DIR + "quicktype
85
98
  const audit = await runs.run('audit', { agent: 'commentator', task: 'Final audit of the uncommitted changes for correctness, plan adherence, and over-engineering (cut bloat, dead flexibility, reinvented stdlib). Plan:\n' + plan.output + '\n\nReport concrete evidence. Do not modify the workspace.' });
86
99
  return { ...loop, audited: true };`;
87
100
  }
101
+
102
+ export const WORKFLOW_MODES: readonly WorkflowMode[] = [
103
+ { command: "workflow-task", description: "Run the task workflow (plan → reuse → handoff → build/review/fix loop).", launch: (goal) => ({ workflowScript: buildTaskScript(goal) }) },
104
+ { command: "workflow-prototype", description: "Run the prototype workflow (parallel research/reuse → plan → handoff → loop → audit).", launch: (goal) => ({ workflowScript: buildPrototypeScript(goal) }) },
105
+ { command: "workflow-quicktype", description: "Run the quicker prototype workflow (plan → reuse → handoff → loop → audit).", launch: (goal) => ({ workflowScript: buildQuicktypeScript(goal) }) },
106
+ { command: "workflow-loop", description: "Run a direct build/review/fix loop for an already-agreed plan.", launch: (goal) => ({ workflowScript: buildLoopScript(goal) }) },
107
+ ];
@@ -0,0 +1,319 @@
1
+ # Kilo Code VS Code Architecture and Selesai Integration Feasibility
2
+
3
+ **Research date:** 2026-08-20
4
+ **Repositories:** [Kilo Code](https://github.com/Kilo-Org/kilocode), this Selesai Code repository
5
+ **Question:** Does Kilo use RPA or another integration mechanism, and can Selesai provide a similar or better VS Code extension?
6
+
7
+ ## Executive conclusion
8
+
9
+ Kilo Code is **not primarily an RPA application** and its core is not an LSP client/server. It is a layered VS Code product:
10
+
11
+ 1. A normal VS Code extension host (`packages/kilo-vscode`) activated through `vscode` APIs.
12
+ 2. Webview-based UI surfaces (sidebar, panels, agent manager, settings, diffs).
13
+ 3. A lazily spawned local Kilo CLI backend (`kilo serve --port 0`).
14
+ 4. An SDK/client connection from the extension host to that backend over authenticated local HTTP APIs and SSE event streaming.
15
+ 5. Separate WebSocket paths for selected features such as PTY terminal streaming and cloud event service.
16
+ 6. MCP client transports for tool/server integration: local stdio subprocesses and remote HTTP/SSE transports.
17
+
18
+ The important correction to a simplistic “Webview only” description is that **Webview `postMessage` is only the UI bridge**. In the current Kilo repository, the extension also connects to a local backend server. The extension spawns the backend, discovers its ephemeral port from stdout, gives it a generated password, and uses an SDK plus HTTP/SSE with Basic authentication.
19
+
20
+ Selesai can absolutely support a comparable VS Code extension. The shortest path is a VS Code extension-host backend that starts `selesai --mode rpc` and adapts the existing strict JSONL RPC client/events into a Webview. The best long-term path is probably a two-tier design:
21
+
22
+ - **MVP:** local child process + existing Selesai RPC over private stdin/stdout.
23
+ - **Better product integration:** a small authenticated local HTTP/SSE or WebSocket server facade, or direct in-process SDK use where safe, with a stable VS Code-oriented protocol.
24
+
25
+ Selesai already has unusually strong agent/session/extension primitives. Its main missing pieces for Kilo-level VS Code integration are a polished VS Code package, a durable UI protocol/adapter, explicit trust UX, and (if remote or multi-client use is desired) authenticated network transport.
26
+
27
+ ## 1. What Kilo actually uses
28
+
29
+ ### 1.1 VS Code extension activation
30
+
31
+ Kilo's extension is a conventional TypeScript VS Code extension. The activation function accepts `vscode.ExtensionContext`, constructs shared services, and registers views/providers through the VS Code API:
32
+
33
+ - [`packages/kilo-vscode/src/extension.ts`](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/extension.ts#L45-L61) — `activate(context)` and shared `KiloConnectionService`.
34
+ - [`packages/kilo-vscode/src/extension.ts`](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/extension.ts#L131-L143) — creates `KiloProvider` and calls `vscode.window.registerWebviewViewProvider(...)`.
35
+ - The extension manifest/build configuration is in [`packages/kilo-vscode/package.json`](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/package.json) and the repository root package configuration.
36
+
37
+ This is ordinary VS Code extension-host architecture, not RPA automation of the VS Code UI.
38
+
39
+ ### 1.2 Webview UI bridge
40
+
41
+ Kilo's chat and panels use Webviews. The extension host attaches a Webview and receives structured messages with `webview.onDidReceiveMessage`; it sends state, stream, and command results back using `webview.postMessage`:
42
+
43
+ - [`KiloProvider.ts`](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/KiloProvider.ts#L987-L1019) — `attachToWebview`, handler setup, interception, and routing.
44
+ - [`KiloProvider.ts`](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/KiloProvider.ts#L1013-L1040) — inbound message handling and dispatch.
45
+ - [`extension.ts`](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/extension.ts#L137-L143) — sidebar Webview registration.
46
+
47
+ The Webview bridge is **structured local message passing across the VS Code extension/Webview boundary**. It is not HTTP RPC, LSP, or RPA.
48
+
49
+ ### 1.3 The extension spawns a local backend
50
+
51
+ The current code explicitly documents that the CLI backend starts lazily rather than during extension activation:
52
+
53
+ > “The CLI backend is NOT spawned here; it starts lazily when a webview connects or when ensureBackendForAutocomplete() triggers it.”
54
+
55
+ Source: [`packages/kilo-vscode/src/extension.ts`](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/extension.ts#L45-L49).
56
+
57
+ The server manager starts a local CLI process with an ephemeral port:
58
+
59
+ - [`server-manager.ts`](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/services/cli-backend/server-manager.ts#L92-L120) — spawns `cliPath serve --port 0` with a workspace-derived cwd.
60
+ - [`server-manager.ts`](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/services/cli-backend/server-manager.ts#L118-L161) — passes environment, including `KILO_SERVER_PASSWORD`, parent PID, VS Code metadata, and `stdio: ["ignore", "pipe", "pipe"]`.
61
+ - [`server-manager.ts`](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/services/cli-backend/server-manager.ts#L167-L177) — reads child stdout and parses the selected port.
62
+
63
+ This is a **local child process plus local server** architecture. It is RPC-like in the broad sense, but it is not RPA.
64
+
65
+ ### 1.4 HTTP/API client and SSE events
66
+
67
+ `KiloConnectionService` owns one backend manager, one Kilo SDK client, and one SSE adapter for multiple UI providers:
68
+
69
+ - [`connection-service.ts`](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/services/cli-backend/connection-service.ts#L80-L99) — describes and declares the shared service, client, and SSE adapter.
70
+ - [`connection-service.ts`](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/services/cli-backend/connection-service.ts#L149-L150) — lazy startup/connection entrypoint.
71
+ - [`types.ts`](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/services/cli-backend/types.ts#L1-L20) — server configuration includes `baseUrl` and `password`.
72
+
73
+ The extension checks backend health over HTTP with Basic authentication:
74
+
75
+ - [`connection-service.ts`](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/services/cli-backend/connection-service.ts#L733-L777) — periodic `GET ${baseUrl}/global/health`, `Authorization: Basic ...`, and reconnect behavior.
76
+
77
+ Thus the current Kilo architecture is more accurately:
78
+
79
+ ```text
80
+ VS Code Webview
81
+ │ postMessage / onDidReceiveMessage
82
+ ▼
83
+ Kilo extension host
84
+ │ Kilo SDK over authenticated localhost HTTP
85
+ │ SSE event stream
86
+ ▼
87
+ local `kilo serve --port 0` process
88
+ │
89
+ ├─ model/provider APIs
90
+ ├─ sessions/tools/filesystem
91
+ └─ MCP clients
92
+ ```
93
+
94
+ ### 1.5 WebSockets are feature-specific, not the basic UI bridge
95
+
96
+ The repository has WebSocket use for feature-specific paths. For example, the Agent Manager terminal routes PTY bytes over a WebSocket rather than through Webview messages:
97
+
98
+ - [`terminal-manager.ts`](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/agent-manager/terminal-manager.ts#L1-L10) — describes PTY output over `/pty/:id/connect` WebSocket.
99
+ - [`terminal-routing.ts`](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/agent-manager/terminal-routing.ts#L242-L248) — builds an authenticated loopback WebSocket URL with `auth_token`.
100
+ - [`event-service-client.ts`](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/kiloclaw/event-service-client.ts#L1-L10) — cloud event-service ticket flow and WebSocket subprotocol.
101
+
102
+ This demonstrates that Kilo chooses transport by use case: Webview messages for UI commands/state, HTTP/SSE for backend API/events, and WebSocket for high-volume or cloud event paths.
103
+
104
+ ### 1.6 MCP is an integration protocol, not the extension connection
105
+
106
+ Kilo's MCP code imports multiple official MCP SDK transports:
107
+
108
+ - [`packages/opencode/src/mcp/index.ts`](https://github.com/Kilo-Org/kilocode/blob/main/packages/opencode/src/mcp/index.ts#L15-L19) — `StreamableHTTPClientTransport`, `SSEClientTransport`, and `StdioClientTransport`.
109
+ - [`packages/opencode/src/mcp/index.ts`](https://github.com/Kilo-Org/kilocode/blob/main/packages/opencode/src/mcp/index.ts#L369-L389) — local MCP config launches a command using `StdioClientTransport` with cwd/environment.
110
+
111
+ MCP lets the coding agent connect to tools/resources exposed by configured MCP servers. It does **not** appear to be the protocol connecting Kilo's VS Code Webview to its extension host. Those are separate layers.
112
+
113
+ ## 2. Is it RPA, LSP, RPC, or something else?
114
+
115
+ | Technology | Kilo role | Conclusion |
116
+ |---|---|---|
117
+ | RPA (Robotic Process Automation) | None identified as a framework/protocol | No. Agentic file/terminal/browser automation is not conventional RPA architecture. |
118
+ | VS Code Extension API | Activation, commands, views, workspace, terminals, Webviews | Yes; foundational. |
119
+ | Webview message bridge | UI ↔ extension host messages | Yes; local structured messaging. |
120
+ | Local child process | Extension starts `kilo serve` | Yes. |
121
+ | HTTP | Extension ↔ local backend API and health checks | Yes; authenticated localhost API. |
122
+ | SSE | Backend event stream to extension | Yes. |
123
+ | WebSocket | PTY and selected cloud/event paths | Yes, feature-specific. |
124
+ | MCP | Agent ↔ configured external tools/resources | Yes; separate integration layer. |
125
+ | LSP | Language server/client lifecycle | No evidence of core LSP architecture. |
126
+ | JSON-RPC | Not the main current VS Code connection identified in Kilo | Do not assume this is Kilo's transport. |
127
+
128
+ ## 3. What Selesai already provides
129
+
130
+ ### 3.1 Existing RPC mode
131
+
132
+ Selesai RPC is a headless JSONL protocol over a child process:
133
+
134
+ - [`docs/rpc.md`](../rpc.md) — protocol, commands, event stream, extension UI protocol, framing, and client examples.
135
+ - [`src/modes/rpc/rpc-mode.ts`](../../src/modes/rpc/rpc-mode.ts) — server implementation over stdin/stdout.
136
+ - [`src/modes/rpc/rpc-types.ts`](../../src/modes/rpc/rpc-types.ts) — typed commands, responses, events, and extension UI messages.
137
+ - [`src/modes/rpc/rpc-client.ts`](../../src/modes/rpc/rpc-client.ts) — typed subprocess client and request correlation.
138
+ - [`src/rpc-entry.ts`](../../src/rpc-entry.ts) — RPC executable entrypoint.
139
+
140
+ The transport is strict LF-delimited JSONL. It has request IDs for responses and asynchronous events for streaming output/tool lifecycle. A VS Code extension can spawn:
141
+
142
+ ```text
143
+ selesai --mode rpc --cwd <workspace> [other options]
144
+ ```
145
+
146
+ (or spawn the packaged Node CLI with equivalent arguments), then:
147
+
148
+ 1. Write one JSON object plus `\n` to stdin.
149
+ 2. Parse stdout one LF record at a time.
150
+ 3. Correlate `type: "response"` records by `id`.
151
+ 4. Render `message_update`, `tool_execution_*`, and lifecycle events.
152
+ 5. Wait for `agent_settled`/`agent_end`, not merely prompt acceptance.
153
+ 6. Answer `extension_ui_request` records when an extension asks for a dialog.
154
+
155
+ This is sufficient for a local VS Code extension MVP.
156
+
157
+ ### 3.2 Direct SDK option
158
+
159
+ Selesai also exports a direct TypeScript/Node SDK:
160
+
161
+ - [`docs/sdk.md`](../sdk.md) — `createAgentSession`, `createAgentSessionRuntime`, event subscription, tools, sessions, cwd, auth, and resources.
162
+ - [`src/core/sdk.ts`](../../src/core/sdk.ts) — SDK options and construction.
163
+ - [`src/index.ts`](../../src/index.ts) — public exports.
164
+
165
+ A VS Code extension host is itself Node-based, so it can potentially use `createAgentSession()` in-process. Advantages are lower latency, no JSON serialization, direct typed events, and custom tool/resource integration. Disadvantages are tighter coupling, more difficult fault isolation, extension-host memory/lifecycle risk, and the need to manage session replacement and extension binding carefully.
166
+
167
+ ### 3.3 Existing extension system and RPC UI support
168
+
169
+ Selesai extensions can register tools, commands, event handlers, and UI operations. In RPC mode, dialog and notification UI is translated into an explicit request/response protocol:
170
+
171
+ - [`docs/extensions.md`](../extensions.md) — extension API and security model.
172
+ - [`docs/rpc.md`](../rpc.md) — `select`, `confirm`, `input`, `editor`, `notify`, status, widget, title, and editor-text requests.
173
+
174
+ This is a useful foundation for mapping agent permission questions to VS Code dialogs, status bar items, notifications, and editor input.
175
+
176
+ ## 4. Selesai versus Kilo: capability comparison
177
+
178
+ | Capability | Kilo current approach | Selesai current position | Assessment |
179
+ |---|---|---|---|
180
+ | VS Code package | Dedicated `packages/kilo-vscode` extension | No dedicated VS Code extension found in this repository | Kilo leads on product packaging. |
181
+ | Main UI | Multiple Webviews/panels/providers | RPC UI protocol can feed a new Webview | Selesai has protocol primitives; Kilo has finished UI. |
182
+ | Agent backend isolation | Spawned local CLI server | Spawned local RPC process | Both isolate backend; Selesai MVP is simpler. |
183
+ | Backend network API | Authenticated localhost HTTP + SSE | No TCP/HTTP RPC server; stdin/stdout JSONL | Kilo leads for multi-client/network-capable architecture. |
184
+ | Transport auth | Generated password; Basic/loopback auth | No RPC auth handshake; private pipes only | Selesai needs auth before exposing network transport. |
185
+ | Streaming | SSE and WebSocket for relevant paths | JSONL event stream | Both support streaming; Selesai JSONL is easy to adapt. |
186
+ | Sessions | Backend session APIs and shared connection service | Rich RPC/SDK session lifecycle and tree operations | Selesai is strong at protocol-level session control. |
187
+ | Extensions/plugins | Agent/runtime extensions and Kilo services | Extension-first tools/events/UI; RPC-compatible UI | Selesai is potentially more extensible, but needs VS Code UX. |
188
+ | MCP | Local stdio and remote transports | Extension/tool ecosystem exists; MCP parity should be audited separately | Kilo has explicit MCP transport integration. |
189
+ | Terminal/PTy | Dedicated terminal manager and WebSocket streaming | RPC bash exists; no Kilo-equivalent VS Code PTY bridge identified | Kilo leads for native terminal UX. |
190
+ | Trust/security | Backend password and process lifecycle controls | Explicit extension trust exists, but RPC has no auth | Selesai needs a dedicated integration security model. |
191
+ | In-process embedding | Backend/client architecture | First-class SDK | Selesai may be better for tightly integrated custom clients. |
192
+
193
+ ## 5. Feasibility: how to build a Selesai VS Code extension
194
+
195
+ ### Phase 1 — local MVP (recommended first)
196
+
197
+ Build a conventional VS Code extension with:
198
+
199
+ ```text
200
+ VS Code Webview UI
201
+ │ vscode.postMessage
202
+ ▼
203
+ Extension host controller
204
+ │ child_process / existing RpcClient
205
+ ▼
206
+ selesai --mode rpc (private stdin/stdout)
207
+ ```
208
+
209
+ Minimum components:
210
+
211
+ - `ExtensionHostController`: owns one Selesai process per workspace/session.
212
+ - `RpcClientAdapter`: wraps the existing `RpcClient`; add or expose extension UI request handling if the current public client does not already do so.
213
+ - `WorkspaceManager`: resolves workspace cwd, session directory, and project trust.
214
+ - `EventStore`: maps streamed events into Webview state, with bounded history and reconnection handling.
215
+ - `WebviewProvider`: chat, streaming text, tool calls, approval questions, model picker, session picker.
216
+ - `Command` registrations: open sidebar, new session, abort, steer, follow-up, model selection.
217
+ - `dispose()` handling: send shutdown/EOF, wait briefly, terminate if necessary.
218
+
219
+ This does **not** need RPA, LSP, MCP, or a network server.
220
+
221
+ ### Phase 2 — native VS Code experience
222
+
223
+ Add the features that make an extension feel better than a terminal wrapper:
224
+
225
+ - Inline editor selection/context actions (“Ask Selesai about this”, “Fix this”).
226
+ - Diff-aware edit review using VS Code `WorkspaceEdit` or controlled file edits.
227
+ - Native permission prompts for writes, shell commands, and external tools.
228
+ - Diagnostics/code actions where agent output can be represented safely.
229
+ - Native terminal/task integration and cancellation.
230
+ - Status bar progress, notifications, output channel, and session restoration.
231
+ - Workspace-specific context files and explicit trust prompts.
232
+ - Webview state restoration and multiple workspace/session routing.
233
+
234
+ ### Phase 3 — Kilo-like backend facade (only if needed)
235
+
236
+ If Selesai needs multiple clients, remote control, or richer WebSocket/SSE behavior, add an authenticated local server mode rather than exposing raw stdin/stdout through an ad-hoc bridge.
237
+
238
+ Suggested shape:
239
+
240
+ ```text
241
+ VS Code extension ── authenticated HTTP/SSE or WebSocket ── Selesai server
242
+ ```
243
+
244
+ Requirements before shipping:
245
+
246
+ - Bind to loopback by default, never `0.0.0.0` by accident.
247
+ - Generate a high-entropy per-process secret.
248
+ - Require authentication on every endpoint/stream.
249
+ - Scope sessions and cwd to the requesting workspace/client.
250
+ - Validate origin and reject cross-workspace access.
251
+ - Provide process parent watchdog and graceful shutdown.
252
+ - Avoid leaking session paths, API keys, prompts, or tool output in logs.
253
+ - Define protocol versioning and event replay/reconnect semantics.
254
+ - Add rate limits and explicit authorization for destructive operations.
255
+
256
+ ## 6. Key risks and design decisions
257
+
258
+ ### Security
259
+
260
+ Selesai's RPC stdin/stdout is relatively safe because it is private to the spawned child process. It has no token, handshake, or authorization layer. It must **not** simply be put behind a TCP port. A Kilo-like server facade needs authentication, loopback binding, workspace isolation, and process lifecycle controls.
261
+
262
+ Selesai extensions execute with full system permissions, as documented in [`docs/extensions.md`](../extensions.md). The VS Code extension should not silently enable untrusted project-local extensions or resources. Mirror VS Code's workspace trust decision in the agent startup path.
263
+
264
+ ### Prompt completion versus acceptance
265
+
266
+ RPC prompt acceptance is not completion. The UI must track the asynchronous event stream and mark a run complete only on the settled/end lifecycle event. This is essential for streaming text, tool progress, approval prompts, and cancellation.
267
+
268
+ ### RPC UI limitations
269
+
270
+ RPC mode supports dialogs and simple status/widget notifications but not all TUI capabilities. A VS Code adapter needs to decide how to map or replace unsupported methods such as custom TUI components, terminal raw input, and theme-specific rendering.
271
+
272
+ ### Process versus in-process SDK
273
+
274
+ Use RPC first when reliability and isolation matter. Use the SDK when the extension needs deep typed access and can own the agent lifecycle. A hybrid is possible: keep the default backend out-of-process, offer an in-process mode for development/advanced integrations.
275
+
276
+ ### Multi-root workspaces
277
+
278
+ Both resource discovery and tool paths depend on cwd. The extension must choose and persist a workspace root/session mapping rather than relying on the extension host's process cwd. For multi-root workspaces, each session should carry an explicit workspace directory.
279
+
280
+ ## 7. Recommended product strategy
281
+
282
+ Selesai can be “similar or better than Kilo” if it focuses on its existing strengths instead of copying every transport immediately:
283
+
284
+ 1. **Ship a focused Webview extension over existing RPC.** This delivers value quickly and preserves process isolation.
285
+ 2. **Use Selesai's extension-first model as the differentiator:** custom tools, lifecycle hooks, prompt/context injection, subagents, workflows, and provider customization.
286
+ 3. **Provide native approvals and diffs**, rather than exposing a terminal transcript inside a Webview.
287
+ 4. **Add robust session restoration and workspace routing.** Selesai's RPC/session APIs already provide a good base.
288
+ 5. **Add an authenticated HTTP/SSE server only when remote/multi-client requirements justify it.** Do not introduce a network daemon solely to imitate Kilo.
289
+ 6. **Add MCP transport parity deliberately** if tool-server interoperability is a product requirement; keep MCP separate from the VS Code UI protocol.
290
+
291
+ ### Final answer to the user's questions
292
+
293
+ - **Did Kilo connect using RPA?** No evidence. It uses VS Code APIs, Webview messages, a spawned local CLI backend, authenticated localhost HTTP/SSE, feature-specific WebSockets, and MCP for external tools.
294
+ - **Can Selesai connect through RPC?** Yes. Existing Selesai RPC is a strong local integration seam: strict JSONL over a private child process with typed commands/events and extension UI requests.
295
+ - **Can Selesai have a similar or better VS Code extension?** Yes. A local RPC-backed extension is straightforward and likely the right MVP. Selesai can exceed Kilo in extensibility and session/workflow customization, while Kilo currently has an advantage in mature VS Code UX, native terminal/PTY integration, backend HTTP/SSE, and polished multi-provider UI.
296
+
297
+ ## Primary-source references
298
+
299
+ ### Kilo Code
300
+
301
+ - [Extension activation](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/extension.ts)
302
+ - [Kilo Webview provider](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/KiloProvider.ts)
303
+ - [CLI server manager](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/services/cli-backend/server-manager.ts)
304
+ - [Shared backend connection service](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/services/cli-backend/connection-service.ts)
305
+ - [Backend server config](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/services/cli-backend/types.ts)
306
+ - [MCP transports](https://github.com/Kilo-Org/kilocode/blob/main/packages/opencode/src/mcp/index.ts)
307
+ - [PTY WebSocket routing](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/agent-manager/terminal-routing.ts)
308
+ - [PTY WebSocket manager](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-vscode/src/agent-manager/terminal-manager.ts)
309
+
310
+ ### Selesai Code
311
+
312
+ - [`docs/rpc.md`](../rpc.md)
313
+ - [`docs/sdk.md`](../sdk.md)
314
+ - [`docs/extensions.md`](../extensions.md)
315
+ - [`src/modes/rpc/rpc-mode.ts`](../../src/modes/rpc/rpc-mode.ts)
316
+ - [`src/modes/rpc/rpc-types.ts`](../../src/modes/rpc/rpc-types.ts)
317
+ - [`src/modes/rpc/rpc-client.ts`](../../src/modes/rpc/rpc-client.ts)
318
+ - [`src/core/sdk.ts`](../../src/core/sdk.ts)
319
+ - [`src/index.ts`](../../src/index.ts)
package/docs/workflows.md CHANGED
@@ -1,250 +1,57 @@
1
1
  # Workflows
2
2
 
3
- Selesai ships a workflow engine under `src/extensions/workflow/`. It powers the built-in `prototype`, `quicktype`, `task`, and `loop` workflows and is designed so you can add a new workflow mode as a thin config file — no engine changes.
3
+ Selesai's built-in workflows are thin slash-command adapters over the **pi-subagents** orchestration runtime. They do not have a separate state machine, artifact protocol, or `workflow.json` format.
4
4
 
5
- ## How it fits together
5
+ ## Run a workflow
6
6
 
7
+ ```text
8
+ /workflow-task <goal>
9
+ /workflow-prototype <goal>
10
+ /workflow-quicktype <goal>
11
+ /workflow-loop <goal>
7
12
  ```
8
- src/extensions/workflow/
9
- package.json pi package manifest; loads ./extension.ts as the single entry
10
- state-machine.ts pure phase state machine (no fs, no pi API)
11
- adapter.ts pi wiring: tools, commands, events, fs, durable-state lifecycle
12
- run-state.ts versioned atomic workflow.json load/save/discovery
13
- extension.ts single pi extension that mounts every workflow mode
14
- modes/
15
- prototype.ts mode config + registration object (exported as `prototypeMode`)
16
- quicktype.ts mode config + registration object (exported as `quicktypeMode`)
17
- task.ts mode config + registration object (exported as `taskMode`)
18
- loop.ts mode config + registration object (exported as `loopMode`)
19
- ```
20
-
21
- - **`state-machine.ts`** is the deep module. It owns the phase graph, artifact gating, skip rules, the terminal close gate, and the reentrancy guard. It imports nothing external — no `node:fs`, no pi API, no `pi-tui`, no `typebox`. Every method returns a `WorkflowEffect` (a discriminated union in domain vocabulary) that the adapter pattern-matches on.
22
- - **`adapter.ts`** is the thin glue. It owns Pi/fs wiring, durable state, explicit resume, loop review persistence, and the git-based `reuse` skip predicate. Parent-written artifacts advance durable phase state and queue hidden engine continuations; every built-in mode flows automatically.
23
- - **`workflow.json`** in each artifact directory is the canonical, versioned run record. It is atomically replaced after state changes; session custom entries are only pointers for UI/history and never reconstruct an active run.
24
- - **`extension.ts`** imports each mode's registration object and calls `createWorkflowExtension(config, options)(pi)` for each. One extension load registers the model-facing artifact writer and `end_workflow` tool. Starting and resuming are user-only actions exposed by each mode's slash command.
25
- - **A mode file** is pure data: the phase list, per-phase artifact filenames, prompt generators, terminal close artifacts, and command/status/entry identities. Prompts receive `{ artifactDir, userPrompt }`. Each mode exports a `WorkflowModeRegistration` object (e.g. `prototypeMode`, `quicktypeMode`); it does not call `createWorkflowExtension` itself.
26
-
27
- ## To add a future mode
28
-
29
- Copy `modes/quicktype.ts` (the smaller one) and change the config. That's the whole change — the engine never needs editing.
30
-
31
- ### 1. Create the mode file
32
-
33
- `src/extensions/workflow/modes/rigorous.ts`:
34
-
35
- ```typescript
36
- import type {
37
- Phase,
38
- PromptContext,
39
- WorkflowConfig,
40
- WorkflowModeRegistration,
41
- } from "../state-machine.ts";
42
-
43
- const phases: Phase[] = [
44
- "grilling",
45
- "spec", // ← new phase, not in the built-in set
46
- "research",
47
- "plan",
48
- "reuse",
49
- "handoff",
50
- "loop",
51
- "audit",
52
- "sign-off", // ← new terminal phase
53
- ];
54
-
55
- const prompts: Partial<Record<Phase, (ctx: PromptContext) => string>> = {
56
- grilling: ({ artifactDir, userPrompt }) => `…grilling prompt…`,
57
- spec: ({ artifactDir }) => `…spec prompt…`,
58
- research: ({ artifactDir }) => `…research prompt…`,
59
- plan: ({ artifactDir }) => `…plan prompt…`,
60
- reuse: ({ artifactDir }) => `…reuse prompt…`,
61
- handoff: ({ artifactDir }) => `…handoff prompt…`,
62
- loop: ({ artifactDir }) => `…loop prompt…`,
63
- audit: ({ artifactDir }) => `…audit prompt…`,
64
- "sign-off": ({ artifactDir }) => `…sign-off prompt…`,
65
- };
66
-
67
- const config: WorkflowConfig = {
68
- mode: "rigorous",
69
- phases,
70
- phaseArtifacts: {
71
- grilling: "requirements.md",
72
- spec: "spec.md",
73
- research: "research.md",
74
- plan: "plan.md",
75
- reuse: "reuse.md",
76
- handoff: "handoff.md",
77
- loop: "loop-complete.md",
78
- audit: "review.md",
79
- "sign-off": "acceptance.md",
80
- },
81
- prompts,
82
- // Files that must exist before end() can close the workflow.
83
- // Config-owned — declare whatever your terminal phase requires.
84
- closeArtifacts: ["acceptance.md", "sign-off-report.md"],
85
- statusKey: "rigorous",
86
- entryType: "rigorous-phase",
87
- footerLabel: "rigorous",
88
- };
89
-
90
- export const rigorousMode: WorkflowModeRegistration = {
91
- config,
92
- commandName: "rigorous",
93
- commandDescription:
94
- "Run the rigorous workflow (grill → spec → research → plan → reuse → handoff → loop → audit → sign-off)",
95
- };
96
-
97
- export default rigorousMode;
98
- ```
99
-
100
- ### 2. Register it in `extension.ts`
101
-
102
- Add the mode to the `MODES` array in `src/extensions/workflow/extension.ts`:
103
13
 
104
- ```typescript
105
- import { rigorousMode } from "./modes/rigorous.ts";
106
-
107
- const MODES = [prototypeMode, quicktypeMode, rigorousMode] as const;
108
- ```
109
-
110
- That's it. The loader picks it up at boot (`package.json` loads only `./extension.ts`), and the `/rigorous` command is registered automatically. There is no model-facing start/resume or `next` tool — users start and resume through `/rigorous`, phases auto-advance as artifacts land, and only `end_workflow({ mode: "rigorous" })` completes the terminal phase.
14
+ Each command starts an async pi-subagents mission. Recover a completed, paused, or confusing run with the pi-subagents mission and status controls (`/subagents`, `/subagents-doctor`, or the corresponding `subagent` tool actions); there is no `/workflow-* resume` command.
111
15
 
112
16
  ## Built-in modes
113
17
 
114
- ### `prototype` and `quicktype` — full vs. quicker prototype
115
-
116
- `quicktype` has the same prototype flow except for research: it goes from grilling directly to planning. Use `prototype` when external research is needed; use `quicktype` when it is not.
117
-
118
- - `prototype`: `grilling → research → plan → reuse → handoff → loop → audit`
119
- - `quicktype`: `grilling → plan → reuse → handoff → loop → audit`
120
-
121
- ### `task` — plan → codebase exploration → handoff → build/review loop
122
-
123
- Task now follows the same phase shape as the other modes, minus grilling/research/audit: an architect subagent produces a validated `plan.md`, an optional explorer subagent produces `reuse.md`, a recapper subagent produces a validated `handoff.md`, and then a builder↔commentator review loop runs (max 3 blocking rounds). A clean review makes the workflow terminal-ready; `end_workflow({ mode: "task" })` completes it.
124
-
125
- Lifecycle: `plan → reuse → handoff → loop (build ↔ review) → terminal-ready → end_workflow({ mode: "task" })`
126
-
127
- - `/workflow-task <goal>` — start a new run
128
- - `/workflow-task resume` — list and resume active runs
129
- - `/workflow-task help` — show the lifecycle
130
- - Valid phase artifacts automatically queue the next phase prompt (the workflow does not pause at artifact boundaries)
131
- - No grilling, research, or audit phases
132
- - `reuse.md` is optional; it is skipped automatically when the project has no git history
133
-
134
- ### `loop` — direct build/review loop
135
-
136
- Use this after the plan was already agreed in the current conversation. It captures the agreed context into a parent-owned `handoff.md` artifact, then runs an engine-owned `loop` phase: builder changes workspace code, commentator independently validates the diff and relevant checks, then blocking feedback returns to the builder (max 3 blocking rounds). A clean review writes `loop-complete.md`, makes the run terminal-ready, and requires explicit completion.
137
-
138
- Fresh subagents do not inherit the parent conversation. The parent forks a `recapper` subagent once to synthesize a concise, self-contained handoff document directly from the inherited conversation. The parent validates the handoff marker and writes `handoff.md` via `write_workflow_artifact`. After that, every builder and commentator call reads `handoff.md` instead of relying on the parent conversation. Persisted `loop-review-N.md` files feed blocking fixes back to the builder.
139
-
140
- Lifecycle: `handoff → loop (build ↔ review) → terminal-ready → end_workflow({ mode: "loop" })`
141
-
142
- - `/workflow-loop <goal>` — start a direct build/review run
143
- - `/workflow-loop resume` / `/workflow-loop resume <id-or-artifact-dir-or-workflow.json>` — list or resume a run
144
- - `/workflow-loop help` — show the lifecycle
145
-
146
- ## Config reference
18
+ | Command | Shape |
19
+ | --- | --- |
20
+ | `/workflow-task` | plan → reuse → handoff → build/review/fix loop |
21
+ | `/workflow-prototype` | parallel research + codebase exploration → plan → handoff → build/review/fix loop → audit |
22
+ | `/workflow-quicktype` | plan → reuse → handoff → build/review/fix loop → audit |
23
+ | `/workflow-loop` | direct build/review/fix loop for an already-agreed plan |
147
24
 
148
- | Field | Type | Description |
149
- |---|---|---|
150
- | `mode` | `string` | Mode name, echoed in entry payloads and messages. |
151
- | `phases` | `Phase[]` | Ordered phase list. `Phase` is `string` — new phase names are allowed. |
152
- | `phaseArtifacts` | `Partial<Record<Phase, string>>` | The artifact file each phase must produce before advancing. Omit a phase to skip its gate. |
153
- | `prompts` | `Partial<Record<Phase, (ctx) => string>>` | Prompt generator per phase. `ctx = { artifactDir, userPrompt }`. |
154
- | `closeArtifacts` | `string[]` | Files that must exist before `end()` succeeds. Config-owned, no built-in default. |
155
- | `skipRules?` | `{ phase, shouldSkip }[]` | Optional per-phase skip rules. `shouldSkip` is a boolean predicate; when true the engine skips to the next phase. Omit to use the adapter's default (skip `reuse` when the project has no git history). |
156
- | `statusKey` | `string` | Footer status key. |
157
- | `entryType` | `string` | Session-history custom-type. It stores a pointer only; `workflow.json` is canonical. |
158
- | `footerLabel` | `string` | Label shown in the footer (`● label · step/total phase`). |
25
+ The prototype mode uses `runs.all` for its independent research and codebase-exploration work. All modes use `runs.run` for ordered handoffs. The build/review/fix loop uses `workflowScript` because its next step depends on the reviewer result; a blocking review gets a scoped fix round, while `clean` plus no remaining work ends the run.
159
26
 
160
- ### Adapter options
27
+ ## Extending workflows
161
28
 
162
- The second argument to `createWorkflowExtension`:
29
+ The extension seam is `src/extensions/workflow/modes.ts`.
163
30
 
164
- | Field | Description |
165
- |---|---|
166
- | `commandName` | The `/<command>` name users type to kick off the workflow. |
167
- | `commandDescription` | Description shown in the command list. |
31
+ Add one `WorkflowMode` entry to `WORKFLOW_MODES`:
168
32
 
169
- ## Durable runs and explicit resume
170
-
171
- Each started workflow receives a UUID artifact directory under `.selesai/artifacts/` and an adjacent `workflow.json`. It stores the mode, phase, armed state, loop round/review path, and timestamps. Writes use a temporary sibling file plus rename, so a crash cannot partially overwrite the canonical record.
172
-
173
- Runs are **never** auto-resumed on session start. At most one run can be attached to a Pi instance, but older active runs remain resumable:
174
-
175
- - Workflow initiation is user-only: `/workflow-prototype <goal>`, `/workflow-quicktype <goal>`, `/workflow-task <goal>`, or `/workflow-loop <goal>`
176
- - `/workflow-prototype resume <id-or-artifact-dir-or-workflow.json>` / `/workflow-quicktype resume ...` / `/workflow-task resume ...` / `/workflow-loop resume ...`
177
- - `/workflow-prototype resume`, `/workflow-quicktype resume`, `/workflow-task resume`, or `/workflow-loop resume` lists active runs (and offers a UI picker when available).
178
- - `/workflow-prototype help`, `/workflow-quicktype help`, `/workflow-task help`, or `/workflow-loop help` shows the start, resume, continue, and explicit-completion lifecycle.
179
-
180
- Resume validates the selected file is under the artifacts base, belongs to that mode, is active, and matches its containing directory. It reconciles the current expected artifact once before emitting the current prompt, covering a crash after `write_workflow_artifact` writes the file but before the phase-state write. Valid artifact writes queue one hidden engine-controlled continuation using `steer` and terminate the current parent turn; invalid writes stay in the current phase and do not terminate. Prompts injected by start, resume, and continue commands are hidden custom messages rather than visible synthetic user messages. Transition-capable calls (`write_workflow_artifact`, loop commentator transitions, and `end_workflow`) must be the sole tool call in their assistant batch; the adapter fails closed when that cannot be proven. Corrupt records are skipped during discovery. Reloads never auto-resume; the user must explicitly resume through a mode's slash command.
181
-
182
- A valid terminal artifact makes a workflow **terminal-ready**; it does not complete the run. Call `end_workflow({ mode })` to write `status: "completed"`, append the done entry, and terminate. This is the only completion path.
183
-
184
- ## Artifact ownership
185
-
186
- Workflow artifacts have one writer: the parent session's `write_workflow_artifact` tool. Every workflow child call uses `output: false` and returns inline. In artifact phases (`plan`, `reuse`, `handoff`, and `audit`), the parent inspects that result, validates any required marker, and immediately passes it to `write_workflow_artifact`. Child output paths and fallback persistence are deliberately disabled; a child result alone cannot create an artifact or advance a phase.
187
-
188
- The implement/review loop is the explicit exception to parent persistence, not inline return: the engine persists `loop-review-<round>.md` and `loop-complete.md` from commentator results so it can manage review rounds. Builders only change workspace code.
189
-
190
- Every state-machine method returns a `WorkflowEffect` — a discriminated union the adapter switches on:
191
-
192
- | Effect | Meaning |
193
- |---|---|
194
- | `started` | `start()` succeeded; first phase prompt + entry + footer. |
195
- | `alreadyActive` | `start()` called while a workflow is active. |
196
- | `advanced` | Phase moved forward (optionally `skipped` a phase). |
197
- | `blocked` | Current phase's artifact is missing. |
198
- | `terminalNeedsArtifacts` | At the last phase; a close artifact is missing. |
199
- | `terminalReady` | At the last phase; all close artifacts present — call `end()`. |
200
- | `closed` | `end()` succeeded; workflow finished. |
201
- | `endBlocked` | `end()` called from the wrong phase or with close artifacts missing. |
202
- | `idle` | No active workflow. |
203
- | `noOp` | Auto-advance checked, nothing to do (not active, not armed, artifact not present, or already advancing). |
204
-
205
- The `tool_result` auto-advance hook is one line:
206
-
207
- ```typescript
208
- const eff = await sm.onArtifactMaybe(deps);
209
- applyEffect(pi, ctx, config, eff);
210
- ```
211
-
212
- The reentrancy guard lives inside `onArtifactMaybe` — concurrent calls return `noOp`, so a double `write` in one turn cannot double-advance the phase.
213
-
214
- ## Skip rules
215
-
216
- By default the adapter skips the `reuse` phase when the project has no git history. To override, supply `skipRules` in your config:
217
-
218
- ```typescript
219
- skipRules: [
220
- { phase: "research", shouldSkip: async () => isWellUnderstoodDomain() },
221
- { phase: "reuse", shouldSkip: async () => isEmptyProject() },
222
- ],
33
+ ```ts
34
+ {
35
+ command: "workflow-rigorous",
36
+ description: "Run the rigorous workflow.",
37
+ launch: (goal) => ({
38
+ workflowScript: `const goal = ${JSON.stringify(goal)};
39
+ return runs.run("plan", { agent: "architect", task: "Plan: " + goal });`,
40
+ }),
41
+ }
223
42
  ```
224
43
 
225
- `shouldSkip` is a boolean predicate. When it returns `true`, the engine skips to the next phase in `phases` — the mode owns the transition graph, not the adapter.
226
-
227
- ## Testing a mode
44
+ `launch(goal)` returns pi-subagents public execution fields. Prefer the native execution shapes where the mode is static:
228
45
 
229
- The state machine is tested directly with in-memory stubs — no filesystem, no pi mock, no events:
46
+ - `chain` for a fixed ordered sequence, including human checkpoints.
47
+ - `tasks` for independent, read-only parallel work.
48
+ - `workflowScript` only when the orchestration is conditional, iterative, needs dynamic fan-out, or combines native run operations.
230
49
 
231
- ```typescript
232
- import { WorkflowStateMachine } from "../extensions/workflow/state-machine.ts";
50
+ `extension.ts` automatically registers every entry in `WORKFLOW_MODES`; no new command plumbing is needed. The mode owns task wording and execution shape. The extension owns only argument validation, async launch, agent scope, and mission creation.
233
51
 
234
- const files = new Set<string>();
235
- const deps = {
236
- async artifactExists(phase, dir) {
237
- const file = config.phaseArtifacts[phase];
238
- return file ? files.has(`${dir}/${file}`) : true;
239
- },
240
- async fileExists(path) { return files.has(path); },
241
- async mkdirArtifactDir() {},
242
- artifactPathFor: (goal) => `/fake/${goal}`,
243
- };
244
-
245
- const sm = new WorkflowStateMachine(config);
246
- const eff = await sm.start("build X", deps);
247
- expect(eff.kind).toBe("started");
248
- ```
52
+ ## Constraints
249
53
 
250
- See `src/__tests__/state-machine.test.ts` for the full set of transition, skip, terminal, rehydrate, and validation tests.
54
+ - `workflowScript`, `chain`, and `tasks` are alternative top-level pi-subagents execution modes. A mode that needs an auto-loop and preceding/following phases should use `workflowScript` and call `runs.run` / `runs.all` within it.
55
+ - Keep one writer at a time. Parallel lanes should be research or review unless they are isolated in worktrees.
56
+ - Workflow progress ledgers are under `.pi-subagents/progress/` and are local runtime artifacts, not durable workflow state.
57
+ - The outer mission and pi-subagents run artifacts are the recovery record.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@selesai/code",
3
- "version": "0.9.3",
3
+ "version": "0.9.4",
4
4
  "description": "Maintained, extension-first Pi coding agent with built-in workflows, subagents, web research, questions, skills, and an enhanced terminal UI.",
5
5
  "type": "module",
6
6
  "repository": {