@fyeeme/pi-dynamic-workflows 0.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +373 -0
- package/README.zh-CN.md +359 -0
- package/index.ts +650 -0
- package/package.json +58 -0
- package/sessions/spawn.ts +15 -0
- package/src/agent/dispatch.ts +76 -0
- package/src/budget/caps.ts +42 -0
- package/src/budget/index.ts +8 -0
- package/src/budget/pool.ts +118 -0
- package/src/cache/index.ts +7 -0
- package/src/cache/journal.ts +184 -0
- package/src/cache/key.ts +97 -0
- package/src/determinism/ast-guard.ts +196 -0
- package/src/errors.ts +55 -0
- package/src/format.ts +27 -0
- package/src/index.ts +28 -0
- package/src/inspect.ts +237 -0
- package/src/lifecycle.ts +75 -0
- package/src/loader.ts +50 -0
- package/src/outcomes.ts +113 -0
- package/src/planner.ts +66 -0
- package/src/runner/index.ts +188 -0
- package/src/runner/stage-executor.ts +1078 -0
- package/src/state/index.ts +1 -0
- package/src/state/names.ts +33 -0
- package/src/types.ts +332 -0
- package/src/ui-groups.ts +43 -0
package/index.ts
ADDED
|
@@ -0,0 +1,650 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* pi-dynamic-workflows — extension entry (Task 8 wiring).
|
|
3
|
+
*
|
|
4
|
+
* Registers the `run_workflow` tool so an agent can construct and execute a
|
|
5
|
+
* workflow from within pi. The engine (src/runner) does the work; this entry
|
|
6
|
+
* only adapts the agent's JSON args into the code-form WorkflowDefinition and
|
|
7
|
+
* runs it with the default dispatch (real `pi --mode json` subprocesses).
|
|
8
|
+
*/
|
|
9
|
+
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
10
|
+
import { Type } from "typebox";
|
|
11
|
+
import { buildRenderGroups, type PhaseDef } from "./src/ui-groups.ts";
|
|
12
|
+
import { BOLD, CYAN, DIM, GREEN, RED, YELLOW, fmtTokens, stepIdOf } from "./src/format.ts";
|
|
13
|
+
import { defineWorkflow, runWorkflow } from "./src/index.ts";
|
|
14
|
+
import type { AgentCallId, Budget, RunResult, StageType, StepContext, StepDefinition, StepResult, StepStats, WorkflowDefinition } from "./src/types.ts";
|
|
15
|
+
import { WorkflowError } from "./src/errors.ts";
|
|
16
|
+
import type { AgentLifecycleListeners } from "./src/lifecycle.ts";
|
|
17
|
+
import { WorkflowInspect } from "./src/inspect.ts";
|
|
18
|
+
|
|
19
|
+
/** Last completed run, exposed to /wf-inspect for interactive review. */
|
|
20
|
+
let lastRunResult: RunResult | null = null;
|
|
21
|
+
|
|
22
|
+
/** Phases of the most recent run — handed to /wf-inspect so the post-run view
|
|
23
|
+
* groups steps the same way the live widget did (C1 consistency). */
|
|
24
|
+
let lastPhases: readonly PhaseDef[] | undefined;
|
|
25
|
+
|
|
26
|
+
/** Active widget during a run — lets /wf-inspect show a live snapshot
|
|
27
|
+
* before the run completes (lastRunResult is only set post-run). */
|
|
28
|
+
let activeWidget: { snapshot(): RunResult } | null = null;
|
|
29
|
+
|
|
30
|
+
// ---------------------------------------------------------------------------
|
|
31
|
+
// Parameter schema (the JSON-serializable workflow subset)
|
|
32
|
+
// ---------------------------------------------------------------------------
|
|
33
|
+
|
|
34
|
+
const BudgetExhaustPolicy = Type.Optional(
|
|
35
|
+
Type.Union([Type.Literal("throw"), Type.Literal("null")], {
|
|
36
|
+
description: "Budget-exhaustion policy for this step: \"throw\" (default) aborts the run; \"null\" degrades this step to a null result so siblings/downstream continue (the run result records degraded steps).",
|
|
37
|
+
}),
|
|
38
|
+
);
|
|
39
|
+
|
|
40
|
+
const StepSchema = Type.Union([
|
|
41
|
+
Type.Object({
|
|
42
|
+
id: Type.String({ description: "Step id; referenceable as {{step.<id>}} in later prompts" }),
|
|
43
|
+
type: Type.Literal("agent"),
|
|
44
|
+
prompt: Type.String({ description: "Prompt text; may use {{input}} / {{step.<id>}}" }),
|
|
45
|
+
model: Type.Optional(Type.String({ description: "Full model id from the session (e.g. claude-sonnet-5). Omit to use the default session model. Invalid ids are dropped." })),
|
|
46
|
+
systemPrompt: Type.Optional(Type.String()),
|
|
47
|
+
onBudgetExhaust: BudgetExhaustPolicy,
|
|
48
|
+
}),
|
|
49
|
+
Type.Object({
|
|
50
|
+
id: Type.String(),
|
|
51
|
+
type: Type.Literal("log"),
|
|
52
|
+
message: Type.String({ description: "Narrative line emitted into the progress widget (zero dispatch / zero tokens)" }),
|
|
53
|
+
onBudgetExhaust: BudgetExhaustPolicy,
|
|
54
|
+
}),
|
|
55
|
+
Type.Object({
|
|
56
|
+
id: Type.String(),
|
|
57
|
+
type: Type.Literal("fan_out"),
|
|
58
|
+
items: Type.Array(Type.Unknown(), { description: "Static list to fan out over" }),
|
|
59
|
+
prompt: Type.String({ description: "Per-item prompt template; {{item}} is the current item" }),
|
|
60
|
+
model: Type.Optional(Type.String()),
|
|
61
|
+
parallelism: Type.Optional(Type.Number()),
|
|
62
|
+
onBudgetExhaust: BudgetExhaustPolicy,
|
|
63
|
+
}),
|
|
64
|
+
Type.Object({
|
|
65
|
+
id: Type.String(),
|
|
66
|
+
type: Type.Literal("adversarial"),
|
|
67
|
+
prompt: Type.String({ description: "Produces the candidate to be judged" }),
|
|
68
|
+
rubric: Type.Array(Type.String()),
|
|
69
|
+
judges: Type.Optional(Type.Number()),
|
|
70
|
+
minPass: Type.Optional(Type.Number()),
|
|
71
|
+
model: Type.Optional(Type.String({ description: "Applies to the produce call AND the judges" })),
|
|
72
|
+
onBudgetExhaust: BudgetExhaustPolicy,
|
|
73
|
+
}),
|
|
74
|
+
Type.Object({
|
|
75
|
+
id: Type.String(),
|
|
76
|
+
type: Type.Literal("tournament"),
|
|
77
|
+
prompt: Type.String({ description: "Candidate producer prompt" }),
|
|
78
|
+
candidates: Type.Number(),
|
|
79
|
+
judges: Type.Number(),
|
|
80
|
+
model: Type.Optional(Type.String({ description: "Applies to the candidate producers AND the judges" })),
|
|
81
|
+
onBudgetExhaust: BudgetExhaustPolicy,
|
|
82
|
+
}),
|
|
83
|
+
Type.Object({
|
|
84
|
+
id: Type.String(),
|
|
85
|
+
type: Type.Literal("classify_route"),
|
|
86
|
+
prompt: Type.String({ description: "Classifier prompt; agent should reply {category: \"...\"}" }),
|
|
87
|
+
routes: Type.Record(
|
|
88
|
+
Type.String(),
|
|
89
|
+
Type.Array(Type.Object({ id: Type.String(), prompt: Type.String(), model: Type.Optional(Type.String()) })),
|
|
90
|
+
),
|
|
91
|
+
fallback: Type.Optional(
|
|
92
|
+
Type.Array(Type.Object({ id: Type.String(), prompt: Type.String(), model: Type.Optional(Type.String()) })),
|
|
93
|
+
),
|
|
94
|
+
model: Type.Optional(Type.String()),
|
|
95
|
+
onBudgetExhaust: BudgetExhaustPolicy,
|
|
96
|
+
}),
|
|
97
|
+
]);
|
|
98
|
+
|
|
99
|
+
const BudgetSchema = Type.Object({
|
|
100
|
+
maxAgents: Type.Optional(Type.Number()),
|
|
101
|
+
maxTokens: Type.Optional(Type.Number()),
|
|
102
|
+
maxDurationMs: Type.Optional(Type.Number()),
|
|
103
|
+
});
|
|
104
|
+
|
|
105
|
+
const PhaseSchema = Type.Object({
|
|
106
|
+
title: Type.String({ description: "Phase display name" }),
|
|
107
|
+
detail: Type.Optional(Type.String({ description: "Short detail shown next to the phase title" })),
|
|
108
|
+
stepIds: Type.Array(Type.String(), { description: "Step ids belonging to this phase" }),
|
|
109
|
+
});
|
|
110
|
+
|
|
111
|
+
const WorkflowSchema = Type.Object({
|
|
112
|
+
name: Type.String(),
|
|
113
|
+
description: Type.Optional(Type.String()),
|
|
114
|
+
steps: Type.Array(StepSchema),
|
|
115
|
+
budget: Type.Optional(BudgetSchema),
|
|
116
|
+
phases: Type.Optional(Type.Array(PhaseSchema, { description: "Group steps into phases for progress-tree UI" })),
|
|
117
|
+
});
|
|
118
|
+
|
|
119
|
+
const RunWorkflowParams = Type.Object({
|
|
120
|
+
workflow: WorkflowSchema,
|
|
121
|
+
input: Type.Optional(Type.String({ description: "Initial ctx.input (also {{input}} in prompts)" })),
|
|
122
|
+
cwd: Type.Optional(Type.String({ description: "Working dir + journal base. Default: session cwd" })),
|
|
123
|
+
now: Type.Optional(Type.Number({ description: "Deterministic inception ms (resume seed). Default: Date.now()" })),
|
|
124
|
+
});
|
|
125
|
+
|
|
126
|
+
// ---------------------------------------------------------------------------
|
|
127
|
+
// Template compilation (data prompt string → code prompt function)
|
|
128
|
+
// ---------------------------------------------------------------------------
|
|
129
|
+
|
|
130
|
+
const HAS_TEMPLATE = /\{\{[^}]+\}\}/;
|
|
131
|
+
const TEMPLATE_TOKEN = /\{\{([^}]+)\}\}/g;
|
|
132
|
+
|
|
133
|
+
function fmt(value: unknown): string {
|
|
134
|
+
if (value === undefined || value === null) return "";
|
|
135
|
+
return typeof value === "string" ? value : JSON.stringify(value);
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
/** Where a template is being filled — gates which tokens are valid.
|
|
139
|
+
* agent: {{input}} and {{step.<id>}} (no {{item}}).
|
|
140
|
+
* fanout-item: {{input}} and {{item}} (no {{step.<id>}} — items run before
|
|
141
|
+
* step results merge, so a step reference is a definition error, not a miss). */
|
|
142
|
+
type TemplateMode = "agent" | "fanout-item";
|
|
143
|
+
|
|
144
|
+
/** Resolve the three template tokens — {{input}}, {{item}} (fan_out item
|
|
145
|
+
* prompts only), {{step.<id>}} (agent prompts only) — in a SINGLE
|
|
146
|
+
* left-to-right pass. Substituted values are opaque: a literal {{...}} inside
|
|
147
|
+
* input/item/step results is emitted verbatim and NEVER re-evaluated, so
|
|
148
|
+
* workflows that process arbitrary text (logs, source, foreign templates)
|
|
149
|
+
* cannot be silently corrupted, value-hijacked, or crashed by their own data.
|
|
150
|
+
* Unknown or out-of-context tokens raise a categorized `compile` error naming
|
|
151
|
+
* the token and the referencing step. */
|
|
152
|
+
export function fill(template: string, mode: TemplateMode, ctx: StepContext, item: unknown | undefined, stepId: string): string {
|
|
153
|
+
let out = "";
|
|
154
|
+
let last = 0;
|
|
155
|
+
for (const m of template.matchAll(TEMPLATE_TOKEN)) {
|
|
156
|
+
const token = m[0];
|
|
157
|
+
const idx = m.index ?? 0;
|
|
158
|
+
out += template.slice(last, idx);
|
|
159
|
+
out += resolveToken(token, m[1].trim(), mode, ctx, item, stepId);
|
|
160
|
+
last = idx + token.length;
|
|
161
|
+
}
|
|
162
|
+
return out + template.slice(last);
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
function resolveToken(
|
|
166
|
+
token: string,
|
|
167
|
+
inner: string,
|
|
168
|
+
mode: TemplateMode,
|
|
169
|
+
ctx: StepContext,
|
|
170
|
+
item: unknown | undefined,
|
|
171
|
+
stepId: string,
|
|
172
|
+
): string {
|
|
173
|
+
if (inner === "input") return fmt(ctx.input);
|
|
174
|
+
if (inner === "item") {
|
|
175
|
+
if (mode !== "fanout-item") throw badToken(token, "{{item}} is only valid in a fan_out item prompt", stepId);
|
|
176
|
+
return fmt(item);
|
|
177
|
+
}
|
|
178
|
+
const stepRef = /^step\.(.+)$/.exec(inner);
|
|
179
|
+
if (stepRef) {
|
|
180
|
+
const id = stepRef[1];
|
|
181
|
+
if (mode === "fanout-item") throw badToken(token, `{{step.${id}}} is not available inside a fan_out item prompt`, stepId, { refId: id });
|
|
182
|
+
try {
|
|
183
|
+
return fmt(ctx.step(id).results);
|
|
184
|
+
} catch (e) {
|
|
185
|
+
throw badToken(token, `{{step.${id}}} could not be resolved: ${(e as Error).message}`, stepId, { refId: id });
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
throw badToken(token, "unknown template token", stepId);
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
function badToken(token: string, reason: string, stepId: string, extra?: Readonly<Record<string, unknown>>): WorkflowError {
|
|
192
|
+
return new WorkflowError(`step "${stepId}": ${reason} (token: ${token})`, {
|
|
193
|
+
category: "compile",
|
|
194
|
+
detail: { token, stepId, ...extra },
|
|
195
|
+
});
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
/** A string prompt becomes a function only if it contains a template token. */
|
|
199
|
+
function promptOf(s: string, stepId: string): string | ((ctx: StepContext) => string) {
|
|
200
|
+
if (!HAS_TEMPLATE.test(s)) return s;
|
|
201
|
+
return (ctx: StepContext) => fill(s, "agent", ctx, undefined, stepId);
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
// ---------------------------------------------------------------------------
|
|
205
|
+
// Data workflow → code WorkflowDefinition
|
|
206
|
+
// ---------------------------------------------------------------------------
|
|
207
|
+
|
|
208
|
+
interface RouteStepData {
|
|
209
|
+
readonly id: string;
|
|
210
|
+
readonly prompt: string;
|
|
211
|
+
readonly model?: string;
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
function routeStepToCode(s: RouteStepData): StepDefinition {
|
|
215
|
+
return { id: s.id, type: "agent", prompt: promptOf(s.prompt, s.id), model: s.model };
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
function buildWorkflow(w: {
|
|
219
|
+
readonly name: string;
|
|
220
|
+
readonly description?: string;
|
|
221
|
+
readonly steps: readonly StepData[];
|
|
222
|
+
readonly budget?: Budget;
|
|
223
|
+
}): WorkflowDefinition {
|
|
224
|
+
return defineWorkflow({
|
|
225
|
+
name: w.name,
|
|
226
|
+
description: w.description,
|
|
227
|
+
budget: w.budget,
|
|
228
|
+
steps: w.steps.map((s): StepDefinition => {
|
|
229
|
+
switch (s.type) {
|
|
230
|
+
case "agent":
|
|
231
|
+
return { id: s.id, type: "agent", prompt: promptOf(s.prompt, s.id), model: s.model, systemPrompt: s.systemPrompt, onBudgetExhaust: s.onBudgetExhaust };
|
|
232
|
+
case "log":
|
|
233
|
+
return { id: s.id, type: "log", message: s.message, onBudgetExhaust: s.onBudgetExhaust };
|
|
234
|
+
case "fan_out":
|
|
235
|
+
return {
|
|
236
|
+
id: s.id,
|
|
237
|
+
type: "fan_out",
|
|
238
|
+
over: () => s.items,
|
|
239
|
+
agent: (item, _index, ctx) => ({ prompt: fill(s.prompt, "fanout-item", ctx, item, s.id), model: s.model }),
|
|
240
|
+
parallelism: s.parallelism,
|
|
241
|
+
onBudgetExhaust: s.onBudgetExhaust,
|
|
242
|
+
};
|
|
243
|
+
case "adversarial":
|
|
244
|
+
return {
|
|
245
|
+
id: s.id,
|
|
246
|
+
type: "adversarial",
|
|
247
|
+
produce: { prompt: promptOf(s.prompt, s.id), model: s.model },
|
|
248
|
+
rubric: [...s.rubric],
|
|
249
|
+
judges: s.judges,
|
|
250
|
+
minPass: s.minPass,
|
|
251
|
+
onBudgetExhaust: s.onBudgetExhaust,
|
|
252
|
+
};
|
|
253
|
+
case "tournament":
|
|
254
|
+
return {
|
|
255
|
+
id: s.id,
|
|
256
|
+
type: "tournament",
|
|
257
|
+
candidates: s.candidates,
|
|
258
|
+
judges: s.judges,
|
|
259
|
+
produce: { prompt: promptOf(s.prompt, s.id), model: s.model },
|
|
260
|
+
onBudgetExhaust: s.onBudgetExhaust,
|
|
261
|
+
};
|
|
262
|
+
case "classify_route": {
|
|
263
|
+
const routes: Record<string, readonly StepDefinition[]> = {};
|
|
264
|
+
for (const [cat, steps] of Object.entries(s.routes)) routes[cat] = steps.map(routeStepToCode);
|
|
265
|
+
const fallback = s.fallback ? s.fallback.map(routeStepToCode) : undefined;
|
|
266
|
+
return { id: s.id, type: "classify_route", classifier: { prompt: promptOf(s.prompt, s.id), model: s.model }, routes, fallback, onBudgetExhaust: s.onBudgetExhaust };
|
|
267
|
+
}
|
|
268
|
+
}
|
|
269
|
+
}),
|
|
270
|
+
});
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
// One discriminated data-step type (kept loose; the TypeBox schema is the contract).
|
|
274
|
+
type StepData =
|
|
275
|
+
| { id: string; type: "agent"; prompt: string; model?: string; systemPrompt?: string; onBudgetExhaust?: "throw" | "null" }
|
|
276
|
+
| { id: string; type: "log"; message: string; onBudgetExhaust?: "throw" | "null" }
|
|
277
|
+
| { id: string; type: "fan_out"; items: readonly unknown[]; prompt: string; model?: string; parallelism?: number; onBudgetExhaust?: "throw" | "null" }
|
|
278
|
+
| { id: string; type: "adversarial"; prompt: string; rubric: readonly string[]; judges?: number; minPass?: number; model?: string; onBudgetExhaust?: "throw" | "null" }
|
|
279
|
+
| { id: string; type: "tournament"; prompt: string; candidates: number; judges: number; model?: string; onBudgetExhaust?: "throw" | "null" }
|
|
280
|
+
| {
|
|
281
|
+
id: string;
|
|
282
|
+
type: "classify_route";
|
|
283
|
+
prompt: string;
|
|
284
|
+
routes: Readonly<Record<string, readonly RouteStepData[]>>;
|
|
285
|
+
fallback?: readonly RouteStepData[];
|
|
286
|
+
model?: string;
|
|
287
|
+
onBudgetExhaust?: "throw" | "null";
|
|
288
|
+
};
|
|
289
|
+
|
|
290
|
+
// ---------------------------------------------------------------------------
|
|
291
|
+
// Progress widget — bridges lifecycle events → TUI setWidget
|
|
292
|
+
// ---------------------------------------------------------------------------
|
|
293
|
+
|
|
294
|
+
type CallStatus = "running" | "done" | "failed" | "skipped" | "retried" | "cached";
|
|
295
|
+
|
|
296
|
+
interface CallInfo {
|
|
297
|
+
readonly stepId: string;
|
|
298
|
+
readonly status: CallStatus;
|
|
299
|
+
readonly tokens: number;
|
|
300
|
+
readonly model?: string;
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
export function buildProgressWidget(
|
|
304
|
+
steps: readonly { id: string; type: string }[],
|
|
305
|
+
setWidget: (lines: string[] | undefined) => void,
|
|
306
|
+
setStatus: (text: string | undefined) => void,
|
|
307
|
+
phases?: readonly PhaseDef[],
|
|
308
|
+
): AgentLifecycleListeners & { cleanup(): void; snapshot(): RunResult } {
|
|
309
|
+
const start = Date.now();
|
|
310
|
+
let lastStreamRender = 0;
|
|
311
|
+
// Once cleanup() runs (run finished), late events from the abort window — the
|
|
312
|
+
// SIGTERM→SIGKILL grace period during which a child can still emit streamed
|
|
313
|
+
// deltas — must not re-create the panel via render()/setWidget.
|
|
314
|
+
let disposed = false;
|
|
315
|
+
const calls = new Map<AgentCallId, CallInfo>();
|
|
316
|
+
// C2: narrative lines emitted by `log` steps, keyed by step id.
|
|
317
|
+
const logLines = new Map<string, string>();
|
|
318
|
+
// C3: accumulated streaming text per in-flight call (delta chunks from onUpdate).
|
|
319
|
+
const streamText = new Map<string, string>();
|
|
320
|
+
// Live output capture: callId → the settled agent's final text (from onAgentEnd
|
|
321
|
+
// `output`), so a live snapshot shows REAL results, not fabricated progress text.
|
|
322
|
+
const callOutputs = new Map<string, string>();
|
|
323
|
+
// Every step starts at 0 expected agents; onAgentStart/onAgentCacheHit
|
|
324
|
+
// increment as calls fire (fan_out totals emerge at runtime). Pre-seeding
|
|
325
|
+
// non-fan_out steps to 1 double-counted (1→2 on start), leaving completed
|
|
326
|
+
// single-agent steps stuck showing [1/2].
|
|
327
|
+
const expected = new Map(steps.map((s) => [s.id, 0]));
|
|
328
|
+
|
|
329
|
+
// C1+D1: phase grouping — shared pure builder (also used by /wf-inspect).
|
|
330
|
+
const renderGroups = buildRenderGroups(steps, (s) => s.id, phases);
|
|
331
|
+
|
|
332
|
+
// callId format: `${stepId}#${n}` (e.g. "fan#2", "adv#produce") — see src/format.ts stepIdOf.
|
|
333
|
+
|
|
334
|
+
function render(): void {
|
|
335
|
+
const picons: Record<string, string> = {
|
|
336
|
+
done: GREEN("✓"),
|
|
337
|
+
failed: RED("✗"),
|
|
338
|
+
skipped: YELLOW("⏭"),
|
|
339
|
+
running: YELLOW("⏳"),
|
|
340
|
+
cached: GREEN("↻"),
|
|
341
|
+
};
|
|
342
|
+
|
|
343
|
+
const lines: string[] = [];
|
|
344
|
+
const renderStep = (s: { id: string; type: string }, indent: boolean): void => {
|
|
345
|
+
// C2: a `log` step renders as a distinct narrative line, not an agent row.
|
|
346
|
+
if (s.type === "log") {
|
|
347
|
+
const msg = logLines.get(s.id);
|
|
348
|
+
if (msg !== undefined) lines.push(`${indent ? " " : " "}${DIM(msg)}`);
|
|
349
|
+
return;
|
|
350
|
+
}
|
|
351
|
+
const total = expected.get(s.id) ?? 0;
|
|
352
|
+
const entries = [...calls.values()].filter((c) => c.stepId === s.id);
|
|
353
|
+
const running = entries.some((c) => c.status === "running");
|
|
354
|
+
const failed = entries.filter((c) => c.status === "failed").length;
|
|
355
|
+
const skipped = entries.filter((c) => c.status === "skipped").length;
|
|
356
|
+
const done = entries.filter((c) => c.status === "done" || c.status === "cached").length;
|
|
357
|
+
const cached = entries.filter((c) => c.status === "cached").length;
|
|
358
|
+
const tokens = entries.reduce((sum, c) => sum + c.tokens, 0);
|
|
359
|
+
const models = new Set(entries.map((c) => c.model).filter((m): m is string => Boolean(m)));
|
|
360
|
+
|
|
361
|
+
const icon = failed > 0 ? picons.failed
|
|
362
|
+
: skipped > 0 && done === 0 ? picons.skipped
|
|
363
|
+
: total > 0 && done >= total ? (cached > 0 ? picons.cached : picons.done)
|
|
364
|
+
: running ? picons.running
|
|
365
|
+
: DIM("○");
|
|
366
|
+
|
|
367
|
+
const progress = total > 1 ? ` [${done}/${total}]` : "";
|
|
368
|
+
const tok = tokens > 0 ? ` · ${fmtTokens(tokens)} tok` : "";
|
|
369
|
+
// A8/C4: show the serving model when known (single model shown; mixed
|
|
370
|
+
// fan_out models collapse to a count to avoid a noisy line).
|
|
371
|
+
const modelTag = models.size === 1 ? ` · ${[...models][0]}` : models.size > 1 ? ` · ${models.size} models` : "";
|
|
372
|
+
const pad = indent ? " " : " ";
|
|
373
|
+
lines.push(`${pad}${icon} ${s.id}${progress}${tok}${modelTag}`);
|
|
374
|
+
};
|
|
375
|
+
|
|
376
|
+
// Every phase group renders its header — a phase interrupted by ungrouped
|
|
377
|
+
// items produces two groups; deduplicating the second header would leave
|
|
378
|
+
// an indented, header-less orphan row (review L2).
|
|
379
|
+
for (const g of renderGroups) {
|
|
380
|
+
if (g.kind === "phase" && g.title) {
|
|
381
|
+
lines.push(` ${BOLD(g.title)}${g.detail ? DIM(` — ${g.detail}`) : ""}`);
|
|
382
|
+
}
|
|
383
|
+
for (const s of g.items) renderStep(s, g.kind === "phase");
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
// C3: streaming tail — the most-recently-started running call's accumulated
|
|
387
|
+
// text, truncated to the last 3 lines, so concurrent fan-out previews only
|
|
388
|
+
// the active call instead of flooding the widget.
|
|
389
|
+
const running = [...calls.entries()].reverse().find(([id, c]) => c.status === "running" && streamText.has(id));
|
|
390
|
+
if (running) {
|
|
391
|
+
const [id] = running;
|
|
392
|
+
const text = streamText.get(id) ?? "";
|
|
393
|
+
const tail = text.split("\n").slice(-3);
|
|
394
|
+
for (const ln of tail) {
|
|
395
|
+
const clipped = ln.length > 100 ? `${ln.slice(0, 99)}…` : ln;
|
|
396
|
+
if (clipped) lines.push(DIM(` ↳ ${clipped}`));
|
|
397
|
+
}
|
|
398
|
+
}
|
|
399
|
+
|
|
400
|
+
|
|
401
|
+
setWidget(lines.length > 0 ? lines : void 0);
|
|
402
|
+
|
|
403
|
+
const all = [...calls.values()];
|
|
404
|
+
const allDone = all.filter((c) => c.status !== "running").length;
|
|
405
|
+
const totalTokens = all.reduce((sum, c) => sum + c.tokens, 0);
|
|
406
|
+
const elapsed = ((Date.now() - start) / 1000).toFixed(0);
|
|
407
|
+
setStatus(`wf ${allDone}/${calls.size} agents · ${fmtTokens(totalTokens)} tok · ${elapsed}s`);
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
const record = (callId: string, status: CallStatus, tokens = 0, model?: string): void => {
|
|
411
|
+
if (disposed) return; // late settle after cleanup — do not re-create the panel
|
|
412
|
+
calls.set(callId, { stepId: stepIdOf(callId), status, tokens, model });
|
|
413
|
+
render();
|
|
414
|
+
};
|
|
415
|
+
|
|
416
|
+
return {
|
|
417
|
+
cleanup() {
|
|
418
|
+
disposed = true;
|
|
419
|
+
calls.clear();
|
|
420
|
+
streamText.clear();
|
|
421
|
+
setWidget(void 0);
|
|
422
|
+
setStatus(void 0);
|
|
423
|
+
},
|
|
424
|
+
/** Build a synthetic RunResult from the live calls map, so /wf-inspect
|
|
425
|
+
* can show in-progress agents before the run finishes. */
|
|
426
|
+
snapshot(): RunResult {
|
|
427
|
+
const snapSteps: StepResult[] = steps.map((s) => {
|
|
428
|
+
const entries = [...calls.values()].filter((c) => c.stepId === s.id);
|
|
429
|
+
const total = expected.get(s.id) ?? 0;
|
|
430
|
+
const done = entries.filter((c) => c.status === "done" || c.status === "cached").length;
|
|
431
|
+
const failed = entries.filter((c) => c.status === "failed").length;
|
|
432
|
+
const running = entries.filter((c) => c.status === "running").length;
|
|
433
|
+
const cached = entries.filter((c) => c.status === "cached").length;
|
|
434
|
+
const tokens = entries.reduce((sum, c) => sum + c.tokens, 0);
|
|
435
|
+
const status: StepResult["status"] = s.type === "log"
|
|
436
|
+
? "done" // narrative line, no agent call — never "skipped"
|
|
437
|
+
: failed > 0 ? "failed" : done >= total && total > 0 ? "done" : running > 0 ? "running" : "skipped";
|
|
438
|
+
// Real outputs from settled calls — NOT the fabricated `[n/m] running`
|
|
439
|
+
// progress string (that leaked into the detail pane as fake results).
|
|
440
|
+
const settled = [...calls.entries()]
|
|
441
|
+
.filter(([callId, c]) => c.stepId === s.id)
|
|
442
|
+
.map(([callId]) => callOutputs.get(callId))
|
|
443
|
+
.filter((o): o is string => Boolean(o));
|
|
444
|
+
const results = settled.length > 0
|
|
445
|
+
? (s.type === "fan_out" ? settled : settled[0])
|
|
446
|
+
: undefined;
|
|
447
|
+
return {
|
|
448
|
+
id: s.id,
|
|
449
|
+
type: s.type as StageType,
|
|
450
|
+
status,
|
|
451
|
+
results,
|
|
452
|
+
stats: { tokens, cost: 0, durationMs: 0, agents: entries.length, failures: failed },
|
|
453
|
+
};
|
|
454
|
+
});
|
|
455
|
+
const all = [...calls.values()];
|
|
456
|
+
const stats: StepStats = {
|
|
457
|
+
tokens: all.reduce((sum, c) => sum + c.tokens, 0),
|
|
458
|
+
cost: 0,
|
|
459
|
+
durationMs: Date.now() - start,
|
|
460
|
+
agents: all.length,
|
|
461
|
+
failures: all.filter((c) => c.status === "failed").length,
|
|
462
|
+
};
|
|
463
|
+
return { runId: "live", status: "completed", steps: snapSteps, stats };
|
|
464
|
+
},
|
|
465
|
+
onAgentStart(callId) {
|
|
466
|
+
const stepId = stepIdOf(callId);
|
|
467
|
+
expected.set(stepId, (expected.get(stepId) ?? 0) + 1);
|
|
468
|
+
record(callId, "running");
|
|
469
|
+
},
|
|
470
|
+
onAgentEnd(callId, ok, stats, model, output) {
|
|
471
|
+
// A skipped/retried/cached call's subprocess still settles (abort →
|
|
472
|
+
// notifyEnd(false)); do not overwrite the already-recorded terminal
|
|
473
|
+
// state with a "failed" stamp — the run's own bookkeeping marks the
|
|
474
|
+
// step skipped/retried, so the widget must show the same.
|
|
475
|
+
const cur = calls.get(callId);
|
|
476
|
+
if (cur && cur.status !== "running") {
|
|
477
|
+
streamText.delete(callId);
|
|
478
|
+
return;
|
|
479
|
+
}
|
|
480
|
+
record(callId, ok ? "done" : "failed", stats?.tokens ?? 0, model);
|
|
481
|
+
streamText.delete(callId); // free the accumulated tail once the call settles
|
|
482
|
+
if (output) callOutputs.set(callId, output);
|
|
483
|
+
},
|
|
484
|
+
onAgentSkip(callId) {
|
|
485
|
+
record(callId, "skipped");
|
|
486
|
+
},
|
|
487
|
+
onAgentRetry(callId) {
|
|
488
|
+
record(callId, "retried");
|
|
489
|
+
},
|
|
490
|
+
onAgentCacheHit(callId) {
|
|
491
|
+
const stepId = stepIdOf(callId);
|
|
492
|
+
expected.set(stepId, (expected.get(stepId) ?? 0) + 1);
|
|
493
|
+
record(callId, "cached");
|
|
494
|
+
},
|
|
495
|
+
onLog(stepId, message) {
|
|
496
|
+
if (disposed) return;
|
|
497
|
+
logLines.set(stepId, message);
|
|
498
|
+
render();
|
|
499
|
+
},
|
|
500
|
+
onUpdate(callId, partial) {
|
|
501
|
+
if (disposed) return;
|
|
502
|
+
// Bound the accumulated tail: the widget only ever renders the last 3
|
|
503
|
+
// lines (each clipped to ~100 chars), so keeping the full stream alive
|
|
504
|
+
// for the call's duration is pure memory growth on long generations.
|
|
505
|
+
streamText.set(callId, ((streamText.get(callId) ?? "") + partial).slice(-4096));
|
|
506
|
+
// Throttle: a high-frequency stream (fan_out × many deltas) would otherwise
|
|
507
|
+
// trigger a full O(steps×calls) render() per chunk. Bound to ~20fps; the
|
|
508
|
+
// final onAgentEnd render always fires, so the settled state is exact.
|
|
509
|
+
const nowMs = Date.now();
|
|
510
|
+
if (nowMs - lastStreamRender >= 50) {
|
|
511
|
+
lastStreamRender = nowMs;
|
|
512
|
+
render();
|
|
513
|
+
}
|
|
514
|
+
},
|
|
515
|
+
};
|
|
516
|
+
}
|
|
517
|
+
|
|
518
|
+
// ---------------------------------------------------------------------------
|
|
519
|
+
// Extension
|
|
520
|
+
// ---------------------------------------------------------------------------
|
|
521
|
+
|
|
522
|
+
/** Drop a step/route model id that isn't in the session registry, recording it. */
|
|
523
|
+
function dropInvalidModel(id: string, model: string | undefined, validIds: Set<string>, dropped: string[]): string | undefined {
|
|
524
|
+
if (model && !validIds.has(model)) {
|
|
525
|
+
dropped.push(`${id}→${model}`);
|
|
526
|
+
return undefined;
|
|
527
|
+
}
|
|
528
|
+
return model;
|
|
529
|
+
}
|
|
530
|
+
|
|
531
|
+
export default function (pi: ExtensionAPI): void {
|
|
532
|
+
pi.registerTool({
|
|
533
|
+
name: "run_workflow",
|
|
534
|
+
label: "Run workflow",
|
|
535
|
+
description: [
|
|
536
|
+
"Run a deterministic multi-agent workflow. Define steps inline and execute them with cache-resume, budget caps, and per-agent abort.",
|
|
537
|
+
"Step types: agent, fan_out (over a static list), adversarial (produce + judges), tournament (candidates + judges), classify_route (classify → route sub-steps), log (narrative line).",
|
|
538
|
+
"Every step accepts onBudgetExhaust: \"throw\" (default) / \"null\" — under \"null\", a step whose budget runs out degrades to a null result instead of aborting the run.",
|
|
539
|
+
"String prompts support templates: {{input}}, {{step.<id>}} (a prior step's result), {{item}} (current fan_out item).",
|
|
540
|
+
"Each agent step spawns a real `pi` subprocess, so pi + a provider must be configured.",
|
|
541
|
+
].join(" "),
|
|
542
|
+
promptSnippet: "run_workflow — execute a declarative multi-agent workflow (agent/fan_out/adversarial/tournament/classify_route/log)",
|
|
543
|
+
parameters: RunWorkflowParams,
|
|
544
|
+
|
|
545
|
+
async execute(_toolCallId, params, signal, _onUpdate, ctx) {
|
|
546
|
+
// Scoped-models: when the session restricts models (--models /
|
|
547
|
+
// enabledModels), validate step models against that scope; otherwise
|
|
548
|
+
// fall back to the full registry. Invalid ids (e.g. "sonnet") are
|
|
549
|
+
// dropped so the subprocess uses the default session model.
|
|
550
|
+
const scoped = ctx.scopedModels;
|
|
551
|
+
const validIds = scoped && scoped.length > 0
|
|
552
|
+
? new Set(scoped.map((s) => s.model.id))
|
|
553
|
+
: new Set(ctx.modelRegistry.getAll().map((m) => m.id));
|
|
554
|
+
const dropped: string[] = [];
|
|
555
|
+
const sanitizedSteps = params.workflow.steps.map((s) => {
|
|
556
|
+
const model = "model" in s ? dropInvalidModel(s.id, s.model, validIds, dropped) : undefined;
|
|
557
|
+
if (s.type === "classify_route") {
|
|
558
|
+
// Route/fallback sub-step models must be sanitized too — the schema
|
|
559
|
+
// promises "Invalid ids are dropped", which routeStepToCode otherwise
|
|
560
|
+
// passes straight through to the subprocess.
|
|
561
|
+
const routes = Object.fromEntries(
|
|
562
|
+
Object.entries(s.routes).map(([cat, rs]) => [
|
|
563
|
+
cat,
|
|
564
|
+
rs.map((r) => ({ ...r, model: dropInvalidModel(`${s.id}.${r.id}`, r.model, validIds, dropped) })),
|
|
565
|
+
]),
|
|
566
|
+
) as typeof s.routes;
|
|
567
|
+
const fallback = s.fallback?.map((r) => ({ ...r, model: dropInvalidModel(`${s.id}.${r.id}`, r.model, validIds, dropped) }));
|
|
568
|
+
return { ...s, model, routes, fallback };
|
|
569
|
+
}
|
|
570
|
+
return { ...s, model };
|
|
571
|
+
});
|
|
572
|
+
if (dropped.length > 0) {
|
|
573
|
+
ctx.ui.notify(`Invalid model(s) dropped, using default: ${dropped.join(", ")}`, "warning");
|
|
574
|
+
}
|
|
575
|
+
const sanitizedWorkflow = { ...params.workflow, steps: sanitizedSteps };
|
|
576
|
+
const widget = buildProgressWidget(
|
|
577
|
+
sanitizedWorkflow.steps,
|
|
578
|
+
(lines) => ctx.ui.setWidget("wf:progress", lines),
|
|
579
|
+
(text) => ctx.ui.setStatus("wf:summary", text),
|
|
580
|
+
sanitizedWorkflow.phases,
|
|
581
|
+
);
|
|
582
|
+
activeWidget = widget;
|
|
583
|
+
lastPhases = sanitizedWorkflow.phases;
|
|
584
|
+
try {
|
|
585
|
+
const workflow = buildWorkflow(sanitizedWorkflow);
|
|
586
|
+
const listeners: AgentLifecycleListeners = {
|
|
587
|
+
onAgentStart: widget.onAgentStart,
|
|
588
|
+
onAgentEnd: widget.onAgentEnd,
|
|
589
|
+
onAgentSkip: widget.onAgentSkip,
|
|
590
|
+
onAgentRetry: widget.onAgentRetry,
|
|
591
|
+
onAgentCacheHit: widget.onAgentCacheHit,
|
|
592
|
+
onLog: widget.onLog,
|
|
593
|
+
onUpdate: widget.onUpdate,
|
|
594
|
+
};
|
|
595
|
+
const result = await runWorkflow({
|
|
596
|
+
workflow,
|
|
597
|
+
input: params.input,
|
|
598
|
+
cwd: params.cwd ?? ctx.cwd,
|
|
599
|
+
now: params.now ?? Date.now(),
|
|
600
|
+
signal,
|
|
601
|
+
listeners,
|
|
602
|
+
});
|
|
603
|
+
lastRunResult = result;
|
|
604
|
+
|
|
605
|
+
const lines = [
|
|
606
|
+
`workflow "${workflow.name}" → ${result.status} (run ${result.runId})`,
|
|
607
|
+
...result.steps.map((s) => ` [${s.status}] ${s.id} (${s.type})${preview(s.results)}`),
|
|
608
|
+
`stats: ${result.stats.agents} agent(s), ${result.stats.tokens} tokens, $${result.stats.cost.toFixed(4)}`,
|
|
609
|
+
];
|
|
610
|
+
if (result.error) lines.push(`error: ${result.error}`);
|
|
611
|
+
return {
|
|
612
|
+
content: [{ type: "text" as const, text: lines.join("\n") }],
|
|
613
|
+
details: result,
|
|
614
|
+
isError: result.status !== "completed",
|
|
615
|
+
};
|
|
616
|
+
} catch (e) {
|
|
617
|
+
const msg = e instanceof Error ? e.message : String(e);
|
|
618
|
+
return { content: [{ type: "text" as const, text: `run_workflow failed: ${msg}` }], details: { error: msg }, isError: true };
|
|
619
|
+
} finally {
|
|
620
|
+
activeWidget = null;
|
|
621
|
+
ctx.ui.setWidget("wf:progress", void 0);
|
|
622
|
+
widget.cleanup();
|
|
623
|
+
}
|
|
624
|
+
},
|
|
625
|
+
});
|
|
626
|
+
|
|
627
|
+
pi.registerCommand("wf-inspect", {
|
|
628
|
+
description: "Inspect the current/last workflow run (↑↓ select, enter detail, esc exit)",
|
|
629
|
+
handler: async (_args, ctx) => {
|
|
630
|
+
// Prefer a live snapshot while a run is in progress; fall back to
|
|
631
|
+
// the last completed result once the run has finished.
|
|
632
|
+
const r = activeWidget?.snapshot() ?? lastRunResult;
|
|
633
|
+
if (!r) {
|
|
634
|
+
ctx.ui.notify("No workflow run yet — run run_workflow first", "warning");
|
|
635
|
+
return;
|
|
636
|
+
}
|
|
637
|
+
await ctx.ui.custom(
|
|
638
|
+
(tui, _theme, _kb, done) =>
|
|
639
|
+
new WorkflowInspect(r, tui, () => done(undefined), lastPhases),
|
|
640
|
+
{ overlay: true, overlayOptions: { anchor: "center", width: "90%", maxHeight: "80%" } },
|
|
641
|
+
);
|
|
642
|
+
},
|
|
643
|
+
});
|
|
644
|
+
}
|
|
645
|
+
|
|
646
|
+
function preview(value: unknown): string {
|
|
647
|
+
const s = typeof value === "string" ? value : JSON.stringify(value);
|
|
648
|
+
if (!s) return "";
|
|
649
|
+
return ` — ${s.length > 100 ? `${s.slice(0, 100)}…` : s}`;
|
|
650
|
+
}
|