@tranhoangnguyen0310/pi-flow-external 2.4.1-external.0 → 2.4.2-external.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -4,6 +4,15 @@ All notable changes to pi-flow external are documented here.
4
4
 
5
5
  ## Unreleased
6
6
 
7
+ ## [2.4.2-external.0] - 2026-09-23
8
+
9
+ ### Fixed
10
+
11
+ - Simplified model-facing run supervision to one `runIds` selector, including singleton output/final inspection and cancellation. Legacy `runId` callers remain supported; conflicting cancellation selectors fail explicitly.
12
+ - Workflow help accepts a supplied harness without blocking, explains its cross-harness scope, and accepts blank filters at the SDK schema boundary.
13
+ - Blank Agent resume arguments no longer conflict with context sharing or trigger resume lookup. Real resume/context conflicts remain errors.
14
+ - Added SDK argument-validation coverage alongside provider-payload tests. Audited workflow source selection: blank sources already normalize away; multiple real sources still fail. Downstream required-field promotion remains unverified, so #62 stays open.
15
+
7
16
  ## [2.4.1-external.0] - 2026-09-23
8
17
 
9
18
  ### Fixed
package/README.md CHANGED
@@ -301,10 +301,10 @@ Agent({
301
301
  Use `external_runs` with these actions:
302
302
 
303
303
  - `list`: current session/project runs; use `cursor` for run pages and `workflowCursor` for workflow pages. `workflowRunId` filters children of one workflow. Rows carry a `timing` projection (`queueDelayMs`, `elapsedMs`, `activityAgeMs` when live, `processDurationMs`) plus `outputAvailable`/`finalAvailable`.
304
- - `inspect`: single `runId` with `view: "summary" | "output" | "diagnostics" | "final"`, and optional opaque `cursor`/`limitBytes` (max 64 KiB, same cap for every view). Follow `nextCursor` to avoid truncation. `summary` includes the same `timing` projection as `list`, plus `output.finalAvailable`. `final` returns only the verified canonical terminal answer — empty with `finalAvailable: false` until a successful terminal boundary exists; it never promotes partial/narration text. `output` stays the combined stream (assistant messages plus canonical result) and is unchanged.
305
- - `inspect` with `runIds` instead of `runId` (up to 20, deduplicated, order preserved): a single bounded batch of `summary`-only projections — one cheap request to see whether several selected background children are queued, running, or terminal, each with `outputRef`/`diagnosticsRef` for follow-up detail. Ownership of every requested ID is validated before any page is returned. Reuses the same `limitBytes` cap as single-run inspection; pages contain whole target entries and continue through `nextCursor`, without invalidation from ordinary live progress. If one compact entry cannot fit, an actionable error asks you to increase `limitBytes` or inspect that run individually; no target is silently dropped. `runId` and `runIds` are mutually exclusive, and batch `view` must stay `summary`.
306
- - `wait`: one `runId` or selected `runIds`, with `mode: "any" | "all"`. It returns terminal outcomes plus still-pending IDs; an unsuccessful workflow returns early even in `all` mode. It never chooses a winner or cancels pending work. While waiting, a bounded heartbeat (independent of any single target settling) reports live progress — watched targets, completed/pending counts, and recent activity — through the tool's update channel; it stops automatically on settlement, error, or interruption. Each settled outcome's `result` is spent from one shared byte budget (`limitBytes`, default 32768) across the whole response, in the requested `runId`/`runIds` order — never settlement race order, so the same targets and final states spend the budget identically regardless of which one happened to settle first: a result that fits is returned complete, one that does not is truncated with `resultTruncated: true` and the existing `outputRef`/`diagnosticsRef` to continue reading it — not a fixed-length teaser regardless of size. A target's evidence is never read from disk once the shared budget is already exhausted.
307
- - `cancel`: one `runId` and optional reason. Whole-workflow cancellation stops active children; targeted child cancellation remains a catchable workflow outcome. Cancellation does not roll back edits or other side effects.
304
+ - `inspect`: single-entry `runIds` with `view: "summary" | "output" | "diagnostics" | "final"`, and optional opaque `cursor`/`limitBytes` (max 64 KiB, same cap for every view). Follow `nextCursor` to avoid truncation. `summary` includes the same `timing` projection as `list`, plus `output.finalAvailable`. `final` returns only the verified canonical terminal answer — empty with `finalAvailable: false` until a successful terminal boundary exists; it never promotes partial/narration text. `output` stays the combined stream (assistant messages plus canonical result) and is unchanged.
305
+ - `inspect` with `runIds` (summary view) (up to 20, deduplicated, order preserved): a single bounded batch of `summary`-only projections — one cheap request to see whether several selected background children are queued, running, or terminal, each with `outputRef`/`diagnosticsRef` for follow-up detail. Ownership of every requested ID is validated before any page is returned. Reuses the same `limitBytes` cap as single-run inspection; pages contain whole target entries and continue through `nextCursor`, without invalidation from ordinary live progress. If one compact entry cannot fit, an actionable error asks you to increase `limitBytes` or inspect that run individually; no target is silently dropped. The legacy `runId` selector remains accepted by programmatic callers but is no longer advertised to models. A singleton list supports other views; multiple targets require `summary`.
306
+ - `wait`: selected `runIds` (one or more), with `mode: "any" | "all"`. It returns terminal outcomes plus still-pending IDs; an unsuccessful workflow returns early even in `all` mode. It never chooses a winner or cancels pending work. While waiting, a bounded heartbeat (independent of any single target settling) reports live progress — watched targets, completed/pending counts, and recent activity — through the tool's update channel; it stops automatically on settlement, error, or interruption. Each settled outcome's `result` is spent from one shared byte budget (`limitBytes`, default 32768) across the whole response, in the requested `runId`/`runIds` order — never settlement race order, so the same targets and final states spend the budget identically regardless of which one happened to settle first: a result that fits is returned complete, one that does not is truncated with `resultTruncated: true` and the existing `outputRef`/`diagnosticsRef` to continue reading it — not a fixed-length teaser regardless of size. A target's evidence is never read from disk once the shared budget is already exhausted.
307
+ - `cancel`: single-entry `runIds` and optional reason. Whole-workflow cancellation stops active children; targeted child cancellation remains a catchable workflow outcome. Cancellation does not roll back edits or other side effects.
308
308
 
309
309
  Interrupting a blocking `Agent`/`workflow` call cancels its work. Interrupting `external_runs wait` stops only that wait. Background work survives its launching tool return and ordinary parent turns, but not the owning session: orderly session shutdown requests cancellation and waits for bounded cleanup. This is not a daemon. After a host crash or unconfirmed shutdown, unfinished evidence is `interrupted_or_uncertain`; restart restores evidence access, never live ownership or guaranteed retrospective process termination. No routine activity wakes the parent, and live steering is not supported.
310
310
 
@@ -328,7 +328,7 @@ const two = Agent({ description: "Audit billing module", prompt: "Audit /absolut
328
328
 
329
329
  // Leave them running and come back later in the conversation (or after
330
330
  // further parent work) to collect final output once each is actually done:
331
- // external_runs({ action: "inspect", runId: one.runId, view: "final" })
331
+ // external_runs({ action: "inspect", runIds: [one.runId], view: "final" })
332
332
  ```
333
333
 
334
334
  A task that sounds like a 30-second command can legitimately take substantially longer end-to-end once queueing and backend overhead are included — inspect and wait, do not assume.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tranhoangnguyen0310/pi-flow-external",
3
- "version": "2.4.1-external.0",
3
+ "version": "2.4.2-external.0",
4
4
  "description": "External Claude Code, Codex CLI, Antigravity, Grok Build CLI, and Muse Code delegation for pi.",
5
5
  "type": "module",
6
6
  "main": "./index.ts",
@@ -45,7 +45,8 @@ export function prepareParentContext(
45
45
  ): { prompt: string; context?: ParentContextReceipt } {
46
46
  const context = parseParentContext(selection);
47
47
  if (!context || context.mode === "none") return { prompt };
48
- if (resume !== undefined) throw new Error("context sharing cannot be combined with resume; continue the child or start a new one");
48
+ const resumeId = typeof resume === "string" && resume.trim() !== "" ? resume.trim() : undefined;
49
+ if (resumeId !== undefined) throw new Error("context sharing cannot be combined with resume; continue the child or start a new one");
49
50
  if (!messages) throw new Error("Parent context is unavailable; use context:none and a self-contained prompt");
50
51
  const compacted = messages.some((message) => message.role === "compactionSummary");
51
52
  let start = 0;
@@ -22,8 +22,7 @@ const externalHelpParameters = Type.Object({
22
22
  description: "Help topic: role descriptions/configured profile availability, harness permissions, or workflow syntax and saved workflows.",
23
23
  }),
24
24
  harness: Type.Optional(Type.String({
25
- minLength: 1,
26
- description: "Optional harness filter for roles or permissions: agy, claude, codex, grok, muse, or a registered pi-* harness. Do not use with topic workflow.",
25
+ description: "Optional harness filter for roles or permissions: agy, claude, codex, grok, muse, or a registered pi-* harness. Workflows orchestrate across harnesses.",
27
26
  })),
28
27
  });
29
28
 
@@ -131,14 +130,18 @@ export function createExternalHelpTool(
131
130
  promptSnippet: EXTERNAL_HELP_PROMPT_SNIPPET,
132
131
  parameters: externalHelpParameters,
133
132
  async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
134
- if (params.topic === "workflow" && params.harness) {
135
- throw new Error("external_help harness is only valid for roles or permissions.");
136
- }
133
+ // A schema-conversion layer downstream of this tool's declaration may
134
+ // present `harness` as required (#62); treat a blank/whitespace
135
+ // placeholder as omitted. Furthermore, do not reject an explicit
136
+ // harness on topic "workflow": workflows orchestrate across all
137
+ // harnesses, so passing a harness filter gracefully returns workflow
138
+ // guidance without error.
139
+ const harness = params.harness?.trim() ? params.harness.trim() : undefined;
137
140
  let text: string;
138
141
  const { harnesses: harnessConfigs } = loadHarnessConfigs(getAgentDir());
139
142
  const configuredPiHarnesses = new Set(harnessConfigs.keys());
140
143
  if (params.topic !== "workflow") {
141
- validateHarnessFilter(params.harness, configuredPiHarnesses);
144
+ validateHarnessFilter(harness, configuredPiHarnesses);
142
145
  }
143
146
  if (params.topic === "roles") {
144
147
  const profiles = mergeSynthesizedPiProfiles(
@@ -148,19 +151,22 @@ export function createExternalHelpTool(
148
151
  ),
149
152
  harnessConfigs,
150
153
  );
151
- text = formatExternalRoleHelp(profiles, options.getDefaultHarness(ctx), params.harness);
154
+ text = formatExternalRoleHelp(profiles, options.getDefaultHarness(ctx), harness);
152
155
  } else if (params.topic === "permissions") {
153
- text = permissionHelp(params.harness, configuredPiHarnesses);
156
+ text = permissionHelp(harness, configuredPiHarnesses);
154
157
  } else {
155
- text = workflowHelp(options.workflowEnabled, listSavedWorkflows({
158
+ const wfText = workflowHelp(options.workflowEnabled, listSavedWorkflows({
156
159
  agentDir: getAgentDir(),
157
160
  cwd: ctx.cwd,
158
161
  projectTrusted: isProjectTrusted(ctx),
159
162
  }));
163
+ text = harness
164
+ ? `${wfText}\n\nNote: Workflows orchestrate across multiple harnesses (including "${harness}"); workflow syntax is uniform across backends.`
165
+ : wfText;
160
166
  }
161
167
  return {
162
168
  content: [{ type: "text" as const, text }],
163
- details: { topic: params.topic, ...(params.harness ? { harness: params.harness } : {}) },
169
+ details: { topic: params.topic, ...(harness ? { harness } : {}) },
164
170
  };
165
171
  },
166
172
  });
@@ -28,13 +28,9 @@ const externalRunsParameters = Type.Object({
28
28
  action: StringEnum(["list", "inspect", "wait", "cancel"] as const, {
29
29
  description: "list: page runs/workflows; inspect: read one run; wait: block until selected terminal outcomes; cancel: stop one run.",
30
30
  }),
31
- runId: Type.Optional(Type.String({
32
- description: "Target run ID (run_... agent, wf_... workflow). Required for inspect/cancel/wait of a single run.",
33
- })),
34
31
  runIds: Type.Optional(Type.Array(Type.String(), {
35
- minItems: 1,
36
32
  maxItems: MAX_TARGETS,
37
- description: "wait: target set for mode any|all, up to 100, deduplicated; already-terminal targets return immediately. inspect: batch summary target set, up to 20, deduplicated, order preserved; cannot combine with runId, and view must stay summary.",
33
+ description: "Target run IDs (run_... agent, wf_... workflow). For inspect: a single-entry list [\"run_...\"] inspects that run with any view (summary, output, diagnostics, final); multiple entries (1-20 targets) batch summary inspect. For cancel: a single-entry list [\"run_...\"]. For wait: any|all of up to 100 targets.",
38
34
  })),
39
35
  view: Type.Optional(StringEnum(["summary", "output", "diagnostics", "final"] as const, {
40
36
  description: "inspect view: summary (state/timing/freshness/refs, default), output (assistant text plus canonical result, partial or final), diagnostics (tool activity/errors), final (only the verified canonical terminal answer, empty until a successful terminal boundary exists). Batch inspect (runIds) only supports summary.",
@@ -67,7 +63,10 @@ const externalRunsParameters = Type.Object({
67
63
  })),
68
64
  });
69
65
 
70
- export type ExternalRunsParams = Static<typeof externalRunsParameters>;
66
+ export type ExternalRunsParams = Static<typeof externalRunsParameters> & {
67
+ /** Legacy single-target selector; preserved for backwards-compatible programmatic/test callers. */
68
+ runId?: string;
69
+ };
71
70
  type ExternalRunsDetails = Record<string, unknown>;
72
71
 
73
72
  export interface CreateExternalRunsToolOptions {
@@ -518,9 +517,12 @@ function renderExternalRunsCall(args: Record<string, unknown>, theme: Theme): Te
518
517
  const ids = Array.isArray(args.runIds) ? args.runIds : args.runId ? [args.runId] : [];
519
518
  detail = `waiting for ${ids.length} task(s) · mode ${typeof args.mode === "string" ? args.mode : "all"}`;
520
519
  } else if (action === "cancel") {
521
- detail = `cancel ${String(args.runId ?? "")}`;
520
+ const target = Array.isArray(args.runIds) && args.runIds.length ? args.runIds[0] : String(args.runId ?? "");
521
+ detail = `cancel ${target}`;
522
522
  } else if (action === "inspect") {
523
- const target = Array.isArray(args.runIds) ? `${args.runIds.length} run(s) (batch)` : String(args.runId ?? "");
523
+ const target = Array.isArray(args.runIds)
524
+ ? args.runIds.length === 1 ? args.runIds[0] : `${args.runIds.length} run(s) (batch)`
525
+ : String(args.runId ?? "");
524
526
  detail = `inspect ${target} · ${typeof args.view === "string" ? args.view : "summary"}`;
525
527
  } else {
526
528
  detail = `list${typeof args.workflowRunId === "string" ? ` · workflow ${args.workflowRunId}` : ""}`;
@@ -626,6 +628,33 @@ function renderExternalRunsResult(toolResult: { content: Array<{ type: string; t
626
628
  return new Text(textFromToolResult(toolResult), 0, 0);
627
629
  }
628
630
 
631
+ /**
632
+ * Reconcile the `runId`/`runIds` selector pair for `inspect` only — the one
633
+ * action that already hard-rejects supplying both (`wait` treats a stray
634
+ * `runId` alongside `runIds` as a harmless one-element convenience, and
635
+ * `cancel` never reads `runIds` at all, so neither gets a new conflict
636
+ * check here). A schema-conversion layer downstream of this tool's
637
+ * declaration may present both mutually exclusive optional selectors as
638
+ * required (#62); a model forced to fill in the one it means to omit
639
+ * typically sends a blank string or an empty array. Neither can name an
640
+ * actual run, so treat that placeholder as omitted rather than a real
641
+ * conflict — this is narrower than guessing between two genuinely
642
+ * populated, disagreeing selectors, which still fails loudly exactly as
643
+ * before (including when both name the very same run: inspect has always
644
+ * rejected supplying the pair at all, on purpose).
645
+ */
646
+ function normalizeInspectSelectors(params: ExternalRunsParams): ExternalRunsParams {
647
+ const runId = params.runId === undefined || params.runId.trim() === "" ? undefined : params.runId;
648
+ const runIds = runId !== undefined && (params.runIds === undefined || params.runIds.length === 0)
649
+ ? undefined
650
+ : params.runIds;
651
+ if (runId !== undefined && runIds !== undefined && runIds.length > 0) {
652
+ throw new Error("inspect accepts either runId or runIds, not both");
653
+ }
654
+ if (runId === params.runId && runIds === params.runIds) return params;
655
+ return { ...params, runId, runIds };
656
+ }
657
+
629
658
  export function createExternalRunsTool(
630
659
  options: CreateExternalRunsToolOptions,
631
660
  ): ToolDefinition<typeof externalRunsParameters, ExternalRunsDetails> {
@@ -636,6 +665,12 @@ export function createExternalRunsTool(
636
665
  promptSnippet: EXTERNAL_RUNS_PROMPT_SNIPPET,
637
666
  parameters: externalRunsParameters,
638
667
  async execute(_toolCallId, params: ExternalRunsParams, signal, onUpdate, ctx) {
668
+ if (params.action === "inspect") {
669
+ params = normalizeInspectSelectors(params);
670
+ if (params.runId === undefined && params.runIds !== undefined && params.runIds.length === 1 && params.view !== undefined && params.view !== "summary") {
671
+ params = { ...params, runId: params.runIds[0], runIds: undefined };
672
+ }
673
+ }
639
674
  const { sessionId, project } = scope(ctx);
640
675
  const runsDirectory = options.runsDirectory();
641
676
 
@@ -670,7 +705,7 @@ export function createExternalRunsTool(
670
705
  }
671
706
 
672
707
  if (params.action === "inspect" && params.runIds !== undefined) {
673
- if (params.runId !== undefined) throw new Error("inspect accepts either runId or runIds, not both");
708
+ // normalizeInspectSelectors already guarantees runId is unset here.
674
709
  if (params.view !== undefined && params.view !== "summary") throw new Error('Batch inspect (runIds) only supports view: "summary"');
675
710
  const runIds = [...new Set(params.runIds)];
676
711
  if (runIds.length === 0 || runIds.length > MAX_BATCH_INSPECT_TARGETS) {
@@ -812,26 +847,35 @@ export function createExternalRunsTool(
812
847
  }
813
848
 
814
849
  if (params.action === "cancel") {
815
- assertRunId(params.runId);
816
- const entry = options.registry.get(params.runId);
850
+ const rawRunIds = params.runIds;
851
+ if (rawRunIds?.length && params.runId?.trim()) {
852
+ throw new Error("cancel accepts either runId or runIds, not both");
853
+ }
854
+ if (rawRunIds && rawRunIds.length > 1) {
855
+ throw new Error("cancel targets one run at a time");
856
+ }
857
+ const targetRunId = (rawRunIds && rawRunIds.length === 1 ? rawRunIds[0] : undefined)
858
+ ?? (typeof params.runId === "string" && params.runId.trim() !== "" ? params.runId.trim() : undefined);
859
+ assertRunId(targetRunId);
860
+ const entry = options.registry.get(targetRunId);
817
861
  if (entry) {
818
862
  assertOwned(entry, sessionId, project);
819
- const status = options.registry.cancel(params.runId, params.reason ?? "cancelled by external_runs");
820
- return result(status === "requested" ? `Cancellation requested for ${params.runId}.` : `${params.runId} is already terminal.`, { runId: params.runId, status });
863
+ const status = options.registry.cancel(targetRunId, params.reason ?? "cancelled by external_runs");
864
+ return result(status === "requested" ? `Cancellation requested for ${targetRunId}.` : `${targetRunId} is already terminal.`, { runId: targetRunId, status });
821
865
  }
822
- const isWorkflowId = WORKFLOW_ID.test(params.runId);
866
+ const isWorkflowId = WORKFLOW_ID.test(targetRunId);
823
867
  const workflowDir = getSessionWorkflowDir(ctx);
824
- const historicalWorkflow = isWorkflowId && workflowDir ? await loadWorkflowJournal(workflowDir, params.runId) : undefined;
868
+ const historicalWorkflow = isWorkflowId && workflowDir ? await loadWorkflowJournal(workflowDir, targetRunId) : undefined;
825
869
  if (historicalWorkflow) {
826
870
  if (historicalWorkflow.project !== project) throw new Error("Run is unknown or unavailable in this session");
827
- if (historicalWorkflow.status !== "running") return result(`${params.runId} is already terminal.`, { runId: params.runId, status: "terminal" });
871
+ if (historicalWorkflow.status !== "running") return result(`${targetRunId} is already terminal.`, { runId: targetRunId, status: "terminal" });
828
872
  throw new Error("Run is no longer live in this session; cancellation cannot be confirmed");
829
873
  }
830
874
  // An unresolved wf_... ID is unknown, not an unowned agent record.
831
875
  if (isWorkflowId) throw new Error("Run is unknown or unavailable in this session");
832
- const durable = await getRunRecord(runsDirectory, params.runId);
876
+ const durable = await getRunRecord(runsDirectory, targetRunId);
833
877
  assertOwnedRecord(durable, sessionId, project);
834
- if (terminalRecord(durable)) return result(`${params.runId} is already terminal.`, { runId: params.runId, status: "terminal" });
878
+ if (terminalRecord(durable)) return result(`${targetRunId} is already terminal.`, { runId: targetRunId, status: "terminal" });
835
879
  throw new Error("Run is no longer live in this session; cancellation cannot be confirmed");
836
880
  }
837
881
 
@@ -358,9 +358,10 @@ function createAgentTool(
358
358
  parameters: agentToolParameters,
359
359
  executionMode: "parallel",
360
360
  async execute(toolCallId, params, signal, onUpdate, ctx) {
361
+ const resume = typeof params.resume === "string" && params.resume.trim() !== "" ? params.resume.trim() : undefined;
361
362
  const briefing = prepareParentContext(params.prompt, params.context,
362
363
  params.context && params.context.mode !== "none" ? captureParentContext(ctx.sessionManager) : undefined,
363
- toolCallId, params.resume);
364
+ toolCallId, resume);
364
365
  const state = getState();
365
366
  const effectiveState: DelegationState = {
366
367
  ...state,
@@ -524,7 +525,7 @@ function createAgentTool(
524
525
  permission: params.permission,
525
526
  defaultPermission,
526
527
  maxBudgetUsd,
527
- resumeRunId: params.resume,
528
+ resumeRunId: resume,
528
529
  executionStartedAt: run.progress.executionStartedAt,
529
530
  onProgress: (partial) => {
530
531
  const details = partial.details as SubagentToolDetails;