@tranhoangnguyen0310/pi-flow-external 2.4.0-external.0 → 2.4.2-external.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/README.md +5 -5
- package/docs/field-testing.md +3 -1
- package/package.json +1 -1
- package/src/core/grok.ts +26 -1
- package/src/core/muse.ts +37 -2
- package/src/core/parent-context.ts +2 -1
- package/src/external-help.ts +16 -10
- package/src/external-runs.ts +62 -18
- package/src/pi-subagent.ts +3 -2
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,23 @@ All notable changes to pi-flow external are documented here.
|
|
|
4
4
|
|
|
5
5
|
## Unreleased
|
|
6
6
|
|
|
7
|
+
## [2.4.2-external.0] - 2026-09-23
|
|
8
|
+
|
|
9
|
+
### Fixed
|
|
10
|
+
|
|
11
|
+
- Simplified model-facing run supervision to one `runIds` selector, including singleton output/final inspection and cancellation. Legacy `runId` callers remain supported; conflicting cancellation selectors fail explicitly.
|
|
12
|
+
- Workflow help accepts a supplied harness without blocking, explains its cross-harness scope, and accepts blank filters at the SDK schema boundary.
|
|
13
|
+
- Blank Agent resume arguments no longer conflict with context sharing or trigger resume lookup. Real resume/context conflicts remain errors.
|
|
14
|
+
- Added SDK argument-validation coverage alongside provider-payload tests. Audited workflow source selection: blank sources already normalize away; multiple real sources still fail. Downstream required-field promotion remains unverified, so #62 stays open.
|
|
15
|
+
|
|
16
|
+
## [2.4.1-external.0] - 2026-09-23
|
|
17
|
+
|
|
18
|
+
### Fixed
|
|
19
|
+
|
|
20
|
+
- Muse rejects effective `thinking: off` before process launch because its `meta` provider does not support `--reasoning-effort none`. The error explains supported profile/parent levels without silently changing reasoning (#60).
|
|
21
|
+
- Grok runtime-socket symlink sandbox failures include actionable guidance while retaining the original error, requested permissions, and fail-closed behavior. No socket changes, retries, or sandbox downgrade (#61).
|
|
22
|
+
- Added an offline test of the actual Pi `openai-completions` provider payload for optional tool selectors. Client optionality is preserved; the downstream model-facing conversion remains unresolved and tracked in #62. See `docs/tool-schema-compatibility.md`.
|
|
23
|
+
|
|
7
24
|
## [2.4.0-external.0] - 2026-09-21
|
|
8
25
|
|
|
9
26
|
Adds `muse` as a fifth external CLI backend, delegating to Muse Code alongside Claude Code, Codex CLI, Antigravity, and Grok Build CLI. Investigation notes: `docs/plans/muse-backend-prep.md`; ex-ante design issue: none filed — additive backend addition following the same shape as the Grok backend.
|
package/README.md
CHANGED
|
@@ -301,10 +301,10 @@ Agent({
|
|
|
301
301
|
Use `external_runs` with these actions:
|
|
302
302
|
|
|
303
303
|
- `list`: current session/project runs; use `cursor` for run pages and `workflowCursor` for workflow pages. `workflowRunId` filters children of one workflow. Rows carry a `timing` projection (`queueDelayMs`, `elapsedMs`, `activityAgeMs` when live, `processDurationMs`) plus `outputAvailable`/`finalAvailable`.
|
|
304
|
-
- `inspect`: single `
|
|
305
|
-
- `inspect` with `runIds`
|
|
306
|
-
- `wait`:
|
|
307
|
-
- `cancel`:
|
|
304
|
+
- `inspect`: single-entry `runIds` with `view: "summary" | "output" | "diagnostics" | "final"`, and optional opaque `cursor`/`limitBytes` (max 64 KiB, same cap for every view). Follow `nextCursor` to avoid truncation. `summary` includes the same `timing` projection as `list`, plus `output.finalAvailable`. `final` returns only the verified canonical terminal answer — empty with `finalAvailable: false` until a successful terminal boundary exists; it never promotes partial/narration text. `output` stays the combined stream (assistant messages plus canonical result) and is unchanged.
|
|
305
|
+
- `inspect` with `runIds` (summary view) (up to 20, deduplicated, order preserved): a single bounded batch of `summary`-only projections — one cheap request to see whether several selected background children are queued, running, or terminal, each with `outputRef`/`diagnosticsRef` for follow-up detail. Ownership of every requested ID is validated before any page is returned. Reuses the same `limitBytes` cap as single-run inspection; pages contain whole target entries and continue through `nextCursor`, without invalidation from ordinary live progress. If one compact entry cannot fit, an actionable error asks you to increase `limitBytes` or inspect that run individually; no target is silently dropped. The legacy `runId` selector remains accepted by programmatic callers but is no longer advertised to models. A singleton list supports other views; multiple targets require `summary`.
|
|
306
|
+
- `wait`: selected `runIds` (one or more), with `mode: "any" | "all"`. It returns terminal outcomes plus still-pending IDs; an unsuccessful workflow returns early even in `all` mode. It never chooses a winner or cancels pending work. While waiting, a bounded heartbeat (independent of any single target settling) reports live progress — watched targets, completed/pending counts, and recent activity — through the tool's update channel; it stops automatically on settlement, error, or interruption. Each settled outcome's `result` is spent from one shared byte budget (`limitBytes`, default 32768) across the whole response, in the requested `runId`/`runIds` order — never settlement race order, so the same targets and final states spend the budget identically regardless of which one happened to settle first: a result that fits is returned complete, one that does not is truncated with `resultTruncated: true` and the existing `outputRef`/`diagnosticsRef` to continue reading it — not a fixed-length teaser regardless of size. A target's evidence is never read from disk once the shared budget is already exhausted.
|
|
307
|
+
- `cancel`: single-entry `runIds` and optional reason. Whole-workflow cancellation stops active children; targeted child cancellation remains a catchable workflow outcome. Cancellation does not roll back edits or other side effects.
|
|
308
308
|
|
|
309
309
|
Interrupting a blocking `Agent`/`workflow` call cancels its work. Interrupting `external_runs wait` stops only that wait. Background work survives its launching tool return and ordinary parent turns, but not the owning session: orderly session shutdown requests cancellation and waits for bounded cleanup. This is not a daemon. After a host crash or unconfirmed shutdown, unfinished evidence is `interrupted_or_uncertain`; restart restores evidence access, never live ownership or guaranteed retrospective process termination. No routine activity wakes the parent, and live steering is not supported.
|
|
310
310
|
|
|
@@ -328,7 +328,7 @@ const two = Agent({ description: "Audit billing module", prompt: "Audit /absolut
|
|
|
328
328
|
|
|
329
329
|
// Leave them running and come back later in the conversation (or after
|
|
330
330
|
// further parent work) to collect final output once each is actually done:
|
|
331
|
-
// external_runs({ action: "inspect",
|
|
331
|
+
// external_runs({ action: "inspect", runIds: [one.runId], view: "final" })
|
|
332
332
|
```
|
|
333
333
|
|
|
334
334
|
A task that sounds like a 30-second command can legitimately take substantially longer end-to-end once queueing and backend overhead are included — inspect and wait, do not assume.
|
package/docs/field-testing.md
CHANGED
|
@@ -51,7 +51,9 @@ Defaults:
|
|
|
51
51
|
|
|
52
52
|
Override with `--model`/`--thinking`. Use `--keep` only when evidence inspection is necessary; it preserves sensitive output and the temporary profile path printed by the runner.
|
|
53
53
|
|
|
54
|
-
Grok's `readonly`/`edit` tiers run under its own kernel sandbox (`--sandbox read-only`/`workspace`).
|
|
54
|
+
Grok's `readonly`/`edit` tiers run under its own kernel sandbox (`--sandbox read-only`/`workspace`). Grok 1.0.40 on macOS can refuse startup with `sandbox could not be applied: socket deny resolution failed: could not resolve runtime-socket deny path /var/run/docker.sock: endpoint is a symlink`. This is an upstream CLI/environment incompatibility, not proof that the adapter's sandbox flags are wrong. The failed receipt retains the diagnostic and requested permission metadata; the adapter never retries with sandbox off. Do **not** remove or alter the Docker socket as a runner workaround. Verify sandbox enforcement on a compatible host or check a newer Grok version with an actual sandboxed invocation, not just `--help`. An explicitly requested `danger` run has no OS isolation and does not validate readonly delegation. Track compatibility and any confirmed upstream fix in [#61](https://github.com/tranhoangnguyen03/pi-flow-external/issues/61); no fixed upstream version has been verified.
|
|
55
|
+
|
|
56
|
+
Muse Code 1.3.0 accepts `none` in help text but its `meta` provider rejects `--reasoning-effort none`. Effective `thinking: off` (including the parent session's inherited default) therefore fails before launch with remediation; it is never silently mapped to minimal. Pin the Muse profile to `minimal`, `low`, `medium`, `high`, or `xhigh`, or select a supported parent thinking level. Removing a profile pin alone does not help when the parent still inherits `off`. Compatibility probes must invoke provider validation, not merely inspect `--help`.
|
|
55
57
|
|
|
56
58
|
Muse's `meta` provider performs its own internal retries (observed up to 10 attempts with growing backoff on transient 503/504 errors) entirely inside the `muse` process; this is unrelated to and invisible from this extension's own no-auto-retry contract, and shows up only as activity narration (e.g. "retrying meta model stream in 60000ms (attempt 3/10)"). A run that never reports usage/cost is expected — Muse has never been observed to report either.
|
|
57
59
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tranhoangnguyen0310/pi-flow-external",
|
|
3
|
-
"version": "2.4.
|
|
3
|
+
"version": "2.4.2-external.0",
|
|
4
4
|
"description": "External Claude Code, Codex CLI, Antigravity, Grok Build CLI, and Muse Code delegation for pi.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "./index.ts",
|
package/src/core/grok.ts
CHANGED
|
@@ -274,6 +274,29 @@ export function extractGrokError(event: Record<string, unknown>): string | undef
|
|
|
274
274
|
return undefined;
|
|
275
275
|
}
|
|
276
276
|
|
|
277
|
+
/**
|
|
278
|
+
* Produces an actionable diagnosis when Grok's kernel sandbox refuses to start
|
|
279
|
+
* because a runtime socket deny path is a symlink.
|
|
280
|
+
*
|
|
281
|
+
* Preserves the exact stderr and explains:
|
|
282
|
+
* - Upstream Grok sandbox initialization failed because a runtime socket deny path is a symlink.
|
|
283
|
+
* - The socket must NOT be removed or altered as a runner workaround.
|
|
284
|
+
* - pi-flow-external preserves requested sandbox permissions and will never automatically downgrade
|
|
285
|
+
* or retry with protections disabled.
|
|
286
|
+
* - Recommends running on a compatible host or upgrading to an upstream Grok release.
|
|
287
|
+
*/
|
|
288
|
+
export function diagnoseGrokSandboxError(stderr: string): string | undefined {
|
|
289
|
+
if (!stderr.includes("runtime-socket deny path") || !stderr.includes("endpoint is a symlink")) {
|
|
290
|
+
return undefined;
|
|
291
|
+
}
|
|
292
|
+
return (
|
|
293
|
+
"Grok sandbox initialization failed because a runtime socket deny path is a symlink. " +
|
|
294
|
+
"Do not remove or alter the socket as a runner workaround. " +
|
|
295
|
+
"pi-flow-external preserves requested sandbox permissions and will not automatically downgrade or retry with protections disabled. " +
|
|
296
|
+
"Run on a compatible host or upgrade to an upstream Grok release that resolves socket symlinks."
|
|
297
|
+
);
|
|
298
|
+
}
|
|
299
|
+
|
|
277
300
|
function getPreviewFromRecord(record: Record<string, unknown>): string {
|
|
278
301
|
const candidates = [
|
|
279
302
|
record.command,
|
|
@@ -560,8 +583,10 @@ export async function spawnGrokSubagent(params: {
|
|
|
560
583
|
}
|
|
561
584
|
if (closeResult.code !== 0) {
|
|
562
585
|
const stderr = stderrBuffer.text().trim();
|
|
586
|
+
const diagnostic = diagnoseGrokSandboxError(stderr);
|
|
587
|
+
const diagnosticSuffix = diagnostic ? `\n\nDiagnostic: ${diagnostic}` : "";
|
|
563
588
|
throw new Error(
|
|
564
|
-
`grok exited with code ${closeResult.code}${closeResult.signal ? ` (signal ${closeResult.signal})` : ""}${stderr ? `: ${stderr}` : ""}`,
|
|
589
|
+
`grok exited with code ${closeResult.code}${closeResult.signal ? ` (signal ${closeResult.signal})` : ""}${stderr ? `: ${stderr}` : ""}${diagnosticSuffix}`,
|
|
565
590
|
);
|
|
566
591
|
}
|
|
567
592
|
|
package/src/core/muse.ts
CHANGED
|
@@ -21,6 +21,7 @@ import { abortChildTree } from "./process-tree.ts";
|
|
|
21
21
|
import { selectorHarness } from "../profiles.ts";
|
|
22
22
|
|
|
23
23
|
const MUSE_COMMAND = "muse";
|
|
24
|
+
const MUSE_PROVIDER = "meta";
|
|
24
25
|
|
|
25
26
|
export type MuseReasoningEffort = "none" | "minimal" | "low" | "medium" | "high" | "xhigh";
|
|
26
27
|
|
|
@@ -49,6 +50,38 @@ export function normalizeMuseReasoningEffort(thinkingLevel: ThinkingLevel | unde
|
|
|
49
50
|
return undefined;
|
|
50
51
|
}
|
|
51
52
|
|
|
53
|
+
/**
|
|
54
|
+
* `muse exec --help` syntactically lists `none` as a valid `--reasoning-effort`
|
|
55
|
+
* value, but `--provider meta` rejects it at launch (verified against
|
|
56
|
+
* installed Muse Code 1.3.0: `--reasoning-effort none is not supported with
|
|
57
|
+
* --provider meta; choose minimal|low|medium|high|xhigh|max|ultra`, exit 2).
|
|
58
|
+
* `thinking: "off"` therefore has no representable effort for this provider —
|
|
59
|
+
* and it is not a rare edge case. `off` is the pinned `pi-agent-core` SDK's
|
|
60
|
+
* own session-level thinking default (`agent.js`/`agent-harness.js`/
|
|
61
|
+
* `session.js`), so any muse call whose profile leaves `thinking` unset
|
|
62
|
+
* inherits `"off"` from the parent session unless a user has explicitly
|
|
63
|
+
* raised it; `pi-subagent.ts`/`workflow/tool.ts` resolve
|
|
64
|
+
* `profile.thinking ?? <session thinking level>` before ever reaching
|
|
65
|
+
* {@link buildMuseArgs}, so a profile pinning `thinking: off` in its own
|
|
66
|
+
* frontmatter reaches this same check the same way. `buildMuseArgs` applies
|
|
67
|
+
* one more `thinkingLevel ?? profile.thinking` fallback of its own (for
|
|
68
|
+
* direct/test callers), so this is checked once on that fully merged,
|
|
69
|
+
* already-normalized value — inherited-default and profile-pinned `off`
|
|
70
|
+
* are indistinguishable by the time they get here, and both are rejected
|
|
71
|
+
* rather than one silently downgraded to `minimal` (a different, undisclosed
|
|
72
|
+
* reasoning level from what was actually requested or inherited).
|
|
73
|
+
*/
|
|
74
|
+
function assertMuseReasoningEffortSupported(effort: MuseReasoningEffort | undefined, profileName: string): void {
|
|
75
|
+
if (effort !== "none") {
|
|
76
|
+
return;
|
|
77
|
+
}
|
|
78
|
+
throw new Error(
|
|
79
|
+
`Muse profile "${profileName}" resolved thinking "off", which has no --reasoning-effort equivalent under --provider ${MUSE_PROVIDER} ` +
|
|
80
|
+
`(muse rejects "--reasoning-effort none"). Pin this Muse profile's thinking to minimal, low, medium, high, or xhigh, ` +
|
|
81
|
+
`or select a supported parent thinking level. An unset profile inherits the parent's level, which may still be off.`,
|
|
82
|
+
);
|
|
83
|
+
}
|
|
84
|
+
|
|
52
85
|
export function buildMuseArgs({
|
|
53
86
|
promptFilePath,
|
|
54
87
|
schemaFilePath,
|
|
@@ -66,7 +99,10 @@ export function buildMuseArgs({
|
|
|
66
99
|
permission?: PermissionTier;
|
|
67
100
|
resumeSessionId?: string;
|
|
68
101
|
}): string[] {
|
|
69
|
-
const
|
|
102
|
+
const effort = normalizeMuseReasoningEffort(thinkingLevel ?? profile.thinking);
|
|
103
|
+
assertMuseReasoningEffortSupported(effort, profile.name);
|
|
104
|
+
|
|
105
|
+
const args: string[] = ["exec", "--json", "--provider", MUSE_PROVIDER, "--workspace", workspace, "--prompt-file", promptFilePath];
|
|
70
106
|
if (schemaFilePath) {
|
|
71
107
|
args.push("--output-schema", schemaFilePath);
|
|
72
108
|
}
|
|
@@ -77,7 +113,6 @@ export function buildMuseArgs({
|
|
|
77
113
|
if (profile.model) {
|
|
78
114
|
args.push("--model", profile.model);
|
|
79
115
|
}
|
|
80
|
-
const effort = normalizeMuseReasoningEffort(thinkingLevel ?? profile.thinking);
|
|
81
116
|
if (effort) {
|
|
82
117
|
args.push("--reasoning-effort", effort);
|
|
83
118
|
}
|
|
@@ -45,7 +45,8 @@ export function prepareParentContext(
|
|
|
45
45
|
): { prompt: string; context?: ParentContextReceipt } {
|
|
46
46
|
const context = parseParentContext(selection);
|
|
47
47
|
if (!context || context.mode === "none") return { prompt };
|
|
48
|
-
|
|
48
|
+
const resumeId = typeof resume === "string" && resume.trim() !== "" ? resume.trim() : undefined;
|
|
49
|
+
if (resumeId !== undefined) throw new Error("context sharing cannot be combined with resume; continue the child or start a new one");
|
|
49
50
|
if (!messages) throw new Error("Parent context is unavailable; use context:none and a self-contained prompt");
|
|
50
51
|
const compacted = messages.some((message) => message.role === "compactionSummary");
|
|
51
52
|
let start = 0;
|
package/src/external-help.ts
CHANGED
|
@@ -22,8 +22,7 @@ const externalHelpParameters = Type.Object({
|
|
|
22
22
|
description: "Help topic: role descriptions/configured profile availability, harness permissions, or workflow syntax and saved workflows.",
|
|
23
23
|
}),
|
|
24
24
|
harness: Type.Optional(Type.String({
|
|
25
|
-
|
|
26
|
-
description: "Optional harness filter for roles or permissions: agy, claude, codex, grok, muse, or a registered pi-* harness. Do not use with topic workflow.",
|
|
25
|
+
description: "Optional harness filter for roles or permissions: agy, claude, codex, grok, muse, or a registered pi-* harness. Workflows orchestrate across harnesses.",
|
|
27
26
|
})),
|
|
28
27
|
});
|
|
29
28
|
|
|
@@ -131,14 +130,18 @@ export function createExternalHelpTool(
|
|
|
131
130
|
promptSnippet: EXTERNAL_HELP_PROMPT_SNIPPET,
|
|
132
131
|
parameters: externalHelpParameters,
|
|
133
132
|
async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
133
|
+
// A schema-conversion layer downstream of this tool's declaration may
|
|
134
|
+
// present `harness` as required (#62); treat a blank/whitespace
|
|
135
|
+
// placeholder as omitted. Furthermore, do not reject an explicit
|
|
136
|
+
// harness on topic "workflow": workflows orchestrate across all
|
|
137
|
+
// harnesses, so passing a harness filter gracefully returns workflow
|
|
138
|
+
// guidance without error.
|
|
139
|
+
const harness = params.harness?.trim() ? params.harness.trim() : undefined;
|
|
137
140
|
let text: string;
|
|
138
141
|
const { harnesses: harnessConfigs } = loadHarnessConfigs(getAgentDir());
|
|
139
142
|
const configuredPiHarnesses = new Set(harnessConfigs.keys());
|
|
140
143
|
if (params.topic !== "workflow") {
|
|
141
|
-
validateHarnessFilter(
|
|
144
|
+
validateHarnessFilter(harness, configuredPiHarnesses);
|
|
142
145
|
}
|
|
143
146
|
if (params.topic === "roles") {
|
|
144
147
|
const profiles = mergeSynthesizedPiProfiles(
|
|
@@ -148,19 +151,22 @@ export function createExternalHelpTool(
|
|
|
148
151
|
),
|
|
149
152
|
harnessConfigs,
|
|
150
153
|
);
|
|
151
|
-
text = formatExternalRoleHelp(profiles, options.getDefaultHarness(ctx),
|
|
154
|
+
text = formatExternalRoleHelp(profiles, options.getDefaultHarness(ctx), harness);
|
|
152
155
|
} else if (params.topic === "permissions") {
|
|
153
|
-
text = permissionHelp(
|
|
156
|
+
text = permissionHelp(harness, configuredPiHarnesses);
|
|
154
157
|
} else {
|
|
155
|
-
|
|
158
|
+
const wfText = workflowHelp(options.workflowEnabled, listSavedWorkflows({
|
|
156
159
|
agentDir: getAgentDir(),
|
|
157
160
|
cwd: ctx.cwd,
|
|
158
161
|
projectTrusted: isProjectTrusted(ctx),
|
|
159
162
|
}));
|
|
163
|
+
text = harness
|
|
164
|
+
? `${wfText}\n\nNote: Workflows orchestrate across multiple harnesses (including "${harness}"); workflow syntax is uniform across backends.`
|
|
165
|
+
: wfText;
|
|
160
166
|
}
|
|
161
167
|
return {
|
|
162
168
|
content: [{ type: "text" as const, text }],
|
|
163
|
-
details: { topic: params.topic, ...(
|
|
169
|
+
details: { topic: params.topic, ...(harness ? { harness } : {}) },
|
|
164
170
|
};
|
|
165
171
|
},
|
|
166
172
|
});
|
package/src/external-runs.ts
CHANGED
|
@@ -28,13 +28,9 @@ const externalRunsParameters = Type.Object({
|
|
|
28
28
|
action: StringEnum(["list", "inspect", "wait", "cancel"] as const, {
|
|
29
29
|
description: "list: page runs/workflows; inspect: read one run; wait: block until selected terminal outcomes; cancel: stop one run.",
|
|
30
30
|
}),
|
|
31
|
-
runId: Type.Optional(Type.String({
|
|
32
|
-
description: "Target run ID (run_... agent, wf_... workflow). Required for inspect/cancel/wait of a single run.",
|
|
33
|
-
})),
|
|
34
31
|
runIds: Type.Optional(Type.Array(Type.String(), {
|
|
35
|
-
minItems: 1,
|
|
36
32
|
maxItems: MAX_TARGETS,
|
|
37
|
-
description: "
|
|
33
|
+
description: "Target run IDs (run_... agent, wf_... workflow). For inspect: a single-entry list [\"run_...\"] inspects that run with any view (summary, output, diagnostics, final); multiple entries (1-20 targets) batch summary inspect. For cancel: a single-entry list [\"run_...\"]. For wait: any|all of up to 100 targets.",
|
|
38
34
|
})),
|
|
39
35
|
view: Type.Optional(StringEnum(["summary", "output", "diagnostics", "final"] as const, {
|
|
40
36
|
description: "inspect view: summary (state/timing/freshness/refs, default), output (assistant text plus canonical result, partial or final), diagnostics (tool activity/errors), final (only the verified canonical terminal answer, empty until a successful terminal boundary exists). Batch inspect (runIds) only supports summary.",
|
|
@@ -67,7 +63,10 @@ const externalRunsParameters = Type.Object({
|
|
|
67
63
|
})),
|
|
68
64
|
});
|
|
69
65
|
|
|
70
|
-
export type ExternalRunsParams = Static<typeof externalRunsParameters
|
|
66
|
+
export type ExternalRunsParams = Static<typeof externalRunsParameters> & {
|
|
67
|
+
/** Legacy single-target selector; preserved for backwards-compatible programmatic/test callers. */
|
|
68
|
+
runId?: string;
|
|
69
|
+
};
|
|
71
70
|
type ExternalRunsDetails = Record<string, unknown>;
|
|
72
71
|
|
|
73
72
|
export interface CreateExternalRunsToolOptions {
|
|
@@ -518,9 +517,12 @@ function renderExternalRunsCall(args: Record<string, unknown>, theme: Theme): Te
|
|
|
518
517
|
const ids = Array.isArray(args.runIds) ? args.runIds : args.runId ? [args.runId] : [];
|
|
519
518
|
detail = `waiting for ${ids.length} task(s) · mode ${typeof args.mode === "string" ? args.mode : "all"}`;
|
|
520
519
|
} else if (action === "cancel") {
|
|
521
|
-
|
|
520
|
+
const target = Array.isArray(args.runIds) && args.runIds.length ? args.runIds[0] : String(args.runId ?? "");
|
|
521
|
+
detail = `cancel ${target}`;
|
|
522
522
|
} else if (action === "inspect") {
|
|
523
|
-
const target = Array.isArray(args.runIds)
|
|
523
|
+
const target = Array.isArray(args.runIds)
|
|
524
|
+
? args.runIds.length === 1 ? args.runIds[0] : `${args.runIds.length} run(s) (batch)`
|
|
525
|
+
: String(args.runId ?? "");
|
|
524
526
|
detail = `inspect ${target} · ${typeof args.view === "string" ? args.view : "summary"}`;
|
|
525
527
|
} else {
|
|
526
528
|
detail = `list${typeof args.workflowRunId === "string" ? ` · workflow ${args.workflowRunId}` : ""}`;
|
|
@@ -626,6 +628,33 @@ function renderExternalRunsResult(toolResult: { content: Array<{ type: string; t
|
|
|
626
628
|
return new Text(textFromToolResult(toolResult), 0, 0);
|
|
627
629
|
}
|
|
628
630
|
|
|
631
|
+
/**
|
|
632
|
+
* Reconcile the `runId`/`runIds` selector pair for `inspect` only — the one
|
|
633
|
+
* action that already hard-rejects supplying both (`wait` treats a stray
|
|
634
|
+
* `runId` alongside `runIds` as a harmless one-element convenience, and
|
|
635
|
+
* `cancel` never reads `runIds` at all, so neither gets a new conflict
|
|
636
|
+
* check here). A schema-conversion layer downstream of this tool's
|
|
637
|
+
* declaration may present both mutually exclusive optional selectors as
|
|
638
|
+
* required (#62); a model forced to fill in the one it means to omit
|
|
639
|
+
* typically sends a blank string or an empty array. Neither can name an
|
|
640
|
+
* actual run, so treat that placeholder as omitted rather than a real
|
|
641
|
+
* conflict — this is narrower than guessing between two genuinely
|
|
642
|
+
* populated, disagreeing selectors, which still fails loudly exactly as
|
|
643
|
+
* before (including when both name the very same run: inspect has always
|
|
644
|
+
* rejected supplying the pair at all, on purpose).
|
|
645
|
+
*/
|
|
646
|
+
function normalizeInspectSelectors(params: ExternalRunsParams): ExternalRunsParams {
|
|
647
|
+
const runId = params.runId === undefined || params.runId.trim() === "" ? undefined : params.runId;
|
|
648
|
+
const runIds = runId !== undefined && (params.runIds === undefined || params.runIds.length === 0)
|
|
649
|
+
? undefined
|
|
650
|
+
: params.runIds;
|
|
651
|
+
if (runId !== undefined && runIds !== undefined && runIds.length > 0) {
|
|
652
|
+
throw new Error("inspect accepts either runId or runIds, not both");
|
|
653
|
+
}
|
|
654
|
+
if (runId === params.runId && runIds === params.runIds) return params;
|
|
655
|
+
return { ...params, runId, runIds };
|
|
656
|
+
}
|
|
657
|
+
|
|
629
658
|
export function createExternalRunsTool(
|
|
630
659
|
options: CreateExternalRunsToolOptions,
|
|
631
660
|
): ToolDefinition<typeof externalRunsParameters, ExternalRunsDetails> {
|
|
@@ -636,6 +665,12 @@ export function createExternalRunsTool(
|
|
|
636
665
|
promptSnippet: EXTERNAL_RUNS_PROMPT_SNIPPET,
|
|
637
666
|
parameters: externalRunsParameters,
|
|
638
667
|
async execute(_toolCallId, params: ExternalRunsParams, signal, onUpdate, ctx) {
|
|
668
|
+
if (params.action === "inspect") {
|
|
669
|
+
params = normalizeInspectSelectors(params);
|
|
670
|
+
if (params.runId === undefined && params.runIds !== undefined && params.runIds.length === 1 && params.view !== undefined && params.view !== "summary") {
|
|
671
|
+
params = { ...params, runId: params.runIds[0], runIds: undefined };
|
|
672
|
+
}
|
|
673
|
+
}
|
|
639
674
|
const { sessionId, project } = scope(ctx);
|
|
640
675
|
const runsDirectory = options.runsDirectory();
|
|
641
676
|
|
|
@@ -670,7 +705,7 @@ export function createExternalRunsTool(
|
|
|
670
705
|
}
|
|
671
706
|
|
|
672
707
|
if (params.action === "inspect" && params.runIds !== undefined) {
|
|
673
|
-
|
|
708
|
+
// normalizeInspectSelectors already guarantees runId is unset here.
|
|
674
709
|
if (params.view !== undefined && params.view !== "summary") throw new Error('Batch inspect (runIds) only supports view: "summary"');
|
|
675
710
|
const runIds = [...new Set(params.runIds)];
|
|
676
711
|
if (runIds.length === 0 || runIds.length > MAX_BATCH_INSPECT_TARGETS) {
|
|
@@ -812,26 +847,35 @@ export function createExternalRunsTool(
|
|
|
812
847
|
}
|
|
813
848
|
|
|
814
849
|
if (params.action === "cancel") {
|
|
815
|
-
|
|
816
|
-
|
|
850
|
+
const rawRunIds = params.runIds;
|
|
851
|
+
if (rawRunIds?.length && params.runId?.trim()) {
|
|
852
|
+
throw new Error("cancel accepts either runId or runIds, not both");
|
|
853
|
+
}
|
|
854
|
+
if (rawRunIds && rawRunIds.length > 1) {
|
|
855
|
+
throw new Error("cancel targets one run at a time");
|
|
856
|
+
}
|
|
857
|
+
const targetRunId = (rawRunIds && rawRunIds.length === 1 ? rawRunIds[0] : undefined)
|
|
858
|
+
?? (typeof params.runId === "string" && params.runId.trim() !== "" ? params.runId.trim() : undefined);
|
|
859
|
+
assertRunId(targetRunId);
|
|
860
|
+
const entry = options.registry.get(targetRunId);
|
|
817
861
|
if (entry) {
|
|
818
862
|
assertOwned(entry, sessionId, project);
|
|
819
|
-
const status = options.registry.cancel(
|
|
820
|
-
return result(status === "requested" ? `Cancellation requested for ${
|
|
863
|
+
const status = options.registry.cancel(targetRunId, params.reason ?? "cancelled by external_runs");
|
|
864
|
+
return result(status === "requested" ? `Cancellation requested for ${targetRunId}.` : `${targetRunId} is already terminal.`, { runId: targetRunId, status });
|
|
821
865
|
}
|
|
822
|
-
const isWorkflowId = WORKFLOW_ID.test(
|
|
866
|
+
const isWorkflowId = WORKFLOW_ID.test(targetRunId);
|
|
823
867
|
const workflowDir = getSessionWorkflowDir(ctx);
|
|
824
|
-
const historicalWorkflow = isWorkflowId && workflowDir ? await loadWorkflowJournal(workflowDir,
|
|
868
|
+
const historicalWorkflow = isWorkflowId && workflowDir ? await loadWorkflowJournal(workflowDir, targetRunId) : undefined;
|
|
825
869
|
if (historicalWorkflow) {
|
|
826
870
|
if (historicalWorkflow.project !== project) throw new Error("Run is unknown or unavailable in this session");
|
|
827
|
-
if (historicalWorkflow.status !== "running") return result(`${
|
|
871
|
+
if (historicalWorkflow.status !== "running") return result(`${targetRunId} is already terminal.`, { runId: targetRunId, status: "terminal" });
|
|
828
872
|
throw new Error("Run is no longer live in this session; cancellation cannot be confirmed");
|
|
829
873
|
}
|
|
830
874
|
// An unresolved wf_... ID is unknown, not an unowned agent record.
|
|
831
875
|
if (isWorkflowId) throw new Error("Run is unknown or unavailable in this session");
|
|
832
|
-
const durable = await getRunRecord(runsDirectory,
|
|
876
|
+
const durable = await getRunRecord(runsDirectory, targetRunId);
|
|
833
877
|
assertOwnedRecord(durable, sessionId, project);
|
|
834
|
-
if (terminalRecord(durable)) return result(`${
|
|
878
|
+
if (terminalRecord(durable)) return result(`${targetRunId} is already terminal.`, { runId: targetRunId, status: "terminal" });
|
|
835
879
|
throw new Error("Run is no longer live in this session; cancellation cannot be confirmed");
|
|
836
880
|
}
|
|
837
881
|
|
package/src/pi-subagent.ts
CHANGED
|
@@ -358,9 +358,10 @@ function createAgentTool(
|
|
|
358
358
|
parameters: agentToolParameters,
|
|
359
359
|
executionMode: "parallel",
|
|
360
360
|
async execute(toolCallId, params, signal, onUpdate, ctx) {
|
|
361
|
+
const resume = typeof params.resume === "string" && params.resume.trim() !== "" ? params.resume.trim() : undefined;
|
|
361
362
|
const briefing = prepareParentContext(params.prompt, params.context,
|
|
362
363
|
params.context && params.context.mode !== "none" ? captureParentContext(ctx.sessionManager) : undefined,
|
|
363
|
-
toolCallId,
|
|
364
|
+
toolCallId, resume);
|
|
364
365
|
const state = getState();
|
|
365
366
|
const effectiveState: DelegationState = {
|
|
366
367
|
...state,
|
|
@@ -524,7 +525,7 @@ function createAgentTool(
|
|
|
524
525
|
permission: params.permission,
|
|
525
526
|
defaultPermission,
|
|
526
527
|
maxBudgetUsd,
|
|
527
|
-
resumeRunId:
|
|
528
|
+
resumeRunId: resume,
|
|
528
529
|
executionStartedAt: run.progress.executionStartedAt,
|
|
529
530
|
onProgress: (partial) => {
|
|
530
531
|
const details = partial.details as SubagentToolDetails;
|