@tranhoangnguyen0310/pi-flow-external 2.4.1-external.0 → 2.5.0-external.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +5 -3
- package/CHANGELOG.md +20 -0
- package/CONTEXT.md +4 -4
- package/README.md +62 -8
- package/package.json +1 -1
- package/src/core/agent-snapshot.ts +2 -0
- package/src/core/capabilities.ts +187 -0
- package/src/core/parent-context.ts +2 -1
- package/src/core/spawn.ts +111 -22
- package/src/external-command.ts +55 -6
- package/src/external-help.ts +21 -16
- package/src/external-runs.ts +62 -18
- package/src/pi-subagent.ts +115 -8
- package/src/profile-creator.ts +59 -18
- package/src/profiles.ts +153 -4
- package/src/settings.ts +167 -2
- package/src/types.ts +39 -0
- package/src/workflow/replay-cache.ts +1 -0
- package/src/workflow/tool.ts +167 -5
- package/src/workflow/types.ts +7 -0
package/AGENTS.md
CHANGED
|
@@ -6,12 +6,14 @@ This fork changes the original pi-flow contract: `Agent` is not a generic Pi sub
|
|
|
6
6
|
|
|
7
7
|
- Ordinary driver tools: `Agent`, read-only `external_help`, `external_runs`, and optional `workflow`. `pi_flow_profile_create` and `pi_flow_harness_create` are active only inside `/external profile create`.
|
|
8
8
|
- User operations use `/external`; `/pi-flow-profile create` is a temporary deprecated alias.
|
|
9
|
-
- Extension-owned concurrency, timeout, and global default-harness settings live in `$PI_CODING_AGENT_DIR/pi-flow-external/settings.json`. Profile backend/model/thinking metadata remains in `subagents/*.md`. Named Pi harness configurations live in the sibling `$PI_CODING_AGENT_DIR/pi-flow-external/harnesses.json` registry (one shared file, `pi-<label>` keys pinning `model`/`thinking`). Registry writes are atomic per file, but there is no inter-process lock: concurrent writers adding distinct names from separate sessions are last-writer-wins and may drop one update. This is an accepted v1 limitation shared with `settings.json` and `subagents/*.md`, not a bug to re-report. A trusted project may override `defaultHarness` via `.pi/pi-flow-external/settings.json` (
|
|
9
|
+
- Extension-owned concurrency, timeout, and global default-harness settings live in `$PI_CODING_AGENT_DIR/pi-flow-external/settings.json`. Profile backend/model/thinking metadata remains in `subagents/*.md`. Named Pi harness configurations live in the sibling `$PI_CODING_AGENT_DIR/pi-flow-external/harnesses.json` registry (one shared file, `pi-<label>` keys pinning `model`/`thinking`). Registry writes are atomic per file, but there is no inter-process lock: concurrent writers adding distinct names from separate sessions are last-writer-wins and may drop one update. This is an accepted v1 limitation shared with `settings.json` and `subagents/*.md`, not a bug to re-report. A trusted project may override `defaultHarness` and/or `piCapabilitySets` via `.pi/pi-flow-external/settings.json` (those two keys only, read-only to the extension); explicit call harness > project default > global default. `defaultHarness` may itself name a registered `pi-*` harness. A project `piCapabilitySets` entry replaces the global entry of the same name in full, never merging its `skills`/`promptTemplates` arrays.
|
|
10
10
|
- Every `Agent` call requires `description`, `prompt`, and either `role` with optional `harness` (`agy`, `claude`, `codex`, `grok`, `muse`, or a registered `pi-*` harness name) or the legacy exact-profile `subagent_type`. The selectors cannot be combined. Without `harness`, use the effective default harness; if that default names a `pi-*` harness that is no longer registered, the call fails with an actionable error rather than silently falling back to `agy`.
|
|
11
11
|
- Valid profiles come from `~/.pi/agent/subagents/*.md` and set `backend: claude`, `backend: codex`, `backend: agy`, `backend: grok`, `backend: muse`, or (only when also declaring `harness: <registered pi-* name>`) `backend: pi`. Use matching names such as `claude-*`, `codex-*`, `agy-*`, `grok-*`, `muse-*`, or `<harness>-*`; the assisted creator enforces that convention.
|
|
12
12
|
- A default roster ships with the extension and seeds once on session start: five code-oriented roles (explorer, planner, implementer, reviewer, qa) plus the generalist worker, one storage profile per CLI backend across five backends (30 files, but six advertised roles). Seeding never overwrites existing files and respects later deletions. Roles outside the default roster are user-created via `/external profile create`. Every registered named Pi harness automatically and immediately gets the same six roles too, synthesized in-memory from the identical canonical role source — no per-harness file needed, and no on-disk seeding for pi harnesses. An installation seeded before Grok support existed migrates forward by adding only the six new `grok-*` profiles on its next seed pass; one seeded before Muse support existed adds only the six new `muse-*` profiles; either migration never resurrects a deleted or customized profile from an earlier backend.
|
|
13
13
|
- A `backend: pi` profile is external-delegation-eligible only when it also declares `harness: <name>` for a name present in the live `harnesses.json` registry; a bare `backend: pi` profile (no `harness`, or one naming an unregistered config) is filtered out and rejected exactly as before — it belongs to Pi's native subagent system and this extension never modifies it.
|
|
14
|
-
- **Named Pi harness configuration:** a `pi-<label>` entry in `harnesses.json` pinning a `provider/model` id (resolved through Pi's own model registry) and a thinking level (`off|minimal|low|medium|high|xhigh`, always persisted explicitly). It runs in-process via Pi's own SDK, not as a spawned CLI. A pi child loads
|
|
14
|
+
- **Named Pi harness configuration:** a `pi-<label>` entry in `harnesses.json` pinning a `provider/model` id (resolved through Pi's own model registry) and a thinking level (`off|minimal|low|medium|high|xhigh`, always persisted explicitly). It runs in-process via Pi's own SDK, not as a spawned CLI. A pi child never loads project/user extensions, MCP, or themes; it additionally loads skills/prompt templates only when its profile declares a `capabilitySet` (below), and even then only the exact resources that set names. Its entire tool surface is otherwise the SDK's own builtins (`read`/`bash`/`edit`/`write`, plus `grep`/`find`/`ls` where a tier's default allow-list adds them), curated but never claimed to be an OS sandbox: `danger` tier `bash` is exactly as exposed as on any external CLI. Retry is disabled per child (in-memory, call-scoped, never touching the user's real settings) to honor this extension's no-auto-retry contract, since the underlying SDK otherwise retries transient provider errors on its own. Pi children cannot resume (no persisted session) and have no enforced budget cap — both are deliberate v1 limitations, not oversights. Follow-up capability expansion (trusted extensions/MCP, resumable sessions, real budget controls) is tracked in issue #43; skills/prompt templates are covered by the slice below.
|
|
15
|
+
- **Reusable capability sets (issue #43, second slice):** a named `piCapabilitySets` entry — `{ skills: string[]; promptTemplates: string[] }`, exact resource names only, never booleans or wildcards — lives in the existing global `pi-flow-external/settings.json` and, for a trusted project, in `.pi/pi-flow-external/settings.json`; a project entry replaces the same-named global entry wholesale (arrays are never merged). A profile opts in via frontmatter `capabilitySet: <name>`, selectable across any harness (including a shared `pi-*` role template, which propagates the field onto every harness it materializes to). Absent, behavior is unchanged: a builtins-only child with nothing loaded. Selected resources are discovered through the installed Pi SDK's own `DefaultResourceLoader` (`noExtensions`/`noThemes` always stay `true`; nothing is executed as a tool), post-filtered to exactly the named skills/prompt templates; project-scope resources are visible only when `ctx.isProjectTrusted()` says so, threaded explicitly rather than relying on the SDK's own trusted-by-default posture. An unknown set name or a selected resource that isn't discoverable fails before any prompt exists, naming exactly what's missing. Because a skill is a lazy file read (the child reads SKILL.md itself via `read`, not a snapshotted string), a workflow freezes one resolved selection per distinct `capabilitySet` up front — hashing the actual selected file bytes plus resource scope, not just names — for its replay fingerprint, and re-resolves and compares that hash immediately before each child spawn, rejecting the child if the content drifted since the run started rather than pretending to run frozen bytes. A direct `Agent` call resolves fresh immediately before its own spawn (no freeze gap to re-verify). Selected names are disclosed on the intent card before launch and recorded as `capabilities` in both the Agent tool's and workflow child receipts, distinct from the tool-name/OS-sandbox boundary tiers already describe. Selection only ever widens what the SDK's own resource loader discovers for the child; it never changes how the SDK exposes what it discovers. A selected prompt template is reachable only through the child's own explicit `/template-name args` task text (`session.prompt`'s built-in expansion); it is never injected into the system prompt, so an unselected or misspelled name is simply never a template and passes through literally. A selected skill's autonomous discoverability (its name/description listed in the system prompt for the model to pick up on its own initiative) is governed entirely by that skill's own `disable-model-invocation` frontmatter flag, exactly as the SDK already enforces (`formatSkillsForPrompt`) — this extension's capabilitySet filtering narrows *which* skills are visible at all, but never overrides that flag; a skill marked `disable-model-invocation: true` still expands via an explicit `/skill:name args` task regardless, since that lookup is by exact selected name, not by the flag. Workflow `agent()` calls append the structured-output contract (or the plain-text-output note) to the task text as `appendInstructions` *after* the caller's own prompt; when that prompt itself is a `/template-name args` or `/skill:name args` invocation, the appended text becomes part of the trailing `args` the SDK substitutes/passes through — expected stock substitution behavior, not something this extension adds, but worth knowing before relying on a workflow-driven template/skill invocation's exact argument boundary.
|
|
16
|
+
- **Shared custom Pi roles:** a custom (non-canonical) role can be authored once as `~/.pi/agent/subagents/pi-<role>.md`, declaring `backend: pi` and the literal marker `harness: "pi-*"` instead of one concrete harness name. Such a template is never itself external-delegation-eligible or admitted as a native profile — the marker deliberately fails `isValidHarnessName`, so it can never satisfy registry membership. Instead, `mergeSynthesizedPiProfiles` materializes it into a concrete `<harness>-<role>` profile for every currently-registered `pi-*` harness that lacks its own on-disk override for that role, pinning that harness's registered model/thinking; precedence is harness-specific on-disk file > shared template > synthesized canonical body. A shared template must not pin `model`/`thinking` itself (registry-authoritative per harness); one that does, or whose filename doesn't match its `pi-<role>` marker, is dropped with a diagnostic surfaced by `/external doctor`. `/external profile create` supports authoring one directly (branch 4), smoke-testing against one representative already-registered harness (at least one must exist).
|
|
15
17
|
- Use Pi's native subagent system for Pi-backed scout/reviewer/planner/worker/oracle work.
|
|
16
18
|
- Permission resolution takes the least restrictive of the profile tier (or default when absent) and the caller's request: `readonly < edit < danger`. Profile permissions are a floor, not an overridable default; a parent cannot handcuff a `danger` worker by requesting `readonly`. Existing backend-specific execution floors still apply. Disclosure and receipts show the resolved tier.
|
|
17
19
|
- External CLI backends use their own tools and permission mechanisms. Codex tiers map to its `--sandbox` axis. Grok tiers also map to its `--sandbox` axis (`read-only`/`workspace`/`off`), always paired with `--permission-mode bypassPermissions`; bypass only skips the interactive prompt, and the kernel sandbox remains the enforced boundary at every tier. Grok's readonly network-blocking guarantee is Linux-only (a no-op on macOS), and sandbox startup can fail closed rather than silently downgrading on some macOS hosts (for example when `/var/run/docker.sock` resolves to a symlink). Claude falls back to `--permission-mode auto` when its effective UID is 0 because Claude refuses bypass mode under root. Execution lanes (implementer, qa, worker) require shell command authority to inspect repositories and run tests; on Claude and on pi harnesses (where `edit` tier excludes `bash`), execution lanes maintain a danger floor so agents are not artificially handcuffed. Grok needs no such floor: its `edit` tier (`--sandbox workspace`) already permits shell execution, sandboxed to writes within the workspace. Antigravity (`agy`) has no granular headless permission mode — its default sandbox denies even read-only tools — so every agy run is unsandboxed (`--dangerously-skip-permissions`) and `readonly`/`edit` on agy are advisory profile-body instructions, not a boundary. Run them only in trusted repositories and state whether the task is read-only or may edit files. Muse (`muse exec`) has approval and its own sandbox ON by default; every tier passes `--disable-approval` so headless runs never hang on an interactive prompt. `readonly` additionally passes `--disable-write --disable-shell`; `edit` leaves the sandbox enabled with only approval bypassed, so shell/write stay available within it (no execution-lane danger floor is needed); `danger` uses `--yolo`, which disables approval and the sandbox and additionally trusts the workspace for this run (loads its skills/rules) — a broader grant than an unsandboxed run alone.
|
|
@@ -24,7 +26,7 @@ This fork changes the original pi-flow contract: `Agent` is not a generic Pi sub
|
|
|
24
26
|
## Delegation transparency invariants
|
|
25
27
|
|
|
26
28
|
- Treat each `description` as a concise user-facing task label. Profile descriptions are also user-visible as the declared reason for profile selection.
|
|
27
|
-
- `unsandboxed external CLI` and `external host access` disclose the real execution boundary for CLI backends; a pi harness delegation discloses `Pi SDK child · host access · curated tools` instead — never call an in-process pi child an "external CLI". Never present a read-only prompt as permission enforcement: on agy every run is unsandboxed regardless of tier, so the profile body — not the tier — is what asks the agent to stay read-only; on pi, the curated tool table bounds which tool *names* exist, but does not make `bash` at `danger` tier any less exposed.
|
|
29
|
+
- `unsandboxed external CLI` and `external host access` disclose the real execution boundary for CLI backends; a pi harness delegation discloses `Pi SDK child · host access · curated tools` instead — never call an in-process pi child an "external CLI". Never present a read-only prompt as permission enforcement: on agy every run is unsandboxed regardless of tier, so the profile body — not the tier — is what asks the agent to stay read-only; on pi, the curated tool table bounds which tool *names* exist, but does not make `bash` at `danger` tier any less exposed. A profile's `capabilitySet` selection adds text resources (skill/prompt-template content the child reads), never new tools or executable extensions/MCP — disclose it as selected skill/prompt-template names, not as expanded tool access.
|
|
28
30
|
- Keep direct intent visible during execution. Workflow access belongs once at the workflow level, not on every child row.
|
|
29
31
|
- Parent-context sharing is a disclosure, not a silent optimization: the intent card names the mode before launch, and receipts name the mode and shared/requested turns. Shared conversation content leaves for the external harness and lands in local evidence, so never describe sharing as internal or free.
|
|
30
32
|
- Keep default progress bounded and human-readable. Expanded terminal output shows bounded canonical output and a full-ID `/external runs` route; record paths, backend-event counts, workflow IDs, and journal paths remain advanced evidence.
|
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,26 @@ All notable changes to pi-flow external are documented here.
|
|
|
4
4
|
|
|
5
5
|
## Unreleased
|
|
6
6
|
|
|
7
|
+
## [2.5.0-external.0] - 2026-09-22
|
|
8
|
+
|
|
9
|
+
Partially addresses issue #43 (named Pi harness follow-up capability expansion): shared custom Pi roles and reusable capability sets. Trusted extensions/MCP, resumable sessions, and real per-child budget controls remain out of scope and stay tracked on issue #43 for a later slice.
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- **Shared custom Pi roles**: a custom (non-canonical) role can now be authored once as `~/.pi/agent/subagents/pi-<role>.md`, declaring `backend: pi` and the literal marker `harness: "pi-*"` instead of one concrete harness name. `mergeSynthesizedPiProfiles` materializes it into a concrete `<harness>-<role>` profile for every currently-registered `pi-*` harness that lacks its own on-disk override for that role, pinned to that harness's registered model/thinking. Precedence: harness-specific on-disk file > shared template > synthesized canonical body. The marker deliberately fails `isValidHarnessName`, so a shared template can never itself become externally selectable or admitted as a native profile. A malformed shared template (wrong `pi-<role>.md` filename, or one that pins `model`/`thinking`) is dropped with a diagnostic surfaced by `/external doctor` rather than failing the whole roster. `/external profile create` supports authoring one directly, smoke-tested against one representative already-registered harness.
|
|
14
|
+
- **Reusable capability sets** (`piCapabilitySets`): a named `{ skills: string[]; promptTemplates: string[] }` entry in the existing global `pi-flow-external/settings.json`, and, for a trusted project, in `.pi/pi-flow-external/settings.json` (a project entry replaces the same-named global entry wholesale, never merging arrays). A profile opts in via frontmatter `capabilitySet: <name>`, on any harness including a shared `pi-*` role template. Selected resources are discovered through the installed Pi SDK's own `DefaultResourceLoader` (extensions/MCP/themes never load), post-filtered to exactly the named skills/prompt templates, with project-scope resources visible only when the caller's project is genuinely trusted. An unknown set name or an undiscoverable resource fails before any prompt/session exists, naming exactly what's missing. A skill is a lazy file read, so both a direct `Agent` call and a workflow run content-hash the selected skill/prompt-template bytes at resolution time and again immediately before each child spawn, refusing to launch a child whose selection drifted since it was resolved/frozen; a workflow's replay fingerprint is widened by that same hash. Selected names are disclosed on the intent card before launch and recorded as `capabilities` in both the `Agent` tool's and workflow child receipts. `/external settings` lists configured sets; `/external doctor` also reports unknown-set references and malformed `capabilitySet` declarations. `/external profiles` includes materialized shared roles.
|
|
15
|
+
- A workflow child that fails before `spawnSubagent` ever runs (unknown `capabilitySet`, or any other pre-spawn resolution error) now still finishes its already-queued durable run record as a recognizable failure, instead of leaving it stuck "incomplete" forever.
|
|
16
|
+
- Offline coverage in `test/capabilities.test.ts`, extended `test/pi-runtime.test.ts` (including real-SDK tests of the installed SDK's own explicit `/name args` prompt-template expansion and `/skill:name` skill invocation semantics, unaffected by `capabilitySet` filtering), `test/profiles.test.ts`, `test/settings.test.ts`, `test/workflow.test.ts`, `test/replay-cache.test.ts`, and `test/external-command.test.ts`. Verified against a real registered `pi-*` harness with real credentials: a real child resolved a project-scope `capabilitySet`, read and followed the selected skill's content (not guessable — an unpredictable nonce), and both the returned result and the persisted receipt disclosed matching `capabilities` (set, skills, content hash).
|
|
17
|
+
|
|
18
|
+
## [2.4.2-external.0] - 2026-09-23
|
|
19
|
+
|
|
20
|
+
### Fixed
|
|
21
|
+
|
|
22
|
+
- Simplified model-facing run supervision to one `runIds` selector, including singleton output/final inspection and cancellation. Legacy `runId` callers remain supported; conflicting cancellation selectors fail explicitly.
|
|
23
|
+
- Workflow help accepts a supplied harness without blocking, explains its cross-harness scope, and accepts blank filters at the SDK schema boundary.
|
|
24
|
+
- Blank Agent resume arguments no longer conflict with context sharing or trigger resume lookup. Real resume/context conflicts remain errors.
|
|
25
|
+
- Added SDK argument-validation coverage alongside provider-payload tests. Audited workflow source selection: blank sources already normalize away; multiple real sources still fail. Downstream required-field promotion remains unverified, so #62 stays open.
|
|
26
|
+
|
|
7
27
|
## [2.4.1-external.0] - 2026-09-23
|
|
8
28
|
|
|
9
29
|
### Fixed
|
package/CONTEXT.md
CHANGED
|
@@ -26,7 +26,7 @@ This package is a fork of pi-flow whose `Agent` and `workflow` tools are reserve
|
|
|
26
26
|
- Native Pi work -> native subagent tool.
|
|
27
27
|
- Claude Code / Codex CLI / Antigravity / Grok Build CLI / Muse Code work, and named Pi harness work -> this extension's `Agent` or `workflow`.
|
|
28
28
|
|
|
29
|
-
The split is global and intentional to avoid tool ambiguity across projects. The ordinary driver sees `Agent`, optional `workflow`, and read-only `external_help`; the profile finalizer is activated only by `/external profile create`. External children start in the requested working directory, but backend-native nested helpers may create or use a separate workspace; prompts that request further nesting should include explicit absolute paths and all required context. Named Pi harness children are the one exception to "backend-native nested helpers may use a different workspace": a pi child runs in-process with no extensions
|
|
29
|
+
The split is global and intentional to avoid tool ambiguity across projects. The ordinary driver sees `Agent`, optional `workflow`, and read-only `external_help`; the profile finalizer is activated only by `/external profile create`. External children start in the requested working directory, but backend-native nested helpers may create or use a separate workspace; prompts that request further nesting should include explicit absolute paths and all required context. Named Pi harness children are the one exception to "backend-native nested helpers may use a different workspace": a pi child runs in-process with no extensions or MCP loaded and the delegation tools excluded, so it structurally cannot itself launch a further nested agent. A profile's `capabilitySet` may additionally load exact named skills/prompt templates (text resources the child reads, never executable extensions), which does not reopen that guarantee.
|
|
30
30
|
|
|
31
31
|
## User control surface
|
|
32
32
|
|
|
@@ -46,7 +46,7 @@ Workflow replay is explicit and successful-prefix-only. A persisted script resum
|
|
|
46
46
|
|
|
47
47
|
Applied to Antigravity (`agy`), that stance is absolute: the harness offers no granular headless permission mode — its default sandbox (`proceed-in-sandbox`) hard-denies even read-only tools like `read_url_content`, and `--dangerously-skip-permissions` is the only unsandboxed mode. So every agy run is unsandboxed, and `readonly`/`edit` on agy profiles are advisory instructions carried by the profile body, never an enforced boundary. We disclose that plainly (`unsandboxed external CLI`) rather than pretend a read-only tier restrains what the harness will actually allow.
|
|
48
48
|
|
|
49
|
-
Applied to named Pi harness configurations, the same "get out of the way" stance takes a different, honestly-scoped shape: a curated, builtins-only tool surface (no project/user extensions,
|
|
49
|
+
Applied to named Pi harness configurations, the same "get out of the way" stance takes a different, honestly-scoped shape: a curated, builtins-only tool surface (no project/user extensions, MCP, or themes) genuinely bounds which tool *names* a pi child can reach — that claim is mechanically true because nothing else loads — but it is not a claim that `bash` at `danger` tier is any less dangerous than on an external CLI once it is granted. A profile's `capabilitySet` selection is a distinct, separately-disclosed axis: it can add exact named skills/prompt templates (text a child reads, never a new tool), gated by explicit project trust and validated before launch, but it never widens the tool surface itself. `readonly`/`edit` tiers get a real default tool allow-list (`read`/`grep`/`find`/`ls`, plus `edit`/`write` at `edit`) so those tiers are actually usable, not just technically restrictive. Execution-lane roles requested at `edit` are still elevated to `danger` (the same floor as Claude), because `edit` excludes `bash` entirely and an execution lane without shell access cannot do its job.
|
|
50
50
|
|
|
51
51
|
Applied to the Grok Build CLI, permission tiers map onto its own kernel-enforced `--sandbox` axis (`read-only`/`workspace`/`off`), always paired with `--permission-mode bypassPermissions` — bypass only removes the interactive approval prompt, never the sandbox itself, so every tier is a genuine boundary rather than an advisory one. `edit` already permits shell execution (sandboxed to workspace writes), so, unlike Claude and pi, Grok execution-lane roles need no `edit`→`danger` floor. `readonly`'s network-blocking guarantee is Linux-only, and sandbox startup can fail closed on some macOS hosts rather than silently running unsandboxed; that fail-closed behavior is preserved rather than weakened for cross-platform convenience.
|
|
52
52
|
|
|
@@ -54,8 +54,8 @@ Applied to Muse Code, approval and its own sandbox are ON by default, so every t
|
|
|
54
54
|
|
|
55
55
|
## Known inelegance
|
|
56
56
|
|
|
57
|
-
<!-- ponytail: one-backend-per-file profile format forces role x backend file duplication; extend src/profiles.ts with multi-backend profiles (e.g. backends: [claude, codex, agy, grok, muse] plus per-backend model map) if maintaining N copies of identical role bodies ever hurts -->
|
|
58
|
-
A standardized role roster needs one file per role per backend (e.g. `claude-qa`, `codex-qa`, `agy-qa`, `grok-qa`, `muse-qa` with identical bodies) because the profile format binds `backend:` and `model:` to a single file. The duplication is accepted for now; a future format extension could let one role file cover all backends. Named Pi harnesses
|
|
57
|
+
<!-- ponytail: one-backend-per-file profile format forces role x backend file duplication across claude/codex/agy/grok/muse; extend src/profiles.ts with multi-backend profiles (e.g. backends: [claude, codex, agy, grok, muse] plus per-backend model map) if maintaining N copies of identical role bodies across CLI backends ever hurts -->
|
|
58
|
+
A standardized role roster needs one file per role per backend (e.g. `claude-qa`, `codex-qa`, `agy-qa`, `grok-qa`, `muse-qa` with identical bodies) because the profile format binds `backend:` and `model:` to a single file. The duplication across the five CLI backends is accepted for now; a future format extension could let one role file cover all backends. Named Pi harnesses resolve the equivalent duplication within the `pi-*` family two ways: the six canonical roles are synthesized once from a single shared source and applied identically to every registered `pi-*` config with zero files written, and a *custom* role can likewise be authored once as a shared `pi-<role>.md` template (`backend: pi`, `harness: "pi-*"`) that `mergeSynthesizedPiProfiles` materializes onto every registered pi harness lacking its own override — a user with several registered pi harnesses no longer needs one file per harness per custom role, only one file per harness when that harness's behavior must actually differ. The same applies to `permission:` tiers: one tier per profile, but backends interpret tiers differently (Claude and pi deny Bash at anything below `danger`), so command-running roles declare `danger` and maintain `danger` as their floor even if an explicit `edit` override is requested. Custom profiles outside the role-name convention get that floor only by declaring `permission: danger`; an undeclared custom profile that meets an `edit` tier still hits headless Bash denials (or, on pi, a curated tool set without `bash`), so `permission: danger` doubles as the lane's "needs shell" capability declaration. Named Pi harnesses also ship with no resume and no enforced budget cap in v1 — both deliberate limitations tracked in issue #43, not oversights.
|
|
59
59
|
|
|
60
60
|
## Evidence boundary
|
|
61
61
|
|
package/README.md
CHANGED
|
@@ -156,7 +156,7 @@ backend: muse
|
|
|
156
156
|
model: muse-spark-1.3-contributor
|
|
157
157
|
```
|
|
158
158
|
|
|
159
|
-
Profile instructions become the external agent's system instructions. A profile's `description` is also shown as the user-visible reason for its selection, so keep it concise and concrete. External CLIs use their own tools, so a profile's `tools:` field does not control them; a pi profile's `tools:` field does apply, intersected with its permission tier's curated tool table. A `backend: pi` profile is
|
|
159
|
+
Profile instructions become the external agent's system instructions. A profile's `description` is also shown as the user-visible reason for its selection, so keep it concise and concrete. External CLIs use their own tools, so a profile's `tools:` field does not control them; a pi profile's `tools:` field does apply, intersected with its permission tier's curated tool table. A `backend: pi` profile is directly selectable by this extension only when it also declares `harness: <name>` for a name registered in `harnesses.json` (see below), or is materialized from a shared role template declaring the literal `harness: "pi-*"` marker (see [Shared custom Pi roles](#shared-custom-pi-roles)); a bare `backend: pi` profile with no `harness`, an unregistered one, or no backend at all belongs to Pi's native subagent system and is never modified by this extension.
|
|
160
160
|
|
|
161
161
|
### Default profiles
|
|
162
162
|
|
|
@@ -192,9 +192,63 @@ Agent({
|
|
|
192
192
|
});
|
|
193
193
|
```
|
|
194
194
|
|
|
195
|
-
A custom (non-canonical) role
|
|
195
|
+
A custom (non-canonical) role can be given its own profile file per harness, the same as for the five CLI backends: `~/.pi/agent/subagents/pi-deepseek-security-reviewer.md` with `backend: pi` and `harness: pi-deepseek`. That file's body/permission/tools may be customized; its `model`/`thinking` are not — they always come from the harness's registered config, and a file that tries to override them to a different value is rejected rather than silently honored.
|
|
196
196
|
|
|
197
|
-
|
|
197
|
+
### Shared custom Pi roles
|
|
198
|
+
|
|
199
|
+
Instead of duplicating a custom role's file per harness, author it once as a **shared role template**: `~/.pi/agent/subagents/pi-security-audit.md` with `backend: pi` and the literal marker `harness: "pi-*"` (not one specific harness name):
|
|
200
|
+
|
|
201
|
+
```md
|
|
202
|
+
---
|
|
203
|
+
description: Shared security audit role.
|
|
204
|
+
backend: pi
|
|
205
|
+
harness: "pi-*"
|
|
206
|
+
---
|
|
207
|
+
|
|
208
|
+
Audit for security defects. Do not modify files.
|
|
209
|
+
```
|
|
210
|
+
|
|
211
|
+
This template is applied to every currently-registered `pi-*` harness that doesn't already have its own `<harness>-security-audit.md` override — `pi-deepseek-security-audit`, `pi-astra-security-audit`, and so on, each pinned to that harness's own registered model/thinking. Precedence: a harness-specific on-disk file wins for that harness, then the shared template, then (for the six canonical role names only) the built-in synthesized body. A shared template must not pin `model` or `thinking` itself — a harness's registry stays authoritative — and its file name must match its `pi-<role>` marker; either mistake drops the template with a diagnostic surfaced by `/external doctor` rather than silently misapplying one harness's model to the rest. `/external profile create` can author one directly (the fourth interview branch); creating one requires at least one already-registered `pi-*` harness, since its smoke test runs against one representative harness while the installed file stays harness-agnostic.
|
|
212
|
+
|
|
213
|
+
**v1 scope, by design:** a pi child's entire tool surface is the SDK's own builtins (`read`/`bash`/`edit`/`write`, plus `grep`/`find`/`ls` at `readonly`/`edit` tiers) — no project/user extensions, MCP, or themes load into it. This bounds which tool *names* exist; it does not make `bash` at `danger` tier any less exposed than on an external CLI. Retry is disabled per pi child (in-memory, never touching your real Pi settings) so a transient provider error fails immediately rather than silently retrying. Pi children cannot resume a prior conversation and have no enforced budget cap. Expanding this capability set (trusted extensions/MCP, resumable sessions, real budget controls) is tracked in [issue #43](https://github.com/tranhoangnguyen03/pi-flow-external/issues/43).
|
|
214
|
+
|
|
215
|
+
### Reusable capability sets (skills & prompt templates)
|
|
216
|
+
|
|
217
|
+
A profile can opt into a named, reusable selection of exact skills and prompt templates via frontmatter:
|
|
218
|
+
|
|
219
|
+
```md
|
|
220
|
+
---
|
|
221
|
+
description: Docs writer.
|
|
222
|
+
backend: pi
|
|
223
|
+
harness: pi-deepseek
|
|
224
|
+
capabilitySet: docs
|
|
225
|
+
---
|
|
226
|
+
|
|
227
|
+
Write documentation.
|
|
228
|
+
```
|
|
229
|
+
|
|
230
|
+
The named set itself lives in settings, not the profile — define it once, reuse it from any profile or harness:
|
|
231
|
+
|
|
232
|
+
```json
|
|
233
|
+
{
|
|
234
|
+
"piCapabilitySets": {
|
|
235
|
+
"docs": {
|
|
236
|
+
"skills": ["technical-writer"],
|
|
237
|
+
"promptTemplates": ["release-notes"]
|
|
238
|
+
}
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
```
|
|
242
|
+
|
|
243
|
+
`skills`/`promptTemplates` are exact resource names only — never booleans, never wildcards — so the set stays a closed, auditable list. Either array may be omitted, and an entry may be the empty object (`"empty": {}`) — that is a valid, explicit selection of nothing, distinct from not declaring `capabilitySet` at all. A trusted project's `.pi/pi-flow-external/settings.json` may define its own `piCapabilitySets`; a project entry with the same name fully replaces the global one (arrays are never merged). Absent `capabilitySet`, behavior is unchanged: nothing loads.
|
|
244
|
+
|
|
245
|
+
Selected resources are discovered through Pi's own SDK — never through project/user extensions or MCP, which stay unconditionally excluded — and filtered to exactly the named skills/prompt templates; project-scope resources are only visible when the project is trusted. An unknown set name, or a selected skill/prompt template that isn't discoverable, fails before any prompt is sent, naming exactly what's missing. Selected names are shown on the intent card before launch and recorded in the run receipt. Because a skill is a lazy file read (the child reads it itself), a workflow run freezes one content hash per selection up front for its replay fingerprint and re-checks it immediately before each child spawn, rejecting a child rather than running it against drifted instructions if the selected file changed mid-run.
|
|
246
|
+
|
|
247
|
+
A frontmatter `capabilitySet` that isn't a non-empty string (a boolean, a number, an empty string) does not silently drop or disable the profile. The profile stays in the roster with the malformed value recorded internally; selecting that exact profile — directly, via a canonical `<harness>-<role>` override, or via a shared `pi-*` role template materialized onto a harness — fails loudly at that point, naming what's wrong. This is deliberate: dropping the whole profile file instead would let an on-disk override or shared template silently vanish and fall back to the built-in canonical role body, hiding a real configuration mistake. Unrelated profiles and roles are never affected.
|
|
248
|
+
|
|
249
|
+
A selected **prompt template** is only reachable through the child's own explicit `/template-name args` task text (the SDK's built-in `/template-name` expansion) — it is never injected into the child's system prompt. A misspelled or unselected template name is therefore not an error; it is simply never expanded and passes through as literal task text.
|
|
250
|
+
|
|
251
|
+
A selected **skill**'s *autonomous* discoverability — whether its name/description is listed in the child's system prompt for the model to invoke on its own initiative — is governed entirely by that skill's own `disable-model-invocation` frontmatter flag, exactly as Pi's SDK already enforces it. `capabilitySet` filtering only narrows *which* skills are visible to the child at all; it never overrides that flag. A skill marked `disable-model-invocation: true` is still reachable via an explicit `/skill:name args` task invocation, since that lookup is by exact selected name, not by the flag.
|
|
198
252
|
|
|
199
253
|
## Agent usage
|
|
200
254
|
|
|
@@ -301,10 +355,10 @@ Agent({
|
|
|
301
355
|
Use `external_runs` with these actions:
|
|
302
356
|
|
|
303
357
|
- `list`: current session/project runs; use `cursor` for run pages and `workflowCursor` for workflow pages. `workflowRunId` filters children of one workflow. Rows carry a `timing` projection (`queueDelayMs`, `elapsedMs`, `activityAgeMs` when live, `processDurationMs`) plus `outputAvailable`/`finalAvailable`.
|
|
304
|
-
- `inspect`: single `
|
|
305
|
-
- `inspect` with `runIds`
|
|
306
|
-
- `wait`:
|
|
307
|
-
- `cancel`:
|
|
358
|
+
- `inspect`: single-entry `runIds` with `view: "summary" | "output" | "diagnostics" | "final"`, and optional opaque `cursor`/`limitBytes` (max 64 KiB, same cap for every view). Follow `nextCursor` to avoid truncation. `summary` includes the same `timing` projection as `list`, plus `output.finalAvailable`. `final` returns only the verified canonical terminal answer — empty with `finalAvailable: false` until a successful terminal boundary exists; it never promotes partial/narration text. `output` stays the combined stream (assistant messages plus canonical result) and is unchanged.
|
|
359
|
+
- `inspect` with `runIds` (summary view) (up to 20, deduplicated, order preserved): a single bounded batch of `summary`-only projections — one cheap request to see whether several selected background children are queued, running, or terminal, each with `outputRef`/`diagnosticsRef` for follow-up detail. Ownership of every requested ID is validated before any page is returned. Reuses the same `limitBytes` cap as single-run inspection; pages contain whole target entries and continue through `nextCursor`, without invalidation from ordinary live progress. If one compact entry cannot fit, an actionable error asks you to increase `limitBytes` or inspect that run individually; no target is silently dropped. The legacy `runId` selector remains accepted by programmatic callers but is no longer advertised to models. A singleton list supports other views; multiple targets require `summary`.
|
|
360
|
+
- `wait`: selected `runIds` (one or more), with `mode: "any" | "all"`. It returns terminal outcomes plus still-pending IDs; an unsuccessful workflow returns early even in `all` mode. It never chooses a winner or cancels pending work. While waiting, a bounded heartbeat (independent of any single target settling) reports live progress — watched targets, completed/pending counts, and recent activity — through the tool's update channel; it stops automatically on settlement, error, or interruption. Each settled outcome's `result` is spent from one shared byte budget (`limitBytes`, default 32768) across the whole response, in the requested `runId`/`runIds` order — never settlement race order, so the same targets and final states spend the budget identically regardless of which one happened to settle first: a result that fits is returned complete, one that does not is truncated with `resultTruncated: true` and the existing `outputRef`/`diagnosticsRef` to continue reading it — not a fixed-length teaser regardless of size. A target's evidence is never read from disk once the shared budget is already exhausted.
|
|
361
|
+
- `cancel`: single-entry `runIds` and optional reason. Whole-workflow cancellation stops active children; targeted child cancellation remains a catchable workflow outcome. Cancellation does not roll back edits or other side effects.
|
|
308
362
|
|
|
309
363
|
Interrupting a blocking `Agent`/`workflow` call cancels its work. Interrupting `external_runs wait` stops only that wait. Background work survives its launching tool return and ordinary parent turns, but not the owning session: orderly session shutdown requests cancellation and waits for bounded cleanup. This is not a daemon. After a host crash or unconfirmed shutdown, unfinished evidence is `interrupted_or_uncertain`; restart restores evidence access, never live ownership or guaranteed retrospective process termination. No routine activity wakes the parent, and live steering is not supported.
|
|
310
364
|
|
|
@@ -328,7 +382,7 @@ const two = Agent({ description: "Audit billing module", prompt: "Audit /absolut
|
|
|
328
382
|
|
|
329
383
|
// Leave them running and come back later in the conversation (or after
|
|
330
384
|
// further parent work) to collect final output once each is actually done:
|
|
331
|
-
// external_runs({ action: "inspect",
|
|
385
|
+
// external_runs({ action: "inspect", runIds: [one.runId], view: "final" })
|
|
332
386
|
```
|
|
333
387
|
|
|
334
388
|
A task that sounds like a 30-second command can legitimately take substantially longer end-to-end once queueing and backend overhead are included — inspect and wait, do not assume.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tranhoangnguyen0310/pi-flow-external",
|
|
3
|
-
"version": "2.
|
|
3
|
+
"version": "2.5.0-external.0",
|
|
4
4
|
"description": "External Claude Code, Codex CLI, Antigravity, Grok Build CLI, and Muse Code delegation for pi.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "./index.ts",
|
|
@@ -24,6 +24,7 @@ export function applySubagentProgressToWorkflowAgent(agent: WorkflowAgentSnapsho
|
|
|
24
24
|
agent.sessionId = progress.sessionId;
|
|
25
25
|
agent.resumedFrom = progress.resumedFrom;
|
|
26
26
|
agent.context = progress.context;
|
|
27
|
+
if (progress.capabilities !== undefined) agent.capabilities = progress.capabilities;
|
|
27
28
|
if (progress.thinkingClamped) agent.thinkingClamped = progress.thinkingClamped;
|
|
28
29
|
}
|
|
29
30
|
|
|
@@ -65,6 +66,7 @@ export function applySubagentResultToWorkflowAgent(agent: WorkflowAgentSnapshot,
|
|
|
65
66
|
if (resultDetails.permissionRequested !== undefined) agent.permissionRequested = resultDetails.permissionRequested;
|
|
66
67
|
if (resultDetails.retries !== undefined) agent.retries = resultDetails.retries;
|
|
67
68
|
if (resultDetails.retryOf !== undefined) agent.retryOf = resultDetails.retryOf;
|
|
69
|
+
if (resultDetails.capabilities !== undefined) agent.capabilities = resultDetails.capabilities;
|
|
68
70
|
const thinkingClamped = resultDetails.thinkingClamped ?? progress?.thinkingClamped;
|
|
69
71
|
if (thinkingClamped) agent.thinkingClamped = thinkingClamped;
|
|
70
72
|
if (progress) {
|
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
import { readFileSync } from "node:fs";
|
|
2
|
+
import { DefaultResourceLoader, SettingsManager, type PromptTemplate, type Skill } from "@earendil-works/pi-coding-agent";
|
|
3
|
+
import { hashStableValue } from "../workflow/replay-cache.ts";
|
|
4
|
+
import type { PiCapabilitySet } from "../settings.ts";
|
|
5
|
+
|
|
6
|
+
export interface CapabilitySelection {
|
|
7
|
+
set: string;
|
|
8
|
+
skills: string[];
|
|
9
|
+
promptTemplates: string[];
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
/** A resolved selection plus a content fingerprint, for workflow freeze/replay invalidation. */
|
|
13
|
+
export interface FrozenCapabilitySelection extends CapabilitySelection {
|
|
14
|
+
/** sha256 over the selected SKILL.md/template raw file bytes plus resource identity/trust (scope). */
|
|
15
|
+
contentHash: string;
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
export interface ResolveCapabilitySelectionParams {
|
|
19
|
+
cwd: string;
|
|
20
|
+
agentDir: string;
|
|
21
|
+
/** Real project-trust decision (ctx.isProjectTrusted()); gates project-scope skill/prompt directories. */
|
|
22
|
+
projectTrusted: boolean;
|
|
23
|
+
/** Name of the capabilitySet being resolved, for error messages. */
|
|
24
|
+
set: string;
|
|
25
|
+
capabilitySet: PiCapabilitySet;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/** Exact selected resource names, already validated once (freeze) or about to be re-validated (actual load). */
|
|
29
|
+
export interface CapabilityNames {
|
|
30
|
+
skills: readonly string[];
|
|
31
|
+
promptTemplates: readonly string[];
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
export interface LoadCapabilityResourcesParams {
|
|
35
|
+
cwd: string;
|
|
36
|
+
agentDir: string;
|
|
37
|
+
settingsManager: SettingsManager;
|
|
38
|
+
/** Name of the capabilitySet being loaded, for error messages. */
|
|
39
|
+
set: string;
|
|
40
|
+
names: CapabilityNames;
|
|
41
|
+
/** Threaded through to the child's resourceLoader when building it for an actual session (spawn.ts). */
|
|
42
|
+
appendSystemPromptOverride?: (base: string[]) => string[];
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
export interface LoadedCapabilityResources {
|
|
46
|
+
/** Already-reloaded loader, ready to hand to createAgentSession as-is. */
|
|
47
|
+
loader: DefaultResourceLoader;
|
|
48
|
+
skills: Skill[];
|
|
49
|
+
prompts: PromptTemplate[];
|
|
50
|
+
/**
|
|
51
|
+
* sha256 fingerprint over the actually-loaded selection's raw file bytes
|
|
52
|
+
* plus resource identity/trust (scope), computed identically to
|
|
53
|
+
* {@link resolveCapabilitySelection}'s frozen contentHash so a caller
|
|
54
|
+
* holding an earlier-frozen hash can detect drift between resolution and
|
|
55
|
+
* this load.
|
|
56
|
+
*/
|
|
57
|
+
contentHash: string;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/**
|
|
61
|
+
* Shared fingerprint formula for a loaded skill/prompt-template selection.
|
|
62
|
+
* Hashes raw on-disk bytes at filePath rather than the SDK's own .content
|
|
63
|
+
* field: skills carry no content field at all (Skill has no `content`
|
|
64
|
+
* property in this SDK layer), and PromptTemplate.content may already have
|
|
65
|
+
* frontmatter stripped, which would hide a frontmatter-only edit from the
|
|
66
|
+
* fingerprint.
|
|
67
|
+
*/
|
|
68
|
+
function hashCapabilitySelection(
|
|
69
|
+
projectTrusted: boolean,
|
|
70
|
+
skills: readonly Skill[],
|
|
71
|
+
prompts: readonly PromptTemplate[],
|
|
72
|
+
): string {
|
|
73
|
+
const skillEntries = [...skills]
|
|
74
|
+
.sort((a, b) => a.name.localeCompare(b.name))
|
|
75
|
+
.map((skill) => ({
|
|
76
|
+
name: skill.name,
|
|
77
|
+
filePath: skill.filePath,
|
|
78
|
+
scope: skill.sourceInfo.scope,
|
|
79
|
+
content: readFileSync(skill.filePath, "utf8"),
|
|
80
|
+
}));
|
|
81
|
+
const promptEntries = [...prompts]
|
|
82
|
+
.sort((a, b) => a.name.localeCompare(b.name))
|
|
83
|
+
.map((prompt) => ({
|
|
84
|
+
name: prompt.name,
|
|
85
|
+
filePath: prompt.filePath,
|
|
86
|
+
scope: prompt.sourceInfo.scope,
|
|
87
|
+
content: readFileSync(prompt.filePath, "utf8"),
|
|
88
|
+
}));
|
|
89
|
+
return hashStableValue({ projectTrusted, skills: skillEntries, prompts: promptEntries });
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/**
|
|
93
|
+
* Authoritative single construct+load+validate step for a capabilitySet's
|
|
94
|
+
* exact resolved skill/prompt-template names, reusing the installed SDK's own
|
|
95
|
+
* resource discovery (DefaultResourceLoader) instead of reimplementing
|
|
96
|
+
* directory scanning. Extensions/MCP/themes never load (noExtensions/noThemes
|
|
97
|
+
* stay true always); discovery for skills/prompts only turns on for the
|
|
98
|
+
* resource kind actually selected, and is immediately post-filtered to
|
|
99
|
+
* exactly those names via skillsOverride/promptsOverride — nothing else
|
|
100
|
+
* discovered is kept.
|
|
101
|
+
*
|
|
102
|
+
* This is the single source of truth for capabilitySet resource loading,
|
|
103
|
+
* used both by {@link resolveCapabilitySelection} (up-front freeze/dry-run
|
|
104
|
+
* validation) and by core/spawn.ts (the actual pi child resource loading at
|
|
105
|
+
* spawn time) — so a resource that disappeared between freeze and spawn
|
|
106
|
+
* fails loudly here instead of a second, unvalidated ad hoc loader silently
|
|
107
|
+
* loading fewer resources than declared.
|
|
108
|
+
*
|
|
109
|
+
* Fails before any prompt/session exists when a selected name is not
|
|
110
|
+
* discoverable, naming exactly what's missing (SDK name collisions are left
|
|
111
|
+
* to the SDK's own first-wins-by-name + collision diagnostic behavior; this
|
|
112
|
+
* loader does not second-guess that).
|
|
113
|
+
*/
|
|
114
|
+
export async function loadCapabilityResources(params: LoadCapabilityResourcesParams): Promise<LoadedCapabilityResources> {
|
|
115
|
+
const { cwd, agentDir, settingsManager, set, names } = params;
|
|
116
|
+
const selectedSkills = new Set(names.skills);
|
|
117
|
+
const selectedPrompts = new Set(names.promptTemplates);
|
|
118
|
+
|
|
119
|
+
let foundSkills: Skill[] = [];
|
|
120
|
+
let foundPrompts: PromptTemplate[] = [];
|
|
121
|
+
const loader = new DefaultResourceLoader({
|
|
122
|
+
cwd,
|
|
123
|
+
agentDir,
|
|
124
|
+
settingsManager,
|
|
125
|
+
noExtensions: true,
|
|
126
|
+
noThemes: true,
|
|
127
|
+
noSkills: selectedSkills.size === 0,
|
|
128
|
+
noPromptTemplates: selectedPrompts.size === 0,
|
|
129
|
+
skillsOverride: (base) => {
|
|
130
|
+
foundSkills = base.skills.filter((skill) => selectedSkills.has(skill.name));
|
|
131
|
+
return { skills: foundSkills, diagnostics: base.diagnostics };
|
|
132
|
+
},
|
|
133
|
+
promptsOverride: (base) => {
|
|
134
|
+
foundPrompts = base.prompts.filter((prompt) => selectedPrompts.has(prompt.name));
|
|
135
|
+
return { prompts: foundPrompts, diagnostics: base.diagnostics };
|
|
136
|
+
},
|
|
137
|
+
...(params.appendSystemPromptOverride ? { appendSystemPromptOverride: params.appendSystemPromptOverride } : {}),
|
|
138
|
+
});
|
|
139
|
+
await loader.reload();
|
|
140
|
+
|
|
141
|
+
const missingSkills = [...selectedSkills].filter((name) => !foundSkills.some((skill) => skill.name === name));
|
|
142
|
+
const missingPrompts = [...selectedPrompts].filter((name) => !foundPrompts.some((prompt) => prompt.name === name));
|
|
143
|
+
if (missingSkills.length || missingPrompts.length) {
|
|
144
|
+
const parts = [
|
|
145
|
+
...missingSkills.map((name) => `skill "${name}"`),
|
|
146
|
+
...missingPrompts.map((name) => `prompt template "${name}"`),
|
|
147
|
+
];
|
|
148
|
+
throw new Error(
|
|
149
|
+
`capabilitySet "${set}" is not fully resolvable: ${parts.join(", ")} ${parts.length === 1 ? "is" : "are"} not discoverable ` +
|
|
150
|
+
`(checked global skills/prompts directories; project directories ${settingsManager.isProjectTrusted() ? "included because the project is trusted" : "excluded because the project is not trusted"}). ` +
|
|
151
|
+
"Fix the set in settings.json, or remove/correct the profile's capabilitySet selection.",
|
|
152
|
+
);
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
const contentHash = hashCapabilitySelection(settingsManager.isProjectTrusted(), foundSkills, foundPrompts);
|
|
156
|
+
return { loader, skills: foundSkills, prompts: foundPrompts, contentHash };
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
/**
|
|
160
|
+
* Up-front freeze/dry-run validation: resolves and content-hashes a
|
|
161
|
+
* capabilitySet's exact selection without needing a live session, via the
|
|
162
|
+
* same authoritative {@link loadCapabilityResources} used at actual spawn
|
|
163
|
+
* time. Project trust is threaded explicitly: settingsManager.setProjectTrusted
|
|
164
|
+
* is called with the caller's real ctx.isProjectTrusted() decision before
|
|
165
|
+
* reload(), since SettingsManager otherwise defaults to trusted on its own.
|
|
166
|
+
*/
|
|
167
|
+
export async function resolveCapabilitySelection(params: ResolveCapabilitySelectionParams): Promise<FrozenCapabilitySelection> {
|
|
168
|
+
const { cwd, agentDir, projectTrusted, set, capabilitySet } = params;
|
|
169
|
+
|
|
170
|
+
const settingsManager = SettingsManager.create(cwd, agentDir);
|
|
171
|
+
settingsManager.setProjectTrusted(projectTrusted);
|
|
172
|
+
|
|
173
|
+
const { skills: foundSkills, prompts: foundPrompts, contentHash } = await loadCapabilityResources({
|
|
174
|
+
cwd,
|
|
175
|
+
agentDir,
|
|
176
|
+
settingsManager,
|
|
177
|
+
set,
|
|
178
|
+
names: capabilitySet,
|
|
179
|
+
});
|
|
180
|
+
|
|
181
|
+
return {
|
|
182
|
+
set,
|
|
183
|
+
skills: [...foundSkills].sort((a, b) => a.name.localeCompare(b.name)).map((skill) => skill.name),
|
|
184
|
+
promptTemplates: [...foundPrompts].sort((a, b) => a.name.localeCompare(b.name)).map((prompt) => prompt.name),
|
|
185
|
+
contentHash,
|
|
186
|
+
};
|
|
187
|
+
}
|
|
@@ -45,7 +45,8 @@ export function prepareParentContext(
|
|
|
45
45
|
): { prompt: string; context?: ParentContextReceipt } {
|
|
46
46
|
const context = parseParentContext(selection);
|
|
47
47
|
if (!context || context.mode === "none") return { prompt };
|
|
48
|
-
|
|
48
|
+
const resumeId = typeof resume === "string" && resume.trim() !== "" ? resume.trim() : undefined;
|
|
49
|
+
if (resumeId !== undefined) throw new Error("context sharing cannot be combined with resume; continue the child or start a new one");
|
|
49
50
|
if (!messages) throw new Error("Parent context is unavailable; use context:none and a self-contained prompt");
|
|
50
51
|
const compacted = messages.some((message) => message.role === "compactionSummary");
|
|
51
52
|
let start = 0;
|