pi-crew 0.9.59 → 0.9.60
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +41 -0
- package/dist/index.mjs +1252 -739
- package/docs/commands-reference.md +5 -0
- package/docs/resource-formats.md +7 -1
- package/package.json +1 -1
- package/skills/real-test-pi-crew/REPORT-TEMPLATE.md +52 -0
- package/skills/real-test-pi-crew/SKILL.md +6 -4
- package/src/config/types.ts +29 -0
- package/src/extension/registration/lifecycle-handlers.ts +33 -0
- package/src/extension/team-tool/chain-dispatch.ts +20 -2
- package/src/extension/team-tool/chain-executor.ts +6 -1
- package/src/extension/team-tool/doctor.ts +43 -0
- package/src/extension/team-tool/run.ts +19 -3
- package/src/extension/team-tool.ts +2 -1
- package/src/runtime/background-runner.ts +26 -0
- package/src/runtime/child-pi/child-pi-spawn.ts +1 -0
- package/src/runtime/child-pi/child-pi.ts +2 -0
- package/src/runtime/live-session/live-session-runtime.ts +83 -16
- package/src/runtime/model/model-fallback.ts +412 -36
- package/src/runtime/model/model-scope.ts +2 -0
- package/src/runtime/model/pi-args.ts +7 -3
- package/src/runtime/model/provider-quota.ts +228 -0
- package/src/runtime/model/session-model.ts +135 -0
- package/src/runtime/task-runner/child-executor.ts +82 -4
- package/src/runtime/task-runner/live-executor.ts +12 -0
- package/src/runtime/task-runner.ts +5 -0
- package/src/runtime/team-runner.ts +18 -4
- package/src/schema/config-schema.ts +12 -0
- package/src/state/types.ts +27 -0
- package/src/teams/discover-teams.ts +16 -2
- package/src/teams/team-config.ts +4 -0
|
@@ -190,6 +190,11 @@ Keeps the 20 most recent runs and deletes the rest.
|
|
|
190
190
|
| `runtime.groupJoinAckTimeoutMs` | number | `300000` | Group join ack timeout (ms) |
|
|
191
191
|
| `runtime.requirePlanApproval` | boolean | `false` | Require approving the plan before execution |
|
|
192
192
|
| `runtime.completionMutationGuard` | string | `"warn"` | `off`, `warn`, `fail` |
|
|
193
|
+
| `runtime.modelFallback.maxAutoFallbacks` | number | — | Cap auto-appended fallback models (default: unbounded) |
|
|
194
|
+
| `runtime.modelFallback.order` | string | `"parentFirst"` | `parentFirst` (same provider first) or `asIs` (catalogue order) |
|
|
195
|
+
| `runtime.modelFallback.requireCredentials` | boolean | `false` | Drop models whose provider has no discoverable credential |
|
|
196
|
+
| `runtime.modelFallback.quotaAwareOrdering` | boolean | `true` | Deprioritize providers near rate-limit/quota |
|
|
197
|
+
| `runtime.modelFallback.defaultSubagentModel` | string | — | Default model when neither caller nor agent specifies one |
|
|
193
198
|
| `limits.maxConcurrentWorkers` | number | `1024` | Max workers running in parallel |
|
|
194
199
|
| `limits.maxTaskDepth` | number | `100` | Max task tree depth |
|
|
195
200
|
| `limits.maxChildrenPerTask` | number | — | Max children per task |
|
package/docs/resource-formats.md
CHANGED
|
@@ -81,9 +81,15 @@ category: implementation
|
|
|
81
81
|
Role line:
|
|
82
82
|
|
|
83
83
|
```text
|
|
84
|
-
- {role-name}: agent={agent-name} [model={provider/model}] [skills={a,b}|false] [maxConcurrency={n}] optional description
|
|
84
|
+
- {role-name}: agent={agent-name} [model={provider/model}] [fallbackModels={a,b}] [thinking={level}] [skills={a,b}|false] [maxConcurrency={n}] optional description
|
|
85
85
|
```
|
|
86
86
|
|
|
87
|
+
- `model` — primary model for this role (e.g. `openai/gpt-5`, `anthropic/claude-sonnet-4-5`)
|
|
88
|
+
- `fallbackModels` — comma-separated fallback models tried in order when the primary fails (e.g. `fallbackModels=openai/gpt-5-mini,anthropic/claude-haiku-4-5`)
|
|
89
|
+
- `thinking` — thinking level override for this role (`high`, `medium`, `low`, `off`)
|
|
90
|
+
- `skills` — additional skills to inject, or `false` to disable role-default skills
|
|
91
|
+
- `maxConcurrency` — max parallel tasks for this role
|
|
92
|
+
|
|
87
93
|
## Workflow files
|
|
88
94
|
|
|
89
95
|
Location:
|
package/package.json
CHANGED
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
# real-test-pi-crew — Run Report
|
|
2
|
+
|
|
3
|
+
<!--
|
|
4
|
+
TEMPLATE — copy this file to docs/real-test/reports/real-test-<YYYY-MM-DD>-<slug>.md
|
|
5
|
+
and fill it in DURING the run (not from memory afterward). Every tier gets a
|
|
6
|
+
row with concrete evidence (counts, md5, runIds, wall-clock). If a tier was
|
|
7
|
+
skipped or incomplete, say so explicitly — do NOT claim "pass" without evidence.
|
|
8
|
+
This artifact exists so past runs are verifiable (see SKILL.md "Output report").
|
|
9
|
+
-->
|
|
10
|
+
|
|
11
|
+
**Date**: YYYY-MM-DD
|
|
12
|
+
**Trigger**: <what prompted this real-test — e.g. "post subagent-model-routing merge 4148540e", "pre-release v0.9.X">
|
|
13
|
+
**Repo HEAD**: `<sha>`
|
|
14
|
+
**Bundle md5 (disk)**: `<md5>`
|
|
15
|
+
**Pi version**: <from `pi --version` or doctor>
|
|
16
|
+
**Run by**: <agent/user>
|
|
17
|
+
|
|
18
|
+
## Tier results
|
|
19
|
+
|
|
20
|
+
| Tier | Status | Evidence |
|
|
21
|
+
|---|---|---|
|
|
22
|
+
| 1 test:critical | ✅/❌/⏭️ | `<pass>/<tests>` pass, `<dur>` |
|
|
23
|
+
| 2 3-path kill-switch | ✅/❌/⏭️ | default + `PI_CREW_BROKER=0` + `=1` all `<n>/<n>` |
|
|
24
|
+
| 3 typecheck + bundle | ✅/❌/⏭️ | tsc exit 0; bundle `<KB>` KB, md5 `<md5>` |
|
|
25
|
+
| 4 bundle md5 sync | ✅/❌/⏭️ | disk = loaded = `<md5>` (or: user must restart) |
|
|
26
|
+
| 5 tmux TUI probe | ✅/❌/⏭️ | `<which slash command reached the screen>` |
|
|
27
|
+
| 6 pty probe | ✅/❌/⏭️ | `<keys reached handleInput / diag lines>` |
|
|
28
|
+
| 7 smoke team run | ✅/❌/⏭️ | runId `<id>`, `<n>/<n>` tasks, verifier `<dur>` (<300s) |
|
|
29
|
+
| 8 final md5 sync | ✅/❌/⏭️ | disk = session = `<md5>` |
|
|
30
|
+
| 9a read-only battery | ✅/❌/⏭️ | list/recommend/health/doctor/status/events/summary/get/explain/worktrees — `<n>/10` |
|
|
31
|
+
| 9b spawn paths | ✅/❌/⏭️ | sync / async / chain / Agent / crew_agent — `<n>/5` |
|
|
32
|
+
| 9c lifecycle | ✅/❌/⏭️ | status-details/cache/checkpoint/steer/retry/resume + live cancel — which ran |
|
|
33
|
+
| 9d destructive | ✅/❌/⏭️ | forget/cleanup/prune — which ran (note data-protection skips) |
|
|
34
|
+
| 9e admin | ✅/❌/⏭️ | team/workflow CRUD round-trip |
|
|
35
|
+
| 9f background | ✅/❌/⏭️ | auto-summarize/anchor/schedule register+remove |
|
|
36
|
+
|
|
37
|
+
Legend: ✅ pass with evidence · ❌ fail (root cause below) · ⏭️ skipped (justify why)
|
|
38
|
+
|
|
39
|
+
## Findings (bugs / quirks / non-blocking notes)
|
|
40
|
+
- <finding 1 — file:line, severity, issue link if filed>
|
|
41
|
+
- <finding 2>
|
|
42
|
+
|
|
43
|
+
## What was NOT run + why
|
|
44
|
+
- <e.g. "prune — protects user run data; handler proven via forget">
|
|
45
|
+
- <e.g. "live mid-run steer race — async tool call blocks; live cancel implicitly verified via …">
|
|
46
|
+
|
|
47
|
+
## Restart needed?
|
|
48
|
+
- [ ] No — session already on the new bundle
|
|
49
|
+
- [ ] Yes — user must `/quit` + reopen; md5 before/after: `<old>` → `<new>`
|
|
50
|
+
|
|
51
|
+
## Verdict
|
|
52
|
+
<one line: e.g. "All required tiers pass; feature safe to ship. Issue #NN tracks <quirk>.">
|
|
@@ -17,7 +17,6 @@ triggers:
|
|
|
17
17
|
- "worker timeout"
|
|
18
18
|
- "verifier hangs"
|
|
19
19
|
- "rebuild and retry"
|
|
20
|
-
- "unknown type" tool error
|
|
21
20
|
- "validation failed for tool"
|
|
22
21
|
- "team tool broken"
|
|
23
22
|
- "schema fix"
|
|
@@ -448,7 +447,7 @@ If the two md5s match → session is on the latest code. If not → user must `/
|
|
|
448
447
|
2. **9b. Spawn paths** (cost tokens — one probe each is enough):
|
|
449
448
|
- `team action='run'` sync (fast-fix, trivial goal) — proves sync run + child-pi spawn + provider-extension loading
|
|
450
449
|
- `team action='run' async=true` — proves background dispatch
|
|
451
|
-
- `team action='run' chain='"A" -> "B"'` — proves sequential handoff (chain runner)
|
|
450
|
+
- `team action='run' chain='"A" -> "B"'` — proves sequential handoff (chain runner). **Omit `workflow`** — passing `workflow:'chain'` forwards it to each step and fails fast (~58ms silent; issue #44).
|
|
452
451
|
- `Agent` direct subagent — proves the direct-subagent tool
|
|
453
452
|
- `crew_agent` `run_in_background=true` then `get_subagent_result` — proves background subagent lifecycle
|
|
454
453
|
3. **Acceptance**: every action returns without `Unknown type` / `Validation failed for tool team` / empty error text; every spawn path completes with `consistency=1` and the expected probe token in the agent output.
|
|
@@ -511,6 +510,8 @@ If the two md5s match → session is on the latest code. If not → user must `/
|
|
|
511
510
|
| Trusting a team-run agent not to edit the repo under test | Agents spawned by `team`/`Agent`/`crew_agent` inherit the session cwd and have `edit`/`write` tools — a proactive LLM (observed with deepseek) will make **unauthorized source edits** to pi-crew during a trivial smoke run (e.g. "improving" `chain-runner.ts` while parsing a chain string). The edit can be correct + green-tested yet still be unintended scope creep that silently lands in your commit. | n/a (permanent) | After EVERY team/subagent run: `git status` and verify each changed file was authored by you. Diff + review any surprise change before staging. Consider `workspaceMode: 'worktree'` for parallel/risky runs to isolate mutations. |
|
|
512
511
|
| `Type.Unsafe({ anyOf/type })` schema field **without** `[TypeBox.Kind]` symbol | `Value.Check` throws `Unknown type` the first time a model emits that field (e.g. `skill`, `config`) — every team action returns `isError:true` text `"Unknown type"`. Tier 1-8 stay green because unit tests never send the offending field. | v0.9.57 | `src/schema/team-tool-schema.ts` — `SkillOverride`/`FreeformConfig` switched from `Type.Unsafe` to TypeBox-native `Type.Union`/`Type.Record`. See Tier 9. |
|
|
513
512
|
| Schema too strict for model-emitted empty strings (`runId:""`, `workspaceMode:""`, `budgetTotal:0`) | pi-ai `validateToolArguments` runs BEFORE the pi-crew handler and rejects `""` against Literal unions / patterns → `Validation failed for tool team` → model loops. | v0.9.57 | `src/schema/team-tool-schema.ts` — added `Literal("")` to unions, `^$|` pattern for runId, `""` to action enum, `0`/Boolean allowances. Handler-side `normalizeTeamParams` drops the empties. |
|
|
513
|
+
| Claiming "all 9 tiers pass" while 9c–9f were never run | Overclaim — once reported "9 tiers pass" when only 9a (8/10) + 9b (4/5) had actually run; 9c–9f were skipped. Past runs then become unverifiable ("did it really pass 9 tiers?"). | n/a (process) | Fill `REPORT-TEMPLATE.md` per-tier DURING the run. "Tier 9 pass" = 9a AND 9b AND the applicable 9c–9f, each with evidence. Round-up-to-pass is the anti-pattern this row exists to prevent. |
|
|
514
|
+
| chain run with `workflow:"chain"` forwarded to steps | Every chain step fails in ~58ms with an EMPTY error string — looks like a parse failure but isn't. `chain-dispatch` forwards `params.workflow` ("chain") into executor overrides; each step then runs the "chain" workflow via the normal `executeTeamRun` path and fails fast + silently. | Open (issue #44) | Omit `workflow` when invoking `action:'run' chain=...` — chain then runs 2/2 success (~308s). See `docs/bugs/chain-workflow-forward-quirk.md`. |
|
|
514
515
|
|
|
515
516
|
---
|
|
516
517
|
|
|
@@ -686,7 +687,7 @@ md5sum "$(npm root -g)"/pi-crew/dist/index.mjs 2>/dev/null \
|
|
|
686
687
|
|
|
687
688
|
Before claiming "tested":
|
|
688
689
|
|
|
689
|
-
- [ ] Tier 1: `test:critical` fresh-run, 97
|
|
690
|
+
- [ ] Tier 1: `test:critical` fresh-run, all pass (<25s). Count varies by release — was 97 at v0.9.46, 101 after the model-routing merge; record the actual count in the report.
|
|
690
691
|
- [ ] Tier 2: 3-path proof all pass — **required if you touched `src/config/defaults.ts` or `src/extension/registration/lifecycle-handlers.ts`**
|
|
691
692
|
- [ ] Tier 3: `npm run typecheck` exit 0, `npm run build:bundle` exit 0
|
|
692
693
|
- [ ] Tier 4: bundle md5 matches what the session loaded (or user has `/quit`-ed + reopened)
|
|
@@ -694,8 +695,9 @@ Before claiming "tested":
|
|
|
694
695
|
- [ ] Tier 7: smoke team run for any `src/runtime/plan-templates.ts` or `workflows/*.workflow.md` change — completed, no hang, verifier output under 60s
|
|
695
696
|
- [ ] Tier 8: final md5 sync check passed
|
|
696
697
|
- [ ] Tier 9: feature battery — **required if you touched `src/schema/team-tool-schema.ts`, `src/extension/registration/team-tool.ts`, or any `Type.Unsafe({...})` schema**. 9a read-only batch all return clean; one probe per 9b spawn path (sync / async / chain / `Agent` / `crew_agent`+`get_subagent_result`) completes with `consistency=1`. Run 9c–9f only when the change touches their code path; 9d (destructive) requires explicit user confirmation. **After every run: `git status` to catch unauthorized agent edits.**
|
|
698
|
+
- [ ] **Output report**: save `docs/real-test/reports/real-test-<YYYY-MM-DD>-<slug>.md` from `skills/real-test-pi-crew/REPORT-TEMPLATE.md`, filled DURING the run with per-tier evidence (counts/md5/runId) — not reconstructed from memory afterward. This is what makes past runs verifiable instead of trust-the-summary.
|
|
697
699
|
|
|
698
|
-
If any required item is unchecked, the answer to "is it tested?" is **no
|
|
700
|
+
**"All 9 tiers pass" is a claim that needs per-row evidence.** Tier 9 means 9a **and** 9b **and** whichever of 9c–9f applies to the change — not "9a passed, therefore 9 passed". If any required item above is unchecked or lacks concrete evidence (a number, an md5, a runId), the answer to "is it tested?" is **no** — say so explicitly instead of rounding up to "pass".
|
|
699
701
|
|
|
700
702
|
---
|
|
701
703
|
|
package/src/config/types.ts
CHANGED
|
@@ -85,6 +85,35 @@ export interface CrewRuntimeConfig {
|
|
|
85
85
|
};
|
|
86
86
|
/** Mark certain bash commands as excludeFromContext to reduce context tokens. Default: false */
|
|
87
87
|
excludeContextBash?: boolean;
|
|
88
|
+
/** Subagent model fallback policy: auto-tail ordering, cap, credential filtering, default model. */
|
|
89
|
+
modelFallback?: CrewModelFallbackConfig;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/**
|
|
93
|
+
* Model fallback policy for subagent model chains. Controls how the auto-tail
|
|
94
|
+
* (models appended from the registry/pi-config that nobody explicitly declared)
|
|
95
|
+
* is ordered, capped, and filtered. Explicit declarations (tool override, step
|
|
96
|
+
* model, team role, agent model, declared fallbackModels) are never affected.
|
|
97
|
+
*/
|
|
98
|
+
export interface CrewModelFallbackConfig {
|
|
99
|
+
/** Cap on auto-appended models. undefined = keep all (legacy). */
|
|
100
|
+
maxAutoFallbacks?: number;
|
|
101
|
+
/** "parentFirst" keeps the auto tail on the same provider as the running model when a policy is configured or quota data enriches it. "asIs" = catalogue order. Without explicit configuration, auto tail stays catalogue order. */
|
|
102
|
+
order?: "parentFirst" | "asIs";
|
|
103
|
+
/** Drop pi-config models whose provider has no discoverable credential. Default: false. */
|
|
104
|
+
requireCredentials?: boolean;
|
|
105
|
+
/**
|
|
106
|
+
* Opt-in quota-aware ordering. When true, provider quota data (when available)
|
|
107
|
+
* influences the auto-tail order — providers near their quota limit are deprioritized.
|
|
108
|
+
* Default: true (default-on with cache, per user preference).
|
|
109
|
+
*/
|
|
110
|
+
quotaAwareOrdering?: boolean;
|
|
111
|
+
/**
|
|
112
|
+
* Default model for subagents when neither the caller (--model) nor the agent
|
|
113
|
+
* frontmatter specifies one. Accepts "provider/id" or bare id. Overrides
|
|
114
|
+
* parent-model inheritance; the inherited parent model becomes the first fallback.
|
|
115
|
+
*/
|
|
116
|
+
defaultSubagentModel?: string;
|
|
88
117
|
}
|
|
89
118
|
|
|
90
119
|
export interface CrewControlConfig {
|
|
@@ -25,7 +25,10 @@ import { CrewBroker } from "../../runtime/broker/crew-broker.ts";
|
|
|
25
25
|
import { terminateActiveChildPiProcesses } from "../../runtime/child-pi/child-pi.ts";
|
|
26
26
|
import { listLiveAgents } from "../../runtime/live-session/live-agent-manager.ts";
|
|
27
27
|
import type { createManifestCache } from "../../runtime/manifest-cache.ts";
|
|
28
|
+
import { providerOfModelRef } from "../../runtime/model/model-fallback.ts";
|
|
28
29
|
import { cleanupLegacyOrphanTempDirs, cleanupOrphanTempDirs, currentCrewDepth } from "../../runtime/model/pi-args.ts";
|
|
30
|
+
import { clearProviderQuotaCache, noteProviderResponse } from "../../runtime/model/provider-quota.ts";
|
|
31
|
+
import { currentSessionModel, noteSessionModel, noteSessionThinking } from "../../runtime/model/session-model.ts";
|
|
29
32
|
import { cleanupOrphanWorkers } from "../../runtime/orphan-worker-registry.ts";
|
|
30
33
|
import { reconcileAllStaleRuns } from "../../runtime/recovery/crash-recovery.ts";
|
|
31
34
|
import { CrewScheduler, type ScheduledJob } from "../../runtime/scheduling/scheduler.ts";
|
|
@@ -68,6 +71,32 @@ export function installSessionLifecycleHandlers(pi: ExtensionAPI, ctx: Registrat
|
|
|
68
71
|
installSessionShutdownHandler(pi, ctx);
|
|
69
72
|
installSessionStartHandler(pi, ctx);
|
|
70
73
|
installSessionBeforeSwitchHandler(pi, ctx);
|
|
74
|
+
installModelTrackingHandlers(pi);
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* model_select / thinking_level_select:
|
|
79
|
+
* Track what the MAIN session is *actually* running so subagents that
|
|
80
|
+
* inherit the parent model (`model: false` — every builtin agent) follow it.
|
|
81
|
+
* `ctx.model` alone is the session's saved model and can point at whatever a
|
|
82
|
+
* previous session persisted, which made inherited models jump around.
|
|
83
|
+
*/
|
|
84
|
+
function installModelTrackingHandlers(pi: ExtensionAPI): void {
|
|
85
|
+
pi.on("model_select", (event) => {
|
|
86
|
+
noteSessionModel(event.model);
|
|
87
|
+
});
|
|
88
|
+
pi.on("thinking_level_select", (event) => {
|
|
89
|
+
noteSessionThinking(event.level);
|
|
90
|
+
});
|
|
91
|
+
// Quota-aware routing: capture rate-limit headers from the main session's
|
|
92
|
+
// provider responses so the fallback chain can deprioritize exhausted
|
|
93
|
+
// providers. The event doesn't carry a provider field, so we attribute it
|
|
94
|
+
// to the currently tracked session model's provider.
|
|
95
|
+
pi.on("after_provider_response", (event) => {
|
|
96
|
+
const model = currentSessionModel();
|
|
97
|
+
const provider = model ? providerOfModelRef(model) : undefined;
|
|
98
|
+
if (provider) noteProviderResponse(provider, event.status, event.headers);
|
|
99
|
+
});
|
|
71
100
|
}
|
|
72
101
|
|
|
73
102
|
/**
|
|
@@ -117,6 +146,7 @@ function installSessionBeforeSwitchHandler(pi: ExtensionAPI, ctx: RegistrationCo
|
|
|
117
146
|
ctx.lifecycleState.deliveryCoordinator?.deactivate();
|
|
118
147
|
resetPowerbarDedupState();
|
|
119
148
|
stopAsyncRunNotifier(ctx.notifierState);
|
|
149
|
+
clearProviderQuotaCache();
|
|
120
150
|
ctx.stopSessionBoundSubagents();
|
|
121
151
|
});
|
|
122
152
|
}
|
|
@@ -160,6 +190,9 @@ function installSessionStartHandler(pi: ExtensionAPI, ctx: RegistrationContext):
|
|
|
160
190
|
ctx.sessionGeneration++;
|
|
161
191
|
const ownerGeneration = ctx.sessionGeneration;
|
|
162
192
|
ctx.currentCtx = extensionCtx;
|
|
193
|
+
// Seed the live-model tracker; a later model_select overrides it.
|
|
194
|
+
noteSessionModel(extensionCtx.model, "session_start");
|
|
195
|
+
noteSessionThinking(extensionCtx.thinkingLevel);
|
|
163
196
|
// Round 13 UX: register the crew natural-language autocomplete provider
|
|
164
197
|
// once we have a UI context. Guarded so repeated session_start events
|
|
165
198
|
// don't stack wrappers (each wrapper delegates, but stacking wastes
|
|
@@ -36,6 +36,18 @@ export async function handleChainRun(params: TeamToolParamsValue, ctx: TeamConte
|
|
|
36
36
|
return result("Chain expression is empty.", { action: "run", status: "error" }, true);
|
|
37
37
|
}
|
|
38
38
|
|
|
39
|
+
// BUG-44 (github #44): reject an explicit `workflow: "chain"` on a chain run with a
|
|
40
|
+
// clear message. `chain` is a dispatcher-only workflow; forwarding it to steps made
|
|
41
|
+
// every step fail fast (~58 ms) with a confusing empty error. Chain runs have no
|
|
42
|
+
// per-step workflow — omit `workflow` (or rely on the step's @team ref) instead.
|
|
43
|
+
if (params.workflow === "chain") {
|
|
44
|
+
return result(
|
|
45
|
+
"Workflow 'chain' cannot be combined with a chain run: the chain runner owns step execution and has no per-step workflow. Omit `workflow` (steps use the team's defaultWorkflow) or use @team references in the chain expression.",
|
|
46
|
+
{ action: "run", status: "error" },
|
|
47
|
+
true,
|
|
48
|
+
);
|
|
49
|
+
}
|
|
50
|
+
|
|
39
51
|
const spec = parseChainString(chainString);
|
|
40
52
|
if (spec.steps.length === 0) {
|
|
41
53
|
return result(
|
|
@@ -47,13 +59,19 @@ export async function handleChainRun(params: TeamToolParamsValue, ctx: TeamConte
|
|
|
47
59
|
|
|
48
60
|
// Construct the concrete executor with per-step overrides forwarded from the
|
|
49
61
|
// chain invocation (overridden by any step that parsed to a @team reference).
|
|
62
|
+
//
|
|
63
|
+
// BUG-44 (github #44): `workflow` is INTENTIONALLY NOT forwarded. The chain
|
|
64
|
+
// runner owns step execution — a chain has no per-step workflow. Forwarding
|
|
65
|
+
// params.workflow (e.g. "chain") made every step execute the `chain` workflow
|
|
66
|
+
// via the normal executeTeamRun path, which fails fast (~58 ms) with a
|
|
67
|
+
// confusing empty error. Steps use the team's defaultWorkflow instead.
|
|
50
68
|
const executor = new ChainTeamRunExecutor({
|
|
51
69
|
handleRun,
|
|
52
70
|
ctx,
|
|
53
71
|
overrides: {
|
|
54
72
|
team: params.team,
|
|
55
|
-
workflow
|
|
56
|
-
model: params.model,
|
|
73
|
+
// workflow intentionally omitted — chain runner owns step execution (bug-44)
|
|
74
|
+
...(params.model ? { model: params.model } : {}),
|
|
57
75
|
},
|
|
58
76
|
});
|
|
59
77
|
|
|
@@ -238,8 +238,13 @@ export class ChainTeamRunExecutor implements ChainTaskRunner {
|
|
|
238
238
|
const enrichedGoal = historyPrefix ? `${historyPrefix}\n\n---\n# Current Chain Step\n${packet.goal}` : packet.goal;
|
|
239
239
|
|
|
240
240
|
// 2. Resolve team/workflow/model: step config (set by executeStep) → overrides → default.
|
|
241
|
+
// BUG-44 (github #44): a workflow override of "chain" is a dispatcher-only marker
|
|
242
|
+
// (the chain runner owns step execution). If it leaks through here (e.g. an older
|
|
243
|
+
// caller forwarded params.workflow), drop it so each step runs the team's
|
|
244
|
+
// defaultWorkflow instead of re-entering the un-runnable `chain` workflow.
|
|
241
245
|
const stepTeam = (context.__chainStepTeam as string | undefined) ?? this.overrides.team ?? "default";
|
|
242
|
-
const
|
|
246
|
+
const rawWorkflow = (context.__chainStepWorkflow as string | undefined) ?? this.overrides.workflow;
|
|
247
|
+
const stepWorkflow = rawWorkflow === "chain" ? undefined : rawWorkflow;
|
|
243
248
|
const stepModel = (context.__chainStepModel as string | undefined) ?? this.overrides.model;
|
|
244
249
|
|
|
245
250
|
// 3. Call handleRun for the heavy lifting. async:false forces each step to
|
|
@@ -5,7 +5,9 @@ import { allAgents, discoverAgents } from "../../agents/discover-agents.ts";
|
|
|
5
5
|
import { loadConfig } from "../../config/config.ts";
|
|
6
6
|
import { DEFAULT_PATHS } from "../../config/defaults.ts";
|
|
7
7
|
import { type DriftReport, detectDrift, formatDriftReport } from "../../config/drift-detector.ts";
|
|
8
|
+
import { buildConfiguredModelRouting, resolveModelFallbackPolicy } from "../../runtime/model/model-fallback.ts";
|
|
8
9
|
import { getRuntimeWarmupStatus } from "../../runtime/model/runtime-warmup.ts";
|
|
10
|
+
import { currentSessionModel, sessionModelSnapshot } from "../../runtime/model/session-model.ts";
|
|
9
11
|
import { getPiSpawnCommand } from "../../runtime/pi-spawn.ts";
|
|
10
12
|
import { formatZombieReport, scanZombieSubagents } from "../../runtime/process/zombie-scanner.ts";
|
|
11
13
|
import type { TeamToolParamsValue } from "../../schema/team-tool-schema.ts";
|
|
@@ -235,6 +237,47 @@ export function buildTeamDoctorReport(input: TeamDoctorReportInput): TeamDoctorR
|
|
|
235
237
|
},
|
|
236
238
|
];
|
|
237
239
|
}),
|
|
240
|
+
section("Model Routing", () => {
|
|
241
|
+
const snapshot = sessionModelSnapshot();
|
|
242
|
+
const liveModel = currentSessionModel();
|
|
243
|
+
const policy = resolveModelFallbackPolicy(loadConfig(input.cwd).config.runtime?.modelFallback);
|
|
244
|
+
// Build a sample chain for a generic agent (no explicit model) to show
|
|
245
|
+
// what the auto tail looks like with the current config.
|
|
246
|
+
const sampleRouting = buildConfiguredModelRouting({
|
|
247
|
+
parentModel: liveModel,
|
|
248
|
+
cwd: input.cwd,
|
|
249
|
+
policy,
|
|
250
|
+
});
|
|
251
|
+
return [
|
|
252
|
+
{
|
|
253
|
+
label: "session model (live)",
|
|
254
|
+
ok: true,
|
|
255
|
+
detail: liveModel ?? "not tracked yet",
|
|
256
|
+
},
|
|
257
|
+
{
|
|
258
|
+
label: "session model (source)",
|
|
259
|
+
ok: true,
|
|
260
|
+
detail: `${snapshot.source}${snapshot.updatedAt ? ` @ ${new Date(snapshot.updatedAt).toISOString()}` : ""}`,
|
|
261
|
+
},
|
|
262
|
+
{
|
|
263
|
+
label: "fallback policy",
|
|
264
|
+
ok: true,
|
|
265
|
+
detail: policy
|
|
266
|
+
? `maxAuto=${policy.maxAutoFallbacks ?? "∞"} order=${policy.order ?? "parentFirst"} creds=${policy.requireCredentials ?? false} quota=${policy.quotaAwareOrdering ?? true}`
|
|
267
|
+
: "legacy (unbounded, unordered)",
|
|
268
|
+
},
|
|
269
|
+
{
|
|
270
|
+
label: "sample chain (no explicit model)",
|
|
271
|
+
ok: true,
|
|
272
|
+
detail: sampleRouting.candidates.length > 0 ? sampleRouting.candidates.join(" → ") : "(empty)",
|
|
273
|
+
},
|
|
274
|
+
{
|
|
275
|
+
label: "auto tail size",
|
|
276
|
+
ok: true,
|
|
277
|
+
detail: `${sampleRouting.autoFallbackCount ?? 0} models`,
|
|
278
|
+
},
|
|
279
|
+
];
|
|
280
|
+
}),
|
|
238
281
|
section("Discovery", () => {
|
|
239
282
|
const agentModelHints = discoveredAgentsAll.filter((agent) => agent.model || agent.fallbackModels?.length).length;
|
|
240
283
|
return [
|
|
@@ -32,6 +32,7 @@ async function executeTeamRun(...args: Parameters<typeof ExecuteTeamRunFn>): Pro
|
|
|
32
32
|
|
|
33
33
|
import { spawnBackgroundTeamRun } from "../../runtime/async-runner.ts";
|
|
34
34
|
import { resolveCrewRuntime, runtimeResolutionState } from "../../runtime/model/runtime-resolver.ts";
|
|
35
|
+
import { captureRunModelContext, resolveParentModel } from "../../runtime/model/session-model.ts";
|
|
35
36
|
import { appendEventAsync, readEventsCursor } from "../../state/event-log/event-log.ts";
|
|
36
37
|
import type { RunMetrics } from "../../state/stores/run-metrics.ts";
|
|
37
38
|
import type { RuntimeResolutionState, TeamRunManifest, TeamTaskState } from "../../state/types.ts";
|
|
@@ -478,7 +479,18 @@ export async function handleRun(params: TeamToolParamsValue, ctx: TeamContext):
|
|
|
478
479
|
}
|
|
479
480
|
: teams.find((item) => item.name === teamName);
|
|
480
481
|
if (!team) return result(`Team '${teamName}' not found.`, { action: "run", status: "error" }, true);
|
|
481
|
-
|
|
482
|
+
// BUG-44 (github #44): `chain` is a dispatcher-only workflow — the chain runner owns
|
|
483
|
+
// step execution, so it is never runnable via the normal executeTeamRun path (its
|
|
484
|
+
// .workflow.md parses to doc headings, and validation fails fast with a confusing
|
|
485
|
+
// error). If a caller (or a chain step forwarding params.workflow) asks for
|
|
486
|
+
// workflow='chain' WITHOUT a chain param, fall back to the team's default workflow
|
|
487
|
+
// instead of failing. Chain steps are the only realistic producers of this state;
|
|
488
|
+
// handleChainRun itself routes via params.chain before reaching here.
|
|
489
|
+
const workflowName = directAgent
|
|
490
|
+
? "direct-agent"
|
|
491
|
+
: params.workflow === "chain" && !params.chain
|
|
492
|
+
? (team.defaultWorkflow ?? "default")
|
|
493
|
+
: (params.workflow ?? team.defaultWorkflow ?? "default");
|
|
482
494
|
const baseWorkflow = directAgent
|
|
483
495
|
? {
|
|
484
496
|
name: "direct-agent",
|
|
@@ -752,6 +764,10 @@ export async function handleRun(params: TeamToolParamsValue, ctx: TeamContext):
|
|
|
752
764
|
...updatedManifest,
|
|
753
765
|
runtimeResolution,
|
|
754
766
|
runConfig: executedConfig,
|
|
767
|
+
// Background/async runs re-enter through background-runner in a detached
|
|
768
|
+
// process with no ExtensionContext. Snapshot the model routing inputs so
|
|
769
|
+
// they survive the hand-off instead of being rediscovered from models.json.
|
|
770
|
+
modelContext: captureRunModelContext(ctx, params.model),
|
|
755
771
|
// Persist budget config on the manifest so it's observable post-run
|
|
756
772
|
// (events.jsonl, status reads, audits). The team-runner reads these
|
|
757
773
|
// from the input, but persisting them means consumers can verify
|
|
@@ -975,7 +991,7 @@ export async function handleRun(params: TeamToolParamsValue, ctx: TeamContext):
|
|
|
975
991
|
runtimeConfig: executedConfig.runtime,
|
|
976
992
|
parentContext: buildParentContext(ctx),
|
|
977
993
|
|
|
978
|
-
parentModel: ctx.model,
|
|
994
|
+
parentModel: resolveParentModel(ctx.model),
|
|
979
995
|
modelRegistry: ctx.modelRegistry,
|
|
980
996
|
modelOverride: params.model,
|
|
981
997
|
skillOverride,
|
|
@@ -1048,7 +1064,7 @@ export async function handleRun(params: TeamToolParamsValue, ctx: TeamContext):
|
|
|
1048
1064
|
runtime,
|
|
1049
1065
|
runtimeConfig: executedConfig.runtime,
|
|
1050
1066
|
parentContext: buildParentContext(ctx),
|
|
1051
|
-
parentModel: ctx.model,
|
|
1067
|
+
parentModel: resolveParentModel(ctx.model),
|
|
1052
1068
|
modelRegistry: ctx.modelRegistry,
|
|
1053
1069
|
modelOverride: params.model,
|
|
1054
1070
|
skillOverride,
|
|
@@ -33,6 +33,7 @@ async function executeTeamRun(...args: Parameters<ExecuteTeamRunFn>): Promise<Aw
|
|
|
33
33
|
|
|
34
34
|
import { directTeamAndWorkflowFromRun } from "../runtime/direct-run.ts";
|
|
35
35
|
import { resolveCrewRuntime, runtimeResolutionState } from "../runtime/model/runtime-resolver.ts";
|
|
36
|
+
import { resolveParentModel } from "../runtime/model/session-model.ts";
|
|
36
37
|
import { parsePiJsonOutput } from "../runtime/output/pi-json-output.ts";
|
|
37
38
|
import { effectiveRunConfig } from "./team-tool/config-patch.ts";
|
|
38
39
|
import { buildParentContext, formatScoped, result, type TeamContext } from "./team-tool/context.ts";
|
|
@@ -497,7 +498,7 @@ export async function handleResume(params: TeamToolParamsValue, ctx: TeamContext
|
|
|
497
498
|
runtime: decision.runtime,
|
|
498
499
|
runtimeConfig: decision.executedConfig.runtime,
|
|
499
500
|
parentContext: buildParentContext(ctx),
|
|
500
|
-
parentModel: ctx.model,
|
|
501
|
+
parentModel: resolveParentModel(ctx.model),
|
|
501
502
|
modelRegistry: ctx.modelRegistry,
|
|
502
503
|
modelOverride: params.model,
|
|
503
504
|
skillOverride: decision.resumeSkillOverride,
|
|
@@ -48,6 +48,7 @@ import { writeAsyncStartMarker } from "./async-marker.ts";
|
|
|
48
48
|
import { terminateActiveChildPiProcesses } from "./child-pi/child-pi.ts";
|
|
49
49
|
import { directTeamAndWorkflowFromRun } from "./direct-run.ts";
|
|
50
50
|
import { resolveCrewRuntime, runtimeResolutionState } from "./model/runtime-resolver.ts";
|
|
51
|
+
import { registryFromModelContext } from "./model/session-model.ts";
|
|
51
52
|
import { unregisterWorker } from "./orphan-worker-registry.ts";
|
|
52
53
|
import { startParentGuard, stopParentGuard } from "./parent-guard.ts";
|
|
53
54
|
import { expandParallelResearchWorkflow } from "./scheduling/parallel-research.ts";
|
|
@@ -61,6 +62,26 @@ function debugLog(message: string): void {
|
|
|
61
62
|
if (process.env.PI_CREW_DEBUG) console.log(message);
|
|
62
63
|
}
|
|
63
64
|
|
|
65
|
+
/**
|
|
66
|
+
* Re-hydrate the model routing inputs a detached background run cannot obtain
|
|
67
|
+
* from an ExtensionContext. Absent `modelContext` (older manifests) yields an
|
|
68
|
+
* empty object, preserving previous behaviour exactly.
|
|
69
|
+
*/
|
|
70
|
+
function restoredModelRouting(manifest: TeamRunManifest): {
|
|
71
|
+
modelOverride?: string;
|
|
72
|
+
parentModel?: string;
|
|
73
|
+
modelRegistry?: { getAvailable: () => unknown[] };
|
|
74
|
+
} {
|
|
75
|
+
const context = manifest.modelContext;
|
|
76
|
+
if (!context) return {};
|
|
77
|
+
const modelRegistry = registryFromModelContext(context);
|
|
78
|
+
return {
|
|
79
|
+
...(context.override ? { modelOverride: context.override } : {}),
|
|
80
|
+
...(context.parentModel ? { parentModel: context.parentModel } : {}),
|
|
81
|
+
...(modelRegistry ? { modelRegistry } : {}),
|
|
82
|
+
};
|
|
83
|
+
}
|
|
84
|
+
|
|
64
85
|
/**
|
|
65
86
|
* Heartbeat mechanism: periodically write a heartbeat file so the stale reconciler
|
|
66
87
|
* can distinguish "process died" from "process still alive but quiet".
|
|
@@ -825,6 +846,11 @@ async function main(): Promise<void> {
|
|
|
825
846
|
runtime,
|
|
826
847
|
runtimeConfig: runConfig.runtime,
|
|
827
848
|
skillOverride: manifest.skillOverride,
|
|
849
|
+
// Restore the caller's model routing inputs (see RunModelContext):
|
|
850
|
+
// this process has no ExtensionContext, so without these the
|
|
851
|
+
// `model=` override and the inherited session model are lost and
|
|
852
|
+
// every worker falls back to the first models.json entry.
|
|
853
|
+
...restoredModelRouting(manifest),
|
|
828
854
|
reliability: runConfig.reliability,
|
|
829
855
|
workspaceId: manifest.ownerSessionId ?? manifest.cwd,
|
|
830
856
|
signal: abortController.signal,
|
|
@@ -252,6 +252,7 @@ export function prepareSpawnContext(
|
|
|
252
252
|
maxDepth: input.maxDepth,
|
|
253
253
|
skillPaths: input.skillPaths,
|
|
254
254
|
role: input.role,
|
|
255
|
+
thinkingOverride: input.thinkingOverride,
|
|
255
256
|
});
|
|
256
257
|
// Pass steering file path to child for real-time steer injection
|
|
257
258
|
if (input.steeringFile) built.env.PI_CREW_STEERING_FILE = input.steeringFile;
|
|
@@ -143,6 +143,8 @@ export interface ChildPiRunInput {
|
|
|
143
143
|
agentId?: string;
|
|
144
144
|
/** Role for tool restrictions (from role-tools.ts) */
|
|
145
145
|
role?: string;
|
|
146
|
+
/** Team-role thinking override (takes precedence over agent.thinking). */
|
|
147
|
+
thinkingOverride?: string;
|
|
146
148
|
/** Root directory for artifacts (used to validate transcriptPath). */
|
|
147
149
|
artifactsRoot?: string;
|
|
148
150
|
/**
|