pi-crew 0.9.59 → 0.9.60

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -190,6 +190,11 @@ Keeps the 20 most recent runs and deletes the rest.
190
190
  | `runtime.groupJoinAckTimeoutMs` | number | `300000` | Group join ack timeout (ms) |
191
191
  | `runtime.requirePlanApproval` | boolean | `false` | Require approving the plan before execution |
192
192
  | `runtime.completionMutationGuard` | string | `"warn"` | `off`, `warn`, `fail` |
193
+ | `runtime.modelFallback.maxAutoFallbacks` | number | — | Cap auto-appended fallback models (default: unbounded) |
194
+ | `runtime.modelFallback.order` | string | `"parentFirst"` | `parentFirst` (same provider first) or `asIs` (catalogue order) |
195
+ | `runtime.modelFallback.requireCredentials` | boolean | `false` | Drop models whose provider has no discoverable credential |
196
+ | `runtime.modelFallback.quotaAwareOrdering` | boolean | `true` | Deprioritize providers near rate-limit/quota |
197
+ | `runtime.modelFallback.defaultSubagentModel` | string | — | Default model when neither caller nor agent specifies one |
193
198
  | `limits.maxConcurrentWorkers` | number | `1024` | Max workers running in parallel |
194
199
  | `limits.maxTaskDepth` | number | `100` | Max task tree depth |
195
200
  | `limits.maxChildrenPerTask` | number | — | Max children per task |
@@ -81,9 +81,15 @@ category: implementation
81
81
  Role line:
82
82
 
83
83
  ```text
84
- - {role-name}: agent={agent-name} [model={provider/model}] [skills={a,b}|false] [maxConcurrency={n}] optional description
84
+ - {role-name}: agent={agent-name} [model={provider/model}] [fallbackModels={a,b}] [thinking={level}] [skills={a,b}|false] [maxConcurrency={n}] optional description
85
85
  ```
86
86
 
87
+ - `model` — primary model for this role (e.g. `openai/gpt-5`, `anthropic/claude-sonnet-4-5`)
88
+ - `fallbackModels` — comma-separated fallback models tried in order when the primary fails (e.g. `fallbackModels=openai/gpt-5-mini,anthropic/claude-haiku-4-5`)
89
+ - `thinking` — thinking level override for this role (`high`, `medium`, `low`, `off`)
90
+ - `skills` — additional skills to inject, or `false` to disable role-default skills
91
+ - `maxConcurrency` — max parallel tasks for this role
92
+
87
93
  ## Workflow files
88
94
 
89
95
  Location:
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-crew",
3
- "version": "0.9.59",
3
+ "version": "0.9.60",
4
4
  "description": "Pi extension for coordinated AI teams, workflows, worktrees, and async task orchestration",
5
5
  "author": "baphuongna",
6
6
  "license": "MIT",
@@ -0,0 +1,52 @@
1
+ # real-test-pi-crew — Run Report
2
+
3
+ <!--
4
+ TEMPLATE — copy this file to docs/real-test/reports/real-test-<YYYY-MM-DD>-<slug>.md
5
+ and fill it in DURING the run (not from memory afterward). Every tier gets a
6
+ row with concrete evidence (counts, md5, runIds, wall-clock). If a tier was
7
+ skipped or incomplete, say so explicitly — do NOT claim "pass" without evidence.
8
+ This artifact exists so past runs are verifiable (see SKILL.md "Output report").
9
+ -->
10
+
11
+ **Date**: YYYY-MM-DD
12
+ **Trigger**: <what prompted this real-test — e.g. "post subagent-model-routing merge 4148540e", "pre-release v0.9.X">
13
+ **Repo HEAD**: `<sha>`
14
+ **Bundle md5 (disk)**: `<md5>`
15
+ **Pi version**: <from `pi --version` or doctor>
16
+ **Run by**: <agent/user>
17
+
18
+ ## Tier results
19
+
20
+ | Tier | Status | Evidence |
21
+ |---|---|---|
22
+ | 1 test:critical | ✅/❌/⏭️ | `<pass>/<tests>` pass, `<dur>` |
23
+ | 2 3-path kill-switch | ✅/❌/⏭️ | default + `PI_CREW_BROKER=0` + `=1` all `<n>/<n>` |
24
+ | 3 typecheck + bundle | ✅/❌/⏭️ | tsc exit 0; bundle `<KB>` KB, md5 `<md5>` |
25
+ | 4 bundle md5 sync | ✅/❌/⏭️ | disk = loaded = `<md5>` (or: user must restart) |
26
+ | 5 tmux TUI probe | ✅/❌/⏭️ | `<which slash command reached the screen>` |
27
+ | 6 pty probe | ✅/❌/⏭️ | `<keys reached handleInput / diag lines>` |
28
+ | 7 smoke team run | ✅/❌/⏭️ | runId `<id>`, `<n>/<n>` tasks, verifier `<dur>` (<300s) |
29
+ | 8 final md5 sync | ✅/❌/⏭️ | disk = session = `<md5>` |
30
+ | 9a read-only battery | ✅/❌/⏭️ | list/recommend/health/doctor/status/events/summary/get/explain/worktrees — `<n>/10` |
31
+ | 9b spawn paths | ✅/❌/⏭️ | sync / async / chain / Agent / crew_agent — `<n>/5` |
32
+ | 9c lifecycle | ✅/❌/⏭️ | status-details/cache/checkpoint/steer/retry/resume + live cancel — which ran |
33
+ | 9d destructive | ✅/❌/⏭️ | forget/cleanup/prune — which ran (note data-protection skips) |
34
+ | 9e admin | ✅/❌/⏭️ | team/workflow CRUD round-trip |
35
+ | 9f background | ✅/❌/⏭️ | auto-summarize/anchor/schedule register+remove |
36
+
37
+ Legend: ✅ pass with evidence · ❌ fail (root cause below) · ⏭️ skipped (justify why)
38
+
39
+ ## Findings (bugs / quirks / non-blocking notes)
40
+ - <finding 1 — file:line, severity, issue link if filed>
41
+ - <finding 2>
42
+
43
+ ## What was NOT run + why
44
+ - <e.g. "prune — protects user run data; handler proven via forget">
45
+ - <e.g. "live mid-run steer race — async tool call blocks; live cancel implicitly verified via …">
46
+
47
+ ## Restart needed?
48
+ - [ ] No — session already on the new bundle
49
+ - [ ] Yes — user must `/quit` + reopen; md5 before/after: `<old>` → `<new>`
50
+
51
+ ## Verdict
52
+ <one line: e.g. "All required tiers pass; feature safe to ship. Issue #NN tracks <quirk>.">
@@ -17,7 +17,6 @@ triggers:
17
17
  - "worker timeout"
18
18
  - "verifier hangs"
19
19
  - "rebuild and retry"
20
- - "unknown type" tool error
21
20
  - "validation failed for tool"
22
21
  - "team tool broken"
23
22
  - "schema fix"
@@ -448,7 +447,7 @@ If the two md5s match → session is on the latest code. If not → user must `/
448
447
  2. **9b. Spawn paths** (cost tokens — one probe each is enough):
449
448
  - `team action='run'` sync (fast-fix, trivial goal) — proves sync run + child-pi spawn + provider-extension loading
450
449
  - `team action='run' async=true` — proves background dispatch
451
- - `team action='run' chain='"A" -> "B"'` — proves sequential handoff (chain runner)
450
+ - `team action='run' chain='"A" -> "B"'` — proves sequential handoff (chain runner). **Omit `workflow`** — passing `workflow:'chain'` forwards it to each step and fails fast (~58ms silent; issue #44).
452
451
  - `Agent` direct subagent — proves the direct-subagent tool
453
452
  - `crew_agent` `run_in_background=true` then `get_subagent_result` — proves background subagent lifecycle
454
453
  3. **Acceptance**: every action returns without `Unknown type` / `Validation failed for tool team` / empty error text; every spawn path completes with `consistency=1` and the expected probe token in the agent output.
@@ -511,6 +510,8 @@ If the two md5s match → session is on the latest code. If not → user must `/
511
510
  | Trusting a team-run agent not to edit the repo under test | Agents spawned by `team`/`Agent`/`crew_agent` inherit the session cwd and have `edit`/`write` tools — a proactive LLM (observed with deepseek) will make **unauthorized source edits** to pi-crew during a trivial smoke run (e.g. "improving" `chain-runner.ts` while parsing a chain string). The edit can be correct + green-tested yet still be unintended scope creep that silently lands in your commit. | n/a (permanent) | After EVERY team/subagent run: `git status` and verify each changed file was authored by you. Diff + review any surprise change before staging. Consider `workspaceMode: 'worktree'` for parallel/risky runs to isolate mutations. |
512
511
  | `Type.Unsafe({ anyOf/type })` schema field **without** `[TypeBox.Kind]` symbol | `Value.Check` throws `Unknown type` the first time a model emits that field (e.g. `skill`, `config`) — every team action returns `isError:true` text `"Unknown type"`. Tier 1-8 stay green because unit tests never send the offending field. | v0.9.57 | `src/schema/team-tool-schema.ts` — `SkillOverride`/`FreeformConfig` switched from `Type.Unsafe` to TypeBox-native `Type.Union`/`Type.Record`. See Tier 9. |
513
512
  | Schema too strict for model-emitted empty strings (`runId:""`, `workspaceMode:""`, `budgetTotal:0`) | pi-ai `validateToolArguments` runs BEFORE the pi-crew handler and rejects `""` against Literal unions / patterns → `Validation failed for tool team` → model loops. | v0.9.57 | `src/schema/team-tool-schema.ts` — added `Literal("")` to unions, `^$|` pattern for runId, `""` to action enum, `0`/Boolean allowances. Handler-side `normalizeTeamParams` drops the empties. |
513
+ | Claiming "all 9 tiers pass" while 9c–9f were never run | Overclaim — once reported "9 tiers pass" when only 9a (8/10) + 9b (4/5) had actually run; 9c–9f were skipped. Past runs then become unverifiable ("did it really pass 9 tiers?"). | n/a (process) | Fill `REPORT-TEMPLATE.md` per-tier DURING the run. "Tier 9 pass" = 9a AND 9b AND the applicable 9c–9f, each with evidence. Round-up-to-pass is the anti-pattern this row exists to prevent. |
514
+ | chain run with `workflow:"chain"` forwarded to steps | Every chain step fails in ~58ms with an EMPTY error string — looks like a parse failure but isn't. `chain-dispatch` forwards `params.workflow` ("chain") into executor overrides; each step then runs the "chain" workflow via the normal `executeTeamRun` path and fails fast + silently. | Open (issue #44) | Omit `workflow` when invoking `action:'run' chain=...` — chain then runs 2/2 success (~308s). See `docs/bugs/chain-workflow-forward-quirk.md`. |
514
515
 
515
516
  ---
516
517
 
@@ -686,7 +687,7 @@ md5sum "$(npm root -g)"/pi-crew/dist/index.mjs 2>/dev/null \
686
687
 
687
688
  Before claiming "tested":
688
689
 
689
- - [ ] Tier 1: `test:critical` fresh-run, 97/97 pass, ~21s
690
+ - [ ] Tier 1: `test:critical` fresh-run, all pass (<25s). Count varies by release — was 97 at v0.9.46, 101 after the model-routing merge; record the actual count in the report.
690
691
  - [ ] Tier 2: 3-path proof all pass — **required if you touched `src/config/defaults.ts` or `src/extension/registration/lifecycle-handlers.ts`**
691
692
  - [ ] Tier 3: `npm run typecheck` exit 0, `npm run build:bundle` exit 0
692
693
  - [ ] Tier 4: bundle md5 matches what the session loaded (or user has `/quit`-ed + reopened)
@@ -694,8 +695,9 @@ Before claiming "tested":
694
695
  - [ ] Tier 7: smoke team run for any `src/runtime/plan-templates.ts` or `workflows/*.workflow.md` change — completed, no hang, verifier output under 60s
695
696
  - [ ] Tier 8: final md5 sync check passed
696
697
  - [ ] Tier 9: feature battery — **required if you touched `src/schema/team-tool-schema.ts`, `src/extension/registration/team-tool.ts`, or any `Type.Unsafe({...})` schema**. 9a read-only batch all return clean; one probe per 9b spawn path (sync / async / chain / `Agent` / `crew_agent`+`get_subagent_result`) completes with `consistency=1`. Run 9c–9f only when the change touches their code path; 9d (destructive) requires explicit user confirmation. **After every run: `git status` to catch unauthorized agent edits.**
698
+ - [ ] **Output report**: save `docs/real-test/reports/real-test-<YYYY-MM-DD>-<slug>.md` from `skills/real-test-pi-crew/REPORT-TEMPLATE.md`, filled DURING the run with per-tier evidence (counts/md5/runId) — not reconstructed from memory afterward. This is what makes past runs verifiable instead of trust-the-summary.
697
699
 
698
- If any required item is unchecked, the answer to "is it tested?" is **no**.
700
+ **"All 9 tiers pass" is a claim that needs per-row evidence.** Tier 9 means 9a **and** 9b **and** whichever of 9c–9f applies to the change — not "9a passed, therefore 9 passed". If any required item above is unchecked or lacks concrete evidence (a number, an md5, a runId), the answer to "is it tested?" is **no** — say so explicitly instead of rounding up to "pass".
699
701
 
700
702
  ---
701
703
 
@@ -85,6 +85,35 @@ export interface CrewRuntimeConfig {
85
85
  };
86
86
  /** Mark certain bash commands as excludeFromContext to reduce context tokens. Default: false */
87
87
  excludeContextBash?: boolean;
88
+ /** Subagent model fallback policy: auto-tail ordering, cap, credential filtering, default model. */
89
+ modelFallback?: CrewModelFallbackConfig;
90
+ }
91
+
92
+ /**
93
+ * Model fallback policy for subagent model chains. Controls how the auto-tail
94
+ * (models appended from the registry/pi-config that nobody explicitly declared)
95
+ * is ordered, capped, and filtered. Explicit declarations (tool override, step
96
+ * model, team role, agent model, declared fallbackModels) are never affected.
97
+ */
98
+ export interface CrewModelFallbackConfig {
99
+ /** Cap on auto-appended models. undefined = keep all (legacy). */
100
+ maxAutoFallbacks?: number;
101
+ /** "parentFirst" keeps the auto tail on the same provider as the running model when a policy is configured or quota data enriches it. "asIs" = catalogue order. Without explicit configuration, auto tail stays catalogue order. */
102
+ order?: "parentFirst" | "asIs";
103
+ /** Drop pi-config models whose provider has no discoverable credential. Default: false. */
104
+ requireCredentials?: boolean;
105
+ /**
106
+ * Opt-in quota-aware ordering. When true, provider quota data (when available)
107
+ * influences the auto-tail order — providers near their quota limit are deprioritized.
108
+ * Default: true (default-on with cache, per user preference).
109
+ */
110
+ quotaAwareOrdering?: boolean;
111
+ /**
112
+ * Default model for subagents when neither the caller (--model) nor the agent
113
+ * frontmatter specifies one. Accepts "provider/id" or bare id. Overrides
114
+ * parent-model inheritance; the inherited parent model becomes the first fallback.
115
+ */
116
+ defaultSubagentModel?: string;
88
117
  }
89
118
 
90
119
  export interface CrewControlConfig {
@@ -25,7 +25,10 @@ import { CrewBroker } from "../../runtime/broker/crew-broker.ts";
25
25
  import { terminateActiveChildPiProcesses } from "../../runtime/child-pi/child-pi.ts";
26
26
  import { listLiveAgents } from "../../runtime/live-session/live-agent-manager.ts";
27
27
  import type { createManifestCache } from "../../runtime/manifest-cache.ts";
28
+ import { providerOfModelRef } from "../../runtime/model/model-fallback.ts";
28
29
  import { cleanupLegacyOrphanTempDirs, cleanupOrphanTempDirs, currentCrewDepth } from "../../runtime/model/pi-args.ts";
30
+ import { clearProviderQuotaCache, noteProviderResponse } from "../../runtime/model/provider-quota.ts";
31
+ import { currentSessionModel, noteSessionModel, noteSessionThinking } from "../../runtime/model/session-model.ts";
29
32
  import { cleanupOrphanWorkers } from "../../runtime/orphan-worker-registry.ts";
30
33
  import { reconcileAllStaleRuns } from "../../runtime/recovery/crash-recovery.ts";
31
34
  import { CrewScheduler, type ScheduledJob } from "../../runtime/scheduling/scheduler.ts";
@@ -68,6 +71,32 @@ export function installSessionLifecycleHandlers(pi: ExtensionAPI, ctx: Registrat
68
71
  installSessionShutdownHandler(pi, ctx);
69
72
  installSessionStartHandler(pi, ctx);
70
73
  installSessionBeforeSwitchHandler(pi, ctx);
74
+ installModelTrackingHandlers(pi);
75
+ }
76
+
77
+ /**
78
+ * model_select / thinking_level_select:
79
+ * Track what the MAIN session is *actually* running so subagents that
80
+ * inherit the parent model (`model: false` — every builtin agent) follow it.
81
+ * `ctx.model` alone is the session's saved model and can point at whatever a
82
+ * previous session persisted, which made inherited models jump around.
83
+ */
84
+ function installModelTrackingHandlers(pi: ExtensionAPI): void {
85
+ pi.on("model_select", (event) => {
86
+ noteSessionModel(event.model);
87
+ });
88
+ pi.on("thinking_level_select", (event) => {
89
+ noteSessionThinking(event.level);
90
+ });
91
+ // Quota-aware routing: capture rate-limit headers from the main session's
92
+ // provider responses so the fallback chain can deprioritize exhausted
93
+ // providers. The event doesn't carry a provider field, so we attribute it
94
+ // to the currently tracked session model's provider.
95
+ pi.on("after_provider_response", (event) => {
96
+ const model = currentSessionModel();
97
+ const provider = model ? providerOfModelRef(model) : undefined;
98
+ if (provider) noteProviderResponse(provider, event.status, event.headers);
99
+ });
71
100
  }
72
101
 
73
102
  /**
@@ -117,6 +146,7 @@ function installSessionBeforeSwitchHandler(pi: ExtensionAPI, ctx: RegistrationCo
117
146
  ctx.lifecycleState.deliveryCoordinator?.deactivate();
118
147
  resetPowerbarDedupState();
119
148
  stopAsyncRunNotifier(ctx.notifierState);
149
+ clearProviderQuotaCache();
120
150
  ctx.stopSessionBoundSubagents();
121
151
  });
122
152
  }
@@ -160,6 +190,9 @@ function installSessionStartHandler(pi: ExtensionAPI, ctx: RegistrationContext):
160
190
  ctx.sessionGeneration++;
161
191
  const ownerGeneration = ctx.sessionGeneration;
162
192
  ctx.currentCtx = extensionCtx;
193
+ // Seed the live-model tracker; a later model_select overrides it.
194
+ noteSessionModel(extensionCtx.model, "session_start");
195
+ noteSessionThinking(extensionCtx.thinkingLevel);
163
196
  // Round 13 UX: register the crew natural-language autocomplete provider
164
197
  // once we have a UI context. Guarded so repeated session_start events
165
198
  // don't stack wrappers (each wrapper delegates, but stacking wastes
@@ -36,6 +36,18 @@ export async function handleChainRun(params: TeamToolParamsValue, ctx: TeamConte
36
36
  return result("Chain expression is empty.", { action: "run", status: "error" }, true);
37
37
  }
38
38
 
39
+ // BUG-44 (github #44): reject an explicit `workflow: "chain"` on a chain run with a
40
+ // clear message. `chain` is a dispatcher-only workflow; forwarding it to steps made
41
+ // every step fail fast (~58 ms) with a confusing empty error. Chain runs have no
42
+ // per-step workflow — omit `workflow` (or rely on the step's @team ref) instead.
43
+ if (params.workflow === "chain") {
44
+ return result(
45
+ "Workflow 'chain' cannot be combined with a chain run: the chain runner owns step execution and has no per-step workflow. Omit `workflow` (steps use the team's defaultWorkflow) or use @team references in the chain expression.",
46
+ { action: "run", status: "error" },
47
+ true,
48
+ );
49
+ }
50
+
39
51
  const spec = parseChainString(chainString);
40
52
  if (spec.steps.length === 0) {
41
53
  return result(
@@ -47,13 +59,19 @@ export async function handleChainRun(params: TeamToolParamsValue, ctx: TeamConte
47
59
 
48
60
  // Construct the concrete executor with per-step overrides forwarded from the
49
61
  // chain invocation (overridden by any step that parsed to a @team reference).
62
+ //
63
+ // BUG-44 (github #44): `workflow` is INTENTIONALLY NOT forwarded. The chain
64
+ // runner owns step execution — a chain has no per-step workflow. Forwarding
65
+ // params.workflow (e.g. "chain") made every step execute the `chain` workflow
66
+ // via the normal executeTeamRun path, which fails fast (~58 ms) with a
67
+ // confusing empty error. Steps use the team's defaultWorkflow instead.
50
68
  const executor = new ChainTeamRunExecutor({
51
69
  handleRun,
52
70
  ctx,
53
71
  overrides: {
54
72
  team: params.team,
55
- workflow: params.workflow,
56
- model: params.model,
73
+ // workflow intentionally omitted — chain runner owns step execution (bug-44)
74
+ ...(params.model ? { model: params.model } : {}),
57
75
  },
58
76
  });
59
77
 
@@ -238,8 +238,13 @@ export class ChainTeamRunExecutor implements ChainTaskRunner {
238
238
  const enrichedGoal = historyPrefix ? `${historyPrefix}\n\n---\n# Current Chain Step\n${packet.goal}` : packet.goal;
239
239
 
240
240
  // 2. Resolve team/workflow/model: step config (set by executeStep) → overrides → default.
241
+ // BUG-44 (github #44): a workflow override of "chain" is a dispatcher-only marker
242
+ // (the chain runner owns step execution). If it leaks through here (e.g. an older
243
+ // caller forwarded params.workflow), drop it so each step runs the team's
244
+ // defaultWorkflow instead of re-entering the un-runnable `chain` workflow.
241
245
  const stepTeam = (context.__chainStepTeam as string | undefined) ?? this.overrides.team ?? "default";
242
- const stepWorkflow = (context.__chainStepWorkflow as string | undefined) ?? this.overrides.workflow;
246
+ const rawWorkflow = (context.__chainStepWorkflow as string | undefined) ?? this.overrides.workflow;
247
+ const stepWorkflow = rawWorkflow === "chain" ? undefined : rawWorkflow;
243
248
  const stepModel = (context.__chainStepModel as string | undefined) ?? this.overrides.model;
244
249
 
245
250
  // 3. Call handleRun for the heavy lifting. async:false forces each step to
@@ -5,7 +5,9 @@ import { allAgents, discoverAgents } from "../../agents/discover-agents.ts";
5
5
  import { loadConfig } from "../../config/config.ts";
6
6
  import { DEFAULT_PATHS } from "../../config/defaults.ts";
7
7
  import { type DriftReport, detectDrift, formatDriftReport } from "../../config/drift-detector.ts";
8
+ import { buildConfiguredModelRouting, resolveModelFallbackPolicy } from "../../runtime/model/model-fallback.ts";
8
9
  import { getRuntimeWarmupStatus } from "../../runtime/model/runtime-warmup.ts";
10
+ import { currentSessionModel, sessionModelSnapshot } from "../../runtime/model/session-model.ts";
9
11
  import { getPiSpawnCommand } from "../../runtime/pi-spawn.ts";
10
12
  import { formatZombieReport, scanZombieSubagents } from "../../runtime/process/zombie-scanner.ts";
11
13
  import type { TeamToolParamsValue } from "../../schema/team-tool-schema.ts";
@@ -235,6 +237,47 @@ export function buildTeamDoctorReport(input: TeamDoctorReportInput): TeamDoctorR
235
237
  },
236
238
  ];
237
239
  }),
240
+ section("Model Routing", () => {
241
+ const snapshot = sessionModelSnapshot();
242
+ const liveModel = currentSessionModel();
243
+ const policy = resolveModelFallbackPolicy(loadConfig(input.cwd).config.runtime?.modelFallback);
244
+ // Build a sample chain for a generic agent (no explicit model) to show
245
+ // what the auto tail looks like with the current config.
246
+ const sampleRouting = buildConfiguredModelRouting({
247
+ parentModel: liveModel,
248
+ cwd: input.cwd,
249
+ policy,
250
+ });
251
+ return [
252
+ {
253
+ label: "session model (live)",
254
+ ok: true,
255
+ detail: liveModel ?? "not tracked yet",
256
+ },
257
+ {
258
+ label: "session model (source)",
259
+ ok: true,
260
+ detail: `${snapshot.source}${snapshot.updatedAt ? ` @ ${new Date(snapshot.updatedAt).toISOString()}` : ""}`,
261
+ },
262
+ {
263
+ label: "fallback policy",
264
+ ok: true,
265
+ detail: policy
266
+ ? `maxAuto=${policy.maxAutoFallbacks ?? "∞"} order=${policy.order ?? "parentFirst"} creds=${policy.requireCredentials ?? false} quota=${policy.quotaAwareOrdering ?? true}`
267
+ : "legacy (unbounded, unordered)",
268
+ },
269
+ {
270
+ label: "sample chain (no explicit model)",
271
+ ok: true,
272
+ detail: sampleRouting.candidates.length > 0 ? sampleRouting.candidates.join(" → ") : "(empty)",
273
+ },
274
+ {
275
+ label: "auto tail size",
276
+ ok: true,
277
+ detail: `${sampleRouting.autoFallbackCount ?? 0} models`,
278
+ },
279
+ ];
280
+ }),
238
281
  section("Discovery", () => {
239
282
  const agentModelHints = discoveredAgentsAll.filter((agent) => agent.model || agent.fallbackModels?.length).length;
240
283
  return [
@@ -32,6 +32,7 @@ async function executeTeamRun(...args: Parameters<typeof ExecuteTeamRunFn>): Pro
32
32
 
33
33
  import { spawnBackgroundTeamRun } from "../../runtime/async-runner.ts";
34
34
  import { resolveCrewRuntime, runtimeResolutionState } from "../../runtime/model/runtime-resolver.ts";
35
+ import { captureRunModelContext, resolveParentModel } from "../../runtime/model/session-model.ts";
35
36
  import { appendEventAsync, readEventsCursor } from "../../state/event-log/event-log.ts";
36
37
  import type { RunMetrics } from "../../state/stores/run-metrics.ts";
37
38
  import type { RuntimeResolutionState, TeamRunManifest, TeamTaskState } from "../../state/types.ts";
@@ -478,7 +479,18 @@ export async function handleRun(params: TeamToolParamsValue, ctx: TeamContext):
478
479
  }
479
480
  : teams.find((item) => item.name === teamName);
480
481
  if (!team) return result(`Team '${teamName}' not found.`, { action: "run", status: "error" }, true);
481
- const workflowName = directAgent ? "direct-agent" : (params.workflow ?? team.defaultWorkflow ?? "default");
482
+ // BUG-44 (github #44): `chain` is a dispatcher-only workflow — the chain runner owns
483
+ // step execution, so it is never runnable via the normal executeTeamRun path (its
484
+ // .workflow.md parses to doc headings, and validation fails fast with a confusing
485
+ // error). If a caller (or a chain step forwarding params.workflow) asks for
486
+ // workflow='chain' WITHOUT a chain param, fall back to the team's default workflow
487
+ // instead of failing. Chain steps are the only realistic producers of this state;
488
+ // handleChainRun itself routes via params.chain before reaching here.
489
+ const workflowName = directAgent
490
+ ? "direct-agent"
491
+ : params.workflow === "chain" && !params.chain
492
+ ? (team.defaultWorkflow ?? "default")
493
+ : (params.workflow ?? team.defaultWorkflow ?? "default");
482
494
  const baseWorkflow = directAgent
483
495
  ? {
484
496
  name: "direct-agent",
@@ -752,6 +764,10 @@ export async function handleRun(params: TeamToolParamsValue, ctx: TeamContext):
752
764
  ...updatedManifest,
753
765
  runtimeResolution,
754
766
  runConfig: executedConfig,
767
+ // Background/async runs re-enter through background-runner in a detached
768
+ // process with no ExtensionContext. Snapshot the model routing inputs so
769
+ // they survive the hand-off instead of being rediscovered from models.json.
770
+ modelContext: captureRunModelContext(ctx, params.model),
755
771
  // Persist budget config on the manifest so it's observable post-run
756
772
  // (events.jsonl, status reads, audits). The team-runner reads these
757
773
  // from the input, but persisting them means consumers can verify
@@ -975,7 +991,7 @@ export async function handleRun(params: TeamToolParamsValue, ctx: TeamContext):
975
991
  runtimeConfig: executedConfig.runtime,
976
992
  parentContext: buildParentContext(ctx),
977
993
 
978
- parentModel: ctx.model,
994
+ parentModel: resolveParentModel(ctx.model),
979
995
  modelRegistry: ctx.modelRegistry,
980
996
  modelOverride: params.model,
981
997
  skillOverride,
@@ -1048,7 +1064,7 @@ export async function handleRun(params: TeamToolParamsValue, ctx: TeamContext):
1048
1064
  runtime,
1049
1065
  runtimeConfig: executedConfig.runtime,
1050
1066
  parentContext: buildParentContext(ctx),
1051
- parentModel: ctx.model,
1067
+ parentModel: resolveParentModel(ctx.model),
1052
1068
  modelRegistry: ctx.modelRegistry,
1053
1069
  modelOverride: params.model,
1054
1070
  skillOverride,
@@ -33,6 +33,7 @@ async function executeTeamRun(...args: Parameters<ExecuteTeamRunFn>): Promise<Aw
33
33
 
34
34
  import { directTeamAndWorkflowFromRun } from "../runtime/direct-run.ts";
35
35
  import { resolveCrewRuntime, runtimeResolutionState } from "../runtime/model/runtime-resolver.ts";
36
+ import { resolveParentModel } from "../runtime/model/session-model.ts";
36
37
  import { parsePiJsonOutput } from "../runtime/output/pi-json-output.ts";
37
38
  import { effectiveRunConfig } from "./team-tool/config-patch.ts";
38
39
  import { buildParentContext, formatScoped, result, type TeamContext } from "./team-tool/context.ts";
@@ -497,7 +498,7 @@ export async function handleResume(params: TeamToolParamsValue, ctx: TeamContext
497
498
  runtime: decision.runtime,
498
499
  runtimeConfig: decision.executedConfig.runtime,
499
500
  parentContext: buildParentContext(ctx),
500
- parentModel: ctx.model,
501
+ parentModel: resolveParentModel(ctx.model),
501
502
  modelRegistry: ctx.modelRegistry,
502
503
  modelOverride: params.model,
503
504
  skillOverride: decision.resumeSkillOverride,
@@ -48,6 +48,7 @@ import { writeAsyncStartMarker } from "./async-marker.ts";
48
48
  import { terminateActiveChildPiProcesses } from "./child-pi/child-pi.ts";
49
49
  import { directTeamAndWorkflowFromRun } from "./direct-run.ts";
50
50
  import { resolveCrewRuntime, runtimeResolutionState } from "./model/runtime-resolver.ts";
51
+ import { registryFromModelContext } from "./model/session-model.ts";
51
52
  import { unregisterWorker } from "./orphan-worker-registry.ts";
52
53
  import { startParentGuard, stopParentGuard } from "./parent-guard.ts";
53
54
  import { expandParallelResearchWorkflow } from "./scheduling/parallel-research.ts";
@@ -61,6 +62,26 @@ function debugLog(message: string): void {
61
62
  if (process.env.PI_CREW_DEBUG) console.log(message);
62
63
  }
63
64
 
65
+ /**
66
+ * Re-hydrate the model routing inputs a detached background run cannot obtain
67
+ * from an ExtensionContext. Absent `modelContext` (older manifests) yields an
68
+ * empty object, preserving previous behaviour exactly.
69
+ */
70
+ function restoredModelRouting(manifest: TeamRunManifest): {
71
+ modelOverride?: string;
72
+ parentModel?: string;
73
+ modelRegistry?: { getAvailable: () => unknown[] };
74
+ } {
75
+ const context = manifest.modelContext;
76
+ if (!context) return {};
77
+ const modelRegistry = registryFromModelContext(context);
78
+ return {
79
+ ...(context.override ? { modelOverride: context.override } : {}),
80
+ ...(context.parentModel ? { parentModel: context.parentModel } : {}),
81
+ ...(modelRegistry ? { modelRegistry } : {}),
82
+ };
83
+ }
84
+
64
85
  /**
65
86
  * Heartbeat mechanism: periodically write a heartbeat file so the stale reconciler
66
87
  * can distinguish "process died" from "process still alive but quiet".
@@ -825,6 +846,11 @@ async function main(): Promise<void> {
825
846
  runtime,
826
847
  runtimeConfig: runConfig.runtime,
827
848
  skillOverride: manifest.skillOverride,
849
+ // Restore the caller's model routing inputs (see RunModelContext):
850
+ // this process has no ExtensionContext, so without these the
851
+ // `model=` override and the inherited session model are lost and
852
+ // every worker falls back to the first models.json entry.
853
+ ...restoredModelRouting(manifest),
828
854
  reliability: runConfig.reliability,
829
855
  workspaceId: manifest.ownerSessionId ?? manifest.cwd,
830
856
  signal: abortController.signal,
@@ -252,6 +252,7 @@ export function prepareSpawnContext(
252
252
  maxDepth: input.maxDepth,
253
253
  skillPaths: input.skillPaths,
254
254
  role: input.role,
255
+ thinkingOverride: input.thinkingOverride,
255
256
  });
256
257
  // Pass steering file path to child for real-time steer injection
257
258
  if (input.steeringFile) built.env.PI_CREW_STEERING_FILE = input.steeringFile;
@@ -143,6 +143,8 @@ export interface ChildPiRunInput {
143
143
  agentId?: string;
144
144
  /** Role for tool restrictions (from role-tools.ts) */
145
145
  role?: string;
146
+ /** Team-role thinking override (takes precedence over agent.thinking). */
147
+ thinkingOverride?: string;
146
148
  /** Root directory for artifacts (used to validate transcriptPath). */
147
149
  artifactsRoot?: string;
148
150
  /**