pi-crew 0.9.59 → 0.9.60

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,228 @@
1
+ /**
2
+ * provider-quota.ts — track provider rate-limit / quota state from response
3
+ * headers so the model fallback chain can deprioritize providers that are
4
+ * near exhaustion.
5
+ *
6
+ * WHY: when a provider returns 429 or its `x-ratelimit-remaining` header
7
+ * drops to zero, every subsequent spawn to that provider wastes a child
8
+ * process + a full retry cycle before the fallback chain moves on. Recording
9
+ * the signal here lets `buildConfiguredModelRouting` push those providers to
10
+ * the back of the auto tail instead of trying them first.
11
+ *
12
+ * Data source: pi's `after_provider_response` event carries `status` and
13
+ * `headers`. Standard rate-limit headers (OpenAI, Anthropic, most proxies):
14
+ * x-ratelimit-remaining-requests / x-ratelimit-remaining-tokens
15
+ * x-ratelimit-reset-requests / x-ratelimit-reset-tokens
16
+ * retry-after (on 429)
17
+ *
18
+ * The tracker is process-local and best-effort: a missed event just means
19
+ * the cache is slightly stale, never a wrong routing decision.
20
+ */
21
+
22
+ interface ProviderQuotaEntry {
23
+ /** Provider key (lowercase). */
24
+ provider: string;
25
+ /** Remaining requests from the most recent response header. */
26
+ remainingRequests?: number;
27
+ /** Remaining tokens from the most recent response header. */
28
+ remainingTokens?: number;
29
+ /** When the rate-limit window resets (epoch ms). */
30
+ resetAtMs?: number;
31
+ /** HTTP status of the last response (429 = rate-limited). */
32
+ lastStatus?: number;
33
+ /** When this entry was last updated (epoch ms). */
34
+ updatedAtMs: number;
35
+ }
36
+
37
+ /** How long a quota entry stays authoritative before we stop trusting it. */
38
+ const QUOTA_TTL_MS = 5 * 60 * 1000; // 5 minutes
39
+
40
+ /** Process-local cache keyed by lowercase provider name. */
41
+ const quotaCache = new Map<string, ProviderQuotaEntry>();
42
+
43
+ /**
44
+ * Parse a numeric header value, returning undefined for missing/invalid.
45
+ */
46
+ function headerNumber(headers: Record<string, string>, ...names: string[]): number | undefined {
47
+ for (const name of names) {
48
+ const raw = headers[name] ?? headers[name.toLowerCase()];
49
+ if (raw === undefined) continue;
50
+ const parsed = Number.parseInt(raw, 10);
51
+ if (Number.isFinite(parsed) && parsed >= 0) return parsed;
52
+ }
53
+ return undefined;
54
+ }
55
+
56
+ /**
57
+ * Read a raw (unparsed) header value — first match wins, case-insensitive.
58
+ */
59
+ function headerRaw(headers: Record<string, string>, ...names: string[]): string | undefined {
60
+ for (const name of names) {
61
+ const raw = headers[name] ?? headers[name.toLowerCase()];
62
+ if (raw !== undefined) return raw;
63
+ }
64
+ return undefined;
65
+ }
66
+
67
+ /** Regex: compact Go-duration like "6m0s", "1s", "1h30m", "120s". */
68
+ const GO_DURATION_RE = /^(\d+[smh])+$/;
69
+ /** Regex: extract individual (\d+)(unit) segments from a Go-duration string. */
70
+ const GO_DURATION_SEG_RE = /(\d+)([smh])/g;
71
+
72
+ /**
73
+ * Parse a reset-timestamp header value into epoch-ms.
74
+ *
75
+ * Three formats are supported (tried in order):
76
+ * (c) RFC3339 absolute timestamp (Anthropic): Date.parse returns epoch-ms directly.
77
+ * (a) Go-duration compact string (OpenAI): "6m0s", "1s", "1h30m", "120s".
78
+ * (b) Pure-seconds integer (retry-after): "30" → nowMs + 30 000.
79
+ *
80
+ * Returns undefined if no format matches.
81
+ */
82
+ function parseResetValue(value: string, nowMs: number): number | undefined {
83
+ const trimmed = value.trim();
84
+ // (b) Pure-seconds integer (retry-after / reset): "0", "30", "120".
85
+ // MUST precede RFC3339 — Date.parse("0") returns 946659600000 (Y2K) in V8,
86
+ // not NaN, so a numeric "0" would otherwise hijack the timestamp branch
87
+ // and yield a resetAtMs in the year 2000.
88
+ if (/^\d+$/.test(trimmed)) {
89
+ const secs = Number.parseInt(trimmed, 10);
90
+ if (Number.isFinite(secs) && secs >= 0) return nowMs + secs * 1000;
91
+ }
92
+ // (a) Go-duration compact: digits+unit pairs ("6m0s", "1s", "1h30m").
93
+ if (GO_DURATION_RE.test(trimmed)) {
94
+ let durationMs = 0;
95
+ for (const seg of trimmed.matchAll(GO_DURATION_SEG_RE)) {
96
+ const n = Number.parseInt(seg[1], 10);
97
+ const unit = seg[2];
98
+ durationMs += unit === "s" ? n * 1000 : unit === "m" ? n * 60_000 : n * 3_600_000;
99
+ }
100
+ return nowMs + durationMs;
101
+ }
102
+ // (c) RFC3339 absolute timestamp (Anthropic). Only non-numeric date-like
103
+ // strings reach here: pure-numeric is handled above, and Date.parse on
104
+ // Go-durations like "6m0s" returns NaN.
105
+ const parsed = Date.parse(trimmed);
106
+ if (Number.isFinite(parsed) && parsed > 0) return parsed;
107
+ return undefined;
108
+ }
109
+
110
+ /**
111
+ * Parse a reset timestamp from headers. Providers use different formats:
112
+ * - OpenAI: Go-duration strings ("6m0s", "1s") in x-ratelimit-reset-requests.
113
+ * - Anthropic: RFC3339 timestamps.
114
+ * - retry-after: pure-seconds integer.
115
+ */
116
+ function headerResetMs(headers: Record<string, string>, nowMs: number): number | undefined {
117
+ const resetRaw = headerRaw(headers, "x-ratelimit-reset-requests", "x-ratelimit-reset");
118
+ if (resetRaw !== undefined) {
119
+ const parsed = parseResetValue(resetRaw, nowMs);
120
+ if (parsed !== undefined) return parsed;
121
+ }
122
+ const retryRaw = headerRaw(headers, "retry-after");
123
+ if (retryRaw !== undefined) {
124
+ return parseResetValue(retryRaw, nowMs);
125
+ }
126
+ return undefined;
127
+ }
128
+
129
+ /**
130
+ * Record a provider response. Called from the `after_provider_response`
131
+ * event handler. The `provider` argument is the pi provider key (e.g.
132
+ * "anthropic", "openai-codex"); it is lowercased for consistent lookup.
133
+ */
134
+ export function noteProviderResponse(provider: string, status: number, headers: Record<string, string>, nowMs: number = Date.now()): void {
135
+ const key = provider.toLowerCase();
136
+ const entry: ProviderQuotaEntry = {
137
+ provider: key,
138
+ remainingRequests: headerNumber(headers, "x-ratelimit-remaining-requests", "x-ratelimit-remaining"),
139
+ remainingTokens: headerNumber(headers, "x-ratelimit-remaining-tokens"),
140
+ resetAtMs: headerResetMs(headers, nowMs),
141
+ lastStatus: status,
142
+ updatedAtMs: nowMs,
143
+ };
144
+ quotaCache.set(key, entry);
145
+ // L1: evict entries older than 2 * QUOTA_TTL_MS (10 min) to bound cache growth.
146
+ const evictionCutoff = nowMs - 2 * QUOTA_TTL_MS;
147
+ for (const [cacheKey, cached] of quotaCache) {
148
+ if (cached.updatedAtMs < evictionCutoff) quotaCache.delete(cacheKey);
149
+ }
150
+ }
151
+
152
+ /**
153
+ * Whether a provider is currently deprioritized (near or at quota limit).
154
+ * A provider is deprioritized when:
155
+ * - its last response was 429, OR
156
+ * - remaining requests/tokens dropped to 0, OR
157
+ * - the reset window hasn't passed yet and remaining is very low (<10% of
158
+ * a typical window, heuristic: remaining < 5)
159
+ */
160
+ export function isProviderDeprioritized(provider: string, nowMs: number = Date.now()): boolean {
161
+ const entry = quotaCache.get(provider.toLowerCase());
162
+ if (!entry) return false;
163
+ // Stale entries are ignored — the provider may have recovered.
164
+ if (nowMs - entry.updatedAtMs > QUOTA_TTL_MS) return false;
165
+ // 429 = actively rate-limited right now.
166
+ if (entry.lastStatus === 429) return true;
167
+ // Explicit zero remaining.
168
+ if (entry.remainingRequests === 0 || entry.remainingTokens === 0) return true;
169
+ // Very low remaining with a future reset = approaching the wall.
170
+ if (entry.resetAtMs !== undefined && entry.resetAtMs > nowMs) {
171
+ if (entry.remainingRequests !== undefined && entry.remainingRequests < 5) return true;
172
+ if (entry.remainingTokens !== undefined && entry.remainingTokens < 1000) return true;
173
+ }
174
+ return false;
175
+ }
176
+
177
+ /**
178
+ * Build the `deprioritizedProviders` list for a set of candidate providers.
179
+ * Only providers with a fresh, deprioritized entry are included.
180
+ */
181
+ export function deprioritizedProviders(providers: string[], nowMs: number = Date.now()): string[] {
182
+ return providers.filter((p) => isProviderDeprioritized(p, nowMs));
183
+ }
184
+
185
+ /**
186
+ * Build a `providerRank` map from quota data. Providers with more remaining
187
+ * capacity get a lower rank (tried earlier). Providers with no data get
188
+ * `Number.MAX_SAFE_INTEGER` (unknown = last). Providers that are
189
+ * deprioritized get `Number.MAX_SAFE_INTEGER - 1` (just above unknown).
190
+ */
191
+ export function providerRankFromQuota(providers: string[], nowMs: number = Date.now()): Record<string, number> {
192
+ const rank: Record<string, number> = {};
193
+ for (const provider of providers) {
194
+ const key = provider.toLowerCase();
195
+ const entry = quotaCache.get(key);
196
+ if (!entry || nowMs - entry.updatedAtMs > QUOTA_TTL_MS) {
197
+ rank[key] = Number.MAX_SAFE_INTEGER;
198
+ continue;
199
+ }
200
+ if (isProviderDeprioritized(key, nowMs)) {
201
+ rank[key] = Number.MAX_SAFE_INTEGER - 1;
202
+ continue;
203
+ }
204
+ // Rank by remaining capacity: more remaining = lower rank = tried first.
205
+ // Use requests as primary signal; fall back to tokens.
206
+ const remaining = entry.remainingRequests ?? entry.remainingTokens;
207
+ rank[key] = remaining !== undefined ? Math.max(0, 1000 - remaining) : Number.MAX_SAFE_INTEGER;
208
+ }
209
+ return rank;
210
+ }
211
+
212
+ /**
213
+ * Clear the provider quota cache. Called on session switch to prevent
214
+ * stale quota data from a previous session leaking into the next.
215
+ */
216
+ export function clearProviderQuotaCache(): void {
217
+ quotaCache.clear();
218
+ }
219
+
220
+ /** @internal Test seam — check if a provider has a cached entry. */
221
+ export function __test_quotaCacheHasProvider(provider: string): boolean {
222
+ return quotaCache.has(provider.toLowerCase());
223
+ }
224
+
225
+ /** @internal Test seam — clear the quota cache. */
226
+ export function __test_resetProviderQuota(): void {
227
+ clearProviderQuotaCache();
228
+ }
@@ -0,0 +1,135 @@
1
+ /**
2
+ * session-model.ts — track the MAIN session's live model + thinking level.
3
+ *
4
+ * WHY: `ctx.model` from the pi extension API is the session's *saved* model,
5
+ * not necessarily the one currently in use. A session that restored stale
6
+ * state can report `anthropic/claude-sonnet-4-5` while actually running
7
+ * `minimax/MiniMax-M3` (the value shown in the footer). Subagents that
8
+ * inherit the parent model (`model: false`, which every builtin agent uses)
9
+ * therefore "jumped" to whatever a previous session had saved.
10
+ *
11
+ * Pi emits `model_select` (with the real `Model` object) on every set / cycle
12
+ * / restore, and `thinking_level_select` for the thinking level. Recording
13
+ * those gives an authoritative view of what the main session is running right
14
+ * now, which is what a subagent should inherit.
15
+ *
16
+ * This module is process-local state with no I/O. `register.ts` feeds it from
17
+ * the pi events; the spawn paths read it through {@link resolveParentModel}.
18
+ */
19
+
20
+ import type { RunModelContext } from "../../state/types.ts";
21
+ import { availableModelInfosFromRegistry, modelRefToString } from "./model-fallback.ts";
22
+
23
+ export type SessionModelSource = "model_select" | "session_start" | "none";
24
+
25
+ interface SessionModelState {
26
+ model?: string;
27
+ thinking?: string;
28
+ source: SessionModelSource;
29
+ updatedAt?: number;
30
+ }
31
+
32
+ const state: SessionModelState = { source: "none" };
33
+
34
+ /**
35
+ * Record the model the main session is running. Accepts pi's `Model` object
36
+ * (`{ provider, id }`) or a `"provider/id"` string; anything unrecognized is
37
+ * ignored so a pi API change can never blank out a known-good value.
38
+ */
39
+ export function noteSessionModel(model: unknown, source: SessionModelSource = "model_select"): void {
40
+ const normalized = modelRefToString(model);
41
+ if (!normalized) return;
42
+ // `session_start` only seeds: pi may emit `model_select` (source "restore")
43
+ // before session_start, and that value is authoritative. Letting the seed
44
+ // overwrite it would put the stale saved model back.
45
+ if (source === "session_start" && state.source === "model_select") return;
46
+ state.model = normalized;
47
+ state.source = source;
48
+ state.updatedAt = Date.now();
49
+ }
50
+
51
+ /**
52
+ * Record the main session's thinking level. An explicit `"off"`/`""` clears it;
53
+ * a non-string (e.g. `ctx.thinkingLevel` on a context that does not expose one)
54
+ * is ignored so seeding cannot erase a tracked value.
55
+ */
56
+ export function noteSessionThinking(level: unknown): void {
57
+ if (typeof level !== "string") return;
58
+ const value = level.trim();
59
+ state.thinking = value && value !== "off" ? value : undefined;
60
+ }
61
+
62
+ /** The main session's live model as `"provider/id"`, when known. */
63
+ export function currentSessionModel(): string | undefined {
64
+ return state.model;
65
+ }
66
+
67
+ /** The main session's live thinking level, when known. */
68
+ export function currentSessionThinking(): string | undefined {
69
+ return state.thinking;
70
+ }
71
+
72
+ /**
73
+ * Resolve the model a subagent should inherit. The live `model_select` value
74
+ * wins over `ctx.model` because the latter can be stale session state; the
75
+ * caller's value is the fallback for contexts that never saw the event
76
+ * (e.g. a fresh headless run).
77
+ */
78
+ export function resolveParentModel(ctxModel: unknown): string | undefined {
79
+ return state.model ?? modelRefToString(ctxModel);
80
+ }
81
+
82
+ /**
83
+ * Snapshot everything the model router needs, for a run that will execute
84
+ * outside this process (background/async). Returns undefined when there is
85
+ * nothing worth persisting, so old manifests stay byte-identical.
86
+ */
87
+ export function captureRunModelContext(
88
+ ctx: { model?: unknown; modelRegistry?: unknown; thinkingLevel?: unknown },
89
+ override?: string,
90
+ ): RunModelContext | undefined {
91
+ const parentModel = resolveParentModel(ctx.model);
92
+ const registryModels = availableModelInfosFromRegistry(ctx.modelRegistry);
93
+ const availableModels = registryModels && registryModels.length > 0 ? registryModels.map((model) => model.fullId) : undefined;
94
+ const parentThinking = currentSessionThinking() ?? (typeof ctx.thinkingLevel === "string" ? ctx.thinkingLevel : undefined);
95
+ const trimmedOverride = override?.trim() || undefined;
96
+ if (!parentModel && !availableModels && !parentThinking && !trimmedOverride) return undefined;
97
+ return {
98
+ ...(trimmedOverride ? { override: trimmedOverride } : {}),
99
+ ...(parentModel ? { parentModel } : {}),
100
+ ...(parentThinking && parentThinking !== "off" ? { parentThinking } : {}),
101
+ ...(availableModels ? { availableModels } : {}),
102
+ };
103
+ }
104
+
105
+ /**
106
+ * Rebuild a minimal `ModelRegistry`-shaped object from a persisted catalogue,
107
+ * so the background path feeds `buildConfiguredModelRouting` the same
108
+ * auth-filtered list the caller had instead of falling back to raw models.json.
109
+ */
110
+ export function registryFromModelContext(context: RunModelContext | undefined): { getAvailable: () => unknown[] } | undefined {
111
+ const models = context?.availableModels;
112
+ if (!models || models.length === 0) return undefined;
113
+ const entries = models
114
+ .map((fullId) => {
115
+ const slashIdx = fullId.indexOf("/");
116
+ if (slashIdx <= 0) return undefined;
117
+ return { provider: fullId.slice(0, slashIdx), id: fullId.slice(slashIdx + 1) };
118
+ })
119
+ .filter((entry): entry is { provider: string; id: string } => entry !== undefined);
120
+ if (entries.length === 0) return undefined;
121
+ return { getAvailable: () => entries };
122
+ }
123
+
124
+ /** Diagnostic snapshot for `team doctor`. */
125
+ export function sessionModelSnapshot(): Readonly<Required<Pick<SessionModelState, "source">> & SessionModelState> {
126
+ return { ...state };
127
+ }
128
+
129
+ /** @internal Test seam — reset tracked state between cases. */
130
+ export function __test_resetSessionModel(): void {
131
+ state.model = undefined;
132
+ state.thinking = undefined;
133
+ state.source = "none";
134
+ state.updatedAt = undefined;
135
+ }
@@ -51,6 +51,10 @@ import {
51
51
  formatModelAttemptNote,
52
52
  isRetryableModelFailure,
53
53
  type ModelAttemptSummary,
54
+ type ModelFallbackPolicy,
55
+ resolveDefaultSubagentModel,
56
+ resolveModelFallbackPolicy,
57
+ warnOutOfScopeSoft,
54
58
  } from "../model/model-fallback.ts";
55
59
  import { readEnabledModelsPatterns } from "../model/model-scope.ts";
56
60
  import { type ParsedPiJsonOutput, parsePiJsonOutput } from "../output/pi-json-output.ts";
@@ -144,6 +148,27 @@ export function computeSpawnBudgetMax(attemptModelsCount: number, configuredMaxA
144
148
  return attemptModelsCount * (configuredMaxAttempts + 1);
145
149
  }
146
150
 
151
+ /**
152
+ * Resolve the model fallback policy from crew config + env. Best-effort: any
153
+ * config read failure returns undefined (legacy unbounded/unordered behaviour).
154
+ */
155
+ function resolveTaskModelFallbackPolicy(cwd: string): ModelFallbackPolicy | undefined {
156
+ try {
157
+ return resolveModelFallbackPolicy(loadConfig(cwd).config.runtime?.modelFallback);
158
+ } catch {
159
+ return undefined;
160
+ }
161
+ }
162
+
163
+ /** Resolve the default subagent model from crew config + env. */
164
+ function resolveTaskDefaultSubagentModel(cwd: string): string | undefined {
165
+ try {
166
+ return resolveDefaultSubagentModel(loadConfig(cwd).config.runtime?.modelFallback);
167
+ } catch {
168
+ return undefined;
169
+ }
170
+ }
171
+
147
172
  /**
148
173
  * 429/rate-limit detection (PI_CREW_TOOLING_429_NOTE.md).
149
174
  *
@@ -256,19 +281,46 @@ export async function runChildProcessTask(ctx: TaskExecutionContext): Promise<Ta
256
281
  let terminalEvidence: OperationTerminalEvidence[] = [];
257
282
  let startupEvidence = ctx.startupEvidence;
258
283
 
284
+ const modelFallbackPolicy = resolveTaskModelFallbackPolicy(task.cwd);
285
+ const defaultSubagentModel = resolveTaskDefaultSubagentModel(task.cwd);
259
286
  const modelRoutingPlan = buildConfiguredModelRouting({
260
287
  overrideModel: input.modelOverride,
261
288
  stepModel: input.step.model,
262
289
  teamRoleModel: input.teamRoleModel,
290
+ teamRoleFallbackModels: input.teamRoleFallbackModels,
263
291
  agentModel: input.agent.model,
292
+ defaultSubagentModel,
264
293
  fallbackModels: input.agent.fallbackModels,
265
294
  parentModel: input.parentModel,
266
295
  modelRegistry: input.modelRegistry,
267
296
  cwd: task.cwd,
297
+ policy: modelFallbackPolicy,
268
298
  scopeModelsPatterns: await resolveTaskScopeModelsPatterns(task.cwd),
269
299
  });
270
300
  const candidates = modelRoutingPlan.candidates;
271
- const attemptModels = candidates.length > 0 ? candidates : [undefined];
301
+ // Surface a warning when the caller's requested model was silently replaced.
302
+ if (modelRoutingPlan.droppedRequested) {
303
+ void appendEventAsync(manifest.eventsPath, {
304
+ type: "task.model_dropped",
305
+ runId: manifest.runId,
306
+ taskId: task.id,
307
+ message: `Requested model "${modelRoutingPlan.droppedRequested}" is not available; using "${candidates[0] ?? "default"}" instead.`,
308
+ data: {
309
+ requested: modelRoutingPlan.droppedRequested,
310
+ resolved: candidates[0],
311
+ fallbackChain: candidates,
312
+ },
313
+ }).catch(() => {
314
+ /* no-op: best-effort diagnostic append, ignore delivery errors */
315
+ });
316
+ }
317
+ // F2: surface a non-silent warning when the INITIAL routing resolved an
318
+ // out-of-scope soft-sourced model (mirrors live-session-runtime.ts). Caller
319
+ // sources already throw inside buildConfiguredModelRouting.
320
+ warnOutOfScopeSoft(modelRoutingPlan.scopeVerdict, "child-executor.initial-out-of-scope");
321
+ // Mutable: the one-shot re-resolve below appends a late-discovered model so
322
+ // the loop actually retries it (see FIX 1 at the end of the attempt loop).
323
+ const attemptModels: (string | undefined)[] = candidates.length > 0 ? [...candidates] : [undefined];
272
324
  // CORE-3: auto-compute per-task spawn budget on first entry.
273
325
  // Budget = attemptModels.length × (maxAttempts + 1) — always one
274
326
  // full attempt-worth above the theoretical maximum of
@@ -283,6 +335,9 @@ export async function runChildProcessTask(ctx: TaskExecutionContext): Promise<Ta
283
335
  }
284
336
  const logs: string[] = [];
285
337
  let finalStderr = "";
338
+ // One-shot: the re-resolve is a safety net for a chain that was computed
339
+ // before the registry was complete, not a retry loop of its own.
340
+ let reResolveUsed = false;
286
341
  modelAttempts = [];
287
342
  let finalCheckpointWritten = false;
288
343
  let lastAgentRecordPersistedAt = 0;
@@ -440,6 +495,7 @@ export async function runChildProcessTask(ctx: TaskExecutionContext): Promise<Ta
440
495
  excludeContextBash: input.runtimeConfig?.excludeContextBash,
441
496
  sessionId: manifest.sessionId,
442
497
  role: task.role,
498
+ thinkingOverride: input.teamRoleThinking,
443
499
  runId: manifest.runId,
444
500
  agentId: task.id,
445
501
  artifactsRoot: manifest.artifactsRoot,
@@ -666,7 +722,8 @@ export async function runChildProcessTask(ctx: TaskExecutionContext): Promise<Ta
666
722
  // after the precompute, or the precompute ran before the parent
667
723
  // model was known). If a different candidate is found, use it as
668
724
  // nextModel; otherwise fall through to the existing break.
669
- if (!nextModel && isRetryableModelFailure(error)) {
725
+ if (!nextModel && !reResolveUsed && isRetryableModelFailure(error)) {
726
+ reResolveUsed = true;
670
727
  const reResolved = buildConfiguredModelRouting({
671
728
  overrideModel: undefined,
672
729
  stepModel: undefined,
@@ -676,10 +733,29 @@ export async function runChildProcessTask(ctx: TaskExecutionContext): Promise<Ta
676
733
  parentModel: attempt.model,
677
734
  modelRegistry: input.modelRegistry,
678
735
  cwd: task.cwd,
736
+ policy: modelFallbackPolicy,
679
737
  scopeModelsPatterns: await resolveTaskScopeModelsPatterns(task.cwd),
680
738
  });
681
- const alt = reResolved.candidates.find((c) => c !== attempt.model);
682
- if (alt) nextModel = alt;
739
+ // Sec-M1: surface a non-silent warning when a soft-sourced re-resolved
740
+ // model is out-of-scope. Only non-caller sources (frontmatter /
741
+ // resolved) reach here — caller sources already throw inside
742
+ // buildConfiguredModelRouting.
743
+ warnOutOfScopeSoft(reResolved.scopeVerdict, "child-executor.re-resolve-out-of-scope", "Re-resolved model");
744
+ // Must exclude EVERY model already tried, not just the last one —
745
+ // otherwise the "alternative" is one that already failed.
746
+ const tried = new Set(modelAttempts.map((a) => a.model));
747
+ const alt = reResolved.candidates.find((candidate) => !tried.has(candidate));
748
+ if (alt) {
749
+ // Append so the loop reaches it. Without this the log claimed a
750
+ // retry that never happened: `nextModel` alone only gates `break`,
751
+ // the next iteration reads `attemptModels[i + 1]`.
752
+ attemptModels.push(alt);
753
+ nextModel = alt;
754
+ // Keep the budget consistent with the extended chain (one extra
755
+ // model = one extra attempt-worth), otherwise the spawn-budget
756
+ // guard aborts the very attempt we just queued.
757
+ if (input.spawnBudget) input.spawnBudget.max += 1;
758
+ }
683
759
  }
684
760
  if (!nextModel || !isRetryableModelFailure(error)) break;
685
761
  logs.push(formatModelAttemptNote(attempt, nextModel), "");
@@ -762,6 +838,8 @@ export async function runChildProcessTask(ctx: TaskExecutionContext): Promise<Ta
762
838
  fallbackChain: candidates,
763
839
  reason: fallbackReason ?? modelRoutingPlan.reason,
764
840
  usedAttempt,
841
+ droppedRequested: modelRoutingPlan.droppedRequested,
842
+ autoFallbackCount: modelRoutingPlan.autoFallbackCount,
765
843
  },
766
844
  };
767
845
  tasks = updateTask(tasks, task);
@@ -35,6 +35,8 @@ export interface RunLiveTaskInput {
35
35
  modelRegistry?: unknown;
36
36
  modelOverride?: string;
37
37
  teamRoleModel?: string;
38
+ teamRoleFallbackModels?: string[];
39
+ teamRoleThinking?: string;
38
40
  isCurrent?: () => boolean;
39
41
  /** Workspace where this task run was initiated — used for session-scoped live-agent visibility. */
40
42
  workspaceId: string;
@@ -134,6 +136,8 @@ export async function runLiveTask(input: RunLiveTaskInput): Promise<RunLiveTaskO
134
136
  modelRegistry: input.modelRegistry,
135
137
  modelOverride: input.modelOverride,
136
138
  teamRoleModel: input.teamRoleModel,
139
+ teamRoleFallbackModels: input.teamRoleFallbackModels,
140
+ teamRoleThinking: input.teamRoleThinking,
137
141
  isCurrent,
138
142
  workspaceId: input.workspaceId,
139
143
  // Phase 2: Pass output schema for yield validation
@@ -181,6 +185,14 @@ export async function runLiveTask(input: RunLiveTaskInput): Promise<RunLiveTaskO
181
185
  usage: liveResult.usage,
182
186
  agentProgress: applyUsageToProgress(task.agentProgress, liveResult.usage),
183
187
  };
188
+ if (liveResult.modelRouting)
189
+ task = {
190
+ ...task,
191
+ modelRouting: {
192
+ ...liveResult.modelRouting,
193
+ usedAttempt: 0,
194
+ },
195
+ };
184
196
  persistLiveProgress({ type: "attempt_finished" }, true);
185
197
  const resultArtifact = writeArtifact(manifest.artifactsRoot, {
186
198
  kind: "result",
@@ -52,6 +52,9 @@ export interface TaskRunnerInput {
52
52
  modelRegistry?: unknown;
53
53
  modelOverride?: string;
54
54
  teamRoleModel?: string;
55
+ teamRoleFallbackModels?: string[];
56
+ /** Per-role thinking level override (teamRole.thinking). Takes precedence over agent.thinking. */
57
+ teamRoleThinking?: string;
55
58
  teamRoleSkills?: string[] | false;
56
59
  skillOverride?: string[] | false;
57
60
  limits?: CrewLimitsConfig;
@@ -152,6 +155,8 @@ export async function runTeamTask(input: TaskRunnerInput): Promise<{ manifest: T
152
155
  modelRegistry: input.modelRegistry,
153
156
  modelOverride: input.modelOverride,
154
157
  teamRoleModel: input.teamRoleModel,
158
+ teamRoleFallbackModels: input.teamRoleFallbackModels,
159
+ teamRoleThinking: input.teamRoleThinking,
155
160
  workspaceId: input.workspaceId,
156
161
  });
157
162
  task = live.task;
@@ -1102,6 +1102,10 @@ interface SchedulerContext {
1102
1102
  queueIndex: TaskGraphIndex;
1103
1103
  wfMachine: WorkflowStateMachine;
1104
1104
  pendingUnits: Map<string, PendingUnit>;
1105
+ /** Task ids ever dispatched (grows monotonically; never removed). Used by
1106
+ * terminaliseRunWithDrain to cancel — not skip — tasks that were in-flight
1107
+ * even after their dispatch unit settled + left pendingUnits (RT-NEW-2 race). */
1108
+ dispatchedTaskIds: Set<string>;
1105
1109
  runController: AbortController;
1106
1110
  runtimeKind: CrewRuntimeKind;
1107
1111
  adaptivePlanInjected: boolean;
@@ -1266,10 +1270,13 @@ async function terminaliseRunWithDrain(
1266
1270
  ctx: SchedulerContext,
1267
1271
  opts: { cancelMessage: string; blockedMessage: string; failedReason: string },
1268
1272
  ): Promise<{ manifest: TeamRunManifest; tasks: TeamTaskState[] }> {
1269
- const inflightTaskIds = new Set<string>();
1270
- for (const unit of ctx.pendingUnits.values()) {
1271
- for (const id of unit.taskIds) inflightTaskIds.add(id);
1272
- }
1273
+ // Ever-dispatched tasks (monotonic set populated at dispatch). Using this
1274
+ // instead of a pendingUnits snapshot closes the RT-NEW-2 race where a task
1275
+ // whose unit settled + left pendingUnits before the abort — but whose task
1276
+ // status isn't terminal yet — would otherwise fall through to markBlocked
1277
+ // and be clobbered to "skipped" (observed CI flake: 02_b skipped on
1278
+ // team-runner-budget-abort-inflight across v0.9.59 / cfd68d06 / 12386af2).
1279
+ const inflightTaskIds = ctx.dispatchedTaskIds;
1273
1280
  const outcomes = await drainPendingUnits(ctx.pendingUnits, ctx.runController);
1274
1281
  const validResults: { manifest: TeamRunManifest; tasks: TeamTaskState[] }[] = [];
1275
1282
  for (const outcome of outcomes) {
@@ -1729,6 +1736,8 @@ async function dispatchBatch(ctx: SchedulerContext, decision: DispatchBatchDecis
1729
1736
  modelRegistry: input.modelRegistry,
1730
1737
  modelOverride: input.modelOverride,
1731
1738
  teamRoleModel: teamRole?.model,
1739
+ teamRoleThinking: teamRole?.thinking,
1740
+ teamRoleFallbackModels: teamRole?.fallbackModels,
1732
1741
  teamRoleSkills: teamRole?.skills,
1733
1742
  skillOverride: input.skillOverride,
1734
1743
  limits: input.limits,
@@ -1926,6 +1935,10 @@ async function dispatchBatch(ctx: SchedulerContext, decision: DispatchBatchDecis
1926
1935
  promise: rawPromise,
1927
1936
  wrapped,
1928
1937
  });
1938
+ // RT-NEW-2 race fix: record ever-dispatched task ids so terminaliseRunWithDrain
1939
+ // cancels (not skips) tasks whose unit settled + left pendingUnits before
1940
+ // the abort fired but whose task status isn't terminal yet.
1941
+ for (const id of unitTaskIds) ctx.dispatchedTaskIds.add(id);
1929
1942
  }
1930
1943
  }
1931
1944
 
@@ -2462,6 +2475,7 @@ async function executeTeamRunCore(
2462
2475
  queueIndex,
2463
2476
  wfMachine,
2464
2477
  pendingUnits,
2478
+ dispatchedTaskIds: new Set(),
2465
2479
  runController,
2466
2480
  runtimeKind,
2467
2481
  adaptivePlanInjected,
@@ -42,6 +42,17 @@ export const PiTeamsLimitsConfigSchema = Type.Object(
42
42
  { additionalProperties: false },
43
43
  );
44
44
 
45
+ export const PiTeamsModelFallbackConfigSchema = Type.Object(
46
+ {
47
+ maxAutoFallbacks: Type.Optional(Type.Integer({ minimum: 0 })),
48
+ order: Type.Optional(Type.Union([Type.Literal("parentFirst"), Type.Literal("asIs")])),
49
+ requireCredentials: Type.Optional(Type.Boolean()),
50
+ quotaAwareOrdering: Type.Optional(Type.Boolean()),
51
+ defaultSubagentModel: Type.Optional(Type.String({ minLength: 1 })),
52
+ },
53
+ { additionalProperties: false },
54
+ );
55
+
45
56
  export const PiTeamsRuntimeConfigSchema = Type.Object(
46
57
  {
47
58
  mode: Type.Optional(
@@ -82,6 +93,7 @@ export const PiTeamsRuntimeConfigSchema = Type.Object(
82
93
  { additionalProperties: false },
83
94
  ),
84
95
  ),
96
+ modelFallback: Type.Optional(PiTeamsModelFallbackConfigSchema),
85
97
  },
86
98
  { additionalProperties: false },
87
99
  );