pi-crew 0.9.59 → 0.9.60
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +41 -0
- package/dist/index.mjs +1252 -739
- package/docs/commands-reference.md +5 -0
- package/docs/resource-formats.md +7 -1
- package/package.json +1 -1
- package/skills/real-test-pi-crew/REPORT-TEMPLATE.md +52 -0
- package/skills/real-test-pi-crew/SKILL.md +6 -4
- package/src/config/types.ts +29 -0
- package/src/extension/registration/lifecycle-handlers.ts +33 -0
- package/src/extension/team-tool/chain-dispatch.ts +20 -2
- package/src/extension/team-tool/chain-executor.ts +6 -1
- package/src/extension/team-tool/doctor.ts +43 -0
- package/src/extension/team-tool/run.ts +19 -3
- package/src/extension/team-tool.ts +2 -1
- package/src/runtime/background-runner.ts +26 -0
- package/src/runtime/child-pi/child-pi-spawn.ts +1 -0
- package/src/runtime/child-pi/child-pi.ts +2 -0
- package/src/runtime/live-session/live-session-runtime.ts +83 -16
- package/src/runtime/model/model-fallback.ts +412 -36
- package/src/runtime/model/model-scope.ts +2 -0
- package/src/runtime/model/pi-args.ts +7 -3
- package/src/runtime/model/provider-quota.ts +228 -0
- package/src/runtime/model/session-model.ts +135 -0
- package/src/runtime/task-runner/child-executor.ts +82 -4
- package/src/runtime/task-runner/live-executor.ts +12 -0
- package/src/runtime/task-runner.ts +5 -0
- package/src/runtime/team-runner.ts +18 -4
- package/src/schema/config-schema.ts +12 -0
- package/src/state/types.ts +27 -0
- package/src/teams/discover-teams.ts +16 -2
- package/src/teams/team-config.ts +4 -0
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* provider-quota.ts — track provider rate-limit / quota state from response
|
|
3
|
+
* headers so the model fallback chain can deprioritize providers that are
|
|
4
|
+
* near exhaustion.
|
|
5
|
+
*
|
|
6
|
+
* WHY: when a provider returns 429 or its `x-ratelimit-remaining` header
|
|
7
|
+
* drops to zero, every subsequent spawn to that provider wastes a child
|
|
8
|
+
* process + a full retry cycle before the fallback chain moves on. Recording
|
|
9
|
+
* the signal here lets `buildConfiguredModelRouting` push those providers to
|
|
10
|
+
* the back of the auto tail instead of trying them first.
|
|
11
|
+
*
|
|
12
|
+
* Data source: pi's `after_provider_response` event carries `status` and
|
|
13
|
+
* `headers`. Standard rate-limit headers (OpenAI, Anthropic, most proxies):
|
|
14
|
+
* x-ratelimit-remaining-requests / x-ratelimit-remaining-tokens
|
|
15
|
+
* x-ratelimit-reset-requests / x-ratelimit-reset-tokens
|
|
16
|
+
* retry-after (on 429)
|
|
17
|
+
*
|
|
18
|
+
* The tracker is process-local and best-effort: a missed event just means
|
|
19
|
+
* the cache is slightly stale, never a wrong routing decision.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
interface ProviderQuotaEntry {
|
|
23
|
+
/** Provider key (lowercase). */
|
|
24
|
+
provider: string;
|
|
25
|
+
/** Remaining requests from the most recent response header. */
|
|
26
|
+
remainingRequests?: number;
|
|
27
|
+
/** Remaining tokens from the most recent response header. */
|
|
28
|
+
remainingTokens?: number;
|
|
29
|
+
/** When the rate-limit window resets (epoch ms). */
|
|
30
|
+
resetAtMs?: number;
|
|
31
|
+
/** HTTP status of the last response (429 = rate-limited). */
|
|
32
|
+
lastStatus?: number;
|
|
33
|
+
/** When this entry was last updated (epoch ms). */
|
|
34
|
+
updatedAtMs: number;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/** How long a quota entry stays authoritative before we stop trusting it. */
|
|
38
|
+
const QUOTA_TTL_MS = 5 * 60 * 1000; // 5 minutes
|
|
39
|
+
|
|
40
|
+
/** Process-local cache keyed by lowercase provider name. */
|
|
41
|
+
const quotaCache = new Map<string, ProviderQuotaEntry>();
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Parse a numeric header value, returning undefined for missing/invalid.
|
|
45
|
+
*/
|
|
46
|
+
function headerNumber(headers: Record<string, string>, ...names: string[]): number | undefined {
|
|
47
|
+
for (const name of names) {
|
|
48
|
+
const raw = headers[name] ?? headers[name.toLowerCase()];
|
|
49
|
+
if (raw === undefined) continue;
|
|
50
|
+
const parsed = Number.parseInt(raw, 10);
|
|
51
|
+
if (Number.isFinite(parsed) && parsed >= 0) return parsed;
|
|
52
|
+
}
|
|
53
|
+
return undefined;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Read a raw (unparsed) header value — first match wins, case-insensitive.
|
|
58
|
+
*/
|
|
59
|
+
function headerRaw(headers: Record<string, string>, ...names: string[]): string | undefined {
|
|
60
|
+
for (const name of names) {
|
|
61
|
+
const raw = headers[name] ?? headers[name.toLowerCase()];
|
|
62
|
+
if (raw !== undefined) return raw;
|
|
63
|
+
}
|
|
64
|
+
return undefined;
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/** Regex: compact Go-duration like "6m0s", "1s", "1h30m", "120s". */
|
|
68
|
+
const GO_DURATION_RE = /^(\d+[smh])+$/;
|
|
69
|
+
/** Regex: extract individual (\d+)(unit) segments from a Go-duration string. */
|
|
70
|
+
const GO_DURATION_SEG_RE = /(\d+)([smh])/g;
|
|
71
|
+
|
|
72
|
+
/**
|
|
73
|
+
* Parse a reset-timestamp header value into epoch-ms.
|
|
74
|
+
*
|
|
75
|
+
* Three formats are supported (tried in order):
|
|
76
|
+
* (c) RFC3339 absolute timestamp (Anthropic): Date.parse returns epoch-ms directly.
|
|
77
|
+
* (a) Go-duration compact string (OpenAI): "6m0s", "1s", "1h30m", "120s".
|
|
78
|
+
* (b) Pure-seconds integer (retry-after): "30" → nowMs + 30 000.
|
|
79
|
+
*
|
|
80
|
+
* Returns undefined if no format matches.
|
|
81
|
+
*/
|
|
82
|
+
function parseResetValue(value: string, nowMs: number): number | undefined {
|
|
83
|
+
const trimmed = value.trim();
|
|
84
|
+
// (b) Pure-seconds integer (retry-after / reset): "0", "30", "120".
|
|
85
|
+
// MUST precede RFC3339 — Date.parse("0") returns 946659600000 (Y2K) in V8,
|
|
86
|
+
// not NaN, so a numeric "0" would otherwise hijack the timestamp branch
|
|
87
|
+
// and yield a resetAtMs in the year 2000.
|
|
88
|
+
if (/^\d+$/.test(trimmed)) {
|
|
89
|
+
const secs = Number.parseInt(trimmed, 10);
|
|
90
|
+
if (Number.isFinite(secs) && secs >= 0) return nowMs + secs * 1000;
|
|
91
|
+
}
|
|
92
|
+
// (a) Go-duration compact: digits+unit pairs ("6m0s", "1s", "1h30m").
|
|
93
|
+
if (GO_DURATION_RE.test(trimmed)) {
|
|
94
|
+
let durationMs = 0;
|
|
95
|
+
for (const seg of trimmed.matchAll(GO_DURATION_SEG_RE)) {
|
|
96
|
+
const n = Number.parseInt(seg[1], 10);
|
|
97
|
+
const unit = seg[2];
|
|
98
|
+
durationMs += unit === "s" ? n * 1000 : unit === "m" ? n * 60_000 : n * 3_600_000;
|
|
99
|
+
}
|
|
100
|
+
return nowMs + durationMs;
|
|
101
|
+
}
|
|
102
|
+
// (c) RFC3339 absolute timestamp (Anthropic). Only non-numeric date-like
|
|
103
|
+
// strings reach here: pure-numeric is handled above, and Date.parse on
|
|
104
|
+
// Go-durations like "6m0s" returns NaN.
|
|
105
|
+
const parsed = Date.parse(trimmed);
|
|
106
|
+
if (Number.isFinite(parsed) && parsed > 0) return parsed;
|
|
107
|
+
return undefined;
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* Parse a reset timestamp from headers. Providers use different formats:
|
|
112
|
+
* - OpenAI: Go-duration strings ("6m0s", "1s") in x-ratelimit-reset-requests.
|
|
113
|
+
* - Anthropic: RFC3339 timestamps.
|
|
114
|
+
* - retry-after: pure-seconds integer.
|
|
115
|
+
*/
|
|
116
|
+
function headerResetMs(headers: Record<string, string>, nowMs: number): number | undefined {
|
|
117
|
+
const resetRaw = headerRaw(headers, "x-ratelimit-reset-requests", "x-ratelimit-reset");
|
|
118
|
+
if (resetRaw !== undefined) {
|
|
119
|
+
const parsed = parseResetValue(resetRaw, nowMs);
|
|
120
|
+
if (parsed !== undefined) return parsed;
|
|
121
|
+
}
|
|
122
|
+
const retryRaw = headerRaw(headers, "retry-after");
|
|
123
|
+
if (retryRaw !== undefined) {
|
|
124
|
+
return parseResetValue(retryRaw, nowMs);
|
|
125
|
+
}
|
|
126
|
+
return undefined;
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
/**
|
|
130
|
+
* Record a provider response. Called from the `after_provider_response`
|
|
131
|
+
* event handler. The `provider` argument is the pi provider key (e.g.
|
|
132
|
+
* "anthropic", "openai-codex"); it is lowercased for consistent lookup.
|
|
133
|
+
*/
|
|
134
|
+
export function noteProviderResponse(provider: string, status: number, headers: Record<string, string>, nowMs: number = Date.now()): void {
|
|
135
|
+
const key = provider.toLowerCase();
|
|
136
|
+
const entry: ProviderQuotaEntry = {
|
|
137
|
+
provider: key,
|
|
138
|
+
remainingRequests: headerNumber(headers, "x-ratelimit-remaining-requests", "x-ratelimit-remaining"),
|
|
139
|
+
remainingTokens: headerNumber(headers, "x-ratelimit-remaining-tokens"),
|
|
140
|
+
resetAtMs: headerResetMs(headers, nowMs),
|
|
141
|
+
lastStatus: status,
|
|
142
|
+
updatedAtMs: nowMs,
|
|
143
|
+
};
|
|
144
|
+
quotaCache.set(key, entry);
|
|
145
|
+
// L1: evict entries older than 2 * QUOTA_TTL_MS (10 min) to bound cache growth.
|
|
146
|
+
const evictionCutoff = nowMs - 2 * QUOTA_TTL_MS;
|
|
147
|
+
for (const [cacheKey, cached] of quotaCache) {
|
|
148
|
+
if (cached.updatedAtMs < evictionCutoff) quotaCache.delete(cacheKey);
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
/**
|
|
153
|
+
* Whether a provider is currently deprioritized (near or at quota limit).
|
|
154
|
+
* A provider is deprioritized when:
|
|
155
|
+
* - its last response was 429, OR
|
|
156
|
+
* - remaining requests/tokens dropped to 0, OR
|
|
157
|
+
* - the reset window hasn't passed yet and remaining is very low (<10% of
|
|
158
|
+
* a typical window, heuristic: remaining < 5)
|
|
159
|
+
*/
|
|
160
|
+
export function isProviderDeprioritized(provider: string, nowMs: number = Date.now()): boolean {
|
|
161
|
+
const entry = quotaCache.get(provider.toLowerCase());
|
|
162
|
+
if (!entry) return false;
|
|
163
|
+
// Stale entries are ignored — the provider may have recovered.
|
|
164
|
+
if (nowMs - entry.updatedAtMs > QUOTA_TTL_MS) return false;
|
|
165
|
+
// 429 = actively rate-limited right now.
|
|
166
|
+
if (entry.lastStatus === 429) return true;
|
|
167
|
+
// Explicit zero remaining.
|
|
168
|
+
if (entry.remainingRequests === 0 || entry.remainingTokens === 0) return true;
|
|
169
|
+
// Very low remaining with a future reset = approaching the wall.
|
|
170
|
+
if (entry.resetAtMs !== undefined && entry.resetAtMs > nowMs) {
|
|
171
|
+
if (entry.remainingRequests !== undefined && entry.remainingRequests < 5) return true;
|
|
172
|
+
if (entry.remainingTokens !== undefined && entry.remainingTokens < 1000) return true;
|
|
173
|
+
}
|
|
174
|
+
return false;
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
/**
|
|
178
|
+
* Build the `deprioritizedProviders` list for a set of candidate providers.
|
|
179
|
+
* Only providers with a fresh, deprioritized entry are included.
|
|
180
|
+
*/
|
|
181
|
+
export function deprioritizedProviders(providers: string[], nowMs: number = Date.now()): string[] {
|
|
182
|
+
return providers.filter((p) => isProviderDeprioritized(p, nowMs));
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
/**
|
|
186
|
+
* Build a `providerRank` map from quota data. Providers with more remaining
|
|
187
|
+
* capacity get a lower rank (tried earlier). Providers with no data get
|
|
188
|
+
* `Number.MAX_SAFE_INTEGER` (unknown = last). Providers that are
|
|
189
|
+
* deprioritized get `Number.MAX_SAFE_INTEGER - 1` (just above unknown).
|
|
190
|
+
*/
|
|
191
|
+
export function providerRankFromQuota(providers: string[], nowMs: number = Date.now()): Record<string, number> {
|
|
192
|
+
const rank: Record<string, number> = {};
|
|
193
|
+
for (const provider of providers) {
|
|
194
|
+
const key = provider.toLowerCase();
|
|
195
|
+
const entry = quotaCache.get(key);
|
|
196
|
+
if (!entry || nowMs - entry.updatedAtMs > QUOTA_TTL_MS) {
|
|
197
|
+
rank[key] = Number.MAX_SAFE_INTEGER;
|
|
198
|
+
continue;
|
|
199
|
+
}
|
|
200
|
+
if (isProviderDeprioritized(key, nowMs)) {
|
|
201
|
+
rank[key] = Number.MAX_SAFE_INTEGER - 1;
|
|
202
|
+
continue;
|
|
203
|
+
}
|
|
204
|
+
// Rank by remaining capacity: more remaining = lower rank = tried first.
|
|
205
|
+
// Use requests as primary signal; fall back to tokens.
|
|
206
|
+
const remaining = entry.remainingRequests ?? entry.remainingTokens;
|
|
207
|
+
rank[key] = remaining !== undefined ? Math.max(0, 1000 - remaining) : Number.MAX_SAFE_INTEGER;
|
|
208
|
+
}
|
|
209
|
+
return rank;
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
/**
|
|
213
|
+
* Clear the provider quota cache. Called on session switch to prevent
|
|
214
|
+
* stale quota data from a previous session leaking into the next.
|
|
215
|
+
*/
|
|
216
|
+
export function clearProviderQuotaCache(): void {
|
|
217
|
+
quotaCache.clear();
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
/** @internal Test seam — check if a provider has a cached entry. */
|
|
221
|
+
export function __test_quotaCacheHasProvider(provider: string): boolean {
|
|
222
|
+
return quotaCache.has(provider.toLowerCase());
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
/** @internal Test seam — clear the quota cache. */
|
|
226
|
+
export function __test_resetProviderQuota(): void {
|
|
227
|
+
clearProviderQuotaCache();
|
|
228
|
+
}
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* session-model.ts — track the MAIN session's live model + thinking level.
|
|
3
|
+
*
|
|
4
|
+
* WHY: `ctx.model` from the pi extension API is the session's *saved* model,
|
|
5
|
+
* not necessarily the one currently in use. A session that restored stale
|
|
6
|
+
* state can report `anthropic/claude-sonnet-4-5` while actually running
|
|
7
|
+
* `minimax/MiniMax-M3` (the value shown in the footer). Subagents that
|
|
8
|
+
* inherit the parent model (`model: false`, which every builtin agent uses)
|
|
9
|
+
* therefore "jumped" to whatever a previous session had saved.
|
|
10
|
+
*
|
|
11
|
+
* Pi emits `model_select` (with the real `Model` object) on every set / cycle
|
|
12
|
+
* / restore, and `thinking_level_select` for the thinking level. Recording
|
|
13
|
+
* those gives an authoritative view of what the main session is running right
|
|
14
|
+
* now, which is what a subagent should inherit.
|
|
15
|
+
*
|
|
16
|
+
* This module is process-local state with no I/O. `register.ts` feeds it from
|
|
17
|
+
* the pi events; the spawn paths read it through {@link resolveParentModel}.
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
import type { RunModelContext } from "../../state/types.ts";
|
|
21
|
+
import { availableModelInfosFromRegistry, modelRefToString } from "./model-fallback.ts";
|
|
22
|
+
|
|
23
|
+
export type SessionModelSource = "model_select" | "session_start" | "none";
|
|
24
|
+
|
|
25
|
+
interface SessionModelState {
|
|
26
|
+
model?: string;
|
|
27
|
+
thinking?: string;
|
|
28
|
+
source: SessionModelSource;
|
|
29
|
+
updatedAt?: number;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
const state: SessionModelState = { source: "none" };
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* Record the model the main session is running. Accepts pi's `Model` object
|
|
36
|
+
* (`{ provider, id }`) or a `"provider/id"` string; anything unrecognized is
|
|
37
|
+
* ignored so a pi API change can never blank out a known-good value.
|
|
38
|
+
*/
|
|
39
|
+
export function noteSessionModel(model: unknown, source: SessionModelSource = "model_select"): void {
|
|
40
|
+
const normalized = modelRefToString(model);
|
|
41
|
+
if (!normalized) return;
|
|
42
|
+
// `session_start` only seeds: pi may emit `model_select` (source "restore")
|
|
43
|
+
// before session_start, and that value is authoritative. Letting the seed
|
|
44
|
+
// overwrite it would put the stale saved model back.
|
|
45
|
+
if (source === "session_start" && state.source === "model_select") return;
|
|
46
|
+
state.model = normalized;
|
|
47
|
+
state.source = source;
|
|
48
|
+
state.updatedAt = Date.now();
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Record the main session's thinking level. An explicit `"off"`/`""` clears it;
|
|
53
|
+
* a non-string (e.g. `ctx.thinkingLevel` on a context that does not expose one)
|
|
54
|
+
* is ignored so seeding cannot erase a tracked value.
|
|
55
|
+
*/
|
|
56
|
+
export function noteSessionThinking(level: unknown): void {
|
|
57
|
+
if (typeof level !== "string") return;
|
|
58
|
+
const value = level.trim();
|
|
59
|
+
state.thinking = value && value !== "off" ? value : undefined;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/** The main session's live model as `"provider/id"`, when known. */
|
|
63
|
+
export function currentSessionModel(): string | undefined {
|
|
64
|
+
return state.model;
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/** The main session's live thinking level, when known. */
|
|
68
|
+
export function currentSessionThinking(): string | undefined {
|
|
69
|
+
return state.thinking;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/**
|
|
73
|
+
* Resolve the model a subagent should inherit. The live `model_select` value
|
|
74
|
+
* wins over `ctx.model` because the latter can be stale session state; the
|
|
75
|
+
* caller's value is the fallback for contexts that never saw the event
|
|
76
|
+
* (e.g. a fresh headless run).
|
|
77
|
+
*/
|
|
78
|
+
export function resolveParentModel(ctxModel: unknown): string | undefined {
|
|
79
|
+
return state.model ?? modelRefToString(ctxModel);
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* Snapshot everything the model router needs, for a run that will execute
|
|
84
|
+
* outside this process (background/async). Returns undefined when there is
|
|
85
|
+
* nothing worth persisting, so old manifests stay byte-identical.
|
|
86
|
+
*/
|
|
87
|
+
export function captureRunModelContext(
|
|
88
|
+
ctx: { model?: unknown; modelRegistry?: unknown; thinkingLevel?: unknown },
|
|
89
|
+
override?: string,
|
|
90
|
+
): RunModelContext | undefined {
|
|
91
|
+
const parentModel = resolveParentModel(ctx.model);
|
|
92
|
+
const registryModels = availableModelInfosFromRegistry(ctx.modelRegistry);
|
|
93
|
+
const availableModels = registryModels && registryModels.length > 0 ? registryModels.map((model) => model.fullId) : undefined;
|
|
94
|
+
const parentThinking = currentSessionThinking() ?? (typeof ctx.thinkingLevel === "string" ? ctx.thinkingLevel : undefined);
|
|
95
|
+
const trimmedOverride = override?.trim() || undefined;
|
|
96
|
+
if (!parentModel && !availableModels && !parentThinking && !trimmedOverride) return undefined;
|
|
97
|
+
return {
|
|
98
|
+
...(trimmedOverride ? { override: trimmedOverride } : {}),
|
|
99
|
+
...(parentModel ? { parentModel } : {}),
|
|
100
|
+
...(parentThinking && parentThinking !== "off" ? { parentThinking } : {}),
|
|
101
|
+
...(availableModels ? { availableModels } : {}),
|
|
102
|
+
};
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Rebuild a minimal `ModelRegistry`-shaped object from a persisted catalogue,
|
|
107
|
+
* so the background path feeds `buildConfiguredModelRouting` the same
|
|
108
|
+
* auth-filtered list the caller had instead of falling back to raw models.json.
|
|
109
|
+
*/
|
|
110
|
+
export function registryFromModelContext(context: RunModelContext | undefined): { getAvailable: () => unknown[] } | undefined {
|
|
111
|
+
const models = context?.availableModels;
|
|
112
|
+
if (!models || models.length === 0) return undefined;
|
|
113
|
+
const entries = models
|
|
114
|
+
.map((fullId) => {
|
|
115
|
+
const slashIdx = fullId.indexOf("/");
|
|
116
|
+
if (slashIdx <= 0) return undefined;
|
|
117
|
+
return { provider: fullId.slice(0, slashIdx), id: fullId.slice(slashIdx + 1) };
|
|
118
|
+
})
|
|
119
|
+
.filter((entry): entry is { provider: string; id: string } => entry !== undefined);
|
|
120
|
+
if (entries.length === 0) return undefined;
|
|
121
|
+
return { getAvailable: () => entries };
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/** Diagnostic snapshot for `team doctor`. */
|
|
125
|
+
export function sessionModelSnapshot(): Readonly<Required<Pick<SessionModelState, "source">> & SessionModelState> {
|
|
126
|
+
return { ...state };
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
/** @internal Test seam — reset tracked state between cases. */
|
|
130
|
+
export function __test_resetSessionModel(): void {
|
|
131
|
+
state.model = undefined;
|
|
132
|
+
state.thinking = undefined;
|
|
133
|
+
state.source = "none";
|
|
134
|
+
state.updatedAt = undefined;
|
|
135
|
+
}
|
|
@@ -51,6 +51,10 @@ import {
|
|
|
51
51
|
formatModelAttemptNote,
|
|
52
52
|
isRetryableModelFailure,
|
|
53
53
|
type ModelAttemptSummary,
|
|
54
|
+
type ModelFallbackPolicy,
|
|
55
|
+
resolveDefaultSubagentModel,
|
|
56
|
+
resolveModelFallbackPolicy,
|
|
57
|
+
warnOutOfScopeSoft,
|
|
54
58
|
} from "../model/model-fallback.ts";
|
|
55
59
|
import { readEnabledModelsPatterns } from "../model/model-scope.ts";
|
|
56
60
|
import { type ParsedPiJsonOutput, parsePiJsonOutput } from "../output/pi-json-output.ts";
|
|
@@ -144,6 +148,27 @@ export function computeSpawnBudgetMax(attemptModelsCount: number, configuredMaxA
|
|
|
144
148
|
return attemptModelsCount * (configuredMaxAttempts + 1);
|
|
145
149
|
}
|
|
146
150
|
|
|
151
|
+
/**
|
|
152
|
+
* Resolve the model fallback policy from crew config + env. Best-effort: any
|
|
153
|
+
* config read failure returns undefined (legacy unbounded/unordered behaviour).
|
|
154
|
+
*/
|
|
155
|
+
function resolveTaskModelFallbackPolicy(cwd: string): ModelFallbackPolicy | undefined {
|
|
156
|
+
try {
|
|
157
|
+
return resolveModelFallbackPolicy(loadConfig(cwd).config.runtime?.modelFallback);
|
|
158
|
+
} catch {
|
|
159
|
+
return undefined;
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
/** Resolve the default subagent model from crew config + env. */
|
|
164
|
+
function resolveTaskDefaultSubagentModel(cwd: string): string | undefined {
|
|
165
|
+
try {
|
|
166
|
+
return resolveDefaultSubagentModel(loadConfig(cwd).config.runtime?.modelFallback);
|
|
167
|
+
} catch {
|
|
168
|
+
return undefined;
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
|
|
147
172
|
/**
|
|
148
173
|
* 429/rate-limit detection (PI_CREW_TOOLING_429_NOTE.md).
|
|
149
174
|
*
|
|
@@ -256,19 +281,46 @@ export async function runChildProcessTask(ctx: TaskExecutionContext): Promise<Ta
|
|
|
256
281
|
let terminalEvidence: OperationTerminalEvidence[] = [];
|
|
257
282
|
let startupEvidence = ctx.startupEvidence;
|
|
258
283
|
|
|
284
|
+
const modelFallbackPolicy = resolveTaskModelFallbackPolicy(task.cwd);
|
|
285
|
+
const defaultSubagentModel = resolveTaskDefaultSubagentModel(task.cwd);
|
|
259
286
|
const modelRoutingPlan = buildConfiguredModelRouting({
|
|
260
287
|
overrideModel: input.modelOverride,
|
|
261
288
|
stepModel: input.step.model,
|
|
262
289
|
teamRoleModel: input.teamRoleModel,
|
|
290
|
+
teamRoleFallbackModels: input.teamRoleFallbackModels,
|
|
263
291
|
agentModel: input.agent.model,
|
|
292
|
+
defaultSubagentModel,
|
|
264
293
|
fallbackModels: input.agent.fallbackModels,
|
|
265
294
|
parentModel: input.parentModel,
|
|
266
295
|
modelRegistry: input.modelRegistry,
|
|
267
296
|
cwd: task.cwd,
|
|
297
|
+
policy: modelFallbackPolicy,
|
|
268
298
|
scopeModelsPatterns: await resolveTaskScopeModelsPatterns(task.cwd),
|
|
269
299
|
});
|
|
270
300
|
const candidates = modelRoutingPlan.candidates;
|
|
271
|
-
|
|
301
|
+
// Surface a warning when the caller's requested model was silently replaced.
|
|
302
|
+
if (modelRoutingPlan.droppedRequested) {
|
|
303
|
+
void appendEventAsync(manifest.eventsPath, {
|
|
304
|
+
type: "task.model_dropped",
|
|
305
|
+
runId: manifest.runId,
|
|
306
|
+
taskId: task.id,
|
|
307
|
+
message: `Requested model "${modelRoutingPlan.droppedRequested}" is not available; using "${candidates[0] ?? "default"}" instead.`,
|
|
308
|
+
data: {
|
|
309
|
+
requested: modelRoutingPlan.droppedRequested,
|
|
310
|
+
resolved: candidates[0],
|
|
311
|
+
fallbackChain: candidates,
|
|
312
|
+
},
|
|
313
|
+
}).catch(() => {
|
|
314
|
+
/* no-op: best-effort diagnostic append, ignore delivery errors */
|
|
315
|
+
});
|
|
316
|
+
}
|
|
317
|
+
// F2: surface a non-silent warning when the INITIAL routing resolved an
|
|
318
|
+
// out-of-scope soft-sourced model (mirrors live-session-runtime.ts). Caller
|
|
319
|
+
// sources already throw inside buildConfiguredModelRouting.
|
|
320
|
+
warnOutOfScopeSoft(modelRoutingPlan.scopeVerdict, "child-executor.initial-out-of-scope");
|
|
321
|
+
// Mutable: the one-shot re-resolve below appends a late-discovered model so
|
|
322
|
+
// the loop actually retries it (see FIX 1 at the end of the attempt loop).
|
|
323
|
+
const attemptModels: (string | undefined)[] = candidates.length > 0 ? [...candidates] : [undefined];
|
|
272
324
|
// CORE-3: auto-compute per-task spawn budget on first entry.
|
|
273
325
|
// Budget = attemptModels.length × (maxAttempts + 1) — always one
|
|
274
326
|
// full attempt-worth above the theoretical maximum of
|
|
@@ -283,6 +335,9 @@ export async function runChildProcessTask(ctx: TaskExecutionContext): Promise<Ta
|
|
|
283
335
|
}
|
|
284
336
|
const logs: string[] = [];
|
|
285
337
|
let finalStderr = "";
|
|
338
|
+
// One-shot: the re-resolve is a safety net for a chain that was computed
|
|
339
|
+
// before the registry was complete, not a retry loop of its own.
|
|
340
|
+
let reResolveUsed = false;
|
|
286
341
|
modelAttempts = [];
|
|
287
342
|
let finalCheckpointWritten = false;
|
|
288
343
|
let lastAgentRecordPersistedAt = 0;
|
|
@@ -440,6 +495,7 @@ export async function runChildProcessTask(ctx: TaskExecutionContext): Promise<Ta
|
|
|
440
495
|
excludeContextBash: input.runtimeConfig?.excludeContextBash,
|
|
441
496
|
sessionId: manifest.sessionId,
|
|
442
497
|
role: task.role,
|
|
498
|
+
thinkingOverride: input.teamRoleThinking,
|
|
443
499
|
runId: manifest.runId,
|
|
444
500
|
agentId: task.id,
|
|
445
501
|
artifactsRoot: manifest.artifactsRoot,
|
|
@@ -666,7 +722,8 @@ export async function runChildProcessTask(ctx: TaskExecutionContext): Promise<Ta
|
|
|
666
722
|
// after the precompute, or the precompute ran before the parent
|
|
667
723
|
// model was known). If a different candidate is found, use it as
|
|
668
724
|
// nextModel; otherwise fall through to the existing break.
|
|
669
|
-
if (!nextModel && isRetryableModelFailure(error)) {
|
|
725
|
+
if (!nextModel && !reResolveUsed && isRetryableModelFailure(error)) {
|
|
726
|
+
reResolveUsed = true;
|
|
670
727
|
const reResolved = buildConfiguredModelRouting({
|
|
671
728
|
overrideModel: undefined,
|
|
672
729
|
stepModel: undefined,
|
|
@@ -676,10 +733,29 @@ export async function runChildProcessTask(ctx: TaskExecutionContext): Promise<Ta
|
|
|
676
733
|
parentModel: attempt.model,
|
|
677
734
|
modelRegistry: input.modelRegistry,
|
|
678
735
|
cwd: task.cwd,
|
|
736
|
+
policy: modelFallbackPolicy,
|
|
679
737
|
scopeModelsPatterns: await resolveTaskScopeModelsPatterns(task.cwd),
|
|
680
738
|
});
|
|
681
|
-
|
|
682
|
-
|
|
739
|
+
// Sec-M1: surface a non-silent warning when a soft-sourced re-resolved
|
|
740
|
+
// model is out-of-scope. Only non-caller sources (frontmatter /
|
|
741
|
+
// resolved) reach here — caller sources already throw inside
|
|
742
|
+
// buildConfiguredModelRouting.
|
|
743
|
+
warnOutOfScopeSoft(reResolved.scopeVerdict, "child-executor.re-resolve-out-of-scope", "Re-resolved model");
|
|
744
|
+
// Must exclude EVERY model already tried, not just the last one —
|
|
745
|
+
// otherwise the "alternative" is one that already failed.
|
|
746
|
+
const tried = new Set(modelAttempts.map((a) => a.model));
|
|
747
|
+
const alt = reResolved.candidates.find((candidate) => !tried.has(candidate));
|
|
748
|
+
if (alt) {
|
|
749
|
+
// Append so the loop reaches it. Without this the log claimed a
|
|
750
|
+
// retry that never happened: `nextModel` alone only gates `break`,
|
|
751
|
+
// the next iteration reads `attemptModels[i + 1]`.
|
|
752
|
+
attemptModels.push(alt);
|
|
753
|
+
nextModel = alt;
|
|
754
|
+
// Keep the budget consistent with the extended chain (one extra
|
|
755
|
+
// model = one extra attempt-worth), otherwise the spawn-budget
|
|
756
|
+
// guard aborts the very attempt we just queued.
|
|
757
|
+
if (input.spawnBudget) input.spawnBudget.max += 1;
|
|
758
|
+
}
|
|
683
759
|
}
|
|
684
760
|
if (!nextModel || !isRetryableModelFailure(error)) break;
|
|
685
761
|
logs.push(formatModelAttemptNote(attempt, nextModel), "");
|
|
@@ -762,6 +838,8 @@ export async function runChildProcessTask(ctx: TaskExecutionContext): Promise<Ta
|
|
|
762
838
|
fallbackChain: candidates,
|
|
763
839
|
reason: fallbackReason ?? modelRoutingPlan.reason,
|
|
764
840
|
usedAttempt,
|
|
841
|
+
droppedRequested: modelRoutingPlan.droppedRequested,
|
|
842
|
+
autoFallbackCount: modelRoutingPlan.autoFallbackCount,
|
|
765
843
|
},
|
|
766
844
|
};
|
|
767
845
|
tasks = updateTask(tasks, task);
|
|
@@ -35,6 +35,8 @@ export interface RunLiveTaskInput {
|
|
|
35
35
|
modelRegistry?: unknown;
|
|
36
36
|
modelOverride?: string;
|
|
37
37
|
teamRoleModel?: string;
|
|
38
|
+
teamRoleFallbackModels?: string[];
|
|
39
|
+
teamRoleThinking?: string;
|
|
38
40
|
isCurrent?: () => boolean;
|
|
39
41
|
/** Workspace where this task run was initiated — used for session-scoped live-agent visibility. */
|
|
40
42
|
workspaceId: string;
|
|
@@ -134,6 +136,8 @@ export async function runLiveTask(input: RunLiveTaskInput): Promise<RunLiveTaskO
|
|
|
134
136
|
modelRegistry: input.modelRegistry,
|
|
135
137
|
modelOverride: input.modelOverride,
|
|
136
138
|
teamRoleModel: input.teamRoleModel,
|
|
139
|
+
teamRoleFallbackModels: input.teamRoleFallbackModels,
|
|
140
|
+
teamRoleThinking: input.teamRoleThinking,
|
|
137
141
|
isCurrent,
|
|
138
142
|
workspaceId: input.workspaceId,
|
|
139
143
|
// Phase 2: Pass output schema for yield validation
|
|
@@ -181,6 +185,14 @@ export async function runLiveTask(input: RunLiveTaskInput): Promise<RunLiveTaskO
|
|
|
181
185
|
usage: liveResult.usage,
|
|
182
186
|
agentProgress: applyUsageToProgress(task.agentProgress, liveResult.usage),
|
|
183
187
|
};
|
|
188
|
+
if (liveResult.modelRouting)
|
|
189
|
+
task = {
|
|
190
|
+
...task,
|
|
191
|
+
modelRouting: {
|
|
192
|
+
...liveResult.modelRouting,
|
|
193
|
+
usedAttempt: 0,
|
|
194
|
+
},
|
|
195
|
+
};
|
|
184
196
|
persistLiveProgress({ type: "attempt_finished" }, true);
|
|
185
197
|
const resultArtifact = writeArtifact(manifest.artifactsRoot, {
|
|
186
198
|
kind: "result",
|
|
@@ -52,6 +52,9 @@ export interface TaskRunnerInput {
|
|
|
52
52
|
modelRegistry?: unknown;
|
|
53
53
|
modelOverride?: string;
|
|
54
54
|
teamRoleModel?: string;
|
|
55
|
+
teamRoleFallbackModels?: string[];
|
|
56
|
+
/** Per-role thinking level override (teamRole.thinking). Takes precedence over agent.thinking. */
|
|
57
|
+
teamRoleThinking?: string;
|
|
55
58
|
teamRoleSkills?: string[] | false;
|
|
56
59
|
skillOverride?: string[] | false;
|
|
57
60
|
limits?: CrewLimitsConfig;
|
|
@@ -152,6 +155,8 @@ export async function runTeamTask(input: TaskRunnerInput): Promise<{ manifest: T
|
|
|
152
155
|
modelRegistry: input.modelRegistry,
|
|
153
156
|
modelOverride: input.modelOverride,
|
|
154
157
|
teamRoleModel: input.teamRoleModel,
|
|
158
|
+
teamRoleFallbackModels: input.teamRoleFallbackModels,
|
|
159
|
+
teamRoleThinking: input.teamRoleThinking,
|
|
155
160
|
workspaceId: input.workspaceId,
|
|
156
161
|
});
|
|
157
162
|
task = live.task;
|
|
@@ -1102,6 +1102,10 @@ interface SchedulerContext {
|
|
|
1102
1102
|
queueIndex: TaskGraphIndex;
|
|
1103
1103
|
wfMachine: WorkflowStateMachine;
|
|
1104
1104
|
pendingUnits: Map<string, PendingUnit>;
|
|
1105
|
+
/** Task ids ever dispatched (grows monotonically; never removed). Used by
|
|
1106
|
+
* terminaliseRunWithDrain to cancel — not skip — tasks that were in-flight
|
|
1107
|
+
* even after their dispatch unit settled + left pendingUnits (RT-NEW-2 race). */
|
|
1108
|
+
dispatchedTaskIds: Set<string>;
|
|
1105
1109
|
runController: AbortController;
|
|
1106
1110
|
runtimeKind: CrewRuntimeKind;
|
|
1107
1111
|
adaptivePlanInjected: boolean;
|
|
@@ -1266,10 +1270,13 @@ async function terminaliseRunWithDrain(
|
|
|
1266
1270
|
ctx: SchedulerContext,
|
|
1267
1271
|
opts: { cancelMessage: string; blockedMessage: string; failedReason: string },
|
|
1268
1272
|
): Promise<{ manifest: TeamRunManifest; tasks: TeamTaskState[] }> {
|
|
1269
|
-
|
|
1270
|
-
|
|
1271
|
-
|
|
1272
|
-
|
|
1273
|
+
// Ever-dispatched tasks (monotonic set populated at dispatch). Using this
|
|
1274
|
+
// instead of a pendingUnits snapshot closes the RT-NEW-2 race where a task
|
|
1275
|
+
// whose unit settled + left pendingUnits before the abort — but whose task
|
|
1276
|
+
// status isn't terminal yet — would otherwise fall through to markBlocked
|
|
1277
|
+
// and be clobbered to "skipped" (observed CI flake: 02_b skipped on
|
|
1278
|
+
// team-runner-budget-abort-inflight across v0.9.59 / cfd68d06 / 12386af2).
|
|
1279
|
+
const inflightTaskIds = ctx.dispatchedTaskIds;
|
|
1273
1280
|
const outcomes = await drainPendingUnits(ctx.pendingUnits, ctx.runController);
|
|
1274
1281
|
const validResults: { manifest: TeamRunManifest; tasks: TeamTaskState[] }[] = [];
|
|
1275
1282
|
for (const outcome of outcomes) {
|
|
@@ -1729,6 +1736,8 @@ async function dispatchBatch(ctx: SchedulerContext, decision: DispatchBatchDecis
|
|
|
1729
1736
|
modelRegistry: input.modelRegistry,
|
|
1730
1737
|
modelOverride: input.modelOverride,
|
|
1731
1738
|
teamRoleModel: teamRole?.model,
|
|
1739
|
+
teamRoleThinking: teamRole?.thinking,
|
|
1740
|
+
teamRoleFallbackModels: teamRole?.fallbackModels,
|
|
1732
1741
|
teamRoleSkills: teamRole?.skills,
|
|
1733
1742
|
skillOverride: input.skillOverride,
|
|
1734
1743
|
limits: input.limits,
|
|
@@ -1926,6 +1935,10 @@ async function dispatchBatch(ctx: SchedulerContext, decision: DispatchBatchDecis
|
|
|
1926
1935
|
promise: rawPromise,
|
|
1927
1936
|
wrapped,
|
|
1928
1937
|
});
|
|
1938
|
+
// RT-NEW-2 race fix: record ever-dispatched task ids so terminaliseRunWithDrain
|
|
1939
|
+
// cancels (not skips) tasks whose unit settled + left pendingUnits before
|
|
1940
|
+
// the abort fired but whose task status isn't terminal yet.
|
|
1941
|
+
for (const id of unitTaskIds) ctx.dispatchedTaskIds.add(id);
|
|
1929
1942
|
}
|
|
1930
1943
|
}
|
|
1931
1944
|
|
|
@@ -2462,6 +2475,7 @@ async function executeTeamRunCore(
|
|
|
2462
2475
|
queueIndex,
|
|
2463
2476
|
wfMachine,
|
|
2464
2477
|
pendingUnits,
|
|
2478
|
+
dispatchedTaskIds: new Set(),
|
|
2465
2479
|
runController,
|
|
2466
2480
|
runtimeKind,
|
|
2467
2481
|
adaptivePlanInjected,
|
|
@@ -42,6 +42,17 @@ export const PiTeamsLimitsConfigSchema = Type.Object(
|
|
|
42
42
|
{ additionalProperties: false },
|
|
43
43
|
);
|
|
44
44
|
|
|
45
|
+
export const PiTeamsModelFallbackConfigSchema = Type.Object(
|
|
46
|
+
{
|
|
47
|
+
maxAutoFallbacks: Type.Optional(Type.Integer({ minimum: 0 })),
|
|
48
|
+
order: Type.Optional(Type.Union([Type.Literal("parentFirst"), Type.Literal("asIs")])),
|
|
49
|
+
requireCredentials: Type.Optional(Type.Boolean()),
|
|
50
|
+
quotaAwareOrdering: Type.Optional(Type.Boolean()),
|
|
51
|
+
defaultSubagentModel: Type.Optional(Type.String({ minLength: 1 })),
|
|
52
|
+
},
|
|
53
|
+
{ additionalProperties: false },
|
|
54
|
+
);
|
|
55
|
+
|
|
45
56
|
export const PiTeamsRuntimeConfigSchema = Type.Object(
|
|
46
57
|
{
|
|
47
58
|
mode: Type.Optional(
|
|
@@ -82,6 +93,7 @@ export const PiTeamsRuntimeConfigSchema = Type.Object(
|
|
|
82
93
|
{ additionalProperties: false },
|
|
83
94
|
),
|
|
84
95
|
),
|
|
96
|
+
modelFallback: Type.Optional(PiTeamsModelFallbackConfigSchema),
|
|
85
97
|
},
|
|
86
98
|
{ additionalProperties: false },
|
|
87
99
|
);
|